diff --git a/src/app/api/chat/route.ts b/src/app/api/chat/route.ts index 18923c4..f4c092d 100644 --- a/src/app/api/chat/route.ts +++ b/src/app/api/chat/route.ts @@ -47,7 +47,7 @@ const zen = (sessionId: string) => }, }); -const MODEL_ID = 'glm-5.3-flash'; +const MODEL_ID = 'gpt-6-luna'; const model = (sessionId: string) => zen(sessionId).chatModel(MODEL_ID); diff --git a/src/lib/abuse/__tests__/cost.test.ts b/src/lib/abuse/__tests__/cost.test.ts index e54a492..0fa9a80 100644 --- a/src/lib/abuse/__tests__/cost.test.ts +++ b/src/lib/abuse/__tests__/cost.test.ts @@ -11,8 +11,8 @@ describe('estimateTokens', () => { describe('calculateCost', () => { it('scales tokens to USD', () => { - expect(calculateCost(1_000_000)).toBeCloseTo(0.28); - expect(calculateCost(500_000)).toBeCloseTo(0.14); + expect(calculateCost(1_000_000)).toBeCloseTo(0.5); + expect(calculateCost(500_000)).toBeCloseTo(0.25); }); }); @@ -31,7 +31,7 @@ describe('checkCostLimit', () => { vi.useFakeTimers(); vi.setSystemTime(new Date('2026-01-01T00:00:00Z')); for (let i = 0; i < 6; i++) { - checkCostLimit('u1', 300_000); // ~$0.084 each + checkCostLimit('u1', 300_000); // ~$0.15 each } const result = checkCostLimit('u1', 300_000); expect(result.allowed).toBe(false); @@ -52,9 +52,9 @@ describe('recordActualUsage', () => { it('corrects the last estimate with actual usage', () => { checkCostLimit('u1', 40_000); recordActualUsage('u1', 10_000); - // Exhausting the budget now: ~$0.0589 spent (actual) + next estimate must - // still fit — verifies the estimate was replaced, not appended. - const result = checkCostLimit('u1', 1_000_000); + // ~$0.005 spent (actual) + next estimate must still fit — verifies the + // estimate was replaced, not appended. + const result = checkCostLimit('u1', 900_000); expect(result.allowed).toBe(true); }); diff --git a/src/lib/abuse/cost.ts b/src/lib/abuse/cost.ts index e4112b7..c47d24e 100644 --- a/src/lib/abuse/cost.ts +++ b/src/lib/abuse/cost.ts @@ -8,7 +8,7 @@ export const MAX_TOKENS_PER_REQUEST = 4000; const MAX_COST_PER_HOUR_USD = 0.5; // ~$0.50/hour budget -const TOKEN_COST_PER_1M = 0.28; // glm-5.3-flash on OpenCode Go (conservative upper bound) +const TOKEN_COST_PER_1M = 0.5; // gpt-6-luna on OpenCode Zen (output upper bound) const USAGE_WINDOW_MS = 60 * 60 * 1000; // 1 hour const CLEANUP_INTERVAL_MS = 300_000; // 5 min