diff --git a/src/backend/drivers/ai-chat/ChatCompletionDriver.test.ts b/src/backend/drivers/ai-chat/ChatCompletionDriver.test.ts index 52b46a806..4cd4fbeef 100644 --- a/src/backend/drivers/ai-chat/ChatCompletionDriver.test.ts +++ b/src/backend/drivers/ai-chat/ChatCompletionDriver.test.ts @@ -495,6 +495,43 @@ describe('ChatCompletionDriver.complete events and cost emission', () => { expect(res.usage.usd_cents).toBe(expectedMicroCents / 1_000_000); }); + it('prices `usd_cents` at long-context rates once input passes the threshold', async () => { + vi.spyOn(FakeChatProvider.prototype, 'models').mockResolvedValueOnce([ + { + id: 'priced', + aliases: [], + costs_currency: 'usd-cents', + costs: { input_tokens: 1000, output_tokens: 2000 }, + long_context_pricing: { + threshold: 5, + input_multiplier: 2, + output_multiplier: 1.5, + }, + max_tokens: 8192, + }, + ]); + const d = await makeDriver(); + + vi.spyOn(FakeChatProvider.prototype, 'complete').mockResolvedValueOnce({ + message: { + role: 'assistant', + content: [{ type: 'text', text: 'ok' }], + }, + usage: { input_tokens: 10, output_tokens: 7 }, + finish_reason: 'stop', + } as never); + + const res = (await withTestActor(() => + d.complete({ + model: 'priced', + messages: [{ role: 'user', content: 'hi' }], + }), + )) as { usage: Record }; + + const expectedMicroCents = 10 * 1000 * 2 + 7 * 2000 * 1.5; + expect(res.usage.usd_cents).toBe(expectedMicroCents / 1_000_000); + }); + it('does not override `usd_cents` when the provider already returned one (e.g. OpenRouter)', async () => { vi.spyOn(FakeChatProvider.prototype, 'complete').mockResolvedValueOnce({ message: { @@ -664,6 +701,54 @@ describe('ChatCompletionDriver.complete credit gate and max_tokens cap', () => { expect(passed.max_tokens!).toBeGreaterThan(0); }); + it('caps `max_tokens` at the long-context output rate for a prompt past the threshold', async () => { + vi.spyOn(FakeChatProvider.prototype, 'models').mockResolvedValueOnce([ + { + id: 'capme', + aliases: [], + costs_currency: 'usd-cents', + costs: { input_tokens: 1000, output_tokens: 2000 }, + long_context_pricing: { + threshold: 10, + input_multiplier: 2, + output_multiplier: 1.5, + }, + max_tokens: 8192, + }, + ]); + const d = await makeDriver(); + + // A ~100-token prompt is past the threshold. 1_000_000 microcents at + // 2000 * 1.5 per output token leaves at most 333 output tokens before + // the prompt is paid for; the standard rate would allow ~450 after it. + vi.spyOn(server.services.metering, 'getRemainingUsage').mockResolvedValue( + 1_000_000, + ); + + const completeSpy = vi + .spyOn(FakeChatProvider.prototype, 'complete') + .mockResolvedValueOnce({ + message: { + role: 'assistant', + content: [{ type: 'text', text: 'ok' }], + }, + usage: { input_tokens: 1, output_tokens: 1 }, + finish_reason: 'stop', + } as never); + + await withTestActor(() => + d.complete({ + model: 'capme', + messages: [{ role: 'user', content: 'x'.repeat(400) }], + max_tokens: 10_000, + }), + ); + + const passed = completeSpy.mock.calls[0]![0] as ICompleteArguments; + expect(passed.max_tokens!).toBeLessThanOrEqual(333); + expect(passed.max_tokens!).toBeGreaterThan(0); + }); + it('throws 402 instead of leaving `max_tokens` unset when credits cannot afford one output token', async () => { // Regression: previously a sub-1 cap set max_tokens to `undefined`, // which let the provider run to the model's full output limit and diff --git a/src/backend/drivers/ai-chat/ChatCompletionDriver.ts b/src/backend/drivers/ai-chat/ChatCompletionDriver.ts index 78f59c832..eaeccd8cf 100644 --- a/src/backend/drivers/ai-chat/ChatCompletionDriver.ts +++ b/src/backend/drivers/ai-chat/ChatCompletionDriver.ts @@ -86,7 +86,13 @@ import { normalizeResultToOpenAI, shouldPresentAsOpenAI, } from './utils/normalizeToOpenAI.js'; -import { costKeys, isFreeModel } from './utils/pricing.js'; +import { + costKeys, + isFreeModel, + isOutputCostKey, + longContextMultipliers, + trackedInputTokens, +} from './utils/pricing.js'; import { isRouteUnhealthy, markRouteUnhealthy, @@ -813,11 +819,11 @@ export class ChatCompletionDriver extends PuterDriver { ? outputRateRaw : undefined; - const isOutputKey = (key: string) => - key === outputKey || - key === 'output_tokens' || - key === 'completion_tokens' || - key === 'thinking_tokens'; + const isOutputKey = (key: string) => isOutputCostKey(key, outputKey); + const multipliers = longContextMultipliers( + model, + trackedInputTokens(usage, model), + ); let inputMicroCents = 0; let outputMicroCents = 0; @@ -851,12 +857,11 @@ export class ChatCompletionDriver extends PuterDriver { } } - const sub = rawAmount * rate; sawAnyRate = true; if (isOutputKey(key)) { - outputMicroCents += sub; + outputMicroCents += rawAmount * rate * multipliers.output; } else { - inputMicroCents += sub; + inputMicroCents += rawAmount * rate * multipliers.input; } } @@ -923,9 +928,14 @@ export class ChatCompletionDriver extends PuterDriver { const metering = this.services.metering; const { promptTokenEstimate, requestedMaxTokens } = estimates; const { inputKey, outputKey } = costKeys(model); + // A prompt estimated past a long-context threshold pays the raised + // rates on input and output alike. + const multipliers = longContextMultipliers(model, promptTokenEstimate); // `|| 0` also catches NaN from a malformed cost table. - const inputTokenCost = Number(model.costs?.[inputKey] ?? 0) || 0; - const outputTokenCost = Number(model.costs?.[outputKey] ?? 0) || 0; + const inputTokenCost = + (Number(model.costs?.[inputKey] ?? 0) || 0) * multipliers.input; + const outputTokenCost = + (Number(model.costs?.[outputKey] ?? 0) || 0) * multipliers.output; const approximateInputCost = promptTokenEstimate * inputTokenCost; const minimumCredits = Number(model.minimumCredits || 1); diff --git a/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.test.ts b/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.test.ts index c6ba39e5f..3e928fcbc 100644 --- a/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.test.ts +++ b/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.test.ts @@ -790,44 +790,48 @@ describe('ClaudeProvider.complete request shape', () => { }); }); - it('omits temperature for fable 5.1 (rejects non-default sampling)', async () => { - const { provider } = makeProvider(); - messagesCreateMock.mockResolvedValueOnce(baseResponse); + it.each(['claude-fable-5-1', 'claude-opus-5-5', 'claude-sonnet-5'])( + 'omits temperature for %s', + async (model) => { + const { provider } = makeProvider(); + messagesCreateMock.mockResolvedValueOnce(baseResponse); - await withTestActor(() => - provider.complete({ - model: 'claude-fable-5-1', - messages: [{ role: 'user', content: 'hi' }], - temperature: 0.5, - }), - ); + await withTestActor(() => + provider.complete({ + model, + messages: [{ role: 'user', content: 'hi' }], + temperature: 0.5, + }), + ); - const [args] = messagesCreateMock.mock.calls[0]!; - expect('temperature' in args).toBe(false); - }); + const [args] = messagesCreateMock.mock.calls[0]!; + expect('temperature' in args).toBe(false); + }, + ); - it('forwards reasoning_effort as adaptive thinking + output_config effort on fable 5.1', async () => { - const { provider } = makeProvider(); - messagesCreateMock.mockResolvedValueOnce(baseResponse); + it.each(['claude-fable-5-1', 'claude-opus-5-5', 'claude-sonnet-5'])( + 'uses adaptive thinking and output effort on %s', + async (model) => { + const { provider } = makeProvider(); + messagesCreateMock.mockResolvedValueOnce(baseResponse); - await withTestActor(() => - provider.complete({ - model: 'claude-fable-5-1', - messages: [{ role: 'user', content: 'hi' }], - reasoning_effort: 'high', - } as never), - ); + await withTestActor(() => + provider.complete({ + model, + messages: [{ role: 'user', content: 'hi' }], + reasoning_effort: 'high', + } as never), + ); - const [args] = messagesCreateMock.mock.calls[0]!; - // Fable 5.1 rejects `budget_tokens` and thinking is always on, so the - // only accepted config is adaptive; effort rides in output_config. - expect(args.thinking).toEqual({ - type: 'adaptive', - display: 'summarized', - }); - expect(args.output_config).toEqual({ effort: 'high' }); - expect('temperature' in args).toBe(false); - }); + const [args] = messagesCreateMock.mock.calls[0]!; + expect(args.thinking).toEqual({ + type: 'adaptive', + display: 'summarized', + }); + expect(args.output_config).toEqual({ effort: 'high' }); + expect('temperature' in args).toBe(false); + }, + ); it('forwards reasoning_effort as adaptive thinking + output_config effort on opus 5.5', async () => { const { provider } = makeProvider(); @@ -957,6 +961,60 @@ describe('ClaudeProvider model resolution', () => { // ── Non-stream completion ─────────────────────────────────────────── describe('ClaudeProvider.complete non-stream output', () => { + it.each([ + ['claude-opus-5-5', 'claude-opus-5-5', 400, 500, 800, 20, 2000], + [ + 'anthropic/claude-opus-5-5', + 'claude-opus-5-5', + 400, + 500, + 800, + 20, + 2000, + ], + ['claude-opus', 'claude-opus-5-5', 400, 500, 800, 20, 2000], + ['claude-opus-latest', 'claude-opus-5-5', 400, 500, 800, 20, 2000], + ['claude-opus-5-latest', 'claude-opus-5', 500, 625, 1000, 50, 2500], + ['claude-sonnet-5', 'claude-sonnet-5', 200, 250, 400, 20, 1000], + ])( + 'resolves and meters %s at its current rates', + async (model, canonicalId, input, write5m, write1h, cached, output) => { + const { provider } = makeProvider(); + messagesCreateMock.mockResolvedValueOnce({ + content: [{ type: 'text', text: 'ok' }], + usage: { + input_tokens: 100, + output_tokens: 50, + cache_creation_input_tokens: 30, + cache_creation: { + ephemeral_5m_input_tokens: 10, + ephemeral_1h_input_tokens: 20, + }, + cache_read_input_tokens: 1000, + }, + }); + await withTestActor(() => + provider.complete({ + model, + messages: [{ role: 'user', content: 'hi' }], + }), + ); + expect(await provider.list()).toContain(model); + const [args] = messagesCreateMock.mock.calls[0]!; + expect(args.model).toBe(canonicalId); + expect(args.max_tokens).toBe(128_000); + const [, , prefix, overrides] = recordSpy.mock.calls[0]!; + expect(prefix).toBe(`claude:${canonicalId}`); + expect(overrides).toMatchObject({ + input_tokens: 100 * Number(input), + output_tokens: 50 * Number(output), + ephemeral_5m_input_tokens: 10 * Number(write5m), + ephemeral_1h_input_tokens: 20 * Number(write1h), + cache_read_input_tokens: 1000 * Number(cached), + }); + }, + ); + it('returns the message verbatim and meters input/output/cache token costs', async () => { const { provider } = makeProvider(); const msg = { diff --git a/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.ts b/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.ts index 8b61360ad..e6e9adcec 100644 --- a/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.ts +++ b/src/backend/drivers/ai-chat/providers/claude/ClaudeProvider.ts @@ -806,14 +806,12 @@ export class ClaudeProvider implements IChatProvider { }) { if (!reasoningEffort) return undefined; - // Fable 5/5.1, Opus 4.7+, 4.6, and Sonnet 4.6 use adaptive thinking - // (`budget_tokens` is deprecated on 4.6/Sonnet 4.6, removed on - // Fable 5+ and Opus 4.7+). Fable 5/5.1 and Opus 4.7+ omit thinking - // content by default; `display: 'summarized'` restores visible - // reasoning in the stream. + // These models reject manual thinking budgets; summarized display + // keeps reasoning visible in the stream. if ( modelId === 'claude-fable-5-1' || modelId === 'claude-fable-5' || + modelId === 'claude-sonnet-5' || modelId === 'claude-opus-5-5' || modelId === 'claude-opus-5' || modelId === 'claude-opus-4-8' || diff --git a/src/backend/drivers/ai-chat/providers/claude/models.ts b/src/backend/drivers/ai-chat/providers/claude/models.ts index d43ef9007..369012601 100644 --- a/src/backend/drivers/ai-chat/providers/claude/models.ts +++ b/src/backend/drivers/ai-chat/providers/claude/models.ts @@ -44,7 +44,7 @@ export const CLAUDE_MODELS: IChatModel[] = [ input_tokens: 1000, ephemeral_5m_input_tokens: 1000 * 1.25, ephemeral_1h_input_tokens: 1000 * 2, - // Fable 5.1 bills cache reads at 0.025x input; every other Claude model is 0.1x. + // Fable 5.1 bills cache reads at 0.025x input. cache_read_input_tokens: 1000 * 0.025, output_tokens: 5000, }, @@ -80,6 +80,7 @@ export const CLAUDE_MODELS: IChatModel[] = [ modalities: { input: ['text', 'image', 'pdf'], output: ['text'] }, open_weights: false, tool_call: true, + knowledge: '2026-01', release_date: '2026-06-30', aliases: [ 'claude-sonnet', @@ -93,14 +94,14 @@ export const CLAUDE_MODELS: IChatModel[] = [ output_cost_key: 'output_tokens', costs: { tokens: 1_000_000, - input_tokens: 300, - ephemeral_5m_input_tokens: 300 * 1.25, - ephemeral_1h_input_tokens: 300 * 2, - cache_read_input_tokens: 300 * 0.1, - output_tokens: 1500, + input_tokens: 200, + ephemeral_5m_input_tokens: 200 * 1.25, + ephemeral_1h_input_tokens: 200 * 2, + cache_read_input_tokens: 200 * 0.1, + output_tokens: 1000, }, context: 1000000, - max_tokens: 64000, + max_tokens: 128000, }, { puterId: 'anthropic:anthropic/claude-opus-5-5', @@ -126,7 +127,7 @@ export const CLAUDE_MODELS: IChatModel[] = [ input_tokens: 400, ephemeral_5m_input_tokens: 400 * 1.25, ephemeral_1h_input_tokens: 400 * 2, - cache_read_input_tokens: 400 * 0.1, + cache_read_input_tokens: 400 * 0.05, output_tokens: 2000, }, context: 1000000, diff --git a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.test.ts b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.test.ts index ec5ed4238..b4ae8f3f3 100644 --- a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.test.ts +++ b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.test.ts @@ -171,6 +171,8 @@ describe('OpenAiChatProvider model catalog', () => { // gpt-5-nano is a Chat-Completions model, must be present. expect(ids).toContain('gpt-5-nano-2025-08-07'); expect(ids).toContain('gpt-6-astra'); + expect(ids).toContain('gpt-6-sol'); + expect(ids).toContain('gpt-6-luna'); }); it('list() flattens canonical ids and aliases', () => { @@ -315,25 +317,30 @@ describe('OpenAiChatProvider.complete request shape', () => { expect(args.safety_identifier).toBe('puter-u42'); }); - it('resolves the namespaced GPT-6 Astra alias', async () => { - const { provider } = makeProvider(); - createMock.mockResolvedValueOnce(baseCompletion); + it.each(['gpt-6-astra', 'gpt-6-sol', 'gpt-6-luna'])( + 'resolves the namespaced %s alias', + async (model) => { + const { provider } = makeProvider(); + createMock.mockResolvedValueOnce(baseCompletion); - await withTestActor(() => - provider.complete({ - model: 'openai/gpt-6-astra', - messages: [{ role: 'user', content: 'hello' }], - }), - ); + await withTestActor(() => + provider.complete({ + model: `openai/${model}`, + messages: [{ role: 'user', content: 'hello' }], + reasoning_effort: 'low', + }), + ); - expect(createMock.mock.calls[0]![0].model).toBe('gpt-6-astra'); - expect(recordSpy).toHaveBeenCalledWith( - expect.any(Object), - expect.anything(), - 'openai:gpt-6-astra', - expect.any(Object), - ); - }); + expect(createMock.mock.calls[0]![0].model).toBe(model); + expect(createMock.mock.calls[0]![0].reasoning_effort).toBe('low'); + expect(recordSpy).toHaveBeenCalledWith( + expect.any(Object), + expect.anything(), + `openai:${model}`, + expect.any(Object), + ); + }, + ); it('forwards temperature 0 and max_tokens 0 instead of dropping them', async () => { const { provider } = makeProvider(); @@ -514,6 +521,48 @@ describe('OpenAiChatProvider.complete non-stream output', () => { ).toBeGreaterThan(0); }); + it('splits cache writes out of prompt_tokens and bills them at 1.25x input', async () => { + const luna = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-luna')!; + const { provider } = makeProvider(); + createMock.mockResolvedValueOnce({ + choices: [ + { + message: { content: 'hi', role: 'assistant' }, + finish_reason: 'stop', + }, + ], + usage: { + prompt_tokens: 5000, + completion_tokens: 12, + prompt_tokens_details: { + cached_tokens: 1000, + cache_write_tokens: 3000, + }, + }, + }); + + await withTestActor(() => + provider.complete({ + model: 'gpt-6-luna', + messages: [{ role: 'user', content: 'hi' }], + }), + ); + + const [usage, , , overrides] = recordSpy.mock.calls[0]!; + expect(usage).toEqual({ + prompt_tokens: 1000, + completion_tokens: 12, + cached_tokens: 1000, + cache_write_tokens: 3000, + }); + expect(overrides).toEqual({ + prompt_tokens: 1000 * Number(luna.costs.prompt_tokens), + completion_tokens: 12 * Number(luna.costs.completion_tokens), + cached_tokens: 1000 * Number(luna.costs.cached_tokens), + cache_write_tokens: 3000 * Number(luna.costs.cache_write_tokens), + }); + }); + it('zeroes cached_tokens when prompt_tokens_details is missing', async () => { const { provider } = makeProvider(); createMock.mockResolvedValueOnce({ diff --git a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.ts b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.ts index 4a9c6ff3b..f3dae0dd2 100644 --- a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.ts +++ b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatCompletionsProvider.ts @@ -217,13 +217,26 @@ export class OpenAiChatProvider implements IChatProvider { return OpenAiUtil.handle_completion_output({ usage_calculator: ({ usage }) => { + const cachedTokens = + usage.prompt_tokens_details?.cached_tokens ?? 0; + // GPT-5.6 and later bill cache writes at 1.25x input. They're + // reported inside `prompt_tokens`, like cached reads. + // The SDK doesn't type `cache_write_tokens` yet. + const cacheWriteTokens = + ( + usage.prompt_tokens_details as + { cache_write_tokens?: number } | undefined + )?.cache_write_tokens ?? 0; const trackedUsage = { prompt_tokens: (usage.prompt_tokens ?? 0) - - (usage.prompt_tokens_details?.cached_tokens ?? 0), + cachedTokens - + cacheWriteTokens, completion_tokens: usage.completion_tokens ?? 0, - cached_tokens: - usage.prompt_tokens_details?.cached_tokens ?? 0, + cached_tokens: cachedTokens, + ...(cacheWriteTokens + ? { cache_write_tokens: cacheWriteTokens } + : {}), }; const costsOverrideFromModel = buildCostsOverride( diff --git a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.test.ts b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.test.ts index 2c5355ea0..784f078bc 100644 --- a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.test.ts +++ b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.test.ts @@ -198,6 +198,38 @@ describe('OpenAiResponsesChatProvider model catalog', () => { expect(ids).not.toContain('gpt-5-nano-2025-08-07'); }); + it.each([ + ['gpt-6-sol', '2026-04-20', 200, 20, 1000], + ['gpt-6-luna', '2026-05-18', 10, 1, 50], + ])( + 'exposes %s with current pricing and limits', + (id, knowledge, input, cached, output) => { + const { provider } = makeProvider(); + expect( + provider.models().find((model) => model.id === id), + ).toMatchObject({ + puterId: `openai:openai/${id}`, + aliases: [`openai/${id}`], + knowledge, + release_date: '2026-09-22', + modalities: { input: ['text', 'image'], output: ['text'] }, + costs_currency: 'usd-cents', + costs: { + tokens: 1_000_000, + prompt_tokens: input, + cached_tokens: cached, + completion_tokens: output, + }, + context: 1_050_000, + max_tokens: 128_000, + responses_api: true, + }); + expect(provider.list()).toEqual( + expect.arrayContaining([id, `openai/${id}`]), + ); + }, + ); + it('models({ no_restrictions: true }) returns the entire catalog (used by complete())', () => { const { provider } = makeProvider(); const ids = provider @@ -386,6 +418,58 @@ describe('OpenAiResponsesChatProvider.complete request shape', () => { expect(o3Args.reasoning_effort).toBe('medium'); expect(o3Args.verbosity).toBe('low'); }); + + it.each(['gpt-6-astra', 'gpt-6-sol', 'gpt-6-luna'])( + 'maps flat controls to Responses options for %s aliases', + async (model) => { + const { provider } = makeProvider(); + responsesCreateMock.mockResolvedValueOnce(baseResponse); + await withTestActor(() => + provider.complete({ + model: `openai/${model}`, + messages: [{ role: 'user', content: 'hi' }], + reasoning_effort: 'high', + verbosity: 'low', + }), + ); + const [args] = responsesCreateMock.mock.calls[0]!; + expect(args.model).toBe(model); + expect(args.reasoning).toEqual({ effort: 'high' }); + expect(args.text).toEqual({ verbosity: 'low' }); + expect(args).not.toHaveProperty('reasoning_effort'); + expect(args).not.toHaveProperty('verbosity'); + expect(recordSpy.mock.calls[0]![2]).toBe(`openai:${model}`); + }, + ); + + it('preserves nested GPT-6 controls and gives flat options precedence', async () => { + const { provider } = makeProvider(); + const reasoning = { effort: 'medium', summary: 'auto' }; + const text = { verbosity: 'high', format: { type: 'text' } }; + for (const flat of [false, true]) { + responsesCreateMock.mockResolvedValueOnce(baseResponse); + await withTestActor(() => + provider.complete({ + model: 'gpt-6-sol', + messages: [{ role: 'user', content: 'hi' }], + reasoning, + text, + ...(flat + ? { reasoning_effort: 'low', verbosity: 'low' } + : {}), + } as never), + ); + const [args] = responsesCreateMock.mock.lastCall!; + expect(args.reasoning).toEqual({ + ...reasoning, + effort: flat ? 'low' : 'medium', + }); + expect(args.text).toEqual({ + ...text, + verbosity: flat ? 'low' : 'high', + }); + } + }); }); // ── Model resolution ──────────────────────────────────────────────── @@ -518,6 +602,89 @@ describe('OpenAiResponsesChatProvider.complete non-stream output', () => { }); }); + it('splits cache writes out of input and bills them at 1.25x input', async () => { + const luna = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-luna')!; + const { provider } = makeProvider(); + responsesCreateMock.mockResolvedValueOnce({ + output: [{ role: 'assistant' }], + output_text: 'hi', + usage: { + input_tokens: 5000, + output_tokens: 20, + input_tokens_details: { + cached_tokens: 1000, + cache_write_tokens: 3000, + }, + }, + }); + + await withTestActor(() => + provider.complete({ + model: 'gpt-6-luna', + messages: [{ role: 'user', content: 'hi' }], + }), + ); + + const [usage, , , overrides] = recordSpy.mock.calls[0]!; + expect(usage).toEqual({ + prompt_tokens: 1000, + completion_tokens: 20, + cached_tokens: 1000, + cache_write_tokens: 3000, + }); + expect(luna.costs.cache_write_tokens).toBe( + Number(luna.costs.prompt_tokens) * 1.25, + ); + expect(overrides).toEqual({ + prompt_tokens: 1000 * Number(luna.costs.prompt_tokens), + completion_tokens: 20 * Number(luna.costs.completion_tokens), + cached_tokens: 1000 * Number(luna.costs.cached_tokens), + cache_write_tokens: 3000 * Number(luna.costs.cache_write_tokens), + }); + }); + + it('bills the whole request at long-context rates past 272K input tokens', async () => { + const sol = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-sol')!; + const { provider } = makeProvider(); + responsesCreateMock.mockResolvedValueOnce({ + output: [{ role: 'assistant' }], + output_text: 'hi', + usage: { + input_tokens: 300_000, + output_tokens: 10_000, + input_tokens_details: { + cached_tokens: 50_000, + cache_write_tokens: 20_000, + }, + }, + }); + + await withTestActor(() => + provider.complete({ + model: 'gpt-6-sol', + messages: [{ role: 'user', content: 'hi' }], + }), + ); + + const [, , , overrides] = recordSpy.mock.calls[0]!; + // $0.92 + $0.02 + $0.10 + $0.15 = $1.19, against $0.62 at the + // standard rates. + expect(overrides).toEqual({ + prompt_tokens: 230_000 * Number(sol.costs.prompt_tokens) * 2, + cached_tokens: 50_000 * Number(sol.costs.cached_tokens) * 2, + cache_write_tokens: + 20_000 * Number(sol.costs.cache_write_tokens) * 2, + completion_tokens: + 10_000 * Number(sol.costs.completion_tokens) * 1.5, + }); + const totalCents = + Object.values(overrides as Record).reduce( + (a, b) => a + b, + 0, + ) / 1_000_000; + expect(totalCents).toBeCloseTo(119); + }); + it('bills cached tokens at the input rate when the model prices no cache read', async () => { // gpt-5.4-pro is responses-API-only and its catalogue entry has no // cached_tokens rate. Cached tokens are subtracted out of the input diff --git a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.ts b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.ts index 6026de0db..5e1924d0f 100644 --- a/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.ts +++ b/src/backend/drivers/ai-chat/providers/openai/OpenAiChatResponsesProvider.ts @@ -175,8 +175,10 @@ export class OpenAiResponsesChatProvider implements IChatProvider { const requestedReasoningEffort = reasoning_effort ?? reasoning?.effort; const requestedVerbosity = verbosity ?? text?.verbosity; + const isGpt6Model = modelUsed.id.startsWith('gpt-6-'); const supportsReasoningControls = - typeof model === 'string' && model.startsWith('gpt-5'); + isGpt6Model || + (typeof model === 'string' && model.startsWith('gpt-5')); // Translate the neutral compaction opt-in (or pass a raw // `context_management` payload through) to OpenAI's Responses shape. @@ -232,6 +234,17 @@ export class OpenAiResponsesChatProvider implements IChatProvider { : {}), }), ...(supportsReasoningControls && reasoning ? { reasoning } : {}), + ...(isGpt6Model && requestedReasoningEffort !== undefined + ? { + reasoning: { + ...reasoning, + effort: requestedReasoningEffort, + }, + } + : {}), + ...(isGpt6Model && requestedVerbosity !== undefined + ? { text: { ...text, verbosity: requestedVerbosity } } + : {}), } as unknown as ResponseCreateParams; // console.log("completion params: ", completionParams) @@ -240,14 +253,23 @@ export class OpenAiResponsesChatProvider implements IChatProvider { // console.log("Completion: ", completion) return OpenAiUtil.handle_completion_output_responses_api({ usage_calculator: ({ usage }) => { + const cachedTokens = + (usage as any).input_tokens_details?.cached_tokens ?? 0; + // GPT-5.6 and later bill cache writes at 1.25x input. They're + // reported inside `input_tokens`, like cached reads. + const cacheWriteTokens = + (usage as any).input_tokens_details?.cache_write_tokens ?? + 0; const trackedUsage = { prompt_tokens: ((usage as any).input_tokens ?? 0) - - ((usage as any).input_tokens_details?.cached_tokens ?? - 0), + cachedTokens - + cacheWriteTokens, completion_tokens: (usage as any).output_tokens ?? 0, - cached_tokens: - (usage as any).input_tokens_details?.cached_tokens ?? 0, + cached_tokens: cachedTokens, + ...(cacheWriteTokens + ? { cache_write_tokens: cacheWriteTokens } + : {}), }; const costsOverrideFromModel = buildCostsOverride( diff --git a/src/backend/drivers/ai-chat/providers/openai/models.ts b/src/backend/drivers/ai-chat/providers/openai/models.ts index f8936942a..099ec70ac 100644 --- a/src/backend/drivers/ai-chat/providers/openai/models.ts +++ b/src/backend/drivers/ai-chat/providers/openai/models.ts @@ -21,8 +21,64 @@ import type { IChatModel } from '../../types.js'; +// Prompts over 272K input tokens are billed at 2x input (cached reads and +// cache writes included) and 1.5x output for the full request. +const GPT_LONG_CONTEXT_PRICING = { + threshold: 272_000, + input_multiplier: 2, + output_multiplier: 1.5, +}; + // Hardcoded from https://models.dev/api.json export const OPEN_AI_MODELS: IChatModel[] = [ + { + puterId: 'openai:openai/gpt-6-sol', + id: 'gpt-6-sol', + modalities: { input: ['text', 'image'], output: ['text'] }, + open_weights: false, + tool_call: true, + knowledge: '2026-04-20', + release_date: '2026-09-22', + aliases: ['openai/gpt-6-sol'], + costs_currency: 'usd-cents', + input_cost_key: 'prompt_tokens', + output_cost_key: 'completion_tokens', + costs: { + tokens: 1_000_000, + prompt_tokens: 200, + cached_tokens: 20, + cache_write_tokens: 200 * 1.25, + completion_tokens: 1000, + }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, + context: 1_050_000, + max_tokens: 128_000, + responses_api: true, + }, + { + puterId: 'openai:openai/gpt-6-luna', + id: 'gpt-6-luna', + modalities: { input: ['text', 'image'], output: ['text'] }, + open_weights: false, + tool_call: true, + knowledge: '2026-05-18', + release_date: '2026-09-22', + aliases: ['openai/gpt-6-luna'], + costs_currency: 'usd-cents', + input_cost_key: 'prompt_tokens', + output_cost_key: 'completion_tokens', + costs: { + tokens: 1_000_000, + prompt_tokens: 10, + cached_tokens: 1, + cache_write_tokens: 10 * 1.25, + completion_tokens: 50, + }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, + context: 1_050_000, + max_tokens: 128_000, + responses_api: true, + }, { puterId: 'openai:openai/gpt-6-astra', id: 'gpt-6-astra', @@ -39,8 +95,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [ tokens: 1_000_000, prompt_tokens: 1000, cached_tokens: 100, + cache_write_tokens: 1000 * 1.25, completion_tokens: 5000, }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, context: 1_050_000, max_tokens: 128_000, responses_api: true, @@ -56,12 +114,15 @@ export const OPEN_AI_MODELS: IChatModel[] = [ costs_currency: 'usd-cents', input_cost_key: 'prompt_tokens', output_cost_key: 'completion_tokens', + // OpenAI's promotional pricing, guaranteed only through 2026-11-21. costs: { tokens: 1_000_000, - prompt_tokens: 500, - cached_tokens: 50, - completion_tokens: 3000, + prompt_tokens: 400, + cached_tokens: 40, + cache_write_tokens: 400 * 1.25, + completion_tokens: 2000, }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, context: 1_050_000, max_tokens: 128_000, responses_api_only: true, @@ -81,8 +142,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [ tokens: 1_000_000, prompt_tokens: 200, cached_tokens: 20, + cache_write_tokens: 200 * 1.25, completion_tokens: 1200, }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, context: 1_050_000, max_tokens: 128_000, responses_api_only: true, @@ -102,8 +165,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [ tokens: 1_000_000, prompt_tokens: 20, cached_tokens: 2, + cache_write_tokens: 20 * 1.25, completion_tokens: 120, }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, context: 1_050_000, max_tokens: 128_000, responses_api_only: true, @@ -126,6 +191,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [ cached_tokens: 50, completion_tokens: 3000, }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, context: 1_050_000, max_tokens: 128_000, }, @@ -169,6 +235,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [ cached_tokens: 25, completion_tokens: 1500, }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, context: 1_050_000, max_tokens: 1_050_000, }, @@ -189,6 +256,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [ prompt_tokens: 3000, completion_tokens: 18000, }, + long_context_pricing: GPT_LONG_CONTEXT_PRICING, context: 1_050_000, max_tokens: 128_000, responses_api_only: true, diff --git a/src/backend/drivers/ai-chat/types.ts b/src/backend/drivers/ai-chat/types.ts index 98510bb88..5a50b31b2 100644 --- a/src/backend/drivers/ai-chat/types.ts +++ b/src/backend/drivers/ai-chat/types.ts @@ -43,6 +43,17 @@ export interface IChatModel extends Record< input_cost_key?: keyof T; output_cost_key?: keyof T; costs: T; + /** + * A request whose input exceeds `threshold` tokens is billed at raised + * rates for the whole request, not only the tokens past the threshold: + * every input-side rate (uncached, cached, cache writes) is multiplied by + * `input_multiplier` and every output-side rate by `output_multiplier`. + */ + long_context_pricing?: { + threshold: number; + input_multiplier: number; + output_multiplier: number; + }; context?: number; max_tokens: number; subscriberOnly?: boolean; diff --git a/src/backend/drivers/ai-chat/utils/pricing.test.ts b/src/backend/drivers/ai-chat/utils/pricing.test.ts index 4acd8e51c..d26c399bb 100644 --- a/src/backend/drivers/ai-chat/utils/pricing.test.ts +++ b/src/backend/drivers/ai-chat/utils/pricing.test.ts @@ -19,7 +19,12 @@ import { describe, expect, it } from 'vitest'; import type { IChatModel } from '../types.js'; -import { buildCostsOverride, isFreeModel, usdPerMToken } from './pricing.js'; +import { + buildCostsOverride, + isFreeModel, + longContextMultipliers, + usdPerMToken, +} from './pricing.js'; const model = (costs: Record): IChatModel => ({ @@ -166,3 +171,91 @@ describe('buildCostsOverride', () => { }); }); }); + +describe('long-context pricing', () => { + const longContext = (costs: Record): IChatModel => ({ + ...model(costs), + long_context_pricing: { + threshold: 272_000, + input_multiplier: 2, + output_multiplier: 1.5, + }, + }); + const rates = { + prompt_tokens: 200, + cached_tokens: 20, + cache_write_tokens: 250, + completion_tokens: 1000, + }; + + it('bills a request at or under the threshold at standard rates', () => { + const overrides = buildCostsOverride( + { + prompt_tokens: 222_000, + cached_tokens: 50_000, + completion_tokens: 10, + }, + longContext(rates), + ); + + expect(overrides).toEqual({ + prompt_tokens: 222_000 * 200, + cached_tokens: 50_000 * 20, + completion_tokens: 10 * 1000, + }); + }); + + it('raises every rate for the whole request once input passes the threshold', () => { + // 230K uncached + 50K cached + 20K cache writes = 300K input. No + // single key crosses 272K; their sum does. + const overrides = buildCostsOverride( + { + prompt_tokens: 230_000, + cached_tokens: 50_000, + cache_write_tokens: 20_000, + completion_tokens: 10_000, + }, + longContext(rates), + ); + + expect(overrides).toEqual({ + prompt_tokens: 230_000 * 200 * 2, + cached_tokens: 50_000 * 20 * 2, + cache_write_tokens: 20_000 * 250 * 2, + completion_tokens: 10_000 * 1000 * 1.5, + }); + }); + + it('raises the fallback rate of an unpriced key too', () => { + const overrides = buildCostsOverride( + { prompt_tokens: 300_000, thinking_tokens: 10 }, + longContext({ prompt_tokens: 200, completion_tokens: 1000 }), + ); + + expect(overrides.thinking_tokens).toBe(10 * 1000 * 1.5); + }); + + it('leaves a model without long-context pricing at standard rates', () => { + const overrides = buildCostsOverride( + { prompt_tokens: 900_000, completion_tokens: 10 }, + model(rates), + ); + + expect(overrides).toEqual({ + prompt_tokens: 900_000 * 200, + completion_tokens: 10 * 1000, + }); + }); + + it('applies the multipliers strictly above the threshold', () => { + const m = longContext(rates); + expect(longContextMultipliers(m, 272_000)).toEqual({ + input: 1, + output: 1, + }); + expect(longContextMultipliers(m, 272_001)).toEqual({ + input: 2, + output: 1.5, + }); + }); +}); diff --git a/src/backend/drivers/ai-chat/utils/pricing.ts b/src/backend/drivers/ai-chat/utils/pricing.ts index accf08ed0..9c116b604 100644 --- a/src/backend/drivers/ai-chat/utils/pricing.ts +++ b/src/backend/drivers/ai-chat/utils/pricing.ts @@ -54,6 +54,56 @@ export const costKeys = ( outputKey: (model.output_cost_key as string | undefined) ?? 'output_tokens', }); +/** + * Whether a usage key is priced at the output rate when the model has no rate + * of its own for it. + */ +export const isOutputCostKey = (key: string, outputKey: string): boolean => + key === outputKey || + key === 'output_tokens' || + key === 'completion_tokens' || + key === 'thinking_tokens'; + +/** + * The rate multipliers a request pays given how many input tokens it sent — + * cached reads and cache writes included. `1`/`1` unless the model has + * long-context pricing and the request is past its threshold. + */ +export const longContextMultipliers = ( + model: IChatModel, + inputTokens: number, +): { input: number; output: number } => { + const pricing = model.long_context_pricing; + if (!pricing || !(inputTokens > pricing.threshold)) { + return { input: 1, output: 1 }; + } + return { + input: pricing.input_multiplier, + output: pricing.output_multiplier, + }; +}; + +/** + * The input tokens a tracked-usage object carries: every key that isn't + * output-side. Providers split one prompt into uncached, cached-read and + * cache-write keys; the long-context threshold is measured on their sum. + */ +export const trackedInputTokens = ( + trackedUsage: Record, + model: IChatModel, +): number => { + const { outputKey } = costKeys(model); + let total = 0; + for (const [key, amount] of Object.entries(trackedUsage)) { + if (key === 'tokens' || key === 'usd_cents') continue; + if (isOutputCostKey(key, outputKey)) continue; + if (typeof amount === 'number' && Number.isFinite(amount)) { + total += amount; + } + } + return total; +}; + /** * Whether a model costs the user nothing to run. * @@ -90,11 +140,10 @@ export const buildCostsOverride = ( const inputRate = isRate(costs[inputKey]) ? costs[inputKey] : undefined; const outputRate = isRate(costs[outputKey]) ? costs[outputKey] : undefined; - const isOutputKey = (key: string) => - key === outputKey || - key === 'output_tokens' || - key === 'completion_tokens' || - key === 'thinking_tokens'; + const multipliers = longContextMultipliers( + model, + trackedInputTokens(trackedUsage, model), + ); const overrides: Record = {}; for (const [key, amount] of Object.entries(trackedUsage)) { @@ -102,11 +151,13 @@ export const buildCostsOverride = ( // not a per-unit rate. if (key === 'tokens') continue; + const isOutput = isOutputCostKey(key, outputKey); const rate = isRate(costs[key]) ? costs[key] - : ((isOutputKey(key) ? outputRate : inputRate) ?? 0); + : ((isOutput ? outputRate : inputRate) ?? 0); - overrides[key] = amount * rate; + overrides[key] = + amount * rate * (isOutput ? multipliers.output : multipliers.input); } return overrides;