feat(ai): add GPT-6 Sol and Luna and sync Claude pricing (#3931)
Maintain Release Merge PR / update-release-pr (push) Canceled after 0s
Notify HeyPuter / notify (push) Canceled after 0s
release-please / release-please (push) Canceled after 0s

* feat(ai): sync GPT-6 Sol, GPT-6 Luna, and Claude Opus 5.5

* chore: remove documentation changes from model sync

* fix(ai): bill OpenAI cache writes and long-context pricing

GPT-5.6 and later bill prompt-cache writes at 1.25x input and report them
in `cache_write_tokens`, inside the input total. Both OpenAI calculators now
split them out of the prompt count and meter them under their own key, and
the six GPT-5.6+ models carry the rate.

GPT-6, GPT-5.6, GPT-5.5 and GPT-5.4 (incl. Pro) bill a request with more
than 272K input tokens at 2x input (cached reads and cache writes included)
and 1.5x output for the whole request. Models declare this as
`long_context_pricing`, and the ledger overrides, the reported `usd_cents`
and the credit gate all apply the multipliers.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix(ai): bill GPT-5.6 Sol at OpenAI's promotional pricing

The catalog carried GPT-5.5's $5/$0.50/$30, but OpenAI bills GPT-5.6 Sol
at $4/$0.40/$20 per million input/cached/output tokens, promotional
through at least 2026-11-21. PUT-1943 tracks re-checking before then.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Filip Kujundžić
2026-09-23 01:59:10 -07:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 17f5454265
commit bc0f4dc279
13 changed files with 719 additions and 93 deletions
@@ -495,6 +495,43 @@ describe('ChatCompletionDriver.complete events and cost emission', () => {
expect(res.usage.usd_cents).toBe(expectedMicroCents / 1_000_000);
});
it('prices `usd_cents` at long-context rates once input passes the threshold', async () => {
vi.spyOn(FakeChatProvider.prototype, 'models').mockResolvedValueOnce([
{
id: 'priced',
aliases: [],
costs_currency: 'usd-cents',
costs: { input_tokens: 1000, output_tokens: 2000 },
long_context_pricing: {
threshold: 5,
input_multiplier: 2,
output_multiplier: 1.5,
},
max_tokens: 8192,
},
]);
const d = await makeDriver();
vi.spyOn(FakeChatProvider.prototype, 'complete').mockResolvedValueOnce({
message: {
role: 'assistant',
content: [{ type: 'text', text: 'ok' }],
},
usage: { input_tokens: 10, output_tokens: 7 },
finish_reason: 'stop',
} as never);
const res = (await withTestActor(() =>
d.complete({
model: 'priced',
messages: [{ role: 'user', content: 'hi' }],
}),
)) as { usage: Record<string, number> };
const expectedMicroCents = 10 * 1000 * 2 + 7 * 2000 * 1.5;
expect(res.usage.usd_cents).toBe(expectedMicroCents / 1_000_000);
});
it('does not override `usd_cents` when the provider already returned one (e.g. OpenRouter)', async () => {
vi.spyOn(FakeChatProvider.prototype, 'complete').mockResolvedValueOnce({
message: {
@@ -664,6 +701,54 @@ describe('ChatCompletionDriver.complete credit gate and max_tokens cap', () => {
expect(passed.max_tokens!).toBeGreaterThan(0);
});
it('caps `max_tokens` at the long-context output rate for a prompt past the threshold', async () => {
vi.spyOn(FakeChatProvider.prototype, 'models').mockResolvedValueOnce([
{
id: 'capme',
aliases: [],
costs_currency: 'usd-cents',
costs: { input_tokens: 1000, output_tokens: 2000 },
long_context_pricing: {
threshold: 10,
input_multiplier: 2,
output_multiplier: 1.5,
},
max_tokens: 8192,
},
]);
const d = await makeDriver();
// A ~100-token prompt is past the threshold. 1_000_000 microcents at
// 2000 * 1.5 per output token leaves at most 333 output tokens before
// the prompt is paid for; the standard rate would allow ~450 after it.
vi.spyOn(server.services.metering, 'getRemainingUsage').mockResolvedValue(
1_000_000,
);
const completeSpy = vi
.spyOn(FakeChatProvider.prototype, 'complete')
.mockResolvedValueOnce({
message: {
role: 'assistant',
content: [{ type: 'text', text: 'ok' }],
},
usage: { input_tokens: 1, output_tokens: 1 },
finish_reason: 'stop',
} as never);
await withTestActor(() =>
d.complete({
model: 'capme',
messages: [{ role: 'user', content: 'x'.repeat(400) }],
max_tokens: 10_000,
}),
);
const passed = completeSpy.mock.calls[0]![0] as ICompleteArguments;
expect(passed.max_tokens!).toBeLessThanOrEqual(333);
expect(passed.max_tokens!).toBeGreaterThan(0);
});
it('throws 402 instead of leaving `max_tokens` unset when credits cannot afford one output token', async () => {
// Regression: previously a sub-1 cap set max_tokens to `undefined`,
// which let the provider run to the model's full output limit and
@@ -86,7 +86,13 @@ import {
normalizeResultToOpenAI,
shouldPresentAsOpenAI,
} from './utils/normalizeToOpenAI.js';
import { costKeys, isFreeModel } from './utils/pricing.js';
import {
costKeys,
isFreeModel,
isOutputCostKey,
longContextMultipliers,
trackedInputTokens,
} from './utils/pricing.js';
import {
isRouteUnhealthy,
markRouteUnhealthy,
@@ -813,11 +819,11 @@ export class ChatCompletionDriver extends PuterDriver {
? outputRateRaw
: undefined;
const isOutputKey = (key: string) =>
key === outputKey ||
key === 'output_tokens' ||
key === 'completion_tokens' ||
key === 'thinking_tokens';
const isOutputKey = (key: string) => isOutputCostKey(key, outputKey);
const multipliers = longContextMultipliers(
model,
trackedInputTokens(usage, model),
);
let inputMicroCents = 0;
let outputMicroCents = 0;
@@ -851,12 +857,11 @@ export class ChatCompletionDriver extends PuterDriver {
}
}
const sub = rawAmount * rate;
sawAnyRate = true;
if (isOutputKey(key)) {
outputMicroCents += sub;
outputMicroCents += rawAmount * rate * multipliers.output;
} else {
inputMicroCents += sub;
inputMicroCents += rawAmount * rate * multipliers.input;
}
}
@@ -923,9 +928,14 @@ export class ChatCompletionDriver extends PuterDriver {
const metering = this.services.metering;
const { promptTokenEstimate, requestedMaxTokens } = estimates;
const { inputKey, outputKey } = costKeys(model);
// A prompt estimated past a long-context threshold pays the raised
// rates on input and output alike.
const multipliers = longContextMultipliers(model, promptTokenEstimate);
// `|| 0` also catches NaN from a malformed cost table.
const inputTokenCost = Number(model.costs?.[inputKey] ?? 0) || 0;
const outputTokenCost = Number(model.costs?.[outputKey] ?? 0) || 0;
const inputTokenCost =
(Number(model.costs?.[inputKey] ?? 0) || 0) * multipliers.input;
const outputTokenCost =
(Number(model.costs?.[outputKey] ?? 0) || 0) * multipliers.output;
const approximateInputCost = promptTokenEstimate * inputTokenCost;
const minimumCredits = Number(model.minimumCredits || 1);
@@ -790,44 +790,48 @@ describe('ClaudeProvider.complete request shape', () => {
});
});
it('omits temperature for fable 5.1 (rejects non-default sampling)', async () => {
const { provider } = makeProvider();
messagesCreateMock.mockResolvedValueOnce(baseResponse);
it.each(['claude-fable-5-1', 'claude-opus-5-5', 'claude-sonnet-5'])(
'omits temperature for %s',
async (model) => {
const { provider } = makeProvider();
messagesCreateMock.mockResolvedValueOnce(baseResponse);
await withTestActor(() =>
provider.complete({
model: 'claude-fable-5-1',
messages: [{ role: 'user', content: 'hi' }],
temperature: 0.5,
}),
);
await withTestActor(() =>
provider.complete({
model,
messages: [{ role: 'user', content: 'hi' }],
temperature: 0.5,
}),
);
const [args] = messagesCreateMock.mock.calls[0]!;
expect('temperature' in args).toBe(false);
});
const [args] = messagesCreateMock.mock.calls[0]!;
expect('temperature' in args).toBe(false);
},
);
it('forwards reasoning_effort as adaptive thinking + output_config effort on fable 5.1', async () => {
const { provider } = makeProvider();
messagesCreateMock.mockResolvedValueOnce(baseResponse);
it.each(['claude-fable-5-1', 'claude-opus-5-5', 'claude-sonnet-5'])(
'uses adaptive thinking and output effort on %s',
async (model) => {
const { provider } = makeProvider();
messagesCreateMock.mockResolvedValueOnce(baseResponse);
await withTestActor(() =>
provider.complete({
model: 'claude-fable-5-1',
messages: [{ role: 'user', content: 'hi' }],
reasoning_effort: 'high',
} as never),
);
await withTestActor(() =>
provider.complete({
model,
messages: [{ role: 'user', content: 'hi' }],
reasoning_effort: 'high',
} as never),
);
const [args] = messagesCreateMock.mock.calls[0]!;
// Fable 5.1 rejects `budget_tokens` and thinking is always on, so the
// only accepted config is adaptive; effort rides in output_config.
expect(args.thinking).toEqual({
type: 'adaptive',
display: 'summarized',
});
expect(args.output_config).toEqual({ effort: 'high' });
expect('temperature' in args).toBe(false);
});
const [args] = messagesCreateMock.mock.calls[0]!;
expect(args.thinking).toEqual({
type: 'adaptive',
display: 'summarized',
});
expect(args.output_config).toEqual({ effort: 'high' });
expect('temperature' in args).toBe(false);
},
);
it('forwards reasoning_effort as adaptive thinking + output_config effort on opus 5.5', async () => {
const { provider } = makeProvider();
@@ -957,6 +961,60 @@ describe('ClaudeProvider model resolution', () => {
// ── Non-stream completion ───────────────────────────────────────────
describe('ClaudeProvider.complete non-stream output', () => {
it.each([
['claude-opus-5-5', 'claude-opus-5-5', 400, 500, 800, 20, 2000],
[
'anthropic/claude-opus-5-5',
'claude-opus-5-5',
400,
500,
800,
20,
2000,
],
['claude-opus', 'claude-opus-5-5', 400, 500, 800, 20, 2000],
['claude-opus-latest', 'claude-opus-5-5', 400, 500, 800, 20, 2000],
['claude-opus-5-latest', 'claude-opus-5', 500, 625, 1000, 50, 2500],
['claude-sonnet-5', 'claude-sonnet-5', 200, 250, 400, 20, 1000],
])(
'resolves and meters %s at its current rates',
async (model, canonicalId, input, write5m, write1h, cached, output) => {
const { provider } = makeProvider();
messagesCreateMock.mockResolvedValueOnce({
content: [{ type: 'text', text: 'ok' }],
usage: {
input_tokens: 100,
output_tokens: 50,
cache_creation_input_tokens: 30,
cache_creation: {
ephemeral_5m_input_tokens: 10,
ephemeral_1h_input_tokens: 20,
},
cache_read_input_tokens: 1000,
},
});
await withTestActor(() =>
provider.complete({
model,
messages: [{ role: 'user', content: 'hi' }],
}),
);
expect(await provider.list()).toContain(model);
const [args] = messagesCreateMock.mock.calls[0]!;
expect(args.model).toBe(canonicalId);
expect(args.max_tokens).toBe(128_000);
const [, , prefix, overrides] = recordSpy.mock.calls[0]!;
expect(prefix).toBe(`claude:${canonicalId}`);
expect(overrides).toMatchObject({
input_tokens: 100 * Number(input),
output_tokens: 50 * Number(output),
ephemeral_5m_input_tokens: 10 * Number(write5m),
ephemeral_1h_input_tokens: 20 * Number(write1h),
cache_read_input_tokens: 1000 * Number(cached),
});
},
);
it('returns the message verbatim and meters input/output/cache token costs', async () => {
const { provider } = makeProvider();
const msg = {
@@ -806,14 +806,12 @@ export class ClaudeProvider implements IChatProvider {
}) {
if (!reasoningEffort) return undefined;
// Fable 5/5.1, Opus 4.7+, 4.6, and Sonnet 4.6 use adaptive thinking
// (`budget_tokens` is deprecated on 4.6/Sonnet 4.6, removed on
// Fable 5+ and Opus 4.7+). Fable 5/5.1 and Opus 4.7+ omit thinking
// content by default; `display: 'summarized'` restores visible
// reasoning in the stream.
// These models reject manual thinking budgets; summarized display
// keeps reasoning visible in the stream.
if (
modelId === 'claude-fable-5-1' ||
modelId === 'claude-fable-5' ||
modelId === 'claude-sonnet-5' ||
modelId === 'claude-opus-5-5' ||
modelId === 'claude-opus-5' ||
modelId === 'claude-opus-4-8' ||
@@ -44,7 +44,7 @@ export const CLAUDE_MODELS: IChatModel[] = [
input_tokens: 1000,
ephemeral_5m_input_tokens: 1000 * 1.25,
ephemeral_1h_input_tokens: 1000 * 2,
// Fable 5.1 bills cache reads at 0.025x input; every other Claude model is 0.1x.
// Fable 5.1 bills cache reads at 0.025x input.
cache_read_input_tokens: 1000 * 0.025,
output_tokens: 5000,
},
@@ -80,6 +80,7 @@ export const CLAUDE_MODELS: IChatModel[] = [
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2026-01',
release_date: '2026-06-30',
aliases: [
'claude-sonnet',
@@ -93,14 +94,14 @@ export const CLAUDE_MODELS: IChatModel[] = [
output_cost_key: 'output_tokens',
costs: {
tokens: 1_000_000,
input_tokens: 300,
ephemeral_5m_input_tokens: 300 * 1.25,
ephemeral_1h_input_tokens: 300 * 2,
cache_read_input_tokens: 300 * 0.1,
output_tokens: 1500,
input_tokens: 200,
ephemeral_5m_input_tokens: 200 * 1.25,
ephemeral_1h_input_tokens: 200 * 2,
cache_read_input_tokens: 200 * 0.1,
output_tokens: 1000,
},
context: 1000000,
max_tokens: 64000,
max_tokens: 128000,
},
{
puterId: 'anthropic:anthropic/claude-opus-5-5',
@@ -126,7 +127,7 @@ export const CLAUDE_MODELS: IChatModel[] = [
input_tokens: 400,
ephemeral_5m_input_tokens: 400 * 1.25,
ephemeral_1h_input_tokens: 400 * 2,
cache_read_input_tokens: 400 * 0.1,
cache_read_input_tokens: 400 * 0.05,
output_tokens: 2000,
},
context: 1000000,
@@ -171,6 +171,8 @@ describe('OpenAiChatProvider model catalog', () => {
// gpt-5-nano is a Chat-Completions model, must be present.
expect(ids).toContain('gpt-5-nano-2025-08-07');
expect(ids).toContain('gpt-6-astra');
expect(ids).toContain('gpt-6-sol');
expect(ids).toContain('gpt-6-luna');
});
it('list() flattens canonical ids and aliases', () => {
@@ -315,25 +317,30 @@ describe('OpenAiChatProvider.complete request shape', () => {
expect(args.safety_identifier).toBe('puter-u42');
});
it('resolves the namespaced GPT-6 Astra alias', async () => {
const { provider } = makeProvider();
createMock.mockResolvedValueOnce(baseCompletion);
it.each(['gpt-6-astra', 'gpt-6-sol', 'gpt-6-luna'])(
'resolves the namespaced %s alias',
async (model) => {
const { provider } = makeProvider();
createMock.mockResolvedValueOnce(baseCompletion);
await withTestActor(() =>
provider.complete({
model: 'openai/gpt-6-astra',
messages: [{ role: 'user', content: 'hello' }],
}),
);
await withTestActor(() =>
provider.complete({
model: `openai/${model}`,
messages: [{ role: 'user', content: 'hello' }],
reasoning_effort: 'low',
}),
);
expect(createMock.mock.calls[0]![0].model).toBe('gpt-6-astra');
expect(recordSpy).toHaveBeenCalledWith(
expect.any(Object),
expect.anything(),
'openai:gpt-6-astra',
expect.any(Object),
);
});
expect(createMock.mock.calls[0]![0].model).toBe(model);
expect(createMock.mock.calls[0]![0].reasoning_effort).toBe('low');
expect(recordSpy).toHaveBeenCalledWith(
expect.any(Object),
expect.anything(),
`openai:${model}`,
expect.any(Object),
);
},
);
it('forwards temperature 0 and max_tokens 0 instead of dropping them', async () => {
const { provider } = makeProvider();
@@ -514,6 +521,48 @@ describe('OpenAiChatProvider.complete non-stream output', () => {
).toBeGreaterThan(0);
});
it('splits cache writes out of prompt_tokens and bills them at 1.25x input', async () => {
const luna = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-luna')!;
const { provider } = makeProvider();
createMock.mockResolvedValueOnce({
choices: [
{
message: { content: 'hi', role: 'assistant' },
finish_reason: 'stop',
},
],
usage: {
prompt_tokens: 5000,
completion_tokens: 12,
prompt_tokens_details: {
cached_tokens: 1000,
cache_write_tokens: 3000,
},
},
});
await withTestActor(() =>
provider.complete({
model: 'gpt-6-luna',
messages: [{ role: 'user', content: 'hi' }],
}),
);
const [usage, , , overrides] = recordSpy.mock.calls[0]!;
expect(usage).toEqual({
prompt_tokens: 1000,
completion_tokens: 12,
cached_tokens: 1000,
cache_write_tokens: 3000,
});
expect(overrides).toEqual({
prompt_tokens: 1000 * Number(luna.costs.prompt_tokens),
completion_tokens: 12 * Number(luna.costs.completion_tokens),
cached_tokens: 1000 * Number(luna.costs.cached_tokens),
cache_write_tokens: 3000 * Number(luna.costs.cache_write_tokens),
});
});
it('zeroes cached_tokens when prompt_tokens_details is missing', async () => {
const { provider } = makeProvider();
createMock.mockResolvedValueOnce({
@@ -217,13 +217,26 @@ export class OpenAiChatProvider implements IChatProvider {
return OpenAiUtil.handle_completion_output({
usage_calculator: ({ usage }) => {
const cachedTokens =
usage.prompt_tokens_details?.cached_tokens ?? 0;
// GPT-5.6 and later bill cache writes at 1.25x input. They're
// reported inside `prompt_tokens`, like cached reads.
// The SDK doesn't type `cache_write_tokens` yet.
const cacheWriteTokens =
(
usage.prompt_tokens_details as
{ cache_write_tokens?: number } | undefined
)?.cache_write_tokens ?? 0;
const trackedUsage = {
prompt_tokens:
(usage.prompt_tokens ?? 0) -
(usage.prompt_tokens_details?.cached_tokens ?? 0),
cachedTokens -
cacheWriteTokens,
completion_tokens: usage.completion_tokens ?? 0,
cached_tokens:
usage.prompt_tokens_details?.cached_tokens ?? 0,
cached_tokens: cachedTokens,
...(cacheWriteTokens
? { cache_write_tokens: cacheWriteTokens }
: {}),
};
const costsOverrideFromModel = buildCostsOverride(
@@ -198,6 +198,38 @@ describe('OpenAiResponsesChatProvider model catalog', () => {
expect(ids).not.toContain('gpt-5-nano-2025-08-07');
});
it.each([
['gpt-6-sol', '2026-04-20', 200, 20, 1000],
['gpt-6-luna', '2026-05-18', 10, 1, 50],
])(
'exposes %s with current pricing and limits',
(id, knowledge, input, cached, output) => {
const { provider } = makeProvider();
expect(
provider.models().find((model) => model.id === id),
).toMatchObject({
puterId: `openai:openai/${id}`,
aliases: [`openai/${id}`],
knowledge,
release_date: '2026-09-22',
modalities: { input: ['text', 'image'], output: ['text'] },
costs_currency: 'usd-cents',
costs: {
tokens: 1_000_000,
prompt_tokens: input,
cached_tokens: cached,
completion_tokens: output,
},
context: 1_050_000,
max_tokens: 128_000,
responses_api: true,
});
expect(provider.list()).toEqual(
expect.arrayContaining([id, `openai/${id}`]),
);
},
);
it('models({ no_restrictions: true }) returns the entire catalog (used by complete())', () => {
const { provider } = makeProvider();
const ids = provider
@@ -386,6 +418,58 @@ describe('OpenAiResponsesChatProvider.complete request shape', () => {
expect(o3Args.reasoning_effort).toBe('medium');
expect(o3Args.verbosity).toBe('low');
});
it.each(['gpt-6-astra', 'gpt-6-sol', 'gpt-6-luna'])(
'maps flat controls to Responses options for %s aliases',
async (model) => {
const { provider } = makeProvider();
responsesCreateMock.mockResolvedValueOnce(baseResponse);
await withTestActor(() =>
provider.complete({
model: `openai/${model}`,
messages: [{ role: 'user', content: 'hi' }],
reasoning_effort: 'high',
verbosity: 'low',
}),
);
const [args] = responsesCreateMock.mock.calls[0]!;
expect(args.model).toBe(model);
expect(args.reasoning).toEqual({ effort: 'high' });
expect(args.text).toEqual({ verbosity: 'low' });
expect(args).not.toHaveProperty('reasoning_effort');
expect(args).not.toHaveProperty('verbosity');
expect(recordSpy.mock.calls[0]![2]).toBe(`openai:${model}`);
},
);
it('preserves nested GPT-6 controls and gives flat options precedence', async () => {
const { provider } = makeProvider();
const reasoning = { effort: 'medium', summary: 'auto' };
const text = { verbosity: 'high', format: { type: 'text' } };
for (const flat of [false, true]) {
responsesCreateMock.mockResolvedValueOnce(baseResponse);
await withTestActor(() =>
provider.complete({
model: 'gpt-6-sol',
messages: [{ role: 'user', content: 'hi' }],
reasoning,
text,
...(flat
? { reasoning_effort: 'low', verbosity: 'low' }
: {}),
} as never),
);
const [args] = responsesCreateMock.mock.lastCall!;
expect(args.reasoning).toEqual({
...reasoning,
effort: flat ? 'low' : 'medium',
});
expect(args.text).toEqual({
...text,
verbosity: flat ? 'low' : 'high',
});
}
});
});
// ── Model resolution ────────────────────────────────────────────────
@@ -518,6 +602,89 @@ describe('OpenAiResponsesChatProvider.complete non-stream output', () => {
});
});
it('splits cache writes out of input and bills them at 1.25x input', async () => {
const luna = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-luna')!;
const { provider } = makeProvider();
responsesCreateMock.mockResolvedValueOnce({
output: [{ role: 'assistant' }],
output_text: 'hi',
usage: {
input_tokens: 5000,
output_tokens: 20,
input_tokens_details: {
cached_tokens: 1000,
cache_write_tokens: 3000,
},
},
});
await withTestActor(() =>
provider.complete({
model: 'gpt-6-luna',
messages: [{ role: 'user', content: 'hi' }],
}),
);
const [usage, , , overrides] = recordSpy.mock.calls[0]!;
expect(usage).toEqual({
prompt_tokens: 1000,
completion_tokens: 20,
cached_tokens: 1000,
cache_write_tokens: 3000,
});
expect(luna.costs.cache_write_tokens).toBe(
Number(luna.costs.prompt_tokens) * 1.25,
);
expect(overrides).toEqual({
prompt_tokens: 1000 * Number(luna.costs.prompt_tokens),
completion_tokens: 20 * Number(luna.costs.completion_tokens),
cached_tokens: 1000 * Number(luna.costs.cached_tokens),
cache_write_tokens: 3000 * Number(luna.costs.cache_write_tokens),
});
});
it('bills the whole request at long-context rates past 272K input tokens', async () => {
const sol = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-sol')!;
const { provider } = makeProvider();
responsesCreateMock.mockResolvedValueOnce({
output: [{ role: 'assistant' }],
output_text: 'hi',
usage: {
input_tokens: 300_000,
output_tokens: 10_000,
input_tokens_details: {
cached_tokens: 50_000,
cache_write_tokens: 20_000,
},
},
});
await withTestActor(() =>
provider.complete({
model: 'gpt-6-sol',
messages: [{ role: 'user', content: 'hi' }],
}),
);
const [, , , overrides] = recordSpy.mock.calls[0]!;
// $0.92 + $0.02 + $0.10 + $0.15 = $1.19, against $0.62 at the
// standard rates.
expect(overrides).toEqual({
prompt_tokens: 230_000 * Number(sol.costs.prompt_tokens) * 2,
cached_tokens: 50_000 * Number(sol.costs.cached_tokens) * 2,
cache_write_tokens:
20_000 * Number(sol.costs.cache_write_tokens) * 2,
completion_tokens:
10_000 * Number(sol.costs.completion_tokens) * 1.5,
});
const totalCents =
Object.values(overrides as Record<string, number>).reduce(
(a, b) => a + b,
0,
) / 1_000_000;
expect(totalCents).toBeCloseTo(119);
});
it('bills cached tokens at the input rate when the model prices no cache read', async () => {
// gpt-5.4-pro is responses-API-only and its catalogue entry has no
// cached_tokens rate. Cached tokens are subtracted out of the input
@@ -175,8 +175,10 @@ export class OpenAiResponsesChatProvider implements IChatProvider {
const requestedReasoningEffort = reasoning_effort ?? reasoning?.effort;
const requestedVerbosity = verbosity ?? text?.verbosity;
const isGpt6Model = modelUsed.id.startsWith('gpt-6-');
const supportsReasoningControls =
typeof model === 'string' && model.startsWith('gpt-5');
isGpt6Model ||
(typeof model === 'string' && model.startsWith('gpt-5'));
// Translate the neutral compaction opt-in (or pass a raw
// `context_management` payload through) to OpenAI's Responses shape.
@@ -232,6 +234,17 @@ export class OpenAiResponsesChatProvider implements IChatProvider {
: {}),
}),
...(supportsReasoningControls && reasoning ? { reasoning } : {}),
...(isGpt6Model && requestedReasoningEffort !== undefined
? {
reasoning: {
...reasoning,
effort: requestedReasoningEffort,
},
}
: {}),
...(isGpt6Model && requestedVerbosity !== undefined
? { text: { ...text, verbosity: requestedVerbosity } }
: {}),
} as unknown as ResponseCreateParams;
// console.log("completion params: ", completionParams)
@@ -240,14 +253,23 @@ export class OpenAiResponsesChatProvider implements IChatProvider {
// console.log("Completion: ", completion)
return OpenAiUtil.handle_completion_output_responses_api({
usage_calculator: ({ usage }) => {
const cachedTokens =
(usage as any).input_tokens_details?.cached_tokens ?? 0;
// GPT-5.6 and later bill cache writes at 1.25x input. They're
// reported inside `input_tokens`, like cached reads.
const cacheWriteTokens =
(usage as any).input_tokens_details?.cache_write_tokens ??
0;
const trackedUsage = {
prompt_tokens:
((usage as any).input_tokens ?? 0) -
((usage as any).input_tokens_details?.cached_tokens ??
0),
cachedTokens -
cacheWriteTokens,
completion_tokens: (usage as any).output_tokens ?? 0,
cached_tokens:
(usage as any).input_tokens_details?.cached_tokens ?? 0,
cached_tokens: cachedTokens,
...(cacheWriteTokens
? { cache_write_tokens: cacheWriteTokens }
: {}),
};
const costsOverrideFromModel = buildCostsOverride(
@@ -21,8 +21,64 @@
import type { IChatModel } from '../../types.js';
// Prompts over 272K input tokens are billed at 2x input (cached reads and
// cache writes included) and 1.5x output for the full request.
const GPT_LONG_CONTEXT_PRICING = {
threshold: 272_000,
input_multiplier: 2,
output_multiplier: 1.5,
};
// Hardcoded from https://models.dev/api.json
export const OPEN_AI_MODELS: IChatModel[] = [
{
puterId: 'openai:openai/gpt-6-sol',
id: 'gpt-6-sol',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2026-04-20',
release_date: '2026-09-22',
aliases: ['openai/gpt-6-sol'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 200,
cached_tokens: 20,
cache_write_tokens: 200 * 1.25,
completion_tokens: 1000,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
responses_api: true,
},
{
puterId: 'openai:openai/gpt-6-luna',
id: 'gpt-6-luna',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2026-05-18',
release_date: '2026-09-22',
aliases: ['openai/gpt-6-luna'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 10,
cached_tokens: 1,
cache_write_tokens: 10 * 1.25,
completion_tokens: 50,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
responses_api: true,
},
{
puterId: 'openai:openai/gpt-6-astra',
id: 'gpt-6-astra',
@@ -39,8 +95,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [
tokens: 1_000_000,
prompt_tokens: 1000,
cached_tokens: 100,
cache_write_tokens: 1000 * 1.25,
completion_tokens: 5000,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
responses_api: true,
@@ -56,12 +114,15 @@ export const OPEN_AI_MODELS: IChatModel[] = [
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
// OpenAI's promotional pricing, guaranteed only through 2026-11-21.
costs: {
tokens: 1_000_000,
prompt_tokens: 500,
cached_tokens: 50,
completion_tokens: 3000,
prompt_tokens: 400,
cached_tokens: 40,
cache_write_tokens: 400 * 1.25,
completion_tokens: 2000,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
responses_api_only: true,
@@ -81,8 +142,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [
tokens: 1_000_000,
prompt_tokens: 200,
cached_tokens: 20,
cache_write_tokens: 200 * 1.25,
completion_tokens: 1200,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
responses_api_only: true,
@@ -102,8 +165,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [
tokens: 1_000_000,
prompt_tokens: 20,
cached_tokens: 2,
cache_write_tokens: 20 * 1.25,
completion_tokens: 120,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
responses_api_only: true,
@@ -126,6 +191,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [
cached_tokens: 50,
completion_tokens: 3000,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
},
@@ -169,6 +235,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [
cached_tokens: 25,
completion_tokens: 1500,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 1_050_000,
},
@@ -189,6 +256,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [
prompt_tokens: 3000,
completion_tokens: 18000,
},
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
context: 1_050_000,
max_tokens: 128_000,
responses_api_only: true,
+11
View File
@@ -43,6 +43,17 @@ export interface IChatModel<T extends ModelCost = ModelCost> extends Record<
input_cost_key?: keyof T;
output_cost_key?: keyof T;
costs: T;
/**
* A request whose input exceeds `threshold` tokens is billed at raised
* rates for the whole request, not only the tokens past the threshold:
* every input-side rate (uncached, cached, cache writes) is multiplied by
* `input_multiplier` and every output-side rate by `output_multiplier`.
*/
long_context_pricing?: {
threshold: number;
input_multiplier: number;
output_multiplier: number;
};
context?: number;
max_tokens: number;
subscriberOnly?: boolean;
@@ -19,7 +19,12 @@
import { describe, expect, it } from 'vitest';
import type { IChatModel } from '../types.js';
import { buildCostsOverride, isFreeModel, usdPerMToken } from './pricing.js';
import {
buildCostsOverride,
isFreeModel,
longContextMultipliers,
usdPerMToken,
} from './pricing.js';
const model = (costs: Record<string, number>): IChatModel =>
({
@@ -166,3 +171,91 @@ describe('buildCostsOverride', () => {
});
});
});
describe('long-context pricing', () => {
const longContext = (costs: Record<string, number>): IChatModel => ({
...model(costs),
long_context_pricing: {
threshold: 272_000,
input_multiplier: 2,
output_multiplier: 1.5,
},
});
const rates = {
prompt_tokens: 200,
cached_tokens: 20,
cache_write_tokens: 250,
completion_tokens: 1000,
};
it('bills a request at or under the threshold at standard rates', () => {
const overrides = buildCostsOverride(
{
prompt_tokens: 222_000,
cached_tokens: 50_000,
completion_tokens: 10,
},
longContext(rates),
);
expect(overrides).toEqual({
prompt_tokens: 222_000 * 200,
cached_tokens: 50_000 * 20,
completion_tokens: 10 * 1000,
});
});
it('raises every rate for the whole request once input passes the threshold', () => {
// 230K uncached + 50K cached + 20K cache writes = 300K input. No
// single key crosses 272K; their sum does.
const overrides = buildCostsOverride(
{
prompt_tokens: 230_000,
cached_tokens: 50_000,
cache_write_tokens: 20_000,
completion_tokens: 10_000,
},
longContext(rates),
);
expect(overrides).toEqual({
prompt_tokens: 230_000 * 200 * 2,
cached_tokens: 50_000 * 20 * 2,
cache_write_tokens: 20_000 * 250 * 2,
completion_tokens: 10_000 * 1000 * 1.5,
});
});
it('raises the fallback rate of an unpriced key too', () => {
const overrides = buildCostsOverride(
{ prompt_tokens: 300_000, thinking_tokens: 10 },
longContext({ prompt_tokens: 200, completion_tokens: 1000 }),
);
expect(overrides.thinking_tokens).toBe(10 * 1000 * 1.5);
});
it('leaves a model without long-context pricing at standard rates', () => {
const overrides = buildCostsOverride(
{ prompt_tokens: 900_000, completion_tokens: 10 },
model(rates),
);
expect(overrides).toEqual({
prompt_tokens: 900_000 * 200,
completion_tokens: 10 * 1000,
});
});
it('applies the multipliers strictly above the threshold', () => {
const m = longContext(rates);
expect(longContextMultipliers(m, 272_000)).toEqual({
input: 1,
output: 1,
});
expect(longContextMultipliers(m, 272_001)).toEqual({
input: 2,
output: 1.5,
});
});
});
+58 -7
View File
@@ -54,6 +54,56 @@ export const costKeys = (
outputKey: (model.output_cost_key as string | undefined) ?? 'output_tokens',
});
/**
* Whether a usage key is priced at the output rate when the model has no rate
* of its own for it.
*/
export const isOutputCostKey = (key: string, outputKey: string): boolean =>
key === outputKey ||
key === 'output_tokens' ||
key === 'completion_tokens' ||
key === 'thinking_tokens';
/**
* The rate multipliers a request pays given how many input tokens it sent —
* cached reads and cache writes included. `1`/`1` unless the model has
* long-context pricing and the request is past its threshold.
*/
export const longContextMultipliers = (
model: IChatModel,
inputTokens: number,
): { input: number; output: number } => {
const pricing = model.long_context_pricing;
if (!pricing || !(inputTokens > pricing.threshold)) {
return { input: 1, output: 1 };
}
return {
input: pricing.input_multiplier,
output: pricing.output_multiplier,
};
};
/**
* The input tokens a tracked-usage object carries: every key that isn't
* output-side. Providers split one prompt into uncached, cached-read and
* cache-write keys; the long-context threshold is measured on their sum.
*/
export const trackedInputTokens = (
trackedUsage: Record<string, unknown>,
model: IChatModel,
): number => {
const { outputKey } = costKeys(model);
let total = 0;
for (const [key, amount] of Object.entries(trackedUsage)) {
if (key === 'tokens' || key === 'usd_cents') continue;
if (isOutputCostKey(key, outputKey)) continue;
if (typeof amount === 'number' && Number.isFinite(amount)) {
total += amount;
}
}
return total;
};
/**
* Whether a model costs the user nothing to run.
*
@@ -90,11 +140,10 @@ export const buildCostsOverride = (
const inputRate = isRate(costs[inputKey]) ? costs[inputKey] : undefined;
const outputRate = isRate(costs[outputKey]) ? costs[outputKey] : undefined;
const isOutputKey = (key: string) =>
key === outputKey ||
key === 'output_tokens' ||
key === 'completion_tokens' ||
key === 'thinking_tokens';
const multipliers = longContextMultipliers(
model,
trackedInputTokens(trackedUsage, model),
);
const overrides: Record<string, number> = {};
for (const [key, amount] of Object.entries(trackedUsage)) {
@@ -102,11 +151,13 @@ export const buildCostsOverride = (
// not a per-unit rate.
if (key === 'tokens') continue;
const isOutput = isOutputCostKey(key, outputKey);
const rate = isRate(costs[key])
? costs[key]
: ((isOutputKey(key) ? outputRate : inputRate) ?? 0);
: ((isOutput ? outputRate : inputRate) ?? 0);
overrides[key] = amount * rate;
overrides[key] =
amount * rate * (isOutput ? multipliers.output : multipliers.input);
}
return overrides;