mirror of
https://github.com/HeyPuter/puter.git
synced 2026-10-02 01:51:55 +00:00
feat(ai): add GPT-6 Sol and Luna and sync Claude pricing (#3931)
* feat(ai): sync GPT-6 Sol, GPT-6 Luna, and Claude Opus 5.5 * chore: remove documentation changes from model sync * fix(ai): bill OpenAI cache writes and long-context pricing GPT-5.6 and later bill prompt-cache writes at 1.25x input and report them in `cache_write_tokens`, inside the input total. Both OpenAI calculators now split them out of the prompt count and meter them under their own key, and the six GPT-5.6+ models carry the rate. GPT-6, GPT-5.6, GPT-5.5 and GPT-5.4 (incl. Pro) bill a request with more than 272K input tokens at 2x input (cached reads and cache writes included) and 1.5x output for the whole request. Models declare this as `long_context_pricing`, and the ledger overrides, the reported `usd_cents` and the credit gate all apply the multipliers. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> * fix(ai): bill GPT-5.6 Sol at OpenAI's promotional pricing The catalog carried GPT-5.5's $5/$0.50/$30, but OpenAI bills GPT-5.6 Sol at $4/$0.40/$20 per million input/cached/output tokens, promotional through at least 2026-11-21. PUT-1943 tracks re-checking before then. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
17f5454265
commit
bc0f4dc279
@@ -495,6 +495,43 @@ describe('ChatCompletionDriver.complete events and cost emission', () => {
|
||||
expect(res.usage.usd_cents).toBe(expectedMicroCents / 1_000_000);
|
||||
});
|
||||
|
||||
it('prices `usd_cents` at long-context rates once input passes the threshold', async () => {
|
||||
vi.spyOn(FakeChatProvider.prototype, 'models').mockResolvedValueOnce([
|
||||
{
|
||||
id: 'priced',
|
||||
aliases: [],
|
||||
costs_currency: 'usd-cents',
|
||||
costs: { input_tokens: 1000, output_tokens: 2000 },
|
||||
long_context_pricing: {
|
||||
threshold: 5,
|
||||
input_multiplier: 2,
|
||||
output_multiplier: 1.5,
|
||||
},
|
||||
max_tokens: 8192,
|
||||
},
|
||||
]);
|
||||
const d = await makeDriver();
|
||||
|
||||
vi.spyOn(FakeChatProvider.prototype, 'complete').mockResolvedValueOnce({
|
||||
message: {
|
||||
role: 'assistant',
|
||||
content: [{ type: 'text', text: 'ok' }],
|
||||
},
|
||||
usage: { input_tokens: 10, output_tokens: 7 },
|
||||
finish_reason: 'stop',
|
||||
} as never);
|
||||
|
||||
const res = (await withTestActor(() =>
|
||||
d.complete({
|
||||
model: 'priced',
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
}),
|
||||
)) as { usage: Record<string, number> };
|
||||
|
||||
const expectedMicroCents = 10 * 1000 * 2 + 7 * 2000 * 1.5;
|
||||
expect(res.usage.usd_cents).toBe(expectedMicroCents / 1_000_000);
|
||||
});
|
||||
|
||||
it('does not override `usd_cents` when the provider already returned one (e.g. OpenRouter)', async () => {
|
||||
vi.spyOn(FakeChatProvider.prototype, 'complete').mockResolvedValueOnce({
|
||||
message: {
|
||||
@@ -664,6 +701,54 @@ describe('ChatCompletionDriver.complete credit gate and max_tokens cap', () => {
|
||||
expect(passed.max_tokens!).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('caps `max_tokens` at the long-context output rate for a prompt past the threshold', async () => {
|
||||
vi.spyOn(FakeChatProvider.prototype, 'models').mockResolvedValueOnce([
|
||||
{
|
||||
id: 'capme',
|
||||
aliases: [],
|
||||
costs_currency: 'usd-cents',
|
||||
costs: { input_tokens: 1000, output_tokens: 2000 },
|
||||
long_context_pricing: {
|
||||
threshold: 10,
|
||||
input_multiplier: 2,
|
||||
output_multiplier: 1.5,
|
||||
},
|
||||
max_tokens: 8192,
|
||||
},
|
||||
]);
|
||||
const d = await makeDriver();
|
||||
|
||||
// A ~100-token prompt is past the threshold. 1_000_000 microcents at
|
||||
// 2000 * 1.5 per output token leaves at most 333 output tokens before
|
||||
// the prompt is paid for; the standard rate would allow ~450 after it.
|
||||
vi.spyOn(server.services.metering, 'getRemainingUsage').mockResolvedValue(
|
||||
1_000_000,
|
||||
);
|
||||
|
||||
const completeSpy = vi
|
||||
.spyOn(FakeChatProvider.prototype, 'complete')
|
||||
.mockResolvedValueOnce({
|
||||
message: {
|
||||
role: 'assistant',
|
||||
content: [{ type: 'text', text: 'ok' }],
|
||||
},
|
||||
usage: { input_tokens: 1, output_tokens: 1 },
|
||||
finish_reason: 'stop',
|
||||
} as never);
|
||||
|
||||
await withTestActor(() =>
|
||||
d.complete({
|
||||
model: 'capme',
|
||||
messages: [{ role: 'user', content: 'x'.repeat(400) }],
|
||||
max_tokens: 10_000,
|
||||
}),
|
||||
);
|
||||
|
||||
const passed = completeSpy.mock.calls[0]![0] as ICompleteArguments;
|
||||
expect(passed.max_tokens!).toBeLessThanOrEqual(333);
|
||||
expect(passed.max_tokens!).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('throws 402 instead of leaving `max_tokens` unset when credits cannot afford one output token', async () => {
|
||||
// Regression: previously a sub-1 cap set max_tokens to `undefined`,
|
||||
// which let the provider run to the model's full output limit and
|
||||
|
||||
@@ -86,7 +86,13 @@ import {
|
||||
normalizeResultToOpenAI,
|
||||
shouldPresentAsOpenAI,
|
||||
} from './utils/normalizeToOpenAI.js';
|
||||
import { costKeys, isFreeModel } from './utils/pricing.js';
|
||||
import {
|
||||
costKeys,
|
||||
isFreeModel,
|
||||
isOutputCostKey,
|
||||
longContextMultipliers,
|
||||
trackedInputTokens,
|
||||
} from './utils/pricing.js';
|
||||
import {
|
||||
isRouteUnhealthy,
|
||||
markRouteUnhealthy,
|
||||
@@ -813,11 +819,11 @@ export class ChatCompletionDriver extends PuterDriver {
|
||||
? outputRateRaw
|
||||
: undefined;
|
||||
|
||||
const isOutputKey = (key: string) =>
|
||||
key === outputKey ||
|
||||
key === 'output_tokens' ||
|
||||
key === 'completion_tokens' ||
|
||||
key === 'thinking_tokens';
|
||||
const isOutputKey = (key: string) => isOutputCostKey(key, outputKey);
|
||||
const multipliers = longContextMultipliers(
|
||||
model,
|
||||
trackedInputTokens(usage, model),
|
||||
);
|
||||
|
||||
let inputMicroCents = 0;
|
||||
let outputMicroCents = 0;
|
||||
@@ -851,12 +857,11 @@ export class ChatCompletionDriver extends PuterDriver {
|
||||
}
|
||||
}
|
||||
|
||||
const sub = rawAmount * rate;
|
||||
sawAnyRate = true;
|
||||
if (isOutputKey(key)) {
|
||||
outputMicroCents += sub;
|
||||
outputMicroCents += rawAmount * rate * multipliers.output;
|
||||
} else {
|
||||
inputMicroCents += sub;
|
||||
inputMicroCents += rawAmount * rate * multipliers.input;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -923,9 +928,14 @@ export class ChatCompletionDriver extends PuterDriver {
|
||||
const metering = this.services.metering;
|
||||
const { promptTokenEstimate, requestedMaxTokens } = estimates;
|
||||
const { inputKey, outputKey } = costKeys(model);
|
||||
// A prompt estimated past a long-context threshold pays the raised
|
||||
// rates on input and output alike.
|
||||
const multipliers = longContextMultipliers(model, promptTokenEstimate);
|
||||
// `|| 0` also catches NaN from a malformed cost table.
|
||||
const inputTokenCost = Number(model.costs?.[inputKey] ?? 0) || 0;
|
||||
const outputTokenCost = Number(model.costs?.[outputKey] ?? 0) || 0;
|
||||
const inputTokenCost =
|
||||
(Number(model.costs?.[inputKey] ?? 0) || 0) * multipliers.input;
|
||||
const outputTokenCost =
|
||||
(Number(model.costs?.[outputKey] ?? 0) || 0) * multipliers.output;
|
||||
const approximateInputCost = promptTokenEstimate * inputTokenCost;
|
||||
const minimumCredits = Number(model.minimumCredits || 1);
|
||||
|
||||
|
||||
@@ -790,44 +790,48 @@ describe('ClaudeProvider.complete request shape', () => {
|
||||
});
|
||||
});
|
||||
|
||||
it('omits temperature for fable 5.1 (rejects non-default sampling)', async () => {
|
||||
const { provider } = makeProvider();
|
||||
messagesCreateMock.mockResolvedValueOnce(baseResponse);
|
||||
it.each(['claude-fable-5-1', 'claude-opus-5-5', 'claude-sonnet-5'])(
|
||||
'omits temperature for %s',
|
||||
async (model) => {
|
||||
const { provider } = makeProvider();
|
||||
messagesCreateMock.mockResolvedValueOnce(baseResponse);
|
||||
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: 'claude-fable-5-1',
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
temperature: 0.5,
|
||||
}),
|
||||
);
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model,
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
temperature: 0.5,
|
||||
}),
|
||||
);
|
||||
|
||||
const [args] = messagesCreateMock.mock.calls[0]!;
|
||||
expect('temperature' in args).toBe(false);
|
||||
});
|
||||
const [args] = messagesCreateMock.mock.calls[0]!;
|
||||
expect('temperature' in args).toBe(false);
|
||||
},
|
||||
);
|
||||
|
||||
it('forwards reasoning_effort as adaptive thinking + output_config effort on fable 5.1', async () => {
|
||||
const { provider } = makeProvider();
|
||||
messagesCreateMock.mockResolvedValueOnce(baseResponse);
|
||||
it.each(['claude-fable-5-1', 'claude-opus-5-5', 'claude-sonnet-5'])(
|
||||
'uses adaptive thinking and output effort on %s',
|
||||
async (model) => {
|
||||
const { provider } = makeProvider();
|
||||
messagesCreateMock.mockResolvedValueOnce(baseResponse);
|
||||
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: 'claude-fable-5-1',
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
reasoning_effort: 'high',
|
||||
} as never),
|
||||
);
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model,
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
reasoning_effort: 'high',
|
||||
} as never),
|
||||
);
|
||||
|
||||
const [args] = messagesCreateMock.mock.calls[0]!;
|
||||
// Fable 5.1 rejects `budget_tokens` and thinking is always on, so the
|
||||
// only accepted config is adaptive; effort rides in output_config.
|
||||
expect(args.thinking).toEqual({
|
||||
type: 'adaptive',
|
||||
display: 'summarized',
|
||||
});
|
||||
expect(args.output_config).toEqual({ effort: 'high' });
|
||||
expect('temperature' in args).toBe(false);
|
||||
});
|
||||
const [args] = messagesCreateMock.mock.calls[0]!;
|
||||
expect(args.thinking).toEqual({
|
||||
type: 'adaptive',
|
||||
display: 'summarized',
|
||||
});
|
||||
expect(args.output_config).toEqual({ effort: 'high' });
|
||||
expect('temperature' in args).toBe(false);
|
||||
},
|
||||
);
|
||||
|
||||
it('forwards reasoning_effort as adaptive thinking + output_config effort on opus 5.5', async () => {
|
||||
const { provider } = makeProvider();
|
||||
@@ -957,6 +961,60 @@ describe('ClaudeProvider model resolution', () => {
|
||||
// ── Non-stream completion ───────────────────────────────────────────
|
||||
|
||||
describe('ClaudeProvider.complete non-stream output', () => {
|
||||
it.each([
|
||||
['claude-opus-5-5', 'claude-opus-5-5', 400, 500, 800, 20, 2000],
|
||||
[
|
||||
'anthropic/claude-opus-5-5',
|
||||
'claude-opus-5-5',
|
||||
400,
|
||||
500,
|
||||
800,
|
||||
20,
|
||||
2000,
|
||||
],
|
||||
['claude-opus', 'claude-opus-5-5', 400, 500, 800, 20, 2000],
|
||||
['claude-opus-latest', 'claude-opus-5-5', 400, 500, 800, 20, 2000],
|
||||
['claude-opus-5-latest', 'claude-opus-5', 500, 625, 1000, 50, 2500],
|
||||
['claude-sonnet-5', 'claude-sonnet-5', 200, 250, 400, 20, 1000],
|
||||
])(
|
||||
'resolves and meters %s at its current rates',
|
||||
async (model, canonicalId, input, write5m, write1h, cached, output) => {
|
||||
const { provider } = makeProvider();
|
||||
messagesCreateMock.mockResolvedValueOnce({
|
||||
content: [{ type: 'text', text: 'ok' }],
|
||||
usage: {
|
||||
input_tokens: 100,
|
||||
output_tokens: 50,
|
||||
cache_creation_input_tokens: 30,
|
||||
cache_creation: {
|
||||
ephemeral_5m_input_tokens: 10,
|
||||
ephemeral_1h_input_tokens: 20,
|
||||
},
|
||||
cache_read_input_tokens: 1000,
|
||||
},
|
||||
});
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model,
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
}),
|
||||
);
|
||||
expect(await provider.list()).toContain(model);
|
||||
const [args] = messagesCreateMock.mock.calls[0]!;
|
||||
expect(args.model).toBe(canonicalId);
|
||||
expect(args.max_tokens).toBe(128_000);
|
||||
const [, , prefix, overrides] = recordSpy.mock.calls[0]!;
|
||||
expect(prefix).toBe(`claude:${canonicalId}`);
|
||||
expect(overrides).toMatchObject({
|
||||
input_tokens: 100 * Number(input),
|
||||
output_tokens: 50 * Number(output),
|
||||
ephemeral_5m_input_tokens: 10 * Number(write5m),
|
||||
ephemeral_1h_input_tokens: 20 * Number(write1h),
|
||||
cache_read_input_tokens: 1000 * Number(cached),
|
||||
});
|
||||
},
|
||||
);
|
||||
|
||||
it('returns the message verbatim and meters input/output/cache token costs', async () => {
|
||||
const { provider } = makeProvider();
|
||||
const msg = {
|
||||
|
||||
@@ -806,14 +806,12 @@ export class ClaudeProvider implements IChatProvider {
|
||||
}) {
|
||||
if (!reasoningEffort) return undefined;
|
||||
|
||||
// Fable 5/5.1, Opus 4.7+, 4.6, and Sonnet 4.6 use adaptive thinking
|
||||
// (`budget_tokens` is deprecated on 4.6/Sonnet 4.6, removed on
|
||||
// Fable 5+ and Opus 4.7+). Fable 5/5.1 and Opus 4.7+ omit thinking
|
||||
// content by default; `display: 'summarized'` restores visible
|
||||
// reasoning in the stream.
|
||||
// These models reject manual thinking budgets; summarized display
|
||||
// keeps reasoning visible in the stream.
|
||||
if (
|
||||
modelId === 'claude-fable-5-1' ||
|
||||
modelId === 'claude-fable-5' ||
|
||||
modelId === 'claude-sonnet-5' ||
|
||||
modelId === 'claude-opus-5-5' ||
|
||||
modelId === 'claude-opus-5' ||
|
||||
modelId === 'claude-opus-4-8' ||
|
||||
|
||||
@@ -44,7 +44,7 @@ export const CLAUDE_MODELS: IChatModel[] = [
|
||||
input_tokens: 1000,
|
||||
ephemeral_5m_input_tokens: 1000 * 1.25,
|
||||
ephemeral_1h_input_tokens: 1000 * 2,
|
||||
// Fable 5.1 bills cache reads at 0.025x input; every other Claude model is 0.1x.
|
||||
// Fable 5.1 bills cache reads at 0.025x input.
|
||||
cache_read_input_tokens: 1000 * 0.025,
|
||||
output_tokens: 5000,
|
||||
},
|
||||
@@ -80,6 +80,7 @@ export const CLAUDE_MODELS: IChatModel[] = [
|
||||
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
|
||||
open_weights: false,
|
||||
tool_call: true,
|
||||
knowledge: '2026-01',
|
||||
release_date: '2026-06-30',
|
||||
aliases: [
|
||||
'claude-sonnet',
|
||||
@@ -93,14 +94,14 @@ export const CLAUDE_MODELS: IChatModel[] = [
|
||||
output_cost_key: 'output_tokens',
|
||||
costs: {
|
||||
tokens: 1_000_000,
|
||||
input_tokens: 300,
|
||||
ephemeral_5m_input_tokens: 300 * 1.25,
|
||||
ephemeral_1h_input_tokens: 300 * 2,
|
||||
cache_read_input_tokens: 300 * 0.1,
|
||||
output_tokens: 1500,
|
||||
input_tokens: 200,
|
||||
ephemeral_5m_input_tokens: 200 * 1.25,
|
||||
ephemeral_1h_input_tokens: 200 * 2,
|
||||
cache_read_input_tokens: 200 * 0.1,
|
||||
output_tokens: 1000,
|
||||
},
|
||||
context: 1000000,
|
||||
max_tokens: 64000,
|
||||
max_tokens: 128000,
|
||||
},
|
||||
{
|
||||
puterId: 'anthropic:anthropic/claude-opus-5-5',
|
||||
@@ -126,7 +127,7 @@ export const CLAUDE_MODELS: IChatModel[] = [
|
||||
input_tokens: 400,
|
||||
ephemeral_5m_input_tokens: 400 * 1.25,
|
||||
ephemeral_1h_input_tokens: 400 * 2,
|
||||
cache_read_input_tokens: 400 * 0.1,
|
||||
cache_read_input_tokens: 400 * 0.05,
|
||||
output_tokens: 2000,
|
||||
},
|
||||
context: 1000000,
|
||||
|
||||
+66
-17
@@ -171,6 +171,8 @@ describe('OpenAiChatProvider model catalog', () => {
|
||||
// gpt-5-nano is a Chat-Completions model, must be present.
|
||||
expect(ids).toContain('gpt-5-nano-2025-08-07');
|
||||
expect(ids).toContain('gpt-6-astra');
|
||||
expect(ids).toContain('gpt-6-sol');
|
||||
expect(ids).toContain('gpt-6-luna');
|
||||
});
|
||||
|
||||
it('list() flattens canonical ids and aliases', () => {
|
||||
@@ -315,25 +317,30 @@ describe('OpenAiChatProvider.complete request shape', () => {
|
||||
expect(args.safety_identifier).toBe('puter-u42');
|
||||
});
|
||||
|
||||
it('resolves the namespaced GPT-6 Astra alias', async () => {
|
||||
const { provider } = makeProvider();
|
||||
createMock.mockResolvedValueOnce(baseCompletion);
|
||||
it.each(['gpt-6-astra', 'gpt-6-sol', 'gpt-6-luna'])(
|
||||
'resolves the namespaced %s alias',
|
||||
async (model) => {
|
||||
const { provider } = makeProvider();
|
||||
createMock.mockResolvedValueOnce(baseCompletion);
|
||||
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: 'openai/gpt-6-astra',
|
||||
messages: [{ role: 'user', content: 'hello' }],
|
||||
}),
|
||||
);
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: `openai/${model}`,
|
||||
messages: [{ role: 'user', content: 'hello' }],
|
||||
reasoning_effort: 'low',
|
||||
}),
|
||||
);
|
||||
|
||||
expect(createMock.mock.calls[0]![0].model).toBe('gpt-6-astra');
|
||||
expect(recordSpy).toHaveBeenCalledWith(
|
||||
expect.any(Object),
|
||||
expect.anything(),
|
||||
'openai:gpt-6-astra',
|
||||
expect.any(Object),
|
||||
);
|
||||
});
|
||||
expect(createMock.mock.calls[0]![0].model).toBe(model);
|
||||
expect(createMock.mock.calls[0]![0].reasoning_effort).toBe('low');
|
||||
expect(recordSpy).toHaveBeenCalledWith(
|
||||
expect.any(Object),
|
||||
expect.anything(),
|
||||
`openai:${model}`,
|
||||
expect.any(Object),
|
||||
);
|
||||
},
|
||||
);
|
||||
|
||||
it('forwards temperature 0 and max_tokens 0 instead of dropping them', async () => {
|
||||
const { provider } = makeProvider();
|
||||
@@ -514,6 +521,48 @@ describe('OpenAiChatProvider.complete non-stream output', () => {
|
||||
).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('splits cache writes out of prompt_tokens and bills them at 1.25x input', async () => {
|
||||
const luna = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-luna')!;
|
||||
const { provider } = makeProvider();
|
||||
createMock.mockResolvedValueOnce({
|
||||
choices: [
|
||||
{
|
||||
message: { content: 'hi', role: 'assistant' },
|
||||
finish_reason: 'stop',
|
||||
},
|
||||
],
|
||||
usage: {
|
||||
prompt_tokens: 5000,
|
||||
completion_tokens: 12,
|
||||
prompt_tokens_details: {
|
||||
cached_tokens: 1000,
|
||||
cache_write_tokens: 3000,
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: 'gpt-6-luna',
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
}),
|
||||
);
|
||||
|
||||
const [usage, , , overrides] = recordSpy.mock.calls[0]!;
|
||||
expect(usage).toEqual({
|
||||
prompt_tokens: 1000,
|
||||
completion_tokens: 12,
|
||||
cached_tokens: 1000,
|
||||
cache_write_tokens: 3000,
|
||||
});
|
||||
expect(overrides).toEqual({
|
||||
prompt_tokens: 1000 * Number(luna.costs.prompt_tokens),
|
||||
completion_tokens: 12 * Number(luna.costs.completion_tokens),
|
||||
cached_tokens: 1000 * Number(luna.costs.cached_tokens),
|
||||
cache_write_tokens: 3000 * Number(luna.costs.cache_write_tokens),
|
||||
});
|
||||
});
|
||||
|
||||
it('zeroes cached_tokens when prompt_tokens_details is missing', async () => {
|
||||
const { provider } = makeProvider();
|
||||
createMock.mockResolvedValueOnce({
|
||||
|
||||
@@ -217,13 +217,26 @@ export class OpenAiChatProvider implements IChatProvider {
|
||||
|
||||
return OpenAiUtil.handle_completion_output({
|
||||
usage_calculator: ({ usage }) => {
|
||||
const cachedTokens =
|
||||
usage.prompt_tokens_details?.cached_tokens ?? 0;
|
||||
// GPT-5.6 and later bill cache writes at 1.25x input. They're
|
||||
// reported inside `prompt_tokens`, like cached reads.
|
||||
// The SDK doesn't type `cache_write_tokens` yet.
|
||||
const cacheWriteTokens =
|
||||
(
|
||||
usage.prompt_tokens_details as
|
||||
{ cache_write_tokens?: number } | undefined
|
||||
)?.cache_write_tokens ?? 0;
|
||||
const trackedUsage = {
|
||||
prompt_tokens:
|
||||
(usage.prompt_tokens ?? 0) -
|
||||
(usage.prompt_tokens_details?.cached_tokens ?? 0),
|
||||
cachedTokens -
|
||||
cacheWriteTokens,
|
||||
completion_tokens: usage.completion_tokens ?? 0,
|
||||
cached_tokens:
|
||||
usage.prompt_tokens_details?.cached_tokens ?? 0,
|
||||
cached_tokens: cachedTokens,
|
||||
...(cacheWriteTokens
|
||||
? { cache_write_tokens: cacheWriteTokens }
|
||||
: {}),
|
||||
};
|
||||
|
||||
const costsOverrideFromModel = buildCostsOverride(
|
||||
|
||||
@@ -198,6 +198,38 @@ describe('OpenAiResponsesChatProvider model catalog', () => {
|
||||
expect(ids).not.toContain('gpt-5-nano-2025-08-07');
|
||||
});
|
||||
|
||||
it.each([
|
||||
['gpt-6-sol', '2026-04-20', 200, 20, 1000],
|
||||
['gpt-6-luna', '2026-05-18', 10, 1, 50],
|
||||
])(
|
||||
'exposes %s with current pricing and limits',
|
||||
(id, knowledge, input, cached, output) => {
|
||||
const { provider } = makeProvider();
|
||||
expect(
|
||||
provider.models().find((model) => model.id === id),
|
||||
).toMatchObject({
|
||||
puterId: `openai:openai/${id}`,
|
||||
aliases: [`openai/${id}`],
|
||||
knowledge,
|
||||
release_date: '2026-09-22',
|
||||
modalities: { input: ['text', 'image'], output: ['text'] },
|
||||
costs_currency: 'usd-cents',
|
||||
costs: {
|
||||
tokens: 1_000_000,
|
||||
prompt_tokens: input,
|
||||
cached_tokens: cached,
|
||||
completion_tokens: output,
|
||||
},
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api: true,
|
||||
});
|
||||
expect(provider.list()).toEqual(
|
||||
expect.arrayContaining([id, `openai/${id}`]),
|
||||
);
|
||||
},
|
||||
);
|
||||
|
||||
it('models({ no_restrictions: true }) returns the entire catalog (used by complete())', () => {
|
||||
const { provider } = makeProvider();
|
||||
const ids = provider
|
||||
@@ -386,6 +418,58 @@ describe('OpenAiResponsesChatProvider.complete request shape', () => {
|
||||
expect(o3Args.reasoning_effort).toBe('medium');
|
||||
expect(o3Args.verbosity).toBe('low');
|
||||
});
|
||||
|
||||
it.each(['gpt-6-astra', 'gpt-6-sol', 'gpt-6-luna'])(
|
||||
'maps flat controls to Responses options for %s aliases',
|
||||
async (model) => {
|
||||
const { provider } = makeProvider();
|
||||
responsesCreateMock.mockResolvedValueOnce(baseResponse);
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: `openai/${model}`,
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
reasoning_effort: 'high',
|
||||
verbosity: 'low',
|
||||
}),
|
||||
);
|
||||
const [args] = responsesCreateMock.mock.calls[0]!;
|
||||
expect(args.model).toBe(model);
|
||||
expect(args.reasoning).toEqual({ effort: 'high' });
|
||||
expect(args.text).toEqual({ verbosity: 'low' });
|
||||
expect(args).not.toHaveProperty('reasoning_effort');
|
||||
expect(args).not.toHaveProperty('verbosity');
|
||||
expect(recordSpy.mock.calls[0]![2]).toBe(`openai:${model}`);
|
||||
},
|
||||
);
|
||||
|
||||
it('preserves nested GPT-6 controls and gives flat options precedence', async () => {
|
||||
const { provider } = makeProvider();
|
||||
const reasoning = { effort: 'medium', summary: 'auto' };
|
||||
const text = { verbosity: 'high', format: { type: 'text' } };
|
||||
for (const flat of [false, true]) {
|
||||
responsesCreateMock.mockResolvedValueOnce(baseResponse);
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: 'gpt-6-sol',
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
reasoning,
|
||||
text,
|
||||
...(flat
|
||||
? { reasoning_effort: 'low', verbosity: 'low' }
|
||||
: {}),
|
||||
} as never),
|
||||
);
|
||||
const [args] = responsesCreateMock.mock.lastCall!;
|
||||
expect(args.reasoning).toEqual({
|
||||
...reasoning,
|
||||
effort: flat ? 'low' : 'medium',
|
||||
});
|
||||
expect(args.text).toEqual({
|
||||
...text,
|
||||
verbosity: flat ? 'low' : 'high',
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// ── Model resolution ────────────────────────────────────────────────
|
||||
@@ -518,6 +602,89 @@ describe('OpenAiResponsesChatProvider.complete non-stream output', () => {
|
||||
});
|
||||
});
|
||||
|
||||
it('splits cache writes out of input and bills them at 1.25x input', async () => {
|
||||
const luna = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-luna')!;
|
||||
const { provider } = makeProvider();
|
||||
responsesCreateMock.mockResolvedValueOnce({
|
||||
output: [{ role: 'assistant' }],
|
||||
output_text: 'hi',
|
||||
usage: {
|
||||
input_tokens: 5000,
|
||||
output_tokens: 20,
|
||||
input_tokens_details: {
|
||||
cached_tokens: 1000,
|
||||
cache_write_tokens: 3000,
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: 'gpt-6-luna',
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
}),
|
||||
);
|
||||
|
||||
const [usage, , , overrides] = recordSpy.mock.calls[0]!;
|
||||
expect(usage).toEqual({
|
||||
prompt_tokens: 1000,
|
||||
completion_tokens: 20,
|
||||
cached_tokens: 1000,
|
||||
cache_write_tokens: 3000,
|
||||
});
|
||||
expect(luna.costs.cache_write_tokens).toBe(
|
||||
Number(luna.costs.prompt_tokens) * 1.25,
|
||||
);
|
||||
expect(overrides).toEqual({
|
||||
prompt_tokens: 1000 * Number(luna.costs.prompt_tokens),
|
||||
completion_tokens: 20 * Number(luna.costs.completion_tokens),
|
||||
cached_tokens: 1000 * Number(luna.costs.cached_tokens),
|
||||
cache_write_tokens: 3000 * Number(luna.costs.cache_write_tokens),
|
||||
});
|
||||
});
|
||||
|
||||
it('bills the whole request at long-context rates past 272K input tokens', async () => {
|
||||
const sol = OPEN_AI_MODELS.find((m) => m.id === 'gpt-6-sol')!;
|
||||
const { provider } = makeProvider();
|
||||
responsesCreateMock.mockResolvedValueOnce({
|
||||
output: [{ role: 'assistant' }],
|
||||
output_text: 'hi',
|
||||
usage: {
|
||||
input_tokens: 300_000,
|
||||
output_tokens: 10_000,
|
||||
input_tokens_details: {
|
||||
cached_tokens: 50_000,
|
||||
cache_write_tokens: 20_000,
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
await withTestActor(() =>
|
||||
provider.complete({
|
||||
model: 'gpt-6-sol',
|
||||
messages: [{ role: 'user', content: 'hi' }],
|
||||
}),
|
||||
);
|
||||
|
||||
const [, , , overrides] = recordSpy.mock.calls[0]!;
|
||||
// $0.92 + $0.02 + $0.10 + $0.15 = $1.19, against $0.62 at the
|
||||
// standard rates.
|
||||
expect(overrides).toEqual({
|
||||
prompt_tokens: 230_000 * Number(sol.costs.prompt_tokens) * 2,
|
||||
cached_tokens: 50_000 * Number(sol.costs.cached_tokens) * 2,
|
||||
cache_write_tokens:
|
||||
20_000 * Number(sol.costs.cache_write_tokens) * 2,
|
||||
completion_tokens:
|
||||
10_000 * Number(sol.costs.completion_tokens) * 1.5,
|
||||
});
|
||||
const totalCents =
|
||||
Object.values(overrides as Record<string, number>).reduce(
|
||||
(a, b) => a + b,
|
||||
0,
|
||||
) / 1_000_000;
|
||||
expect(totalCents).toBeCloseTo(119);
|
||||
});
|
||||
|
||||
it('bills cached tokens at the input rate when the model prices no cache read', async () => {
|
||||
// gpt-5.4-pro is responses-API-only and its catalogue entry has no
|
||||
// cached_tokens rate. Cached tokens are subtracted out of the input
|
||||
|
||||
@@ -175,8 +175,10 @@ export class OpenAiResponsesChatProvider implements IChatProvider {
|
||||
|
||||
const requestedReasoningEffort = reasoning_effort ?? reasoning?.effort;
|
||||
const requestedVerbosity = verbosity ?? text?.verbosity;
|
||||
const isGpt6Model = modelUsed.id.startsWith('gpt-6-');
|
||||
const supportsReasoningControls =
|
||||
typeof model === 'string' && model.startsWith('gpt-5');
|
||||
isGpt6Model ||
|
||||
(typeof model === 'string' && model.startsWith('gpt-5'));
|
||||
|
||||
// Translate the neutral compaction opt-in (or pass a raw
|
||||
// `context_management` payload through) to OpenAI's Responses shape.
|
||||
@@ -232,6 +234,17 @@ export class OpenAiResponsesChatProvider implements IChatProvider {
|
||||
: {}),
|
||||
}),
|
||||
...(supportsReasoningControls && reasoning ? { reasoning } : {}),
|
||||
...(isGpt6Model && requestedReasoningEffort !== undefined
|
||||
? {
|
||||
reasoning: {
|
||||
...reasoning,
|
||||
effort: requestedReasoningEffort,
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
...(isGpt6Model && requestedVerbosity !== undefined
|
||||
? { text: { ...text, verbosity: requestedVerbosity } }
|
||||
: {}),
|
||||
} as unknown as ResponseCreateParams;
|
||||
|
||||
// console.log("completion params: ", completionParams)
|
||||
@@ -240,14 +253,23 @@ export class OpenAiResponsesChatProvider implements IChatProvider {
|
||||
// console.log("Completion: ", completion)
|
||||
return OpenAiUtil.handle_completion_output_responses_api({
|
||||
usage_calculator: ({ usage }) => {
|
||||
const cachedTokens =
|
||||
(usage as any).input_tokens_details?.cached_tokens ?? 0;
|
||||
// GPT-5.6 and later bill cache writes at 1.25x input. They're
|
||||
// reported inside `input_tokens`, like cached reads.
|
||||
const cacheWriteTokens =
|
||||
(usage as any).input_tokens_details?.cache_write_tokens ??
|
||||
0;
|
||||
const trackedUsage = {
|
||||
prompt_tokens:
|
||||
((usage as any).input_tokens ?? 0) -
|
||||
((usage as any).input_tokens_details?.cached_tokens ??
|
||||
0),
|
||||
cachedTokens -
|
||||
cacheWriteTokens,
|
||||
completion_tokens: (usage as any).output_tokens ?? 0,
|
||||
cached_tokens:
|
||||
(usage as any).input_tokens_details?.cached_tokens ?? 0,
|
||||
cached_tokens: cachedTokens,
|
||||
...(cacheWriteTokens
|
||||
? { cache_write_tokens: cacheWriteTokens }
|
||||
: {}),
|
||||
};
|
||||
|
||||
const costsOverrideFromModel = buildCostsOverride(
|
||||
|
||||
@@ -21,8 +21,64 @@
|
||||
|
||||
import type { IChatModel } from '../../types.js';
|
||||
|
||||
// Prompts over 272K input tokens are billed at 2x input (cached reads and
|
||||
// cache writes included) and 1.5x output for the full request.
|
||||
const GPT_LONG_CONTEXT_PRICING = {
|
||||
threshold: 272_000,
|
||||
input_multiplier: 2,
|
||||
output_multiplier: 1.5,
|
||||
};
|
||||
|
||||
// Hardcoded from https://models.dev/api.json
|
||||
export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
{
|
||||
puterId: 'openai:openai/gpt-6-sol',
|
||||
id: 'gpt-6-sol',
|
||||
modalities: { input: ['text', 'image'], output: ['text'] },
|
||||
open_weights: false,
|
||||
tool_call: true,
|
||||
knowledge: '2026-04-20',
|
||||
release_date: '2026-09-22',
|
||||
aliases: ['openai/gpt-6-sol'],
|
||||
costs_currency: 'usd-cents',
|
||||
input_cost_key: 'prompt_tokens',
|
||||
output_cost_key: 'completion_tokens',
|
||||
costs: {
|
||||
tokens: 1_000_000,
|
||||
prompt_tokens: 200,
|
||||
cached_tokens: 20,
|
||||
cache_write_tokens: 200 * 1.25,
|
||||
completion_tokens: 1000,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api: true,
|
||||
},
|
||||
{
|
||||
puterId: 'openai:openai/gpt-6-luna',
|
||||
id: 'gpt-6-luna',
|
||||
modalities: { input: ['text', 'image'], output: ['text'] },
|
||||
open_weights: false,
|
||||
tool_call: true,
|
||||
knowledge: '2026-05-18',
|
||||
release_date: '2026-09-22',
|
||||
aliases: ['openai/gpt-6-luna'],
|
||||
costs_currency: 'usd-cents',
|
||||
input_cost_key: 'prompt_tokens',
|
||||
output_cost_key: 'completion_tokens',
|
||||
costs: {
|
||||
tokens: 1_000_000,
|
||||
prompt_tokens: 10,
|
||||
cached_tokens: 1,
|
||||
cache_write_tokens: 10 * 1.25,
|
||||
completion_tokens: 50,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api: true,
|
||||
},
|
||||
{
|
||||
puterId: 'openai:openai/gpt-6-astra',
|
||||
id: 'gpt-6-astra',
|
||||
@@ -39,8 +95,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
tokens: 1_000_000,
|
||||
prompt_tokens: 1000,
|
||||
cached_tokens: 100,
|
||||
cache_write_tokens: 1000 * 1.25,
|
||||
completion_tokens: 5000,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api: true,
|
||||
@@ -56,12 +114,15 @@ export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
costs_currency: 'usd-cents',
|
||||
input_cost_key: 'prompt_tokens',
|
||||
output_cost_key: 'completion_tokens',
|
||||
// OpenAI's promotional pricing, guaranteed only through 2026-11-21.
|
||||
costs: {
|
||||
tokens: 1_000_000,
|
||||
prompt_tokens: 500,
|
||||
cached_tokens: 50,
|
||||
completion_tokens: 3000,
|
||||
prompt_tokens: 400,
|
||||
cached_tokens: 40,
|
||||
cache_write_tokens: 400 * 1.25,
|
||||
completion_tokens: 2000,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api_only: true,
|
||||
@@ -81,8 +142,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
tokens: 1_000_000,
|
||||
prompt_tokens: 200,
|
||||
cached_tokens: 20,
|
||||
cache_write_tokens: 200 * 1.25,
|
||||
completion_tokens: 1200,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api_only: true,
|
||||
@@ -102,8 +165,10 @@ export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
tokens: 1_000_000,
|
||||
prompt_tokens: 20,
|
||||
cached_tokens: 2,
|
||||
cache_write_tokens: 20 * 1.25,
|
||||
completion_tokens: 120,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api_only: true,
|
||||
@@ -126,6 +191,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
cached_tokens: 50,
|
||||
completion_tokens: 3000,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
},
|
||||
@@ -169,6 +235,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
cached_tokens: 25,
|
||||
completion_tokens: 1500,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 1_050_000,
|
||||
},
|
||||
@@ -189,6 +256,7 @@ export const OPEN_AI_MODELS: IChatModel[] = [
|
||||
prompt_tokens: 3000,
|
||||
completion_tokens: 18000,
|
||||
},
|
||||
long_context_pricing: GPT_LONG_CONTEXT_PRICING,
|
||||
context: 1_050_000,
|
||||
max_tokens: 128_000,
|
||||
responses_api_only: true,
|
||||
|
||||
@@ -43,6 +43,17 @@ export interface IChatModel<T extends ModelCost = ModelCost> extends Record<
|
||||
input_cost_key?: keyof T;
|
||||
output_cost_key?: keyof T;
|
||||
costs: T;
|
||||
/**
|
||||
* A request whose input exceeds `threshold` tokens is billed at raised
|
||||
* rates for the whole request, not only the tokens past the threshold:
|
||||
* every input-side rate (uncached, cached, cache writes) is multiplied by
|
||||
* `input_multiplier` and every output-side rate by `output_multiplier`.
|
||||
*/
|
||||
long_context_pricing?: {
|
||||
threshold: number;
|
||||
input_multiplier: number;
|
||||
output_multiplier: number;
|
||||
};
|
||||
context?: number;
|
||||
max_tokens: number;
|
||||
subscriberOnly?: boolean;
|
||||
|
||||
@@ -19,7 +19,12 @@
|
||||
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import type { IChatModel } from '../types.js';
|
||||
import { buildCostsOverride, isFreeModel, usdPerMToken } from './pricing.js';
|
||||
import {
|
||||
buildCostsOverride,
|
||||
isFreeModel,
|
||||
longContextMultipliers,
|
||||
usdPerMToken,
|
||||
} from './pricing.js';
|
||||
|
||||
const model = (costs: Record<string, number>): IChatModel =>
|
||||
({
|
||||
@@ -166,3 +171,91 @@ describe('buildCostsOverride', () => {
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('long-context pricing', () => {
|
||||
const longContext = (costs: Record<string, number>): IChatModel => ({
|
||||
...model(costs),
|
||||
long_context_pricing: {
|
||||
threshold: 272_000,
|
||||
input_multiplier: 2,
|
||||
output_multiplier: 1.5,
|
||||
},
|
||||
});
|
||||
const rates = {
|
||||
prompt_tokens: 200,
|
||||
cached_tokens: 20,
|
||||
cache_write_tokens: 250,
|
||||
completion_tokens: 1000,
|
||||
};
|
||||
|
||||
it('bills a request at or under the threshold at standard rates', () => {
|
||||
const overrides = buildCostsOverride(
|
||||
{
|
||||
prompt_tokens: 222_000,
|
||||
cached_tokens: 50_000,
|
||||
completion_tokens: 10,
|
||||
},
|
||||
longContext(rates),
|
||||
);
|
||||
|
||||
expect(overrides).toEqual({
|
||||
prompt_tokens: 222_000 * 200,
|
||||
cached_tokens: 50_000 * 20,
|
||||
completion_tokens: 10 * 1000,
|
||||
});
|
||||
});
|
||||
|
||||
it('raises every rate for the whole request once input passes the threshold', () => {
|
||||
// 230K uncached + 50K cached + 20K cache writes = 300K input. No
|
||||
// single key crosses 272K; their sum does.
|
||||
const overrides = buildCostsOverride(
|
||||
{
|
||||
prompt_tokens: 230_000,
|
||||
cached_tokens: 50_000,
|
||||
cache_write_tokens: 20_000,
|
||||
completion_tokens: 10_000,
|
||||
},
|
||||
longContext(rates),
|
||||
);
|
||||
|
||||
expect(overrides).toEqual({
|
||||
prompt_tokens: 230_000 * 200 * 2,
|
||||
cached_tokens: 50_000 * 20 * 2,
|
||||
cache_write_tokens: 20_000 * 250 * 2,
|
||||
completion_tokens: 10_000 * 1000 * 1.5,
|
||||
});
|
||||
});
|
||||
|
||||
it('raises the fallback rate of an unpriced key too', () => {
|
||||
const overrides = buildCostsOverride(
|
||||
{ prompt_tokens: 300_000, thinking_tokens: 10 },
|
||||
longContext({ prompt_tokens: 200, completion_tokens: 1000 }),
|
||||
);
|
||||
|
||||
expect(overrides.thinking_tokens).toBe(10 * 1000 * 1.5);
|
||||
});
|
||||
|
||||
it('leaves a model without long-context pricing at standard rates', () => {
|
||||
const overrides = buildCostsOverride(
|
||||
{ prompt_tokens: 900_000, completion_tokens: 10 },
|
||||
model(rates),
|
||||
);
|
||||
|
||||
expect(overrides).toEqual({
|
||||
prompt_tokens: 900_000 * 200,
|
||||
completion_tokens: 10 * 1000,
|
||||
});
|
||||
});
|
||||
|
||||
it('applies the multipliers strictly above the threshold', () => {
|
||||
const m = longContext(rates);
|
||||
expect(longContextMultipliers(m, 272_000)).toEqual({
|
||||
input: 1,
|
||||
output: 1,
|
||||
});
|
||||
expect(longContextMultipliers(m, 272_001)).toEqual({
|
||||
input: 2,
|
||||
output: 1.5,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -54,6 +54,56 @@ export const costKeys = (
|
||||
outputKey: (model.output_cost_key as string | undefined) ?? 'output_tokens',
|
||||
});
|
||||
|
||||
/**
|
||||
* Whether a usage key is priced at the output rate when the model has no rate
|
||||
* of its own for it.
|
||||
*/
|
||||
export const isOutputCostKey = (key: string, outputKey: string): boolean =>
|
||||
key === outputKey ||
|
||||
key === 'output_tokens' ||
|
||||
key === 'completion_tokens' ||
|
||||
key === 'thinking_tokens';
|
||||
|
||||
/**
|
||||
* The rate multipliers a request pays given how many input tokens it sent —
|
||||
* cached reads and cache writes included. `1`/`1` unless the model has
|
||||
* long-context pricing and the request is past its threshold.
|
||||
*/
|
||||
export const longContextMultipliers = (
|
||||
model: IChatModel,
|
||||
inputTokens: number,
|
||||
): { input: number; output: number } => {
|
||||
const pricing = model.long_context_pricing;
|
||||
if (!pricing || !(inputTokens > pricing.threshold)) {
|
||||
return { input: 1, output: 1 };
|
||||
}
|
||||
return {
|
||||
input: pricing.input_multiplier,
|
||||
output: pricing.output_multiplier,
|
||||
};
|
||||
};
|
||||
|
||||
/**
|
||||
* The input tokens a tracked-usage object carries: every key that isn't
|
||||
* output-side. Providers split one prompt into uncached, cached-read and
|
||||
* cache-write keys; the long-context threshold is measured on their sum.
|
||||
*/
|
||||
export const trackedInputTokens = (
|
||||
trackedUsage: Record<string, unknown>,
|
||||
model: IChatModel,
|
||||
): number => {
|
||||
const { outputKey } = costKeys(model);
|
||||
let total = 0;
|
||||
for (const [key, amount] of Object.entries(trackedUsage)) {
|
||||
if (key === 'tokens' || key === 'usd_cents') continue;
|
||||
if (isOutputCostKey(key, outputKey)) continue;
|
||||
if (typeof amount === 'number' && Number.isFinite(amount)) {
|
||||
total += amount;
|
||||
}
|
||||
}
|
||||
return total;
|
||||
};
|
||||
|
||||
/**
|
||||
* Whether a model costs the user nothing to run.
|
||||
*
|
||||
@@ -90,11 +140,10 @@ export const buildCostsOverride = (
|
||||
const inputRate = isRate(costs[inputKey]) ? costs[inputKey] : undefined;
|
||||
const outputRate = isRate(costs[outputKey]) ? costs[outputKey] : undefined;
|
||||
|
||||
const isOutputKey = (key: string) =>
|
||||
key === outputKey ||
|
||||
key === 'output_tokens' ||
|
||||
key === 'completion_tokens' ||
|
||||
key === 'thinking_tokens';
|
||||
const multipliers = longContextMultipliers(
|
||||
model,
|
||||
trackedInputTokens(trackedUsage, model),
|
||||
);
|
||||
|
||||
const overrides: Record<string, number> = {};
|
||||
for (const [key, amount] of Object.entries(trackedUsage)) {
|
||||
@@ -102,11 +151,13 @@ export const buildCostsOverride = (
|
||||
// not a per-unit rate.
|
||||
if (key === 'tokens') continue;
|
||||
|
||||
const isOutput = isOutputCostKey(key, outputKey);
|
||||
const rate = isRate(costs[key])
|
||||
? costs[key]
|
||||
: ((isOutputKey(key) ? outputRate : inputRate) ?? 0);
|
||||
: ((isOutput ? outputRate : inputRate) ?? 0);
|
||||
|
||||
overrides[key] = amount * rate;
|
||||
overrides[key] =
|
||||
amount * rate * (isOutput ? multipliers.output : multipliers.input);
|
||||
}
|
||||
|
||||
return overrides;
|
||||
|
||||
Reference in New Issue
Block a user