fix(ai-chat): derive Azure model pricing from the OpenAI and xAI catalogs

Azure entries that front an OpenAI or xAI model now come from a `mirror`
helper that copies every non-identity field from the source catalog entry.
The hand-copied entries had drifted: none carried `long_context_pricing`, so
long prompts on Azure-served gpt-5.4 and Grok 4.3/4.20 were billed at 1x;
grok-4.3 had lost `pdf` input and the Grok 4.20 deployments still listed the
old 2M context. Models no longer in the direct catalogs stay inline.
This commit is contained in:
Daniel Salazar committed 2026-10-10 15:47:50 -04:00
1 parent d4589caefe
commit 911e220362
3 files changed
+236 -231

No files matched your search

@@ -45,6 +45,7 @@ import { setupTestServer } from '../../testUtil.js';
import { kv } from '../../util/kvSingleton.js';
import { withTestActor } from '../integrationTestUtil.js';
import { ChatCompletionDriver } from './ChatCompletionDriver.js';
import { OPEN_AI_MODELS } from './providers/openai/models.js';
import {
clearUnhealthyRoutes,
isRouteUnhealthy,
@@ -564,3 +565,83 @@ describe('ChatCompletionDriver timeout classification across the chain', () => {
});
});
});
// Azure is served ahead of OpenAI for the models it fronts, at OpenAI's list
// price — the long-context tier included.
describe('ChatCompletionDriver Azure-served OpenAI pricing', () => {
let azureDriver: ChatCompletionDriver;
beforeAll(async () => {
azureDriver = new ChatCompletionDriver(
{
providers: {
'azure-openai': {
apiKey: 'test-key',
apiURL: 'https://azure.test/openai/v1',
},
'openai-completion': { apiKey: 'test-key' },
ollama: { enabled: false },
},
} as never,
server.clients,
server.stores,
server.services,
);
azureDriver.onServerStart();
for (let i = 0; i < 200; i++) {
if ((await azureDriver.list()).includes('azure:openai/gpt-5.4')) {
break;
}
await new Promise((r) => setTimeout(r, 5));
}
});
it('bills a gpt-5.4 prompt past the long-context threshold at the raised rates', async () => {
const recorded: MockInstance = vi.spyOn(
server.services.metering,
'utilRecordUsageObject',
);
// 300K input tokens (100K of them cached) is past OpenAI's 272K
// threshold: 2x every input rate and 1.5x output.
createMock.mockResolvedValueOnce({
choices: [
{
message: { role: 'assistant', content: 'ok' },
finish_reason: 'stop',
},
],
usage: {
prompt_tokens: 300_000,
completion_tokens: 1_000,
prompt_tokens_details: { cached_tokens: 100_000 },
},
});
const res = (await withTestActor(() =>
azureDriver.complete({
model: 'gpt-5.4',
messages: [{ role: 'user', content: 'summarize this' }],
}),
)) as { usage: Record<string, number> };
const { costs } = OPEN_AI_MODELS.find(
(m) => m.puterId === 'openai:openai/gpt-5.4',
)!;
const expected = {
prompt_tokens: 200_000 * costs.prompt_tokens * 2,
completion_tokens: 1_000 * costs.completion_tokens * 1.5,
cached_tokens: 100_000 * costs.cached_tokens * 2,
};
const call = recorded.mock.calls.find(
([, , prefix]) => prefix === 'azure-openai:gpt-5.4',
);
expect(call?.[3]).toEqual(expected);
expect(res.usage.usd_cents).toBeCloseTo(
(expected.prompt_tokens +
expected.completion_tokens +
expected.cached_tokens) /
1_000_000,
);
recorded.mockRestore();
});
});
@@ -0,0 +1,106 @@
/*
* Copyright (C) 2024-present Puter Technologies Inc.
*
* This file is part of Puter.
*
* Puter is free software: you can redistribute it and/or modify
* it under the terms of the GNU Affero General Public License as published
* by the Free Software Foundation, either version 3 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU Affero General Public License for more details.
*
* You should have received a copy of the GNU Affero General Public License
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*/
/**
* Azure deployments bill at the direct provider's list price, so an entry that
* mirrors an OpenAI or xAI model must match it on every field but its
* identity.
*/
import { describe, expect, it } from 'vitest';
import type { IChatModel } from '../../types.js';
import { OPEN_AI_MODELS } from '../openai/models.js';
import { XAI_MODELS } from '../xai/models.js';
import { AZURE_MODELS } from './models.js';
// Deployments of models the direct providers have retired: nothing to mirror.
const STANDALONE = [
'azure:x-ai/grok-4-1-fast-non-reasoning',
'azure:x-ai/grok-4-1-fast-reasoning',
'azure:openai/gpt-5',
'azure:openai/gpt-5-nano',
'azure:openai/gpt-5-mini',
];
// Fields an Azure deployment genuinely differs from its source on, by puterId.
const OVERRIDES: Record<string, string[]> = {};
const IDENTITY = ['id', 'puterId', 'aliases'];
// `azure:openai/gpt-5.4` mirrors `openai:openai/gpt-5.4`.
const unprefixed = (puterId: string | undefined) =>
puterId?.slice(puterId.indexOf(':') + 1);
const sourceOf = (azure: IChatModel): IChatModel | undefined =>
[...OPEN_AI_MODELS, ...XAI_MODELS].find(
(m) => unprefixed(m.puterId) === unprefixed(azure.puterId),
);
const sharedFields = (model: IChatModel, skip: string[]) =>
Object.fromEntries(
Object.entries(model).filter(
([key]) => !IDENTITY.includes(key) && !skip.includes(key),
),
);
const mirrored = AZURE_MODELS.filter(
(m) => !STANDALONE.includes(m.puterId!),
).map((m) => [m.puterId!, m] as const);
describe('AZURE_MODELS mirroring', () => {
it('has a direct-provider source for every entry not listed as standalone', () => {
const orphans = mirrored
.filter(([, m]) => !sourceOf(m))
.map(([puterId]) => puterId);
expect(orphans).toEqual([]);
expect(mirrored.length).toBeGreaterThan(0);
});
it('lists no standalone entry that a direct provider still serves', () => {
const stillServed = AZURE_MODELS.filter(
(m) => STANDALONE.includes(m.puterId!) && sourceOf(m),
).map((m) => m.puterId);
expect(stillServed).toEqual([]);
});
it.each(mirrored)(
'%s prices and caps exactly like its source',
(puterId, azure) => {
const source = sourceOf(azure)!;
const skip = OVERRIDES[puterId] ?? [];
// The fields billing and the credit gate read, spelled out so a
// failure names them.
for (const field of [
'costs',
'long_context_pricing',
'modalities',
'context',
'max_tokens',
]) {
if (skip.includes(field)) continue;
expect(azure[field], field).toEqual(source[field]);
}
expect(sharedFields(azure, skip)).toEqual(
sharedFields(source, skip),
);
},
);
});
@@ -18,22 +18,49 @@
*/
import type { IChatModel } from '../../types.js';
import { OPEN_AI_MODELS } from '../openai/models.js';
import { XAI_MODELS } from '../xai/models.js';
/**
* An Azure deployment of a model we also serve direct. Every field but the
* identity comes from the source entry, so prices, long-context tiers and
* limits can't drift from it. `name` is the vendor-qualified name both catalogs
* share (`openai/gpt-5.4`); extra fields override the source.
*/
const mirror = (
source: readonly IChatModel[],
name: string,
azure: { id: string } & Partial<IChatModel>,
): IChatModel => {
const model = source.find((m) => m.puterId?.endsWith(`:${name}`));
if (!model) {
throw new Error(`Azure model ${name} has no source entry to mirror`);
}
const { id: _id, puterId: _puterId, aliases: _aliases, ...shared } = model;
return {
puterId: `azure:${name}`,
id: azure.id,
...shared,
aliases: [name],
...azure,
};
};
// Models served through our Azure AI Foundry deployment. This is NOT just
// OpenAI — Azure AI also fronts xAI's Grok models — so the list lives in its
// own provider folder rather than sharing the OpenAI list.
//
// IMPORTANT: the `costs` below intentionally mirror the public list prices of
// the equivalent OpenAI / xAI models (see `../openai/models.ts` and
// `../xai/models.ts`). Azure is subsidised for us, so our actual spend is
// lower — but we bill users at the standard model price, which is the whole
// reason we route through Azure. Do NOT replace these with Azure's own rates.
// IMPORTANT: we bill users at the public list price of the equivalent OpenAI /
// xAI model, which `mirror` copies from `../openai/models.ts` and
// `../xai/models.ts`. Azure is subsidised for us, so our actual spend is lower
// — but billing at the standard model price is the whole reason we route
// through Azure. Do NOT replace these with Azure's own rates.
//
// `id` is the Azure deployment name and is what we send upstream.
export const AZURE_MODELS: IChatModel[] = [
// -- xAI Grok (via Azure AI Foundry) -----------------------------------
{
// Costs mirror xai grok-4-1-fast-non-reasoning.
// Not in the direct xAI catalog any more; last xAI list price.
puterId: 'azure:x-ai/grok-4-1-fast-non-reasoning',
id: 'grok-4-1-fast-non-reasoning',
modalities: { input: ['text', 'image'], output: ['text'] },
@@ -56,7 +83,7 @@ export const AZURE_MODELS: IChatModel[] = [
max_tokens: 2_000_000,
},
{
// Costs mirror xai grok-4-1-fast (alias grok-4-1-fast-reasoning).
// Not in the direct xAI catalog any more; last xAI list price.
puterId: 'azure:x-ai/grok-4-1-fast-reasoning',
id: 'grok-4-1-fast-reasoning',
modalities: { input: ['text', 'image'], output: ['text'] },
@@ -78,78 +105,17 @@ export const AZURE_MODELS: IChatModel[] = [
},
max_tokens: 2_000_000,
},
{
// Costs mirror xai grok-4.3.
puterId: 'azure:x-ai/grok-4.3',
id: 'grok-4.3',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
release_date: '2026-05-01',
name: 'Grok 4.3',
aliases: ['x-ai/grok-4.3'],
context: 1_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
{
// Costs mirror xai grok-4.20 (grok-4.20-0309-non-reasoning).
puterId: 'azure:x-ai/grok-4-20-non-reasoning',
mirror(XAI_MODELS, 'x-ai/grok-4.3', { id: 'grok-4.3' }),
mirror(XAI_MODELS, 'x-ai/grok-4-20-non-reasoning', {
id: 'grok-4-20-non-reasoning',
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2026-03-09',
name: 'Grok 4.20 (Non-Reasoning)',
aliases: ['x-ai/grok-4-20-non-reasoning'],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
{
// Costs mirror xai grok-4.20 (grok-4.20-0309-reasoning).
puterId: 'azure:x-ai/grok-4-20-reasoning',
}),
mirror(XAI_MODELS, 'x-ai/grok-4-20-reasoning', {
id: 'grok-4-20-reasoning',
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2026-03-09',
name: 'Grok 4.20 (Reasoning)',
aliases: ['x-ai/grok-4-20-reasoning'],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
}),
// -- OpenAI (via Azure AI Foundry) -------------------------------------
{
// Costs mirror openai gpt-5.
// Not in the direct OpenAI catalog any more; last OpenAI list price.
puterId: 'azure:openai/gpt-5',
id: 'gpt-5',
modalities: { input: ['text', 'image'], output: ['text'] },
@@ -171,7 +137,7 @@ export const AZURE_MODELS: IChatModel[] = [
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5-nano.
// Not in the direct OpenAI catalog any more; last OpenAI list price.
puterId: 'azure:openai/gpt-5-nano',
id: 'gpt-5-nano',
modalities: { input: ['text', 'image'], output: ['text'] },
@@ -193,7 +159,7 @@ export const AZURE_MODELS: IChatModel[] = [
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5-mini.
// Not in the direct OpenAI catalog any more; last OpenAI list price.
puterId: 'azure:openai/gpt-5-mini',
id: 'gpt-5-mini',
modalities: { input: ['text', 'image'], output: ['text'] },
@@ -214,159 +180,11 @@ export const AZURE_MODELS: IChatModel[] = [
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-4o.
puterId: 'azure:openai/gpt-4o',
id: 'gpt-4o',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2023-09',
release_date: '2024-05-13',
aliases: ['openai/gpt-4o'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 250,
cached_tokens: 125,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 16384,
},
{
// Costs mirror openai gpt-5.1.
puterId: 'azure:openai/gpt-5.1',
id: 'gpt-5.1',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-09-30',
release_date: '2025-11-13',
aliases: ['openai/gpt-5.1'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
cached_tokens: 13,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.2.
puterId: 'azure:openai/gpt-5.2',
id: 'gpt-5.2',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
release_date: '2025-12-11',
aliases: ['openai/gpt-5.2'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 175,
cached_tokens: 17.5,
completion_tokens: 1400,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.4-nano.
puterId: 'azure:openai/gpt-5.4-nano',
id: 'gpt-5.4-nano',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
release_date: '2026-03-19',
aliases: ['openai/gpt-5.4-nano'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 20,
cached_tokens: 2,
completion_tokens: 125,
},
context: 400_000,
max_tokens: 128_000,
},
{
// Costs mirror openai gpt-5.4-mini.
puterId: 'azure:openai/gpt-5.4-mini',
id: 'gpt-5.4-mini',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
aliases: ['openai/gpt-5.4-mini'],
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 75,
cached_tokens: 7.5,
completion_tokens: 450,
},
context: 400_000,
max_tokens: 128_000,
},
{
// Costs mirror openai gpt-5.3-codex.
puterId: 'azure:openai/gpt-5.3-codex',
id: 'gpt-5.3-codex',
responsesSampling: 'never',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
aliases: ['openai/gpt-5.3-codex'],
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 175,
cached_tokens: 17.5,
completion_tokens: 1400,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.4.
puterId: 'azure:openai/gpt-5.4',
id: 'gpt-5.4',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
release_date: '2026-03-05',
aliases: ['openai/gpt-5.4'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 250,
cached_tokens: 25,
completion_tokens: 1500,
},
context: 1_050_000,
max_tokens: 1_050_000,
},
mirror(OPEN_AI_MODELS, 'openai/gpt-4o', { id: 'gpt-4o' }),
mirror(OPEN_AI_MODELS, 'openai/gpt-5.1', { id: 'gpt-5.1' }),
mirror(OPEN_AI_MODELS, 'openai/gpt-5.2', { id: 'gpt-5.2' }),
mirror(OPEN_AI_MODELS, 'openai/gpt-5.4-nano', { id: 'gpt-5.4-nano' }),
mirror(OPEN_AI_MODELS, 'openai/gpt-5.4-mini', { id: 'gpt-5.4-mini' }),
mirror(OPEN_AI_MODELS, 'openai/gpt-5.3-codex', { id: 'gpt-5.3-codex' }),
mirror(OPEN_AI_MODELS, 'openai/gpt-5.4', { id: 'gpt-5.4' }),
];