fix: quote and pin Infron service tiers

Infron sells the same model at several service tiers. `min_prompt_price`
and `min_completion_price` are the floor across all of them, so every
model with a flex tier was advertised at a batch-job price nobody gets
by default: 25 of 286 chat models, among them gpt-6-astra at $5/$25
against the $7.5/$37.5 a default request actually bills.

Price each tier from its own row in `providers[]` and pin that tier on
the request, so the price quoted is the price charged. Tiers beyond the
default are listed under their own `<model>:<tier>` ids, letting callers
opt into flex or priority by model name:

    puter.ai.chat(prompt, { model: 'infron:openai/gpt-6-astra:flex' })

Suffix parsing matches the catalog exactly before reading a trailing
segment as a tier, since catalog ids can carry a colon of their own
(`deepseek/deepseek-v4-flash:free`). Tier variants take the context
window of their own offering, which differs from the model-level figure
for 11 of them.

Billing was never wrong — it bills Infron's reported `cost` — but the
understated prices fed the credit gate's output cap, which let a request
run roughly 50% past the balance it was gated against, and the fallback
path that prices per token when a response carries no cost.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Filip Kujundžić
2026-09-11 14:47:41 +02:00
co-authored by Claude Opus 5
parent a8a78736bb
commit 3d2bdfceef
2 changed files with 378 additions and 43 deletions
@@ -120,6 +120,82 @@ const SAMPLE_API_MODELS = [
min_prompt_price: 2,
min_completion_price: 10,
},
{
// Sold at several service tiers; the flex tier undercuts standard.
id: 'openai/gpt-6-astra',
display_name: 'OpenAI: GPT-6 Astra',
category_type: 'LLM',
supported_endpoint_types: ['openai'],
context_length: 1050000,
max_output_tokens: 128000,
min_prompt_price: 5,
min_completion_price: 25,
providers: [
{
provider_slug: 'openai/flex',
service_tier: 'flex',
prompt_price: 5,
completion_price: 25,
context_length: 400000,
},
{
provider_slug: 'azure',
service_tier: 'standard',
prompt_price: 7.5,
completion_price: 37.5,
},
{
provider_slug: 'openai',
service_tier: 'standard',
prompt_price: 10,
completion_price: 50,
},
{
provider_slug: 'openai/priority',
service_tier: 'priority',
prompt_price: 20,
completion_price: 100,
},
],
},
{
// Flex is the only tier on offer — nothing to pin.
id: 'example/flex-only-model',
display_name: 'Flex Only',
category_type: 'LLM',
supported_endpoint_types: ['openai'],
context_length: 128000,
max_output_tokens: 4096,
min_prompt_price: 1,
min_completion_price: 2,
providers: [
{
provider_slug: 'example/flex',
service_tier: 'flex',
prompt_price: 1,
completion_price: 2,
},
],
},
{
// Catalog ids can contain a colon of their own.
id: 'example/colon-model:free',
display_name: 'Colon Model (free)',
category_type: 'LLM',
supported_endpoint_types: ['openai'],
context_length: 64000,
max_output_tokens: 4096,
min_prompt_price: 0,
min_completion_price: 0,
providers: [
{
provider_slug: 'example',
service_tier: 'standard',
prompt_price: 0,
completion_price: 0,
},
],
},
{
// Non-chat modality — filtered out.
id: 'black-forest-labs/flux-2.1',
@@ -262,10 +338,87 @@ describe('InfronProvider model catalog', () => {
expect(axiosRequestMock).toHaveBeenCalledTimes(1);
});
it('quotes the tier Infron routes to by default, not the cheaper flex tier', async () => {
const { provider } = makeProvider();
const models = await provider.models();
// Cheapest standard offering is Azure at $7.5/$37.5 per million,
// not the $5/$25 flex tier the catalog floor reports.
expect(models).toEqual(
expect.arrayContaining([
expect.objectContaining({
id: 'infron:openai/gpt-6-astra',
costs: expect.objectContaining({
prompt: 750,
completion: 3750,
input_cache_read: 750,
}),
}),
]),
);
});
it('lists a :flex id alongside the default-tier id, each at its own price', async () => {
const { provider } = makeProvider();
const models = await provider.models();
expect(models).toEqual(
expect.arrayContaining([
expect.objectContaining({
id: 'infron:openai/gpt-6-astra',
costs: expect.objectContaining({
prompt: 750,
completion: 3750,
}),
}),
expect.objectContaining({
id: 'infron:openai/gpt-6-astra:flex',
costs: expect.objectContaining({
prompt: 500,
completion: 2500,
}),
}),
expect.objectContaining({
id: 'infron:openai/gpt-6-astra:priority',
costs: expect.objectContaining({
prompt: 2000,
completion: 10000,
}),
}),
]),
);
});
it('takes a tier variant context window from its own offering', async () => {
const { provider } = makeProvider();
const models = await provider.models();
const flex = models.find(
(m) => m.id === 'infron:openai/gpt-6-astra:flex',
);
// The flex offering caps context below the model-level figure.
expect(flex?.context).toBe(400000);
});
it('accepts the explicit :standard spelling as an alias of the plain id', async () => {
const { provider } = makeProvider();
const models = await provider.models();
const base = models.find((m) => m.id === 'infron:openai/gpt-6-astra');
expect(base?.aliases).toContain('openai/gpt-6-astra:standard');
});
it('still offers an explicit tier id when no default tier is sold', async () => {
const { provider } = makeProvider();
const ids = await provider.list();
// The plain id leaves routing to Infron; the :flex id pins the one
// tier on offer. Same price, different routing guarantee.
expect(ids).toContain('infron:example/flex-only-model');
expect(ids).toContain('infron:example/flex-only-model:flex');
});
it('converts USD-per-million-token prices to microcents per token', async () => {
const { provider } = makeProvider();
const models = await provider.models();
// $10/M tokens → 10 * 100 = 1000 microcents per token.
// $10/M tokens → 10 * 100 = 1000 microcents per token. This model
// carries no per-tier breakdown, so the catalog floor is used.
expect(models).toEqual(
expect.arrayContaining([
expect.objectContaining({
@@ -314,6 +467,68 @@ describe('InfronProvider.complete request shape', () => {
expect(args.usage).toEqual({ include: true });
});
it('pins the standard service tier so billing matches the quoted price', async () => {
const { provider } = makeProvider();
createMock.mockResolvedValueOnce(baseCompletion);
await withTestActor(() =>
provider.complete({
model: 'infron:openai/gpt-6-astra',
messages: [{ role: 'user', content: 'hello' }],
}),
);
expect(createMock.mock.calls[0]![0].provider).toEqual({
service_tier: 'standard',
});
});
it('leaves routing to Infron for models with no standard tier', async () => {
const { provider } = makeProvider();
createMock.mockResolvedValueOnce(baseCompletion);
await withTestActor(() =>
provider.complete({
model: 'infron:example/flex-only-model',
messages: [{ role: 'user', content: 'hello' }],
}),
);
expect('provider' in createMock.mock.calls[0]![0]).toBe(false);
});
it('pins the tier named by a :flex model id and strips it from the wire id', async () => {
const { provider } = makeProvider();
createMock.mockResolvedValueOnce(baseCompletion);
await withTestActor(() =>
provider.complete({
model: 'infron:openai/gpt-6-astra:flex',
messages: [{ role: 'user', content: 'hello' }],
}),
);
const [args] = createMock.mock.calls[0]!;
expect(args.model).toBe('openai/gpt-6-astra');
expect(args.provider).toEqual({ service_tier: 'flex' });
});
it('keeps a colon-bearing catalog id intact rather than reading it as a tier', async () => {
const { provider } = makeProvider();
createMock.mockResolvedValueOnce(baseCompletion);
await withTestActor(() =>
provider.complete({
model: 'infron:example/colon-model:free',
messages: [{ role: 'user', content: 'hello' }],
}),
);
const [args] = createMock.mock.calls[0]!;
expect(args.model).toBe('example/colon-model:free');
expect(args.provider).toEqual({ service_tier: 'standard' });
});
it('only sets stream_options.include_usage when streaming', async () => {
const { provider } = makeProvider();
@@ -31,12 +31,25 @@ import type {
ICompleteArguments,
} from '../../types.js';
/**
* One upstream offering of a model in Infron's catalog. The same model is often
* served at several service tiers with different prices; `flex` is
* batch-oriented and cheapest, `priority` is the fastest and dearest.
*/
type InfronApiProviderOffer = {
provider_slug?: string;
service_tier?: string;
prompt_price?: number;
completion_price?: number;
context_length?: number;
};
/**
* Shape of one entry in Infron's `GET /v1/models` catalog. Unlike OpenRouter
* there is no `pricing` object; prices are USD per million tokens in
* `min_prompt_price` / `min_completion_price`, and the catalog mixes non-chat
* modalities (image, video, embeddings) that this provider filters out via
* `category_type`.
* there is no `pricing` object; prices are USD per million tokens, per offering
* in `providers` and as a catalog-wide floor in `min_prompt_price` /
* `min_completion_price`. The catalog mixes non-chat modalities (image, video,
* embeddings) that this provider filters out via `category_type`.
*/
type InfronApiModel = {
id: string;
@@ -49,6 +62,117 @@ type InfronApiModel = {
min_prompt_price?: number;
min_completion_price?: number;
min_request_price?: number;
providers?: InfronApiProviderOffer[];
};
/** Tier Infron routes to when a request carries no explicit `service_tier`. */
const DEFAULT_SERVICE_TIER = 'standard';
/** What one tier of a model costs, in USD per million tokens. */
type InfronTierPrices = {
prompt: number;
completion: number;
context?: number;
};
/**
* Cheapest offering of each service tier a model sells. Tiers differ in price
* and sometimes in context window, so each one is quoted from its own row
* rather than from the catalog-wide `min_*` floor.
*/
const pricesByTier = (model: InfronApiModel) => {
const byTier = new Map<string, InfronTierPrices>();
for (const offer of model.providers ?? []) {
const tier = offer.service_tier;
if (!tier) continue;
const prompt = offer.prompt_price ?? 0;
const completion = offer.completion_price ?? 0;
const seen = byTier.get(tier);
if (seen && seen.prompt + seen.completion <= prompt + completion) {
continue;
}
byTier.set(tier, { prompt, completion, context: offer.context_length });
}
return byTier;
};
/**
* Splits a model id into the id Infron expects on the wire and the tier to pin.
* Catalog ids can contain a colon themselves (`…:free`), so an exact catalog
* match always wins over reading the last segment as a tier suffix.
*/
const resolveTier = (id: string, catalog: InfronApiModel[]) => {
const tiersOf = (model: InfronApiModel) =>
new Set((model.providers ?? []).map((offer) => offer.service_tier));
const exact = catalog.find((model) => model.id === id);
if (exact) {
return {
wireModelId: id,
// Unsuffixed ids are quoted at the default tier, so pin it — left
// unset, Infron load-balances across tiers and could bill another.
tier: tiersOf(exact).has(DEFAULT_SERVICE_TIER)
? DEFAULT_SERVICE_TIER
: undefined,
};
}
const cut = id.lastIndexOf(':');
const base =
cut > 0 ? catalog.find((m) => m.id === id.slice(0, cut)) : undefined;
const tier = id.slice(cut + 1);
if (base && tiersOf(base).has(tier)) return { wireModelId: base.id, tier };
return { wireModelId: id, tier: undefined };
};
/**
* One listed model: the default tier under the plain id, or a single service
* tier under a `…:<tier>` id priced from that tier's own offering.
*/
const coerceModel = (
model: InfronApiModel,
prices: InfronTierPrices,
tier?: string,
): IChatModel => {
const suffix = tier ? `:${tier}` : '';
const shortId = model.id.split('/').slice(1).join('/');
// Catalog prices are USD per million tokens; costs are microcents per
// token, so the conversion is ×100.
const promptCost = Math.round(prices.prompt * 100);
return {
id: `infron:${model.id}${suffix}`,
name: `${model.display_name || model.id} (Infron${tier ? `, ${tier}` : ''})`,
aliases: [
`${model.id}${suffix}`,
...(model.display_name && !tier ? [model.display_name] : []),
`infron/${model.id}${suffix}`,
`${shortId}${suffix}`,
// The plain id already means the default tier; accept the
// explicit spelling of it too.
...(tier ? [] : [`${model.id}:${DEFAULT_SERVICE_TIER}`]),
],
context: tier
? (prices.context ?? model.context_length)
: model.context_length,
max_tokens: model.max_output_tokens ?? 0,
costs_currency: 'usd-cents',
input_cost_key: 'prompt',
output_cost_key: 'completion',
costs: {
tokens: 1_000_000,
prompt: promptCost,
completion: Math.round(prices.completion * 100),
// The catalog carries no cache-read price; charge the full
// prompt rate in the fallback path so cached tokens are never
// billed below list. The normal path bills the
// gateway-reported cost instead.
input_cache_read: promptCost,
// USD per request → microcents per request.
request: Math.round(
(model.min_request_price ?? 0) * 1_000_000 * 100,
),
},
};
};
type InfronUsage = OpenAI.Completions.CompletionUsage & {
@@ -112,17 +236,25 @@ export class InfronProvider implements IChatProvider {
[m.id, ...(m.aliases || [])].includes(model),
) || availableModels.find((m) => m.id === this.getDefaultModel())!;
const modelIdForParams = modelUsed.id.startsWith('infron:')
const catalogId = modelUsed.id.startsWith('infron:')
? modelUsed.id.slice('infron:'.length)
: modelUsed.id;
// A `…:<tier>` id carries the tier to pin; a plain id means the
// default tier. Only tiers the model actually sells are pinned —
// Infron routes freely (and reports the tier back) when unset.
const { wireModelId, tier } = resolveTier(
catalogId,
await this.#rawModels(),
);
const actor = Context.get('actor');
messages = await OpenAIUtil.process_input_messages(messages);
const completionParams = {
messages,
model: modelIdForParams,
model: wireModelId,
...(tools ? { tools } : {}),
max_tokens,
temperature,
@@ -132,6 +264,10 @@ export class InfronProvider implements IChatProvider {
stream_options: { include_usage: true },
}
: {}),
// Without this Infron load-balances across service tiers, so a
// request could be billed at a tier other than the one whose
// price we quote in the catalog.
...(tier ? { provider: { service_tier: tier } } : {}),
// Surfaces the authoritative `cost` field (USD) on the
// response so metering doesn't depend on catalog prices.
usage: { include: true },
@@ -216,7 +352,8 @@ export class InfronProvider implements IChatProvider {
});
}
async models() {
/** The catalog as Infron returns it, kv-cached and shared by callers. */
async #rawModels(): Promise<InfronApiModel[]> {
let models = kv.get(KV_MODELS_KEY) as InfronApiModel[] | undefined;
if (!models) {
try {
@@ -235,7 +372,11 @@ export class InfronProvider implements IChatProvider {
console.log(e);
}
}
if (!models) return [];
return models ?? [];
}
async models() {
const models = await this.#rawModels();
const coerced_models: IChatModel[] = [];
for (const model of models) {
// The catalog mixes chat with image/video/embedding/search
@@ -245,40 +386,19 @@ export class InfronProvider implements IChatProvider {
if (!(model.supported_endpoint_types ?? []).includes('openai')) {
continue;
}
// Catalog prices are USD per million tokens; costs are
// microcents per token, so the conversion is ×100.
const promptCost = Math.round((model.min_prompt_price ?? 0) * 100);
coerced_models.push({
id: `infron:${model.id}`,
name: `${model.display_name || model.id} (Infron)`,
aliases: [
model.id,
...(model.display_name ? [model.display_name] : []),
`infron/${model.id}`,
model.id.split('/').slice(1).join('/'),
],
context: model.context_length,
max_tokens: model.max_output_tokens ?? 0,
costs_currency: 'usd-cents',
input_cost_key: 'prompt',
output_cost_key: 'completion',
costs: {
tokens: 1_000_000,
prompt: promptCost,
completion: Math.round(
(model.min_completion_price ?? 0) * 100,
),
// The catalog carries no cache-read price; charge the
// full prompt rate in the fallback path so cached
// tokens are never billed below list. The normal path
// bills the gateway-reported cost instead.
input_cache_read: promptCost,
// USD per request → microcents per request.
request: Math.round(
(model.min_request_price ?? 0) * 1_000_000 * 100,
),
},
});
const byTier = pricesByTier(model);
const defaultPrices = byTier.get(DEFAULT_SERVICE_TIER) ?? {
prompt: model.min_prompt_price ?? 0,
completion: model.min_completion_price ?? 0,
};
// The unsuffixed id is the default tier; every other tier the
// model sells gets its own `…:<tier>` id so callers can ask for
// one by name and see what it costs.
coerced_models.push(coerceModel(model, defaultPrices));
for (const [tier, prices] of byTier) {
if (tier === DEFAULT_SERVICE_TIER) continue;
coerced_models.push(coerceModel(model, prices, tier));
}
}
return coerced_models;
}