From 52f0176854f246df8e1517902cb8c0fc995f457b Mon Sep 17 00:00:00 2001 From: 404oops Date: Tue, 22 Sep 2026 12:36:39 +0200 Subject: [PATCH] fix(ai): classify and sanitize upstream credit exhaustion --- .../drivers/DriverController.errors.test.ts | 22 +++ .../controllers/drivers/DriverController.ts | 22 ++- .../ChatCompletionDriver.edges.test.ts | 136 +++++++++++++++++- .../ChatCompletionDriver.routing.test.ts | 75 +++++++++- .../drivers/ai-chat/ChatCompletionDriver.ts | 23 ++- .../drivers/util/upstreamErrors.test.ts | 61 ++++++++ src/backend/drivers/util/upstreamErrors.ts | 24 +++- src/backend/server.test.ts | 48 ++++++- src/backend/server.ts | 2 + 9 files changed, 396 insertions(+), 17 deletions(-) diff --git a/src/backend/controllers/drivers/DriverController.errors.test.ts b/src/backend/controllers/drivers/DriverController.errors.test.ts index fd9b3db9d..ad67bcaf6 100644 --- a/src/backend/controllers/drivers/DriverController.errors.test.ts +++ b/src/backend/controllers/drivers/DriverController.errors.test.ts @@ -179,6 +179,28 @@ const throwing = (payload: unknown) => () => { // -- Upstream status extraction -------------------------------------- describe('DriverController upstream error translation', () => { + it('maps an upstream 402 to sanitized credit exhaustion', async () => { + const { err } = await callWith( + throwing( + Object.assign( + new Error( + 'Insufficient credits. Add more using https://openrouter.ai/settings/credits (request id: req-secret)', + ), + { status: 402 }, + ), + ), + ); + + expect(err).toMatchObject({ + statusCode: 503, + legacyCode: 'upstream_credits_exhausted', + message: 'AI provider out of credits', + fields: { upstreamStatus: 402 }, + }); + expect(JSON.stringify(err)).not.toContain('http'); + expect(JSON.stringify(err)).not.toMatch(/request id|/i); + }); + it('maps an upstream 429 to a Puter 429 with upstream_rate_limited', async () => { const { err } = await callWith( throwing( diff --git a/src/backend/controllers/drivers/DriverController.ts b/src/backend/controllers/drivers/DriverController.ts index be3dba5c2..7cd5885f6 100644 --- a/src/backend/controllers/drivers/DriverController.ts +++ b/src/backend/controllers/drivers/DriverController.ts @@ -31,7 +31,11 @@ import { } from '../../core/http/middleware/rateLimit.js'; import type { PuterRouter } from '../../core/http/PuterRouter.js'; import type { DriverMeta } from '../../drivers/meta.js'; -import { isUpstreamTimeoutError } from '../../drivers/util/upstreamErrors.js'; +import { + isCreditExhaustion, + isUpstreamTimeoutError, + sanitizeUpstreamMessage, +} from '../../drivers/util/upstreamErrors.js'; import { isDriverStreamResult, resolveCallableMethods, @@ -120,6 +124,18 @@ const translateProviderError = (err: unknown): unknown => { cause?: unknown; }; const status = extractUpstreamStatus(e); + const msg = sanitizeUpstreamMessage( + e.error?.message ?? e.message ?? 'Upstream provider error', + ); + const upstreamCode = e.error?.code ?? e.code; + const fields = { upstreamStatus: status, upstreamCode }; + + if (isCreditExhaustion(status, upstreamCode, msg)) { + return new HttpError(503, 'AI provider out of credits', { + legacyCode: 'upstream_credits_exhausted', + fields, + }); + } if (typeof status !== 'number') { if (isUpstreamTimeoutError(e)) { const cause = e.cause as { code?: string } | undefined; @@ -132,10 +148,6 @@ const translateProviderError = (err: unknown): unknown => { return err; } - const msg = e.error?.message ?? e.message ?? 'Upstream provider error'; - const upstreamCode = e.error?.code ?? e.code; - const fields = { upstreamStatus: status, upstreamCode }; - if (status === 429) { return new HttpError(429, msg, { legacyCode: 'upstream_rate_limited', diff --git a/src/backend/drivers/ai-chat/ChatCompletionDriver.edges.test.ts b/src/backend/drivers/ai-chat/ChatCompletionDriver.edges.test.ts index 272af255d..faa0cc269 100644 --- a/src/backend/drivers/ai-chat/ChatCompletionDriver.edges.test.ts +++ b/src/backend/drivers/ai-chat/ChatCompletionDriver.edges.test.ts @@ -257,6 +257,126 @@ describe('ChatCompletionDriver provider registration', () => { // -- Failure classification ------------------------------------------ describe('ChatCompletionDriver exhausted-chain classification', () => { + const attemptsFrom = (err: HttpError) => + ( + err as unknown as { + fields: { + attempts: Array<{ + error: string; + status?: number; + code?: string; + }>; + }; + } + ).fields.attempts; + + it('maps an upstream 402 to alerted credit exhaustion and redacts its URL', async () => { + const warn = vi + .spyOn(console, 'warn') + .mockImplementation(() => undefined); + const err = await errorFor( + Object.assign( + new Error( + 'Insufficient credits. Add more using https://openrouter.ai/settings/credits', + ), + { status: 402 }, + ), + ); + + expect(err).toMatchObject({ + statusCode: 503, + legacyCode: 'upstream_credits_exhausted', + message: 'AI provider out of credits', + }); + expect(err.noAlarm).toBeFalsy(); + expect(attemptsFrom(err)[0]!.error).toBe( + 'Insufficient credits. Add more using', + ); + expect(JSON.stringify(attemptsFrom(err))).not.toContain('http'); + const failureLog = warn.mock.calls.find((call) => + String(call[0]).startsWith('[ai-chat] all routes failed'), + ); + expect(failureLog).toBeDefined(); + expect(String(failureLog![0])).not.toContain('http'); + }); + + it('classifies a 403 used-up credit message before authentication failures', async () => { + const err = await errorFor( + Object.assign( + new Error( + 'Your current credits have been used up and we are unable to process further requests. Please visit https://openrouter.ai/settings/credits to add credits.', + ), + { status: 403 }, + ), + ); + + expect(err).toMatchObject({ + statusCode: 503, + legacyCode: 'upstream_credits_exhausted', + }); + expect(err.legacyCode).not.toBe('upstream_auth_failed'); + }); + + it('alerts on a free model whose required team balance is missing', async () => { + const warn = vi + .spyOn(console, 'warn') + .mockImplementation(() => undefined); + const err = await errorFor( + Object.assign( + new Error( + 'Free model requires Team balance greater than $4.999999. (request id: 20260921210512505339070jYB)', + ), + { status: 429 }, + ), + ); + + expect(err).toMatchObject({ + statusCode: 503, + legacyCode: 'upstream_credits_exhausted', + }); + expect(err.noAlarm).toBeFalsy(); + expect(attemptsFrom(err)[0]!.error).toBe( + 'Free model requires Team balance greater than $4.999999.', + ); + expect(JSON.stringify(attemptsFrom(err))).not.toMatch(/request id/i); + expect( + warn.mock.calls.some((call) => + String(call[0]).startsWith('[ai-chat] all routes failed'), + ), + ).toBe(true); + }); + + it('classifies the OpenAI insufficient_quota code before rate limits', async () => { + const err = await errorFor({ + status: 429, + message: 'quota exhausted', + error: { code: 'insufficient_quota' }, + }); + + expect(err).toMatchObject({ + statusCode: 503, + legacyCode: 'upstream_credits_exhausted', + }); + expect(err.legacyCode).not.toBe('upstream_rate_limited'); + }); + + it('strips HTML, URLs and request ids from serialized attempt details', async () => { + const err = await errorFor( + Object.assign( + new Error( + 'Insufficient credits. Visit https://vendor.test/top-up (request id: req-secret)', + ), + { status: 402 }, + ), + ); + const serialized = JSON.stringify(attemptsFrom(err)); + + expect(err.legacyCode).toBe('upstream_credits_exhausted'); + expect(serialized).not.toContain('<'); + expect(serialized).not.toContain('http'); + expect(serialized).not.toMatch(/request id/i); + }); + it('maps an upstream 429 to 429 upstream_rate_limited', async () => { const err = await errorFor( Object.assign(new Error('slow down'), { status: 429 }), @@ -307,6 +427,16 @@ describe('ChatCompletionDriver exhausted-chain classification', () => { expect(err).toMatchObject({ legacyCode: 'upstream_rate_limited' }); }); + it('keeps a plain 429 in the rate-limit bucket', async () => { + const err = await errorFor( + Object.assign(new Error('Rate limit exceeded'), { status: 429 }), + ); + expect(err).toMatchObject({ + statusCode: 429, + legacyCode: 'upstream_rate_limited', + }); + }); + it('maps an upstream 401 to a 500 upstream_auth_failed — our misconfiguration, not the callerdispute', async () => { const err = await errorFor( Object.assign(new Error('invalid api key'), { statusCode: 401 }), @@ -488,7 +618,9 @@ describe('ChatCompletionDriver streaming failure handling', () => { vi.spyOn(FakeChatProvider.prototype, 'complete').mockResolvedValue({ stream: true, init_chat_stream: async () => { - throw new Error('populator exploded'); + throw new Error( + 'populator exploded; see https://vendor.test/request/secret', + ); }, finally_fn: cleanup, } as never); @@ -503,7 +635,7 @@ describe('ChatCompletionDriver streaming failure handling', () => { const events = await collect(result.stream); expect(events).toEqual([ - { type: 'error', message: 'populator exploded' }, + { type: 'error', message: 'populator exploded; see' }, ]); // The provider's cleanup hook still runs on the failure path. expect(cleanup).toHaveBeenCalledTimes(1); diff --git a/src/backend/drivers/ai-chat/ChatCompletionDriver.routing.test.ts b/src/backend/drivers/ai-chat/ChatCompletionDriver.routing.test.ts index 473ae64b5..c7235dd4b 100644 --- a/src/backend/drivers/ai-chat/ChatCompletionDriver.routing.test.ts +++ b/src/backend/drivers/ai-chat/ChatCompletionDriver.routing.test.ts @@ -202,8 +202,11 @@ afterAll(async () => { * reject is what makes the whole chain observable: the driver records every * attempt on the thrown error, and `attempts[0]` is who it chose first. */ -const attemptsFor = async (model: string) => { - createMock.mockRejectedValue(new Error('upstream down')); +const attemptsFor = async ( + model: string, + thrown: unknown = new Error('upstream down'), +) => { + createMock.mockRejectedValue(thrown); let caught: HttpError | undefined; try { await withTestActor(() => @@ -405,6 +408,74 @@ describe('ChatCompletionDriver unhealthy-route skipping', () => { expect(next[0].provider).not.toBe('deepseek'); }); + it('marks a credit-exhausted 402 route unhealthy for the next caller', async () => { + const first = await attemptsFor( + 'deepseek-v4-pro', + Object.assign( + new Error( + 'Insufficient credits. Add more using https://openrouter.ai/settings/credits', + ), + { status: 402 }, + ), + ); + const next = await attemptsFor('deepseek-v4-pro'); + + expect(next[0]).not.toMatchObject({ + provider: first[0]!.provider, + model: first[0]!.model, + }); + }); + + it('classifies differing credit-exhaustion statuses as one exhausted chain', async () => { + createMock + .mockRejectedValueOnce( + Object.assign( + new Error( + 'Insufficient credits. Add more using https://openrouter.ai/settings/credits', + ), + { status: 402 }, + ), + ) + .mockRejectedValueOnce( + Object.assign( + new Error( + 'Your current credits have been used up and we are unable to process further requests. Please visit https://openrouter.ai/settings/credits to add credits.', + ), + { status: 403 }, + ), + ) + .mockRejectedValueOnce( + Object.assign( + new Error( + 'Free model requires Team balance greater than $4.999999. (request id: 20260921210512505339070jYB)', + ), + { status: 429 }, + ), + ); + + const err = await withTestActor(() => + driver + .complete({ + model: 'deepseek-v4-pro', + messages: [{ role: 'user', content: 'hi' }], + }) + .catch((e: unknown) => e as HttpError), + ); + + expect(err).toMatchObject({ + statusCode: 503, + legacyCode: 'upstream_credits_exhausted', + }); + const attempts = ( + err as unknown as { + fields: { attempts: Array<{ status?: number }> }; + } + ).fields.attempts; + expect(attempts.map((attempt) => attempt.status)).toEqual([ + 402, 403, 429, + ]); + }); + it('still serves a marked route when it is the only one left', async () => { // gemini-2.5-flash-image-preview has a single route; marking it must // degrade to trying it anyway rather than failing with no attempt. diff --git a/src/backend/drivers/ai-chat/ChatCompletionDriver.ts b/src/backend/drivers/ai-chat/ChatCompletionDriver.ts index 2e4136109..78f59c832 100644 --- a/src/backend/drivers/ai-chat/ChatCompletionDriver.ts +++ b/src/backend/drivers/ai-chat/ChatCompletionDriver.ts @@ -30,7 +30,11 @@ import type { MeteringService } from '../../services/metering/MeteringService.js import type { DriverStreamResult } from '../meta.js'; import { PuterDriver } from '../types.js'; import { AI_CONCURRENT, AI_RATE_LIMIT } from '../util/aiLimits.js'; -import { isUpstreamTimeoutError } from '../util/upstreamErrors.js'; +import { + isCreditExhaustion as isUpstreamCreditExhaustion, + isUpstreamTimeoutError, + sanitizeUpstreamMessage, +} from '../util/upstreamErrors.js'; import { AlibabaProvider } from './providers/alibaba/AlibabaProvider.js'; import { AzureChatProvider } from './providers/azure/AzureChatProvider.js'; import { AzureResponsesProvider } from './providers/azure/AzureResponsesProvider.js'; @@ -153,11 +157,14 @@ const toAttempt = ( provider: providerId, status, code: e?.error?.code ?? e?.code, - error: message, + error: sanitizeUpstreamMessage(message), ...(isUpstreamTimeoutError(err) ? { timedOut: true } : {}), }; }; +const isCreditExhaustion = (a: ProviderAttempt) => + isUpstreamCreditExhaustion(a.status, a.code, a.error); + const isRateLimit = (a: ProviderAttempt) => a.status === 429 || /rate[\s_-]?limit|too many requests|quota/i.test(a.error); @@ -184,6 +191,7 @@ const isUpstream5xx = (a: ProviderAttempt) => */ const isRouteLevelFailure = (a: ProviderAttempt) => a.status === undefined || + isCreditExhaustion(a) || isRateLimit(a) || isAuthFailure(a) || isUpstream5xx(a); @@ -197,6 +205,7 @@ const routeId = (provider: string, modelId: string) => `${provider}:${modelId}`; * * Per-class rules (see also alarm gate in server.ts): * + * - All credit-exhausted → 503 `upstream_credits_exhausted` (alerted) * - All rate-limited → 429 `upstream_rate_limited` (alerted, unless every attempt * was on a free model — see `allModelsFree`) * - All auth failures → 500 `upstream_auth_failed` (paged: our config) @@ -217,6 +226,12 @@ const classifyAttempts = ( }); } + if (attempts.every(isCreditExhaustion)) { + return new HttpError(503, 'AI provider out of credits', { + legacyCode: 'upstream_credits_exhausted', + fields, + }); + } if (attempts.every(isRateLimit)) { return new HttpError(429, 'AI provider rate limit exceeded', { legacyCode: 'upstream_rate_limited', @@ -659,7 +674,9 @@ export class ChatCompletionDriver extends PuterDriver { passthrough.write( `${JSON.stringify({ type: 'error', - message: (e as Error).message, + message: sanitizeUpstreamMessage( + e instanceof Error ? e.message : String(e), + ), })}\n`, ); passthrough.end(); diff --git a/src/backend/drivers/util/upstreamErrors.test.ts b/src/backend/drivers/util/upstreamErrors.test.ts index 4c1ed8abc..0f9415f2b 100644 --- a/src/backend/drivers/util/upstreamErrors.test.ts +++ b/src/backend/drivers/util/upstreamErrors.test.ts @@ -22,6 +22,8 @@ import { describe, expect, it } from 'vitest'; import { HttpError } from '../../core/http/HttpError.js'; import { CONTENT_FILTER_PATTERN, + CREDIT_EXHAUSTION_PATTERN, + isCreditExhaustion, isTransientUpstreamError, isUpstreamTimeoutError, sanitizeUpstreamMessage, @@ -112,6 +114,57 @@ describe('CONTENT_FILTER_PATTERN', () => { }); }); +describe('credit exhaustion detection', () => { + it('matches provider credit and billing failures', () => { + for (const { status, code, message } of [ + { + status: 403, + code: undefined, + message: + 'Your current credits have been used up and we are unable to process further requests. Please visit https://openrouter.ai/settings/credits to add credits.', + }, + { + status: 402, + code: undefined, + message: + 'Insufficient credits. Add more using https://openrouter.ai/settings/credits', + }, + { + status: 429, + code: undefined, + message: + 'Free model requires Team balance greater than $4.999999. (request id: 20260921210512505339070jYB)', + }, + { + status: 403, + code: 'insufficient_user_quota', + message: 'request rejected', + }, + { + status: 429, + code: 'insufficient_quota', + message: 'request rejected', + }, + ]) { + if (message !== 'request rejected') { + expect(message).toMatch(CREDIT_EXHAUSTION_PATTERN); + } + expect(isCreditExhaustion(status, code, message)).toBe(true); + } + }); + + it('does not mistake ordinary rate limits for exhausted credits', () => { + for (const message of [ + 'Rate limit exceeded', + 'Too many requests', + 'Quota exceeded for this key', + ]) { + expect(message).not.toMatch(CREDIT_EXHAUSTION_PATTERN); + expect(isCreditExhaustion(429, undefined, message)).toBe(false); + } + }); +}); + describe('sanitizeUpstreamMessage', () => { it('strips markup and collapses whitespace', () => { expect( @@ -126,4 +179,12 @@ describe('sanitizeUpstreamMessage', () => { expect(out.length).toBe(300); expect(out.endsWith('...')).toBe(true); }); + + it('redacts URLs and request identifiers', () => { + expect( + sanitizeUpstreamMessage( + 'Add credits at https://vendor.test/billing (request id: req-parenthesized) request_id: req_standalone', + ), + ).toBe('Add credits at'); + }); }); diff --git a/src/backend/drivers/util/upstreamErrors.ts b/src/backend/drivers/util/upstreamErrors.ts index 482932ca2..eb6ad8e67 100644 --- a/src/backend/drivers/util/upstreamErrors.ts +++ b/src/backend/drivers/util/upstreamErrors.ts @@ -113,14 +113,34 @@ const MAX_UPSTREAM_MESSAGE_LENGTH = 300; export const CONTENT_FILTER_PATTERN = /\bnsfw\b|sensitive|content[\s_-]?policy|moderation|safety|\bunsafe\b|\bflagged\b|prohibited|\bE005\b/i; +/** Provider messages that indicate the account cannot fund another request. */ +export const CREDIT_EXHAUSTION_PATTERN = + /insufficient[\s_-]?(credits?|quota|funds|balance)|credits? (have been|are) used up|out of credits|(team )?balance (greater than|below|too low)|billing hard limit/i; + +/** A provider account has exhausted its credits or billing allowance. */ +export const isCreditExhaustion = ( + status: number | undefined, + code: string | undefined, + message: string, +): boolean => + status === 402 || + (code !== undefined && + /insufficient_(user_)?quota|insufficient_credits|billing/i.test( + code, + )) || + CREDIT_EXHAUSTION_PATTERN.test(message); + /** - * Strips markup and bounds length so an upstream HTML error page never rides - * through into a response body or an alarm signature. + * Strips markup, URLs and request ids, then bounds length so provider details + * never ride through into a response body or an alarm signature. */ export const sanitizeUpstreamMessage = (raw: string): string => { const text = raw .replace(/<(style|script)[\s\S]*?<\/\1>/gi, ' ') .replace(/<[^>]*>/g, ' ') + .replace(/https?:\/\/\S+/gi, ' ') + .replace(/\(\s*request[\s_-]?id\s*:\s*[^)]*\)/gi, ' ') + .replace(/\brequest[\s_-]?id\s*:\s*\S+/gi, ' ') .replace(/\s+/g, ' ') .trim(); return text.length > MAX_UPSTREAM_MESSAGE_LENGTH diff --git a/src/backend/server.test.ts b/src/backend/server.test.ts index 1119cf09f..c77e71a96 100644 --- a/src/backend/server.test.ts +++ b/src/backend/server.test.ts @@ -527,6 +527,26 @@ describe('PuterServer HTTP alarm gate', () => { throw new Error('kaboom'); }) as unknown as RequestHandler, }, + { + method: 'get', + path: '/credits-exhausted', + options: {}, + handler: (() => { + throw new HttpError(503, 'AI provider out of credits', { + legacyCode: 'upstream_credits_exhausted', + fields: { + attempts: [ + { + model: 'm', + provider: 'a', + status: 402, + error: 'Insufficient credits.', + }, + ], + }, + }); + }) as unknown as RequestHandler, + }, ); port = await allocateEphemeralPort(); server = await setupTestServer( @@ -548,19 +568,20 @@ describe('PuterServer HTTP alarm gate', () => { vi.restoreAllMocks(); }); - const raisedFor = async (path: string) => { + const raisedFor = async (path: string, status = 500) => { const alarm = vi .spyOn(server.clients.alarm, 'create') .mockImplementation(() => undefined); const res = await rawRequest(port, path, { host: 'puter.localhost' }); - expect(res.status).toBe(500); + expect(res.status).toBe(status); const raised = alarm.mock.calls.find((c) => - String(c[0]).startsWith(`http_500:GET:${path}:`), + String(c[0]).startsWith(`http_${status}:GET:${path}:`), ); expect(raised).toBeTruthy(); return { id: raised![0] as string, fields: raised![2] as Record, + severity: raised![3] as string, }; }; @@ -585,6 +606,27 @@ describe('PuterServer HTTP alarm gate', () => { expect(fields.error).toBeInstanceOf(Error); expect(fields).not.toHaveProperty('details'); }); + + it('raises a warning when an upstream account is out of credits', async () => { + const { id, fields, severity } = await raisedFor( + '/credits-exhausted', + 503, + ); + expect(id).toBe( + 'http_503:GET:/credits-exhausted:upstream_credits_exhausted:AI provider out of credits', + ); + expect(severity).toBe('warning'); + expect(fields.details).toEqual({ + attempts: [ + { + model: 'm', + provider: 'a', + status: 402, + error: 'Insufficient credits.', + }, + ], + }); + }); }); /** diff --git a/src/backend/server.ts b/src/backend/server.ts index ad9429379..3ecf0dbed 100644 --- a/src/backend/server.ts +++ b/src/backend/server.ts @@ -821,6 +821,8 @@ export class PuterServer { // Our credentials for a provider stopped working — // everything through it fails until someone looks. ['upstream_auth_failed', 'warning'], + // A vendor account is dry — everything through it fails until someone tops up. + ['upstream_credits_exhausted', 'warning'], ]); const SKIP_ALERT_PREFIXES = /^(upstream_|client_)/; const isHttp = isHttpError(err);