fix: stop video generation timeouts and failures from paging (PUT-1620)

A video job that outlived its poll window, an SDK request that timed out,
or a Veo operation that finished with an error all reached the HTTP error
handler as plain Errors. Each became an unhandled 500 with critical
severity and paged on-call for what is the provider's pace or the
provider's fault.

Video providers now share one poll loop that gives up with a 504
`upstream_timeout`, treats a transient poll failure (timeout, dropped
connection, 408/429/5xx) as a missed poll rather than a failed job, and
stops polling with a 400 `client_aborted` when the caller disconnects, so
nothing is metered for a clip nobody will receive. The driver controller
exposes the disconnect as an `abortSignal` on the request context. The
window is ten minutes for every provider; Together and BytePlus move up
from five.

Failed jobs are classified: content-filter refusals become a 400
`bad_request` with `errorCode: moderation_flagged`, rejected parameters a
400 `upstream_bad_request`, and anything else a 502 `upstream_failed`,
each carrying the provider's own code. Veo's filtered output keeps
`disallowed_value` and gains the same `errorCode`. The sanitizer and
content-filter pattern move from the Replicate provider into a shared
util so image and video agree.

Status-less SDK connection timeouts are translated to a 504
`upstream_timeout` at the driver boundary, and the chat driver records
them per attempt so an all-timeout chain is a 504 and a mixed chain is
`upstream_failed` instead of an `internal_error` 500. The Together chat
client gets the same ten-minute request timeout as the other providers.

The OpenAI video provider is left alone beyond an import path: its API is
scheduled to shut down on 2026-09-24.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
404oops
2026-09-03 16:47:16 +02:00
co-authored by Claude Fable 5.1
parent 3de6eeb474
commit 627a5b2d5e
23 changed files with 1334 additions and 126 deletions
@@ -35,9 +35,10 @@
/* eslint-disable @typescript-eslint/no-explicit-any */
import { Readable, Writable } from 'node:stream';
import type { Request, RequestHandler, Response } from 'express';
import { APIConnectionTimeoutError } from 'openai';
import { beforeEach, describe, expect, it, vi } from 'vitest';
import type { DriverMethodLifecycleEvent } from '../../clients/event/types.js';
import { runWithContext } from '../../core/context.js';
import { Context, runWithContext } from '../../core/context.js';
import { configureRateLimit } from '../../core/http/middleware/rateLimit.js';
import { DriverController } from './DriverController.js';
@@ -325,6 +326,32 @@ describe('DriverController upstream error translation', () => {
expect(err).toBe('a bare string');
});
it('maps an SDK connection timeout, which carries no status, to a 504 upstream_timeout', async () => {
const raw = new APIConnectionTimeoutError();
const { err } = await callWith(throwing(raw));
expect(err).toMatchObject({
statusCode: 504,
legacyCode: 'upstream_timeout',
message: 'AI provider timed out',
cause: raw,
});
});
it('maps a fetch timeout that undici wraps in a `fetch failed` TypeError to a 504 upstream_timeout', async () => {
const raw = new TypeError('fetch failed', {
cause: Object.assign(new Error('Headers Timeout Error'), {
name: 'HeadersTimeoutError',
code: 'UND_ERR_HEADERS_TIMEOUT',
}),
});
const { err } = await callWith(throwing(raw));
expect(err).toMatchObject({
statusCode: 504,
legacyCode: 'upstream_timeout',
fields: { upstreamCode: 'UND_ERR_HEADERS_TIMEOUT' },
});
});
it('passes an error with a sub-400 status through untranslated', async () => {
const raw = { status: 302, message: 'redirected' };
const { err } = await callWith(throwing(raw));
@@ -435,3 +462,53 @@ describe('DriverController per-method rate limiting', () => {
expect(alarms).toEqual([]);
});
});
// -- Client disconnect --------------------------------------------------
describe('DriverController client disconnect', () => {
it('exposes an abort signal in the request context that fires when the client leaves early', async () => {
let seen: AbortSignal | undefined;
const { handler } = build({
run: async () => {
seen = Context.get('abortSignal');
await new Promise((r) => setImmediate(r));
return { ok: true };
},
});
const res = new MockRes();
const done = runWithContext({}, () =>
handler(
makeReq({ interface: 'test-iface', method: 'run' }),
res as unknown as Response,
() => {},
),
);
res.destroy();
await done;
expect(seen).toBeInstanceOf(AbortSignal);
expect(seen?.aborted).toBe(true);
});
it('does not abort when the response simply finished', async () => {
let seen: AbortSignal | undefined;
const { handler } = build({
run: () => {
seen = Context.get('abortSignal');
return { ok: true };
},
});
const res = new MockRes();
await runWithContext({}, () =>
handler(
makeReq({ interface: 'test-iface', method: 'run' }),
res as unknown as Response,
() => {},
),
);
res.end();
await new Promise((r) => setImmediate(r));
expect(seen?.aborted).toBe(false);
});
});