mirror of
https://github.com/HeyPuter/puter.git
synced 2026-10-10 13:51:41 +00:00
Merge pull request #3949 from HeyPuter/fix/ai-ocr-model-sync-and-normalization
This commit is contained in:
16 files changed
+1038
-205
No files matched your search
@@ -43,6 +43,7 @@ import {
|
||||
vi,
|
||||
type MockInstance,
|
||||
} from 'vitest';
|
||||
import { UnsupportedDocumentException } from '@aws-sdk/client-textract';
|
||||
import { v4 as uuidv4 } from 'uuid';
|
||||
|
||||
import type { Actor } from '../../core/actor.js';
|
||||
@@ -242,12 +243,11 @@ describe('OCRDriver.recognize (aws-textract)', () => {
|
||||
const sampleTextractResponse = {
|
||||
Blocks: [
|
||||
{ BlockType: 'PAGE' },
|
||||
{ BlockType: 'PAGE' }, // 2 pages
|
||||
{ BlockType: 'WORD', Text: 'should-be-skipped' },
|
||||
{ BlockType: 'TABLE' }, // skipped
|
||||
{ BlockType: 'LINE', Text: 'hello world', Confidence: 99.5 },
|
||||
{ BlockType: 'WORD', Text: 'should-be-skipped' },
|
||||
{ BlockType: 'LAYOUT_TITLE', Confidence: 85 }, // no Text
|
||||
{ BlockType: 'PAGE' }, // 2 pages
|
||||
{ BlockType: 'LINE', Text: 'second line', Confidence: 80 },
|
||||
{ BlockType: 'LAYOUT_TITLE', Text: 'Title!', Confidence: 85 },
|
||||
],
|
||||
};
|
||||
|
||||
@@ -279,33 +279,73 @@ describe('OCRDriver.recognize (aws-textract)', () => {
|
||||
provider: 'aws-textract',
|
||||
}),
|
||||
)) as {
|
||||
model: string;
|
||||
blocks: Array<{ type: string; text: string; confidence: number }>;
|
||||
text: string;
|
||||
};
|
||||
|
||||
// Driver issued AnalyzeDocumentCommand{ Bytes: <buffer> }.
|
||||
// Driver issued DetectDocumentTextCommand{ Bytes: <buffer> } — the
|
||||
// API billed at the detect-document-text rate, with no paid features.
|
||||
const sentCmd = textractSendMock.mock.calls[0]![0];
|
||||
expect(sentCmd.constructor.name).toBe('DetectDocumentTextCommand');
|
||||
expect(sentCmd.input.Document.Bytes).toEqual(buf);
|
||||
expect(sentCmd.input.FeatureTypes).toEqual(['LAYOUT']);
|
||||
expect(sentCmd.input.FeatureTypes).toBeUndefined();
|
||||
|
||||
// PAGE/WORD/TABLE/etc. are skipped; LINE and LAYOUT_TITLE pass
|
||||
// through with `text/textract:<BlockType>` namespacing.
|
||||
// Only LINE blocks carry text; each is tagged with its 0-based page.
|
||||
expect(result.blocks).toEqual([
|
||||
{
|
||||
type: 'text/textract:LINE',
|
||||
text: 'hello world',
|
||||
confidence: 99.5,
|
||||
page: 0,
|
||||
},
|
||||
{
|
||||
type: 'text/textract:LINE',
|
||||
text: 'second line',
|
||||
confidence: 80,
|
||||
},
|
||||
{
|
||||
type: 'text/textract:LAYOUT_TITLE',
|
||||
text: 'Title!',
|
||||
confidence: 85,
|
||||
page: 1,
|
||||
},
|
||||
]);
|
||||
expect(result.text).toBe('hello world\nsecond line');
|
||||
expect(result.model).toBe('aws-textract');
|
||||
});
|
||||
|
||||
it('rejects multi-page and unsupported documents with a 400 that names the alternative', async () => {
|
||||
const { actor } = await makeUser();
|
||||
textractSendMock.mockRejectedValueOnce(
|
||||
new UnsupportedDocumentException({
|
||||
message: 'Request has unsupported document format',
|
||||
$metadata: {},
|
||||
}),
|
||||
);
|
||||
|
||||
await expect(
|
||||
withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('%PDF-1.4'), 'application/pdf'),
|
||||
provider: 'aws-textract',
|
||||
}),
|
||||
),
|
||||
).rejects.toMatchObject({
|
||||
statusCode: 400,
|
||||
message: expect.stringContaining('Mistral OCR model'),
|
||||
});
|
||||
expect(incrementUsageSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('refuses input over the 10 MB Textract limit before any upstream call', async () => {
|
||||
const { actor } = await makeUser();
|
||||
const oversize = Buffer.alloc(10 * 1024 * 1024 + 1);
|
||||
|
||||
await expect(
|
||||
withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(oversize, 'image/png'),
|
||||
provider: 'aws-textract',
|
||||
}),
|
||||
),
|
||||
).rejects.toMatchObject({ statusCode: 413 });
|
||||
expect(textractSendMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
// Note: the S3Object-source branch (driver picks `Document.S3Object`
|
||||
@@ -374,6 +414,12 @@ it('meters one usage line per detected page at the per-page rate from costs.ts',
|
||||
// ── Mistral OCR ─────────────────────────────────────────────────────
|
||||
|
||||
describe('OCRDriver.recognize (mistral)', () => {
|
||||
const bboxSchema = { type: 'object', properties: {} };
|
||||
const docSchema = {
|
||||
type: 'object',
|
||||
properties: { total: { type: 'string' } },
|
||||
};
|
||||
|
||||
it('throws 402 when the actor does not have enough credits', async () => {
|
||||
hasCreditsSpy.mockResolvedValueOnce(false);
|
||||
const { actor } = await makeUser();
|
||||
@@ -408,7 +454,8 @@ describe('OCRDriver.recognize (mistral)', () => {
|
||||
);
|
||||
|
||||
const payload = mistralOcrProcessMock.mock.calls[0]![0];
|
||||
expect(payload.model).toBe('mistral-ocr-latest');
|
||||
// The floating alias is pinned to the model the catalog prices.
|
||||
expect(payload.model).toBe('mistral-ocr-4-1');
|
||||
// Mistral's SDK uses camelCase imageUrl on this chunk shape.
|
||||
expect(payload.document).toEqual({
|
||||
type: 'image_url',
|
||||
@@ -450,6 +497,45 @@ describe('OCRDriver.recognize (mistral)', () => {
|
||||
);
|
||||
});
|
||||
|
||||
it('sends non-image documents such as DOCX as a document_url chunk', async () => {
|
||||
const { actor } = await makeUser();
|
||||
mistralOcrProcessMock.mockResolvedValueOnce({ pages: [] });
|
||||
const docx =
|
||||
'application/vnd.openxmlformats-officedocument.wordprocessingml.document';
|
||||
|
||||
await withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('PK'), docx),
|
||||
provider: 'mistral',
|
||||
}),
|
||||
);
|
||||
|
||||
const payload = mistralOcrProcessMock.mock.calls[0]![0];
|
||||
expect(payload.document.type).toBe('document_url');
|
||||
expect(payload.document.documentUrl).toMatch(/^data:application\/vnd/);
|
||||
});
|
||||
|
||||
it('sends untyped PDF bytes as a document and untyped other bytes as an image', async () => {
|
||||
const { actor } = await makeUser();
|
||||
mistralOcrProcessMock.mockResolvedValue({ pages: [] });
|
||||
|
||||
for (const bytes of ['%PDF-1.7', 'png-ish']) {
|
||||
await withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(
|
||||
Buffer.from(bytes),
|
||||
'application/octet-stream',
|
||||
),
|
||||
provider: 'mistral',
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
const [pdfCall, imageCall] = mistralOcrProcessMock.mock.calls;
|
||||
expect(pdfCall![0].document.type).toBe('document_url');
|
||||
expect(imageCall![0].document.type).toBe('image_url');
|
||||
});
|
||||
|
||||
it('forwards page filters and annotation options to Mistral when supplied', async () => {
|
||||
const { actor } = await makeUser();
|
||||
mistralOcrProcessMock.mockResolvedValueOnce({
|
||||
@@ -465,8 +551,23 @@ describe('OCRDriver.recognize (mistral)', () => {
|
||||
includeImageBase64: true,
|
||||
imageLimit: 10,
|
||||
imageMinSize: 64,
|
||||
bboxAnnotationFormat: { schema: 'bbox' },
|
||||
documentAnnotationFormat: { schema: 'doc' },
|
||||
// REST spelling and SDK spelling are both accepted.
|
||||
bboxAnnotationFormat: {
|
||||
type: 'json_schema',
|
||||
json_schema: { name: 'bbox', schema: bboxSchema },
|
||||
},
|
||||
documentAnnotationFormat: {
|
||||
type: 'json_schema',
|
||||
jsonSchema: {
|
||||
name: 'doc',
|
||||
schemaDefinition: docSchema,
|
||||
strict: true,
|
||||
},
|
||||
},
|
||||
documentAnnotationPrompt: 'Extract the invoice fields',
|
||||
tableFormat: 'html',
|
||||
extractHeader: true,
|
||||
extractFooter: false,
|
||||
}),
|
||||
);
|
||||
|
||||
@@ -475,8 +576,44 @@ describe('OCRDriver.recognize (mistral)', () => {
|
||||
expect(payload.includeImageBase64).toBe(true);
|
||||
expect(payload.imageLimit).toBe(10);
|
||||
expect(payload.imageMinSize).toBe(64);
|
||||
expect(payload.bboxAnnotationFormat).toEqual({ schema: 'bbox' });
|
||||
expect(payload.documentAnnotationFormat).toEqual({ schema: 'doc' });
|
||||
expect(payload.bboxAnnotationFormat).toEqual({
|
||||
type: 'json_schema',
|
||||
jsonSchema: { name: 'bbox', schemaDefinition: bboxSchema },
|
||||
});
|
||||
expect(payload.documentAnnotationFormat).toEqual({
|
||||
type: 'json_schema',
|
||||
jsonSchema: {
|
||||
name: 'doc',
|
||||
schemaDefinition: docSchema,
|
||||
strict: true,
|
||||
},
|
||||
});
|
||||
expect(payload.documentAnnotationPrompt).toBe(
|
||||
'Extract the invoice fields',
|
||||
);
|
||||
expect(payload.tableFormat).toBe('html');
|
||||
expect(payload.extractHeader).toBe(true);
|
||||
expect(payload.extractFooter).toBe(false);
|
||||
});
|
||||
|
||||
it.each([
|
||||
['documentAnnotationFormat', { schema: 'doc' }],
|
||||
['documentAnnotationFormat', 'json'],
|
||||
['bboxAnnotationFormat', { type: 'json_object' }],
|
||||
['tableFormat', 'csv'],
|
||||
])('rejects an invalid %s before calling Mistral', async (name, value) => {
|
||||
const { actor } = await makeUser();
|
||||
|
||||
await expect(
|
||||
withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('x'), 'application/pdf'),
|
||||
provider: 'mistral',
|
||||
[name]: value,
|
||||
}),
|
||||
),
|
||||
).rejects.toMatchObject({ statusCode: 400 });
|
||||
expect(mistralOcrProcessMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('normalises the response: each markdown line becomes a LINE block on its source page', async () => {
|
||||
@@ -527,6 +664,30 @@ describe('OCRDriver.recognize (mistral)', () => {
|
||||
expect(result.usage_info).toEqual({ pagesProcessed: 2 });
|
||||
});
|
||||
|
||||
it('returns the document annotation Mistral produced', async () => {
|
||||
const { actor } = await makeUser();
|
||||
mistralOcrProcessMock.mockResolvedValueOnce({
|
||||
model: 'mistral-ocr-4-1',
|
||||
pages: [{ index: 0, markdown: 'Total: 42' }],
|
||||
documentAnnotation: '{"total": "42"}',
|
||||
usageInfo: { pagesProcessed: 1 },
|
||||
});
|
||||
|
||||
const result = (await withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('x'), 'application/pdf'),
|
||||
provider: 'mistral',
|
||||
documentAnnotationFormat: {
|
||||
type: 'json_schema',
|
||||
json_schema: { name: 'doc', schema: docSchema },
|
||||
},
|
||||
}),
|
||||
)) as { document_annotation?: string; text: string };
|
||||
|
||||
expect(result.document_annotation).toBe('{"total": "42"}');
|
||||
expect(result.text).toBe('Total: 42');
|
||||
});
|
||||
|
||||
it('meters per-page Mistral OCR usage from costs.ts', async () => {
|
||||
const { actor } = await makeUser();
|
||||
mistralOcrProcessMock.mockResolvedValueOnce({
|
||||
@@ -545,12 +706,12 @@ describe('OCRDriver.recognize (mistral)', () => {
|
||||
);
|
||||
|
||||
const ocrCalls = incrementUsageSpy.mock.calls.filter(
|
||||
([, type]) => type === 'mistral-ocr:ocr:page',
|
||||
([, type]) => type === 'mistral-ocr:mistral-ocr-4-1:page',
|
||||
);
|
||||
expect(ocrCalls).toHaveLength(1);
|
||||
const [, , count, cost] = ocrCalls[0]!;
|
||||
expect(count).toBe(2);
|
||||
expect(cost).toBe(OCR_COSTS['mistral-ocr:ocr:page'] * 2);
|
||||
expect(cost).toBe(OCR_COSTS['mistral-ocr:mistral-ocr-4-1:page'] * 2);
|
||||
});
|
||||
|
||||
it('also meters annotations when bboxAnnotationFormat or documentAnnotationFormat is requested', async () => {
|
||||
@@ -564,24 +725,54 @@ describe('OCRDriver.recognize (mistral)', () => {
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('x'), 'application/pdf'),
|
||||
provider: 'mistral',
|
||||
bboxAnnotationFormat: { schema: 'bbox' },
|
||||
bboxAnnotationFormat: {
|
||||
type: 'json_schema',
|
||||
json_schema: { name: 'bbox', schema: bboxSchema },
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const pageType = 'mistral-ocr:mistral-ocr-4-1:page';
|
||||
const annotationType = 'mistral-ocr:mistral-ocr-4-1:annotations:page';
|
||||
// The pre-flight covers the page plus its annotation.
|
||||
expect(hasCreditsSpy.mock.calls[0]![1]).toBe(
|
||||
OCR_COSTS[pageType] + OCR_COSTS[annotationType],
|
||||
);
|
||||
const ocrCalls = incrementUsageSpy.mock.calls.filter(
|
||||
([, type]) => type === 'mistral-ocr:ocr:page',
|
||||
([, type]) => type === pageType,
|
||||
);
|
||||
const annotationCalls = incrementUsageSpy.mock.calls.filter(
|
||||
([, type]) => type === 'mistral-ocr:annotations:page',
|
||||
([, type]) => type === annotationType,
|
||||
);
|
||||
expect(ocrCalls).toHaveLength(1);
|
||||
expect(annotationCalls).toHaveLength(1);
|
||||
expect(ocrCalls[0]![2]).toBe(1);
|
||||
expect(ocrCalls[0]![3]).toBe(OCR_COSTS['mistral-ocr:ocr:page']);
|
||||
expect(ocrCalls[0]![3]).toBe(OCR_COSTS[pageType]);
|
||||
expect(annotationCalls[0]![2]).toBe(1);
|
||||
expect(annotationCalls[0]![3]).toBe(
|
||||
OCR_COSTS['mistral-ocr:annotations:page'],
|
||||
expect(annotationCalls[0]![3]).toBe(OCR_COSTS[annotationType]);
|
||||
});
|
||||
|
||||
it('bills OCR 3 at its own rate', async () => {
|
||||
const { actor } = await makeUser();
|
||||
mistralOcrProcessMock.mockResolvedValueOnce({
|
||||
pages: [{ index: 0, markdown: 'a' }],
|
||||
usageInfo: { pagesProcessed: 1 },
|
||||
});
|
||||
|
||||
await withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('x'), 'application/pdf'),
|
||||
model: 'mistral-ocr-3',
|
||||
}),
|
||||
);
|
||||
|
||||
expect(mistralOcrProcessMock.mock.calls[0]![0].model).toBe(
|
||||
'mistral-ocr-2512',
|
||||
);
|
||||
const [, type, count, cost] = incrementUsageSpy.mock.calls[0]!;
|
||||
expect(type).toBe('mistral-ocr:mistral-ocr-2512:page');
|
||||
expect(count).toBe(1);
|
||||
expect(cost).toBe(OCR_COSTS['mistral-ocr:mistral-ocr-2512:page']);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -620,6 +811,79 @@ describe('OCRDriver provider aliases', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('OCRDriver model routing', () => {
|
||||
it.each([
|
||||
['aws-textract', 'textract'],
|
||||
['textract', 'textract'],
|
||||
['mistral-ocr-latest', 'mistral'],
|
||||
['mistral-ocr-4-0', 'mistral'],
|
||||
['MISTRAL-OCR-2512', 'mistral'],
|
||||
])('model %s alone selects the %s backend', async (model, expected) => {
|
||||
const { actor } = await makeUser();
|
||||
textractSendMock.mockResolvedValueOnce({
|
||||
Blocks: [{ BlockType: 'PAGE' }],
|
||||
});
|
||||
mistralOcrProcessMock.mockResolvedValueOnce({ pages: [] });
|
||||
|
||||
await withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('img'), 'image/png'),
|
||||
model,
|
||||
}),
|
||||
);
|
||||
|
||||
if (expected === 'textract') {
|
||||
expect(textractSendMock).toHaveBeenCalledTimes(1);
|
||||
expect(mistralOcrProcessMock).not.toHaveBeenCalled();
|
||||
} else {
|
||||
expect(mistralOcrProcessMock).toHaveBeenCalledTimes(1);
|
||||
expect(textractSendMock).not.toHaveBeenCalled();
|
||||
}
|
||||
});
|
||||
|
||||
it.each([
|
||||
[{ model: 'mistral-ocr-2505' }, 'no longer available'],
|
||||
[{ model: 'gpt-4o' }, 'Unknown OCR model'],
|
||||
[{ model: '' }, 'non-empty string'],
|
||||
[
|
||||
{ model: 'mistral-ocr-latest', provider: 'aws-textract' },
|
||||
'not served by provider',
|
||||
],
|
||||
])('rejects %j with a 400 before any upstream call', async (args, message) => {
|
||||
const { actor } = await makeUser();
|
||||
|
||||
await expect(
|
||||
withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('img'), 'image/png'),
|
||||
...args,
|
||||
}),
|
||||
),
|
||||
).rejects.toMatchObject({
|
||||
statusCode: 400,
|
||||
message: expect.stringContaining(message),
|
||||
});
|
||||
expect(textractSendMock).not.toHaveBeenCalled();
|
||||
expect(mistralOcrProcessMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('keeps the retired mistral-ocr-2503 working on the model Mistral serves it with', async () => {
|
||||
const { actor } = await makeUser();
|
||||
mistralOcrProcessMock.mockResolvedValueOnce({ pages: [] });
|
||||
|
||||
await withActor(actor, () =>
|
||||
driver.recognize({
|
||||
source: dataUrl(Buffer.from('img'), 'image/png'),
|
||||
model: 'mistral-ocr-2503',
|
||||
}),
|
||||
);
|
||||
|
||||
expect(mistralOcrProcessMock.mock.calls[0]![0].model).toBe(
|
||||
'mistral-ocr-4-1',
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('OCRDriver default provider selection', () => {
|
||||
it('defaults to aws-textract when both providers are configured', async () => {
|
||||
const { actor } = await makeUser();
|
||||
|
||||
@@ -18,9 +18,10 @@
|
||||
*/
|
||||
|
||||
import {
|
||||
AnalyzeDocumentCommand,
|
||||
DetectDocumentTextCommand,
|
||||
InvalidS3ObjectException,
|
||||
TextractClient,
|
||||
UnsupportedDocumentException,
|
||||
} from '@aws-sdk/client-textract';
|
||||
import { Mistral } from '@mistralai/mistralai';
|
||||
import { Actor } from '../../core/actor.js';
|
||||
@@ -32,6 +33,14 @@ import { PuterDriver } from '../types.js';
|
||||
import { AI_CONCURRENT, AI_RATE_LIMIT } from '../util/aiLimits.js';
|
||||
import { loadFileInput, type LoadedFile } from '../util/fileInput.js';
|
||||
import { OCR_COSTS } from './costs.js';
|
||||
import {
|
||||
DEFAULT_OCR_MODEL,
|
||||
findOcrModel,
|
||||
OCR_MAX_INPUT_BYTES,
|
||||
RETIRED_OCR_MODELS,
|
||||
type OcrModel,
|
||||
type OcrProviderId,
|
||||
} from './models.js';
|
||||
|
||||
/**
|
||||
* Driver implementing `puter-ocr` — document OCR. Two providers: •
|
||||
@@ -42,17 +51,29 @@ interface RecognizeArgs {
|
||||
source?: unknown;
|
||||
file?: unknown;
|
||||
provider?: string;
|
||||
// Mistral-specific options — ignored by Textract.
|
||||
model?: string;
|
||||
// Mistral-specific options — ignored by Textract.
|
||||
pages?: number[];
|
||||
includeImageBase64?: boolean;
|
||||
imageLimit?: number;
|
||||
imageMinSize?: number;
|
||||
bboxAnnotationFormat?: unknown;
|
||||
documentAnnotationFormat?: unknown;
|
||||
documentAnnotationPrompt?: string;
|
||||
tableFormat?: string;
|
||||
extractHeader?: boolean;
|
||||
extractFooter?: boolean;
|
||||
test_mode?: boolean;
|
||||
}
|
||||
|
||||
/** One recognized line, in the same shape for every provider. */
|
||||
interface OcrBlock {
|
||||
type: string;
|
||||
text: string;
|
||||
confidence?: number;
|
||||
page: number;
|
||||
}
|
||||
|
||||
interface TextractBlock {
|
||||
BlockType?: string;
|
||||
Confidence?: number;
|
||||
@@ -67,6 +88,7 @@ interface MistralOcrResponse {
|
||||
images?: unknown[];
|
||||
dimensions?: unknown;
|
||||
}>;
|
||||
documentAnnotation?: string | null;
|
||||
usageInfo?: { pagesProcessed?: number };
|
||||
}
|
||||
|
||||
@@ -82,7 +104,7 @@ const OCR_PROVIDERS = ['aws-textract', 'mistral'] as const;
|
||||
|
||||
// Aliases callers may use in place of a canonical provider id. Resolved here
|
||||
// rather than in the SDK so a new alias reaches every caller at once.
|
||||
const PROVIDER_BY_ALIAS: Record<string, string> = {
|
||||
const PROVIDER_BY_ALIAS: Record<string, OcrProviderId> = {
|
||||
aws: 'aws-textract',
|
||||
'aws-textract': 'aws-textract',
|
||||
textract: 'aws-textract',
|
||||
@@ -90,11 +112,58 @@ const PROVIDER_BY_ALIAS: Record<string, string> = {
|
||||
'mistral-ocr': 'mistral',
|
||||
};
|
||||
|
||||
const normalizeOcrProvider = (value: unknown): string | undefined =>
|
||||
const normalizeOcrProvider = (value: unknown): OcrProviderId | undefined =>
|
||||
typeof value === 'string'
|
||||
? PROVIDER_BY_ALIAS[value.trim().toLowerCase()]
|
||||
: undefined;
|
||||
|
||||
const TABLE_FORMATS = new Set(['markdown', 'html']);
|
||||
|
||||
const badRequest = (message: string) =>
|
||||
new HttpError(400, message, { legacyCode: 'bad_request' });
|
||||
|
||||
/**
|
||||
* Accept a Mistral annotation format in the REST spelling
|
||||
* (`json_schema.schema`) or the SDK spelling (`jsonSchema.schemaDefinition`)
|
||||
* and return the SDK shape.
|
||||
*/
|
||||
const toMistralResponseFormat = (
|
||||
name: string,
|
||||
value: unknown,
|
||||
): Record<string, unknown> => {
|
||||
const format = value as Record<string, unknown> | null;
|
||||
const jsonSchema = (format?.jsonSchema ?? format?.json_schema) as
|
||||
Record<string, unknown> | undefined;
|
||||
const schema = jsonSchema?.schemaDefinition ?? jsonSchema?.schema;
|
||||
if (
|
||||
!format ||
|
||||
typeof format !== 'object' ||
|
||||
(format.type !== undefined && format.type !== 'json_schema') ||
|
||||
!schema ||
|
||||
typeof schema !== 'object'
|
||||
) {
|
||||
throw badRequest(
|
||||
`\`${name}\` must be { type: 'json_schema', json_schema: { name, schema } }`,
|
||||
);
|
||||
}
|
||||
return {
|
||||
type: 'json_schema',
|
||||
jsonSchema: {
|
||||
name:
|
||||
typeof jsonSchema?.name === 'string'
|
||||
? jsonSchema.name
|
||||
: 'annotation',
|
||||
schemaDefinition: schema,
|
||||
...(typeof jsonSchema?.description === 'string' && {
|
||||
description: jsonSchema.description,
|
||||
}),
|
||||
...(typeof jsonSchema?.strict === 'boolean' && {
|
||||
strict: jsonSchema.strict,
|
||||
}),
|
||||
},
|
||||
};
|
||||
};
|
||||
|
||||
export class OCRDriver extends PuterDriver {
|
||||
readonly driverInterface = 'puter-ocr';
|
||||
readonly driverName = 'ai-ocr';
|
||||
@@ -118,7 +187,7 @@ export class OCRDriver extends PuterDriver {
|
||||
}
|
||||
|
||||
// Older SDK bundles name the provider in the driver slot instead of
|
||||
// passing `{ provider }`; `#resolveProvider` reads the requested alias
|
||||
// passing `{ provider }`; `#resolveModel` reads the requested alias
|
||||
// back off the Context.
|
||||
readonly driverAliases = [...OCR_PROVIDERS];
|
||||
readonly isDefault = true;
|
||||
@@ -175,11 +244,7 @@ export class OCRDriver extends PuterDriver {
|
||||
async recognize(args: RecognizeArgs) {
|
||||
if (args.test_mode) return sampleResponse();
|
||||
|
||||
const provider = this.#resolveProvider(args);
|
||||
if (!provider)
|
||||
throw new HttpError(500, 'No OCR provider configured', {
|
||||
legacyCode: 'internal_error',
|
||||
});
|
||||
const model = this.#resolveModel(args);
|
||||
|
||||
const actor = Context.get('actor');
|
||||
if (!actor)
|
||||
@@ -188,9 +253,15 @@ export class OCRDriver extends PuterDriver {
|
||||
});
|
||||
|
||||
const input = args.source ?? args.file;
|
||||
if (!input)
|
||||
throw new HttpError(400, '`source` is required', {
|
||||
legacyCode: 'bad_request',
|
||||
if (!input) throw badRequest('`source` is required');
|
||||
|
||||
if (model.provider === 'aws-textract' && !this.#awsConfig)
|
||||
throw new HttpError(500, 'AWS credentials not configured', {
|
||||
legacyCode: 'internal_error',
|
||||
});
|
||||
if (model.provider === 'mistral' && !this.#mistral)
|
||||
throw new HttpError(500, 'Mistral OCR not configured', {
|
||||
legacyCode: 'internal_error',
|
||||
});
|
||||
|
||||
const loaded = await loadFileInput(
|
||||
@@ -198,57 +269,80 @@ export class OCRDriver extends PuterDriver {
|
||||
this.services.fs,
|
||||
actor,
|
||||
input,
|
||||
{ acceptWebInput: true },
|
||||
{
|
||||
acceptWebInput: true,
|
||||
maxBytes: OCR_MAX_INPUT_BYTES[model.provider],
|
||||
},
|
||||
);
|
||||
|
||||
if (provider === 'aws-textract') {
|
||||
if (!this.#awsConfig)
|
||||
throw new HttpError(500, 'AWS credentials not configured', {
|
||||
legacyCode: 'internal_error',
|
||||
});
|
||||
return this.#textractRecognize(loaded, actor);
|
||||
}
|
||||
if (provider === 'mistral') {
|
||||
if (!this.#mistral)
|
||||
throw new HttpError(500, 'Mistral OCR not configured', {
|
||||
legacyCode: 'internal_error',
|
||||
});
|
||||
return this.#mistralRecognize(loaded, args, actor);
|
||||
}
|
||||
throw new HttpError(400, `Unknown OCR provider: ${provider}`, {
|
||||
legacyCode: 'bad_request',
|
||||
});
|
||||
return model.provider === 'aws-textract'
|
||||
? this.#textractRecognize(loaded, model, actor)
|
||||
: this.#mistralRecognize(loaded, args, model, actor);
|
||||
}
|
||||
|
||||
/**
|
||||
* Decide which provider handles a call: an explicit `provider` wins, then
|
||||
* the legacy driver alias the caller dispatched through, then whichever
|
||||
* provider is configured.
|
||||
* Decide which model handles a call. A `model` picks its own provider; an
|
||||
* explicit `provider`, then the legacy driver alias the caller dispatched
|
||||
* through, then whichever provider is configured pick that provider's
|
||||
* default model.
|
||||
*/
|
||||
#resolveProvider(args: RecognizeArgs): string | null {
|
||||
#resolveModel(args: RecognizeArgs): OcrModel {
|
||||
let provider: OcrProviderId | undefined;
|
||||
if (args.provider) {
|
||||
const named = normalizeOcrProvider(args.provider);
|
||||
if (!named) {
|
||||
throw new HttpError(
|
||||
400,
|
||||
provider = normalizeOcrProvider(args.provider);
|
||||
if (!provider) {
|
||||
throw badRequest(
|
||||
`Unknown OCR provider: ${args.provider}. Available: ${OCR_PROVIDERS.join(', ')}`,
|
||||
{ legacyCode: 'bad_request' },
|
||||
);
|
||||
}
|
||||
return named;
|
||||
}
|
||||
return (
|
||||
|
||||
if (args.model !== undefined) {
|
||||
if (typeof args.model !== 'string' || !args.model.trim())
|
||||
throw badRequest('`model` must be a non-empty string');
|
||||
const retired = RETIRED_OCR_MODELS[args.model.trim().toLowerCase()];
|
||||
if (retired)
|
||||
throw badRequest(
|
||||
`${args.model} is no longer available: ${retired}`,
|
||||
);
|
||||
const model = findOcrModel(args.model);
|
||||
if (!model) throw badRequest(`Unknown OCR model: ${args.model}`);
|
||||
if (provider && model.provider !== provider)
|
||||
throw badRequest(
|
||||
`OCR model ${args.model} is not served by provider ${args.provider}`,
|
||||
);
|
||||
return model;
|
||||
}
|
||||
|
||||
provider ??=
|
||||
normalizeOcrProvider(Context.get('driverName')) ??
|
||||
this.#defaultProvider()
|
||||
);
|
||||
this.#defaultProvider();
|
||||
if (!provider)
|
||||
throw new HttpError(500, 'No OCR provider configured', {
|
||||
legacyCode: 'internal_error',
|
||||
});
|
||||
return findOcrModel(DEFAULT_OCR_MODEL[provider])!;
|
||||
}
|
||||
|
||||
#defaultProvider(): 'aws-textract' | 'mistral' | null {
|
||||
#defaultProvider(): OcrProviderId | null {
|
||||
if (this.#awsConfig) return 'aws-textract';
|
||||
if (this.#mistral) return 'mistral';
|
||||
return null;
|
||||
}
|
||||
|
||||
async #assertCredits(actor: Actor, costPerPage: number) {
|
||||
// Page count is only known once the provider answers, so pre-flight
|
||||
// one page's cost and meter the real total afterward.
|
||||
const hasCredits = await this.services.metering.hasEnoughCredits(
|
||||
actor,
|
||||
costPerPage,
|
||||
);
|
||||
if (!hasCredits)
|
||||
throw new HttpError(402, 'Insufficient credits', {
|
||||
legacyCode: 'insufficient_funds',
|
||||
});
|
||||
}
|
||||
|
||||
// -- AWS Textract -------------------------------------------------
|
||||
|
||||
#textractClientFor(region: string): TextractClient {
|
||||
@@ -265,17 +359,13 @@ export class OCRDriver extends PuterDriver {
|
||||
return client;
|
||||
}
|
||||
|
||||
async #textractRecognize(loaded: LoadedFile, actor: Actor) {
|
||||
const usageType = 'aws-textract:detect-document-text:page';
|
||||
const costPerPage = OCR_COSTS[usageType];
|
||||
const hasCredits = await this.services.metering.hasEnoughCredits(
|
||||
actor!,
|
||||
costPerPage,
|
||||
);
|
||||
if (!hasCredits)
|
||||
throw new HttpError(402, 'Insufficient credits', {
|
||||
legacyCode: 'insufficient_funds',
|
||||
});
|
||||
async #textractRecognize(
|
||||
loaded: LoadedFile,
|
||||
model: OcrModel,
|
||||
actor: Actor,
|
||||
) {
|
||||
const costPerPage = OCR_COSTS[model.pageUsageType];
|
||||
await this.#assertCredits(actor, costPerPage);
|
||||
|
||||
// Prefer S3 direct source if the file is FS-backed; fall back to raw bytes.
|
||||
const s3Info =
|
||||
@@ -300,61 +390,55 @@ export class OCRDriver extends PuterDriver {
|
||||
? { S3Object: { Bucket: s3Info.bucket, Name: s3Info.key } }
|
||||
: { Bytes: loaded.buffer };
|
||||
return client.send(
|
||||
new AnalyzeDocumentCommand({
|
||||
Document: document,
|
||||
FeatureTypes: ['LAYOUT'],
|
||||
}),
|
||||
new DetectDocumentTextCommand({ Document: document }),
|
||||
);
|
||||
};
|
||||
|
||||
let response;
|
||||
try {
|
||||
response = await tryRun(Boolean(s3Info));
|
||||
} catch (err) {
|
||||
if (s3Info && err instanceof InvalidS3ObjectException) {
|
||||
try {
|
||||
response = await tryRun(Boolean(s3Info));
|
||||
} catch (err) {
|
||||
if (!(s3Info && err instanceof InvalidS3ObjectException))
|
||||
throw err;
|
||||
response = await tryRun(false);
|
||||
} else {
|
||||
throw err;
|
||||
}
|
||||
} catch (err) {
|
||||
if (err instanceof UnsupportedDocumentException)
|
||||
throw badRequest(
|
||||
'AWS Textract reads JPEG, PNG, TIFF and single-page PDF documents; use a Mistral OCR model for multi-page PDFs and other formats',
|
||||
);
|
||||
throw err;
|
||||
}
|
||||
|
||||
const blocks: Array<{
|
||||
type: string;
|
||||
confidence: number;
|
||||
text: string;
|
||||
}> = [];
|
||||
const blocks: OcrBlock[] = [];
|
||||
let pageCount = 0;
|
||||
for (const block of (response.Blocks ?? []) as TextractBlock[]) {
|
||||
if (block.BlockType === 'PAGE') {
|
||||
pageCount += 1;
|
||||
continue;
|
||||
}
|
||||
if (
|
||||
[
|
||||
'CELL',
|
||||
'TABLE',
|
||||
'MERGED_CELL',
|
||||
'LAYOUT_FIGURE',
|
||||
'LAYOUT_TEXT',
|
||||
'WORD',
|
||||
].includes(block.BlockType ?? '')
|
||||
)
|
||||
continue;
|
||||
if (block.BlockType !== 'LINE' || !block.Text) continue;
|
||||
blocks.push({
|
||||
type: `text/textract:${block.BlockType ?? 'UNKNOWN'}`,
|
||||
type: 'text/textract:LINE',
|
||||
text: block.Text,
|
||||
confidence: Number(block.Confidence ?? 0),
|
||||
text: block.Text ?? '',
|
||||
page: Math.max(pageCount - 1, 0),
|
||||
});
|
||||
}
|
||||
|
||||
const pages = pageCount || 1;
|
||||
this.#aiMetering.incrementUsage(
|
||||
actor,
|
||||
usageType,
|
||||
model.pageUsageType,
|
||||
pages,
|
||||
costPerPage * pages,
|
||||
);
|
||||
return { blocks };
|
||||
return {
|
||||
model: model.id,
|
||||
blocks,
|
||||
text: blocks.map((b) => b.text).join('\n'),
|
||||
};
|
||||
}
|
||||
|
||||
// -- Mistral OCR --------------------------------------------------
|
||||
@@ -362,23 +446,13 @@ export class OCRDriver extends PuterDriver {
|
||||
async #mistralRecognize(
|
||||
loaded: LoadedFile,
|
||||
args: RecognizeArgs,
|
||||
model: OcrModel,
|
||||
actor: Actor,
|
||||
) {
|
||||
// Gate on credits before the paid upstream call, mirroring the
|
||||
// Textract branch. Page count isn't known until Mistral responds, so
|
||||
// pre-flight one page's cost and meter the real total afterward.
|
||||
const hasCredits = await this.services.metering.hasEnoughCredits(
|
||||
actor,
|
||||
OCR_COSTS['mistral-ocr:ocr:page'],
|
||||
);
|
||||
if (!hasCredits)
|
||||
throw new HttpError(402, 'Insufficient credits', {
|
||||
legacyCode: 'insufficient_funds',
|
||||
});
|
||||
|
||||
const model = args.model ?? 'mistral-ocr-latest';
|
||||
const chunk = this.#mistralBuildChunk(loaded);
|
||||
const payload: Record<string, unknown> = { model, document: chunk };
|
||||
const payload: Record<string, unknown> = {
|
||||
model: model.id,
|
||||
document: this.#mistralBuildChunk(loaded),
|
||||
};
|
||||
if (args.pages) payload.pages = args.pages;
|
||||
if (args.includeImageBase64 !== undefined)
|
||||
payload.includeImageBase64 = args.includeImageBase64;
|
||||
@@ -387,16 +461,41 @@ export class OCRDriver extends PuterDriver {
|
||||
if (typeof args.imageMinSize === 'number')
|
||||
payload.imageMinSize = args.imageMinSize;
|
||||
if (args.bboxAnnotationFormat !== undefined)
|
||||
payload.bboxAnnotationFormat = args.bboxAnnotationFormat;
|
||||
payload.bboxAnnotationFormat = toMistralResponseFormat(
|
||||
'bboxAnnotationFormat',
|
||||
args.bboxAnnotationFormat,
|
||||
);
|
||||
if (args.documentAnnotationFormat !== undefined)
|
||||
payload.documentAnnotationFormat = args.documentAnnotationFormat;
|
||||
payload.documentAnnotationFormat = toMistralResponseFormat(
|
||||
'documentAnnotationFormat',
|
||||
args.documentAnnotationFormat,
|
||||
);
|
||||
if (typeof args.documentAnnotationPrompt === 'string')
|
||||
payload.documentAnnotationPrompt = args.documentAnnotationPrompt;
|
||||
if (args.tableFormat !== undefined) {
|
||||
if (!TABLE_FORMATS.has(args.tableFormat))
|
||||
throw badRequest("`tableFormat` must be 'markdown' or 'html'");
|
||||
payload.tableFormat = args.tableFormat;
|
||||
}
|
||||
if (typeof args.extractHeader === 'boolean')
|
||||
payload.extractHeader = args.extractHeader;
|
||||
if (typeof args.extractFooter === 'boolean')
|
||||
payload.extractFooter = args.extractFooter;
|
||||
|
||||
const response = await this.#mistral!.ocr.process(payload);
|
||||
const annotations =
|
||||
payload.documentAnnotationFormat !== undefined ||
|
||||
payload.bboxAnnotationFormat !== undefined;
|
||||
this.#recordMistralUsage(response, actor, annotations);
|
||||
return this.#normalizeMistralResponse(response);
|
||||
await this.#assertCredits(
|
||||
actor,
|
||||
OCR_COSTS[model.pageUsageType] +
|
||||
(annotations && model.annotationUsageType
|
||||
? OCR_COSTS[model.annotationUsageType]
|
||||
: 0),
|
||||
);
|
||||
|
||||
const response = await this.#mistral!.ocr.process(payload);
|
||||
this.#recordMistralUsage(response, model, actor, annotations);
|
||||
return this.#normalizeMistralResponse(response, model);
|
||||
}
|
||||
|
||||
#mistralBuildChunk(loaded: LoadedFile): Record<string, unknown> {
|
||||
@@ -404,11 +503,14 @@ export class OCRDriver extends PuterDriver {
|
||||
loaded.mimeType ??
|
||||
mimeFromName(loaded.filename) ??
|
||||
'application/octet-stream';
|
||||
const isPdf =
|
||||
mime.includes('pdf') ||
|
||||
loaded.filename.toLowerCase().endsWith('.pdf');
|
||||
// Declared documents (PDF, DOCX, PPTX, ...) and PDF bytes go as a
|
||||
// document; images and untyped bytes keep the image chunk.
|
||||
const isDocument =
|
||||
loaded.buffer.subarray(0, 4).toString('latin1') === '%PDF' ||
|
||||
loaded.filename.toLowerCase().endsWith('.pdf') ||
|
||||
(!mime.startsWith('image/') && mime !== 'application/octet-stream');
|
||||
const dataUrl = `data:${mime};base64,${loaded.buffer.toString('base64')}`;
|
||||
if (isPdf) {
|
||||
if (isDocument) {
|
||||
return {
|
||||
type: 'document_url',
|
||||
documentUrl: dataUrl,
|
||||
@@ -418,10 +520,10 @@ export class OCRDriver extends PuterDriver {
|
||||
return { type: 'image_url', imageUrl: { url: dataUrl } };
|
||||
}
|
||||
|
||||
#normalizeMistralResponse(response: MistralOcrResponse) {
|
||||
#normalizeMistralResponse(response: MistralOcrResponse, model: OcrModel) {
|
||||
const pages = response?.pages ?? [];
|
||||
const blocks: Array<{ type: string; text: string; page?: number }> = [];
|
||||
for (const page of pages) {
|
||||
const blocks: OcrBlock[] = [];
|
||||
for (const [position, page] of pages.entries()) {
|
||||
if (typeof page?.markdown !== 'string') continue;
|
||||
const lines = page.markdown
|
||||
.split('\n')
|
||||
@@ -431,28 +533,25 @@ export class OCRDriver extends PuterDriver {
|
||||
blocks.push({
|
||||
type: 'text/mistral:LINE',
|
||||
text: line,
|
||||
page: page.index,
|
||||
page: page.index ?? position,
|
||||
});
|
||||
}
|
||||
}
|
||||
const text =
|
||||
blocks.length > 0
|
||||
? blocks.map((b) => b.text).join('\n')
|
||||
: pages
|
||||
.map((p) => p?.markdown ?? '')
|
||||
.join('\n\n')
|
||||
.trim();
|
||||
return {
|
||||
model: response?.model,
|
||||
model: response?.model ?? model.id,
|
||||
pages,
|
||||
usage_info: response?.usageInfo,
|
||||
...(typeof response?.documentAnnotation === 'string' && {
|
||||
document_annotation: response.documentAnnotation,
|
||||
}),
|
||||
blocks,
|
||||
text,
|
||||
text: blocks.map((b) => b.text).join('\n'),
|
||||
};
|
||||
}
|
||||
|
||||
#recordMistralUsage(
|
||||
response: MistralOcrResponse,
|
||||
model: OcrModel,
|
||||
actor: Actor,
|
||||
annotations: boolean,
|
||||
) {
|
||||
@@ -462,16 +561,16 @@ export class OCRDriver extends PuterDriver {
|
||||
(Array.isArray(response?.pages) ? response.pages.length : 1);
|
||||
this.#aiMetering.incrementUsage(
|
||||
actor,
|
||||
'mistral-ocr:ocr:page',
|
||||
model.pageUsageType,
|
||||
pagesProcessed,
|
||||
OCR_COSTS['mistral-ocr:ocr:page'] * pagesProcessed,
|
||||
OCR_COSTS[model.pageUsageType] * pagesProcessed,
|
||||
);
|
||||
if (annotations) {
|
||||
if (annotations && model.annotationUsageType) {
|
||||
this.#aiMetering.incrementUsage(
|
||||
actor,
|
||||
'mistral-ocr:annotations:page',
|
||||
model.annotationUsageType,
|
||||
pagesProcessed,
|
||||
OCR_COSTS['mistral-ocr:annotations:page'] * pagesProcessed,
|
||||
OCR_COSTS[model.annotationUsageType] * pagesProcessed,
|
||||
);
|
||||
}
|
||||
} catch {
|
||||
@@ -482,12 +581,15 @@ export class OCRDriver extends PuterDriver {
|
||||
|
||||
function sampleResponse() {
|
||||
return {
|
||||
model: 'test-mode',
|
||||
blocks: [
|
||||
{
|
||||
type: 'text/puter:sample-output',
|
||||
confidence: 1,
|
||||
text: 'test_mode is enabled; this is a sample OCR response.',
|
||||
page: 0,
|
||||
},
|
||||
],
|
||||
text: 'test_mode is enabled; this is a sample OCR response.',
|
||||
};
|
||||
}
|
||||
@@ -17,10 +17,16 @@
|
||||
* along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||
*/
|
||||
|
||||
// Microcents per page — Textract $1.50/1000 pages = 150,000 µ¢/page.
|
||||
// Mistral OCR $1/1000 pages, annotations $3/1000 pages.
|
||||
// Microcents per page. Textract DetectDocumentText $1.50/1000 pages. Mistral
|
||||
// OCR 4.x $4/1000 pages with annotations $5/1000; OCR 3 $2 and $3.
|
||||
export const OCR_COSTS = {
|
||||
'aws-textract:detect-document-text:page': 150000,
|
||||
'mistral-ocr:ocr:page': 100000,
|
||||
'mistral-ocr:annotations:page': 300000,
|
||||
'mistral-ocr:mistral-ocr-4-1:page': 400000,
|
||||
'mistral-ocr:mistral-ocr-4-1:annotations:page': 500000,
|
||||
'mistral-ocr:mistral-ocr-4-0:page': 400000,
|
||||
'mistral-ocr:mistral-ocr-4-0:annotations:page': 500000,
|
||||
'mistral-ocr:mistral-ocr-2512:page': 200000,
|
||||
'mistral-ocr:mistral-ocr-2512:annotations:page': 300000,
|
||||
} as const;
|
||||
|
||||
export type OcrUsageType = keyof typeof OCR_COSTS;
|
||||
@@ -0,0 +1,91 @@
|
||||
/*
|
||||
* Copyright (C) 2024-present Puter Technologies Inc.
|
||||
*
|
||||
* This file is part of Puter.
|
||||
*
|
||||
* Puter is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Affero General Public License as published
|
||||
* by the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Affero General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Affero General Public License
|
||||
* along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||
*/
|
||||
|
||||
import type { OcrUsageType } from './costs.js';
|
||||
|
||||
export type OcrProviderId = 'aws-textract' | 'mistral';
|
||||
|
||||
export interface OcrModel {
|
||||
/** Canonical id; for Mistral also the pinned id sent upstream. */
|
||||
id: string;
|
||||
provider: OcrProviderId;
|
||||
/** Other spellings callers may pass, resolved to this entry. */
|
||||
aliases?: readonly string[];
|
||||
pageUsageType: OcrUsageType;
|
||||
annotationUsageType?: OcrUsageType;
|
||||
}
|
||||
|
||||
// Floating vendor aliases (`mistral-ocr-latest`) are pinned to the entry
|
||||
// they currently point at, so a vendor-side move never changes what a call
|
||||
// costs before the catalog is synced.
|
||||
export const OCR_MODELS: readonly OcrModel[] = [
|
||||
{
|
||||
id: 'aws-textract',
|
||||
provider: 'aws-textract',
|
||||
aliases: ['textract'],
|
||||
pageUsageType: 'aws-textract:detect-document-text:page',
|
||||
},
|
||||
{
|
||||
id: 'mistral-ocr-4-1',
|
||||
provider: 'mistral',
|
||||
// Preserve Puter's old 2503 spelling by routing it to OCR 4.1.
|
||||
aliases: ['mistral-ocr-latest', 'mistral-ocr-4', 'mistral-ocr-2503'],
|
||||
pageUsageType: 'mistral-ocr:mistral-ocr-4-1:page',
|
||||
annotationUsageType: 'mistral-ocr:mistral-ocr-4-1:annotations:page',
|
||||
},
|
||||
{
|
||||
id: 'mistral-ocr-4-0',
|
||||
provider: 'mistral',
|
||||
pageUsageType: 'mistral-ocr:mistral-ocr-4-0:page',
|
||||
annotationUsageType: 'mistral-ocr:mistral-ocr-4-0:annotations:page',
|
||||
},
|
||||
{
|
||||
id: 'mistral-ocr-2512',
|
||||
provider: 'mistral',
|
||||
aliases: ['mistral-ocr-3', 'mistral-ocr-3-0'],
|
||||
pageUsageType: 'mistral-ocr:mistral-ocr-2512:page',
|
||||
annotationUsageType: 'mistral-ocr:mistral-ocr-2512:annotations:page',
|
||||
},
|
||||
];
|
||||
|
||||
/** Models the vendor no longer serves, with the reason callers see. */
|
||||
export const RETIRED_OCR_MODELS: Readonly<Record<string, string>> = {
|
||||
'mistral-ocr-2505':
|
||||
'Puter no longer supports this deprecated model; use mistral-ocr-latest.',
|
||||
};
|
||||
|
||||
export const DEFAULT_OCR_MODEL: Record<OcrProviderId, string> = {
|
||||
'aws-textract': 'aws-textract',
|
||||
mistral: 'mistral-ocr-4-1',
|
||||
};
|
||||
|
||||
/** Largest input each provider accepts (Textract sync: 10 MB; Mistral: 50 MB). */
|
||||
export const OCR_MAX_INPUT_BYTES: Record<OcrProviderId, number> = {
|
||||
'aws-textract': 10 * 1024 * 1024,
|
||||
mistral: 50 * 1024 * 1024,
|
||||
};
|
||||
|
||||
const MODEL_BY_NAME = new Map<string, OcrModel>();
|
||||
for (const model of OCR_MODELS) {
|
||||
MODEL_BY_NAME.set(model.id, model);
|
||||
for (const alias of model.aliases ?? []) MODEL_BY_NAME.set(alias, model);
|
||||
}
|
||||
|
||||
export const findOcrModel = (name: string): OcrModel | undefined =>
|
||||
MODEL_BY_NAME.get(name.trim().toLowerCase());
|
||||
@@ -17,7 +17,7 @@
|
||||
* along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||
*/
|
||||
|
||||
import { afterAll, beforeAll, describe, expect, it } from 'vitest';
|
||||
import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest';
|
||||
import { v4 as uuidv4 } from 'uuid';
|
||||
import type { Actor } from '../../core/actor.js';
|
||||
import { runWithContext } from '../../core/context.js';
|
||||
@@ -26,6 +26,14 @@ import { setupTestServer } from '../../testUtil.js';
|
||||
import { generateDefaultFsentries } from '../../util/userProvisioning.js';
|
||||
import { inferFilenameFromUrlOrPath, loadFileInput } from './fileInput.js';
|
||||
|
||||
// Web inputs leave the process through secureFetch; stub that boundary so
|
||||
// the tests control the remote response without real network access.
|
||||
const { secureFetchMock } = vi.hoisted(() => ({ secureFetchMock: vi.fn() }));
|
||||
vi.mock('../../util/secureHttp.js', async (importOriginal) => ({
|
||||
...(await importOriginal<typeof import('../../util/secureHttp.js')>()),
|
||||
secureFetch: secureFetchMock,
|
||||
}));
|
||||
|
||||
// ── Test harness ────────────────────────────────────────────────────
|
||||
//
|
||||
// Boots one real PuterServer (in-memory sqlite + dynamo + s3 + mock
|
||||
@@ -391,3 +399,78 @@ describe('loadFileInput FS path', () => {
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
// ── loadFileInput web URL ──────────────────────────────────────────
|
||||
|
||||
/** A web stream that yields `chunks` one pull at a time, counting pulls. */
|
||||
const chunkedBody = (chunks: Buffer[]) => {
|
||||
const state = { pulls: 0, cancelled: false };
|
||||
const body = new ReadableStream<Uint8Array>({
|
||||
pull(controller) {
|
||||
const next = chunks[state.pulls++];
|
||||
if (next) controller.enqueue(new Uint8Array(next));
|
||||
else controller.close();
|
||||
},
|
||||
cancel() {
|
||||
state.cancelled = true;
|
||||
},
|
||||
});
|
||||
return { body, state };
|
||||
};
|
||||
|
||||
describe('loadFileInput web URL', () => {
|
||||
it('returns the fetched bytes with the response content type', async () => {
|
||||
const { actor } = await makeUser();
|
||||
const { body } = chunkedBody([Buffer.from('hello '), Buffer.from('web')]);
|
||||
secureFetchMock.mockResolvedValueOnce(
|
||||
new Response(body, {
|
||||
headers: { 'content-type': 'image/png; charset=binary' },
|
||||
}),
|
||||
);
|
||||
|
||||
const loaded = await callLoadFileInput(
|
||||
actor,
|
||||
'https://example.com/scan.png',
|
||||
{ acceptWebInput: true, maxBytes: 64 },
|
||||
);
|
||||
|
||||
expect(loaded.buffer.toString()).toBe('hello web');
|
||||
expect(loaded.mimeType).toBe('image/png');
|
||||
expect(loaded.filename).toBe('scan.png');
|
||||
});
|
||||
|
||||
it('refuses a declared content-length over maxBytes without reading the body', async () => {
|
||||
const { actor } = await makeUser();
|
||||
const { body, state } = chunkedBody([Buffer.alloc(32)]);
|
||||
secureFetchMock.mockResolvedValueOnce(
|
||||
new Response(body, { headers: { 'content-length': '4096' } }),
|
||||
);
|
||||
|
||||
await expect(
|
||||
callLoadFileInput(actor, 'https://example.com/big.pdf', {
|
||||
acceptWebInput: true,
|
||||
maxBytes: 64,
|
||||
}),
|
||||
).rejects.toMatchObject({
|
||||
statusCode: 413,
|
||||
legacyCode: 'storage_limit_reached',
|
||||
});
|
||||
expect(state.cancelled).toBe(true);
|
||||
});
|
||||
|
||||
it('stops reading an undeclared-length body once it passes maxBytes', async () => {
|
||||
const { actor } = await makeUser();
|
||||
const chunks = Array.from({ length: 50 }, () => Buffer.alloc(40));
|
||||
const { body, state } = chunkedBody(chunks);
|
||||
secureFetchMock.mockResolvedValueOnce(new Response(body));
|
||||
|
||||
await expect(
|
||||
callLoadFileInput(actor, 'https://example.com/stream', {
|
||||
acceptWebInput: true,
|
||||
maxBytes: 64,
|
||||
}),
|
||||
).rejects.toMatchObject({ statusCode: 413 });
|
||||
expect(state.cancelled).toBe(true);
|
||||
expect(state.pulls).toBeLessThan(chunks.length);
|
||||
});
|
||||
});
|
||||
@@ -19,6 +19,7 @@
|
||||
|
||||
import { posix as pathPosix } from 'node:path';
|
||||
import { Readable } from 'node:stream';
|
||||
import type { ReadableStream as WebReadableStream } from 'node:stream/web';
|
||||
import type { Actor } from '../../core/actor.js';
|
||||
import { HttpError } from '../../core/http/HttpError.js';
|
||||
import type { FSService } from '../../services/fs/FSService.js';
|
||||
@@ -65,8 +66,7 @@ export interface LoadedFile {
|
||||
* uuid? }`.
|
||||
*/
|
||||
export type FileInputRef =
|
||||
| string
|
||||
| { path?: string; uid?: string; uuid?: string };
|
||||
string | { path?: string; uid?: string; uuid?: string };
|
||||
|
||||
export interface OpenedFileInput {
|
||||
body: Readable;
|
||||
@@ -132,9 +132,18 @@ export async function loadFileInput(
|
||||
{ legacyCode: 'bad_request' },
|
||||
);
|
||||
}
|
||||
const arrayBuf = await response.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuf);
|
||||
assertMax(buffer, options.maxBytes);
|
||||
// Stream under the cap: a remote body is never buffered past maxBytes.
|
||||
const declaredLength = Number(response.headers.get('content-length'));
|
||||
if (options.maxBytes && declaredLength > options.maxBytes) {
|
||||
await response.body?.cancel();
|
||||
throw tooLarge(options.maxBytes);
|
||||
}
|
||||
const buffer = response.body
|
||||
? await collectStream(
|
||||
Readable.fromWeb(response.body as WebReadableStream),
|
||||
options.maxBytes,
|
||||
)
|
||||
: Buffer.alloc(0);
|
||||
const contentType = response.headers.get('content-type');
|
||||
const mime =
|
||||
contentType?.split(';')[0]?.trim() ||
|
||||
|
||||
+108
-27
@@ -1,10 +1,10 @@
|
||||
---
|
||||
title: puter.ai.img2txt()
|
||||
description: Extract text from images using OCR to read printed text, handwriting, and any text-based content.
|
||||
description: Extract text from images and documents using OCR to read printed text, handwriting, and any text-based content.
|
||||
platforms: [websites, apps, nodejs, workers]
|
||||
---
|
||||
|
||||
Given an image, returns the text contained in the image. Also known as OCR (Optical Character Recognition), this API can be used to extract text from images of printed text, handwriting, or any other text-based content. You can choose between AWS Textract (default) or Mistral’s OCR service when you need multilingual or richer annotation output.
|
||||
Given an image or document, returns the text it contains. Also known as OCR (Optical Character Recognition), this API can be used to extract text from images of printed text, handwriting, or any other text-based content. AWS Textract is the default; Mistral OCR reads multi-page documents and more file formats, returns Markdown, and can fill a JSON schema from the document.
|
||||
|
||||
## Syntax
|
||||
|
||||
@@ -18,7 +18,7 @@ puter.ai.img2txt({ source: image, ...options })
|
||||
|
||||
#### `image` / `source` (String|File|Blob) (required)
|
||||
|
||||
A string containing the URL or Puter path, or a `File`/`Blob` object containing the source image or file. When calling with an options object, pass it as `{ source: ... }`. Maximum input size at 10MB.
|
||||
A string containing the URL, Puter path, or data URI, or a `File`/`Blob` object containing the source image or document. When calling with an options object, pass it as `{ source: ... }`. See [Input limits](#input-limits) for the accepted formats and sizes.
|
||||
|
||||
#### `testMode` (Boolean) (Optional)
|
||||
|
||||
@@ -26,47 +26,81 @@ A boolean indicating whether you want to use the test API. Defaults to `false`.
|
||||
|
||||
#### `options` (Object) (Optional)
|
||||
|
||||
Additional settings for the OCR request. Available options depend on the provider.
|
||||
Every call has the same shape; only the model name changes which service reads the input.
|
||||
|
||||
| Option | Type | Description |
|
||||
|--------|------|-------------|
|
||||
| `provider` | `String` | The OCR backend to use. `'aws-textract'` (default) \| `'mistral'`. Aliases `'aws'`, `'textract'` and `'mistral-ocr'` are also accepted; anything else is rejected with a `bad_request` error |
|
||||
| `model` | `String` | OCR model to use (provider-specific) |
|
||||
| `model` | `String` | The OCR model to use (see [Models](#models)). The model picks its provider, so `provider` is not needed alongside it. Lookups are case-insensitive. |
|
||||
| `provider` | `String` | `'aws-textract'` (default) or `'mistral'`. Aliases `'aws'`, `'textract'` and `'mistral-ocr'` are also accepted. Without a `model`, the provider's default model runs. When both are given, the model must belong to the provider. |
|
||||
| `testMode` | `Boolean` | When `true`, returns a sample response without using credits. Defaults to `false` |
|
||||
|
||||
#### AWS Textract Options
|
||||
#### Models
|
||||
|
||||
Available when `provider: 'aws-textract'` (default):
|
||||
| Model | Provider | Notes |
|
||||
|-------|----------|-------|
|
||||
| `aws-textract` (alias `textract`) | AWS Textract | Default. Plain text, one line per detected line. |
|
||||
| `mistral-ocr-latest` (aliases `mistral-ocr-4`, `mistral-ocr-4-1`) | Mistral | Mistral OCR 4.1, the default Mistral model. Markdown output. |
|
||||
| `mistral-ocr-4-0` | Mistral | Mistral OCR 4.0. |
|
||||
| `mistral-ocr-2512` (aliases `mistral-ocr-3`, `mistral-ocr-3-0`) | Mistral | Mistral OCR 3, at a lower per-page rate than OCR 4. |
|
||||
|
||||
`mistral-ocr-latest` is pinned to OCR 4.1 and moves to a newer model only when Puter adds it. Mistral has deprecated `mistral-ocr-2503`; Puter keeps that name as a compatibility alias for OCR 4.1. Puter rejects the deprecated `mistral-ocr-2505` with `bad_request`. Per-page prices for each model are listed by the API at `GET /metering/allCosts`.
|
||||
|
||||
#### AWS Textract options
|
||||
|
||||
AWS Textract takes no options beyond `model` and `provider`. It reads a single-page document as a whole, so there is nothing to select or tune; the Mistral options below are ignored.
|
||||
|
||||
#### Mistral options
|
||||
|
||||
These options apply to Mistral models. AWS Textract ignores them.
|
||||
|
||||
| Option | Type | Description |
|
||||
|--------|------|-------------|
|
||||
| `pages` | `Array<Number>` | Limit processing to specific page numbers (multi-page PDFs) |
|
||||
| `pages` | `Array<Number>` | Pages to process, as 0-based indexes |
|
||||
| `includeImageBase64` | `Boolean` | Include extracted images in the provider response |
|
||||
| `imageLimit` | `Number` | Maximum number of images to extract |
|
||||
| `imageMinSize` | `Number` | Minimum height and width of an image to extract |
|
||||
| `tableFormat` | `String` | `'markdown'` or `'html'`: extract tables in that format instead of leaving them inline |
|
||||
| `extractHeader` | `Boolean` | Move page headers out of the returned text |
|
||||
| `extractFooter` | `Boolean` | Move page footers out of the returned text |
|
||||
| `documentAnnotationFormat` | `Object` | A JSON schema for a document-level annotation: `{ type: 'json_schema', json_schema: { name, schema } }`. When set, `img2txt()` resolves to the annotation instead of the text. Billed per page on top of OCR. |
|
||||
| `documentAnnotationPrompt` | `String` | Instructions that guide the document annotation |
|
||||
| `bboxAnnotationFormat` | `Object` | A JSON schema, in the same shape, for annotating each extracted image. Billed per page on top of OCR. |
|
||||
|
||||
For more details about each option, see the [AWS Textract documentation](https://docs.aws.amazon.com/textract/latest/dg/what-is.html).
|
||||
|
||||
#### Mistral Options
|
||||
|
||||
Available when `provider: 'mistral'`:
|
||||
|
||||
| Option | Type | Description |
|
||||
|--------|------|-------------|
|
||||
| `model` | `String` | Mistral OCR model to use |
|
||||
| `pages` | `Array<Number>` | Specific pages to process. Starts from 0 |
|
||||
| `includeImageBase64` | `Boolean` | Include image URLs in response |
|
||||
| `imageLimit` | `Number` | Max images to extract |
|
||||
| `imageMinSize` | `Number` | Minimum height and width of image to extract |
|
||||
| `bboxAnnotationFormat` | `String` | Specify the format that the model must output for bounding-box annotations |
|
||||
| `documentAnnotationFormat` | `String` | Specify the format that the model must output for document-level annotations |
|
||||
In both annotation formats, `json_schema` may also be spelled `jsonSchema` and `schema` may be spelled `schemaDefinition`. Any other shape is rejected with `bad_request` before anything is billed.
|
||||
|
||||
For more details about each option, see the [Mistral OCR documentation](https://docs.mistral.ai/api/endpoint/ocr).
|
||||
|
||||
Any properties not set fall back to provider defaults.
|
||||
#### Input limits
|
||||
|
||||
| Model | Formats | Maximum size |
|
||||
|-------|---------|--------------|
|
||||
| AWS Textract | JPEG, PNG, TIFF, and **single-page** PDF | 10 MB |
|
||||
| Mistral | PDF (up to 1,000 pages), images (JPEG, PNG, AVIF, TIFF, GIF, HEIC, BMP, WebP), and documents such as DOCX, PPTX, XLSX, EPUB and RTF | 50 MB |
|
||||
|
||||
The SDK checks the decoded size of data URI inputs against the selected model's limit before upload: 10 MB for Textract and 36 MB when a Mistral model or provider is specified (a data URI is a third larger than the file, and a request body is capped at 50 MB). URLs and Puter paths are limited by the backend. When neither model nor provider is specified, the SDK checks against 10 MB; explicitly select Mistral for larger inline inputs. SDK size failures use `input_too_large`; backend size failures use `storage_limit_reached`. Textract rejects multi-page PDFs and other formats with `bad_request`; use a Mistral model for those.
|
||||
|
||||
## Return value
|
||||
|
||||
A `Promise` that will resolve to a string containing the text contained in the image.
|
||||
A `Promise` that resolves to a string.
|
||||
|
||||
In case of an error, the `Promise` will reject with an error message.
|
||||
- By default the string is the recognized text, one line per line: plain text from AWS Textract, Markdown from Mistral.
|
||||
- When `documentAnnotationFormat` is set, the string is the document annotation: JSON that follows your schema.
|
||||
|
||||
The return shape is the same for every model.
|
||||
|
||||
## Errors
|
||||
|
||||
A rejection carries the error body as the backend sent it: `{ message, code }`.
|
||||
|
||||
| Code | Meaning |
|
||||
| --- | --- |
|
||||
| `arguments_required`, `source_required` | Raised by the SDK before any request is made: the call had no arguments, or no source. |
|
||||
| `input_too_large` | Raised by the SDK before any request is made: a `File`, `Blob` or data URI input exceeds the selected model's limit. |
|
||||
| `storage_limit_reached` | The input is larger than the model accepts (HTTP 413). |
|
||||
| `bad_request` | The provider or model is unknown or retired, the model does not belong to the named provider, an option is invalid, or Textract cannot read the document. |
|
||||
| `insufficient_funds` | Your balance cannot cover the first page. Arrives as HTTP 402. |
|
||||
|
||||
Other `upstream_*` codes mean the provider rejected the request or was unavailable; the `message` carries the provider's reason.
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -82,3 +116,50 @@ In case of an error, the `Promise` will reject with an error message.
|
||||
</body>
|
||||
</html>
|
||||
```
|
||||
|
||||
<strong class="example-title">Read the same image with Mistral OCR</strong>
|
||||
|
||||
```html;ai-img2txt-mistral
|
||||
<html>
|
||||
<body>
|
||||
<script src="https://js.puter.com/v2/"></script>
|
||||
<script>
|
||||
// Only the model name changes; the result is still a string.
|
||||
puter.ai.img2txt('https://assets.puter.site/letter.png', { model: 'mistral-ocr-latest' })
|
||||
.then(puter.print);
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
```
|
||||
|
||||
<strong class="example-title">Extract structured data with a document annotation</strong>
|
||||
|
||||
```html;ai-img2txt-annotation
|
||||
<html>
|
||||
<body>
|
||||
<script src="https://js.puter.com/v2/"></script>
|
||||
<script>
|
||||
(async () => {
|
||||
const annotation = await puter.ai.img2txt('https://assets.puter.site/letter.png', {
|
||||
model: 'mistral-ocr-latest',
|
||||
documentAnnotationFormat: {
|
||||
type: 'json_schema',
|
||||
json_schema: {
|
||||
name: 'letter',
|
||||
schema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
greeting: { type: 'string' },
|
||||
signature: { type: 'string' },
|
||||
},
|
||||
required: ['greeting', 'signature'],
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
puter.print(JSON.parse(annotation).signature);
|
||||
})();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
```
|
||||
@@ -145,6 +145,18 @@ const examples = [
|
||||
slug: 'ai-img2txt',
|
||||
source: '/playground/examples/ai-img2txt.html',
|
||||
},
|
||||
{
|
||||
title: 'Extract Text with Mistral OCR',
|
||||
description: 'Extract text from images with Mistral OCR using Puter.js AI API. Run and modify this OCR example instantly in your browser.',
|
||||
slug: 'ai-img2txt-mistral',
|
||||
source: '/playground/examples/ai-img2txt-mistral.html',
|
||||
},
|
||||
{
|
||||
title: 'Extract Structured Data with OCR',
|
||||
description: 'Fill a JSON schema from a document with Mistral OCR annotations using Puter.js AI API. Run and modify this example in the playground.',
|
||||
slug: 'ai-img2txt-annotation',
|
||||
source: '/playground/examples/ai-img2txt-annotation.html',
|
||||
},
|
||||
{
|
||||
title: 'Text to Image',
|
||||
description: 'Generate images from text with Puter.js AI API. Run and experiment with this text-to-image example in the playground.',
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
<html>
|
||||
<body>
|
||||
<script src="https://js.puter.com/v2/"></script>
|
||||
<script>
|
||||
(async () => {
|
||||
// Loading ...
|
||||
puter.print(`Loading...`);
|
||||
|
||||
// Ask Mistral OCR to fill a JSON schema from the document
|
||||
const annotation = await puter.ai.img2txt('https://assets.puter.site/letter.png', {
|
||||
model: 'mistral-ocr-latest',
|
||||
documentAnnotationFormat: {
|
||||
type: 'json_schema',
|
||||
json_schema: {
|
||||
name: 'letter',
|
||||
schema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
greeting: { type: 'string' },
|
||||
signature: { type: 'string' },
|
||||
},
|
||||
required: ['greeting', 'signature'],
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
puter.print(JSON.parse(annotation).signature);
|
||||
})();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,13 @@
|
||||
<html>
|
||||
<body>
|
||||
<script src="https://js.puter.com/v2/"></script>
|
||||
<script>
|
||||
// Loading ...
|
||||
puter.print(`Loading...`);
|
||||
|
||||
// Only the model name changes; the result is still a string.
|
||||
puter.ai.img2txt('https://assets.puter.site/letter.png', { model: 'mistral-ocr-latest' })
|
||||
.then(puter.print);
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -69,6 +69,17 @@ Sharing a file or folder with **anyone with the link** ([`puter.fs.share()`](/FS
|
||||
|
||||
See [`txt2img()`](/AI/txt2img) for provider-specific options and supported models.
|
||||
|
||||
### OCR
|
||||
|
||||
`puter.ai.img2txt()` input limits apply in addition to the shared AI limits above:
|
||||
|
||||
| Provider | Limit |
|
||||
|----------|-------|
|
||||
| AWS Textract | 10 MB per input. JPEG, PNG, TIFF, or a single-page PDF. |
|
||||
| Mistral | 50 MB per input. PDFs up to 1,000 pages. |
|
||||
|
||||
`File`, `Blob` and data URI inputs are checked by the SDK before upload: 10 MB for Textract, and 36 MB when a Mistral model or provider is named, since the base64 upload must fit the 50 MB request body. URLs and Puter paths are read up to the provider's limit and rejected with `413 storage_limit_reached` beyond it. See [`img2txt()`](/AI/img2txt) for models and options.
|
||||
|
||||
### Key-value store
|
||||
|
||||
| Limit | Paid | Free | Anonymous |
|
||||
|
||||
Vendored
+1
@@ -42,6 +42,7 @@ export type {
|
||||
Img2TxtOptions,
|
||||
ListTTSEnginesOptions,
|
||||
ListTTSVoicesOptions,
|
||||
OcrAnnotationFormat,
|
||||
Speech2SpeechOptions,
|
||||
Speech2TxtOptions,
|
||||
Speech2TxtResult,
|
||||
|
||||
@@ -346,6 +346,24 @@ describe('ai.img2txt driver payloads', () => {
|
||||
expect(text).toBe('line one\nline two\n');
|
||||
});
|
||||
|
||||
it('img2txt resolves to the document annotation when one was requested', async () => {
|
||||
FakeXHR.respondWith = () => ({
|
||||
success: true,
|
||||
result: {
|
||||
blocks: [{ type: 'text/mistral:LINE', text: 'Total: 42' }],
|
||||
document_annotation: '{"total":"42"}',
|
||||
},
|
||||
});
|
||||
const annotation = await ai.img2txt('https://example.com/invoice.pdf', {
|
||||
model: 'mistral-ocr-latest',
|
||||
documentAnnotationFormat: { type: 'json_schema', json_schema: { name: 'invoice', schema: { type: 'object' } } },
|
||||
});
|
||||
expect(annotation).toBe('{"total":"42"}');
|
||||
// Without the option the same result still reads as text.
|
||||
const text = await ai.img2txt('https://example.com/invoice.pdf');
|
||||
expect(text).toBe('Total: 42\n');
|
||||
});
|
||||
|
||||
it('img2txt rejects without a source', async () => {
|
||||
await expect(ai.img2txt({})).rejects.toMatchObject({ code: 'source_required' });
|
||||
});
|
||||
@@ -369,6 +387,33 @@ describe('ai.img2txt driver payloads', () => {
|
||||
const uri = prefix + 'A'.repeat(base64Len);
|
||||
await expect(ai.img2txt(uri)).rejects.toMatchObject({ code: 'input_too_large' });
|
||||
});
|
||||
|
||||
it('img2txt accepts an inline document over 10MB for a Mistral model or provider', async () => {
|
||||
FakeXHR.respondWith = () => ({ success: true, result: { text: 'recognized' } });
|
||||
const uri = `data:application/pdf;base64,${'A'.repeat(14 * 1024 * 1024)}`;
|
||||
for (const options of [{ model: 'mistral-ocr-latest' }, { provider: 'mistral' }]) {
|
||||
await expect(ai.img2txt(uri, options)).resolves.toBe('recognized');
|
||||
}
|
||||
});
|
||||
|
||||
it('img2txt rejects a Mistral inline source that would not fit the 50MB request body', async () => {
|
||||
let base64Len = Math.ceil(((36 * 1024 * 1024) + 1) * 4 / 3);
|
||||
base64Len += (4 - (base64Len % 4)) % 4;
|
||||
const uri = `data:application/pdf;base64,${'A'.repeat(base64Len)}`;
|
||||
await expect(ai.img2txt(uri, { model: 'mistral-ocr-latest' }))
|
||||
.rejects.toMatchObject({ code: 'input_too_large' });
|
||||
});
|
||||
|
||||
it('img2txt keeps its string result when normalization is requested', async () => {
|
||||
FakeXHR.respondWith = () => ({ success: true, result: { text: 'recognized' } });
|
||||
ai.normalize = true;
|
||||
try {
|
||||
await expect(ai.img2txt('https://example.com/scan.png', { normalize: false }))
|
||||
.resolves.toBe('recognized');
|
||||
} finally {
|
||||
ai.normalize = undefined;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('ai.txt2speech driver payloads', () => {
|
||||
|
||||
@@ -13,18 +13,25 @@ import { dataUriByteLength, isBlobLike, isPlainObject } from './lib/args.js';
|
||||
* }} OcrResult
|
||||
*/
|
||||
|
||||
const MAX_INPUT_SIZE = 10 * 1024 * 1024;
|
||||
const DEFAULT_MAX_INPUT_SIZE = 10 * 1024 * 1024;
|
||||
// Inline sources travel as base64 in a JSON body capped at 50 MB.
|
||||
const MISTRAL_MAX_INPUT_SIZE = 36 * 1024 * 1024;
|
||||
|
||||
// The unified OCR driver picks the provider from `options.provider`.
|
||||
const OCR_DRIVER = 'ai-ocr';
|
||||
|
||||
/**
|
||||
* Reduce the provider-specific recognition result to plain text.
|
||||
* Reduce the recognition result to a string: the requested document
|
||||
* annotation when there is one, the recognized text otherwise.
|
||||
* @param {OcrResult | null | undefined} result
|
||||
* @param {boolean} [wantsAnnotation]
|
||||
* @returns {string}
|
||||
*/
|
||||
const toText = (result) => {
|
||||
const toText = (result, wantsAnnotation = false) => {
|
||||
if ( ! result ) return '';
|
||||
if ( wantsAnnotation && typeof result.document_annotation === 'string' ) {
|
||||
return result.document_annotation;
|
||||
}
|
||||
if ( Array.isArray(result.blocks) && result.blocks.length ) {
|
||||
let str = '';
|
||||
for ( const block of result.blocks ) {
|
||||
@@ -122,10 +129,17 @@ export async function img2txt (sourceOrOptions, optionsOrTestMode, testModeOrOpt
|
||||
options.source = await utils.blobToDataUri(options.source.source);
|
||||
}
|
||||
|
||||
const requestedModel = typeof options.model === 'string' ? options.model.trim().toLowerCase() : '';
|
||||
const requestedProvider = typeof options.provider === 'string' ? options.provider.trim().toLowerCase() : '';
|
||||
const maxInputSize = requestedModel.startsWith('mistral-ocr-') ||
|
||||
(!requestedModel && ['mistral', 'mistral-ocr'].includes(requestedProvider))
|
||||
? MISTRAL_MAX_INPUT_SIZE
|
||||
: DEFAULT_MAX_INPUT_SIZE;
|
||||
|
||||
if ( typeof options.source === 'string' &&
|
||||
options.source.startsWith('data:') &&
|
||||
dataUriByteLength(options.source) > MAX_INPUT_SIZE ) {
|
||||
throw { message: `Input size cannot be larger than ${ MAX_INPUT_SIZE}`, code: 'input_too_large' };
|
||||
dataUriByteLength(options.source) > maxInputSize ) {
|
||||
throw { message: `Input size cannot be larger than ${ maxInputSize}`, code: 'input_too_large' };
|
||||
}
|
||||
|
||||
return await utils.makeDriverMethod({
|
||||
@@ -135,6 +149,6 @@ export async function img2txt (sourceOrOptions, optionsOrTestMode, testModeOrOpt
|
||||
argNames: ['source'],
|
||||
puter,
|
||||
testMode: testMode ?? false,
|
||||
transform: async (result) => toText(result),
|
||||
transform: async (result) => toText(result, options.documentAnnotationFormat !== undefined),
|
||||
})(options);
|
||||
}
|
||||
@@ -143,21 +143,39 @@
|
||||
* @property {string} [message] Error description. Present on `"error"` chunks, which end the stream.
|
||||
*/
|
||||
|
||||
/**
|
||||
* A Mistral OCR annotation format: the JSON schema the annotation must follow.
|
||||
* `jsonSchema` / `schemaDefinition` are accepted as aliases of `json_schema` / `schema`.
|
||||
*
|
||||
* @typedef {Object} OcrAnnotationFormat
|
||||
* @property {'json_schema'} [type]
|
||||
* @property {{ name?: string, description?: string, schema: Record<string, unknown>, strict?: boolean }} [json_schema]
|
||||
* @property {{ name?: string, description?: string, schemaDefinition: Record<string, unknown>, strict?: boolean }} [jsonSchema]
|
||||
*/
|
||||
|
||||
/**
|
||||
* Options for `img2txt()` (OCR).
|
||||
*
|
||||
* @typedef {Object} Img2TxtOptions
|
||||
* @property {string | File | Blob} [source]
|
||||
* @property {string} [provider]
|
||||
* @property {string | File | Blob} [source] Image or document: URL, Puter path, data URI, `File` or `Blob`.
|
||||
* @property {string} [model] OCR model: `'aws-textract'` (alias `'textract'`), `'mistral-ocr-latest'`
|
||||
* (OCR 4.1; aliases `'mistral-ocr-4'`, `'mistral-ocr-4-1'`), `'mistral-ocr-4-0'`, or `'mistral-ocr-2512'`
|
||||
* (OCR 3; aliases `'mistral-ocr-3'`, `'mistral-ocr-3-0'`). The model picks its provider.
|
||||
* @property {string} [provider] `'aws-textract'` (default) or `'mistral'`; aliases `'aws'`, `'textract'`,
|
||||
* `'mistral-ocr'`. Without a `model`, the provider's default model runs.
|
||||
* @property {boolean} [testMode]
|
||||
* @property {boolean} [test_mode] `snake_case` spelling of `testMode`, forwarded to the driver as-is.
|
||||
* @property {string} [model]
|
||||
* @property {number[]} [pages]
|
||||
* @property {boolean} [includeImageBase64]
|
||||
* @property {number} [imageLimit]
|
||||
* @property {number} [imageMinSize]
|
||||
* @property {string} [bboxAnnotationFormat]
|
||||
* @property {string} [documentAnnotationFormat]
|
||||
* @property {number[]} [pages] Mistral: 0-based page indexes to process.
|
||||
* @property {boolean} [includeImageBase64] Mistral: include extracted images in the provider response.
|
||||
* @property {number} [imageLimit] Mistral: maximum number of images to extract.
|
||||
* @property {number} [imageMinSize] Mistral: minimum height and width of an image to extract.
|
||||
* @property {OcrAnnotationFormat} [bboxAnnotationFormat] Mistral: schema for per-image annotations.
|
||||
* @property {OcrAnnotationFormat} [documentAnnotationFormat] Mistral: schema for a document-level
|
||||
* annotation. When set, `img2txt()` resolves to the annotation (a JSON string) instead of the text.
|
||||
* @property {string} [documentAnnotationPrompt] Mistral: instructions for the document annotation.
|
||||
* @property {'markdown' | 'html'} [tableFormat] Mistral: extract tables in this format.
|
||||
* @property {boolean} [extractHeader] Mistral: move page headers out of the text.
|
||||
* @property {boolean} [extractFooter] Mistral: move page footers out of the text.
|
||||
*/
|
||||
|
||||
/**
|
||||
|
||||
@@ -668,6 +668,49 @@ export default suite('ai', {
|
||||
}
|
||||
},
|
||||
|
||||
'img2txt routes by model name alone': async (t) => {
|
||||
useApiToken(t);
|
||||
// Keyless: each model resolves to its provider, then fails on that
|
||||
// provider's missing credentials — never as an unknown model.
|
||||
const models = [
|
||||
'aws-textract',
|
||||
'textract',
|
||||
'mistral-ocr-latest',
|
||||
'mistral-ocr-4-0',
|
||||
'mistral-ocr-2512',
|
||||
];
|
||||
for (const model of models) {
|
||||
const text = await rejectionText(() =>
|
||||
t.puter.ai.img2txt(TINY_PNG, { model }),
|
||||
);
|
||||
t.assert.ok(
|
||||
!text.includes('Unknown OCR model'),
|
||||
`model "${model}" should resolve to a provider, got ${text}`,
|
||||
);
|
||||
}
|
||||
},
|
||||
|
||||
'img2txt rejects retired, unknown and mismatched OCR models': async (t) => {
|
||||
useApiToken(t);
|
||||
const cases: Array<[Record<string, string>, string]> = [
|
||||
[{ model: 'mistral-ocr-2505' }, 'no longer available'],
|
||||
[{ model: 'ai-suite-no-such-ocr-model' }, 'Unknown OCR model'],
|
||||
[
|
||||
{ model: 'mistral-ocr-latest', provider: 'aws-textract' },
|
||||
'not served by provider',
|
||||
],
|
||||
];
|
||||
for (const [options, expected] of cases) {
|
||||
const text = await rejectionText(() =>
|
||||
t.puter.ai.img2txt(TINY_PNG, options),
|
||||
);
|
||||
t.assert.ok(
|
||||
text.includes(expected),
|
||||
`${JSON.stringify(options)} should reject with "${expected}", got ${text}`,
|
||||
);
|
||||
}
|
||||
},
|
||||
|
||||
'the legacy per-provider OCR driver names still route': async (t) => {
|
||||
useApiToken(t);
|
||||
for (const driver of ['aws-textract', 'mistral']) {
|
||||
@@ -1132,6 +1175,15 @@ export default suite('ai', {
|
||||
);
|
||||
},
|
||||
|
||||
'img2txt accepts a Mistral inline source above the Textract limit': async (t) => {
|
||||
useApiToken(t);
|
||||
const text = await t.puter.ai.img2txt(
|
||||
oversizedDataUri('image/png', 10 * 1024 * 1024, 2),
|
||||
{ model: 'mistral-ocr-latest', testMode: true },
|
||||
);
|
||||
t.assert.ok(text.includes('sample OCR response'));
|
||||
},
|
||||
|
||||
// -- speech2txt --------------------------------------------------
|
||||
|
||||
'speech2txt rejects a call with no arguments': async (t) => {
|
||||
|
||||
Reference in new issue
Block a user