azure ai provider (#3232)

This commit is contained in:
ProgrammerIn-wonderland
2026-06-08 18:31:49 -04:00
committed by GitHub
parent e782fa832e
commit 708fe1bc4e
11 changed files with 1328 additions and 6 deletions
+4
View File
@@ -260,6 +260,10 @@
// ─ Chat / completion ─
"claude": { "apiKey": "" },
"openai-completion": { "apiKey": "" },
"azure-openai": {
"apiKey": "",
"apiURL": ""
},
"gemini": { "apiKey": "" },
"groq": { "apiKey": "" },
"deepseek": { "apiKey": "" },
+1 -1
View File
@@ -17696,7 +17696,7 @@
},
"src/cli": {
"name": "@heyputer/cli",
"version": "0.1.1",
"version": "0.1.2",
"license": "MIT",
"dependencies": {
"@clack/prompts": "^0.7.0",
@@ -17,8 +17,9 @@
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*/
import { PassThrough } from 'node:stream';
import crypto from 'node:crypto';
import { PassThrough } from 'node:stream';
import { EventMap } from '../../clients/event/types.js';
import { Context } from '../../core/context.js';
import { HttpError } from '../../core/http/HttpError.js';
import {
@@ -28,12 +29,17 @@ import {
import type { DriverStreamResult } from '../meta.js';
import { PuterDriver } from '../types.js';
import { AI_CONCURRENT, AI_RATE_LIMIT } from '../util/aiLimits.js';
import { AlibabaProvider } from './providers/alibaba/AlibabaProvider.js';
import { AzureChatProvider } from './providers/azure/AzureChatProvider.js';
import { AzureResponsesProvider } from './providers/azure/AzureResponsesProvider.js';
import { ClaudeProvider } from './providers/claude/ClaudeProvider.js';
import { DeepSeekProvider } from './providers/deepseek/DeepSeekProvider.js';
import { FakeChatProvider } from './providers/FakeChatProvider.js';
import { GeminiChatProvider } from './providers/gemini/GeminiChatProvider.js';
import { GroqAIProvider } from './providers/groq/GroqAIProvider.js';
import { MiniMaxProvider } from './providers/minimax/MiniMaxProvider.js';
import { MistralAIProvider } from './providers/mistral/MistralAiProvider.js';
import { MoonshotProvider } from './providers/moonshot/MoonshotProvider.js';
import { OllamaChatProvider } from './providers/ollama/OllamaProvider.js';
import { OpenAiChatProvider } from './providers/openai/OpenAiChatCompletionsProvider.js';
import { OpenAiResponsesChatProvider } from './providers/openai/OpenAiChatResponsesProvider.js';
@@ -41,9 +47,6 @@ import { OpenRouterProvider } from './providers/openrouter/OpenRouterProvider.js
import { TogetherAIProvider } from './providers/together/TogetherAIProvider.js';
import { XAIProvider } from './providers/xai/XAIProvider.js';
import { ZAIProvider } from './providers/zai/ZAIProvider.js';
import { AlibabaProvider } from './providers/alibaba/AlibabaProvider.js';
import { MoonshotProvider } from './providers/moonshot/MoonshotProvider.js';
import { MiniMaxProvider } from './providers/minimax/MiniMaxProvider.js';
import type {
IChatCompleteResult,
IChatModel,
@@ -57,7 +60,6 @@ import {
normalize_single_message,
} from './utils/Messages.js';
import { AIChatStream } from './utils/Streaming.js';
import { EventMap } from '../../clients/event/types.js';
const MAX_FALLBACKS = 4; // includes first attempt
@@ -796,6 +798,43 @@ export class ChatCompletionDriver extends PuterDriver {
);
}
// Azure AI Foundry (OpenAI + xAI Grok). Registered before the regular
// OpenAI/xAI providers so that since its costs mirror theirs but
// Azure is preferred for us, it takes precedence in the per-model
// bucket
const azureOpenai = providers['azure-openai'];
const azureOpenaiKey = readKey(azureOpenai);
const azureOpenaiURL = azureOpenai?.apiURL as string | undefined;
if (azureOpenaiKey && azureOpenaiURL) {
const azureStores = {
fsEntry: this.stores.fsEntry,
s3Object: this.stores.s3Object,
};
const azureConfig = {
apiKey: azureOpenaiKey,
apiURL: azureOpenaiURL,
};
const azureCompletions = new AzureChatProvider(
metering,
azureStores,
this.services.fs,
azureConfig,
);
// Codex / Responses-API-only models can't use Chat Completions, so
// they route through a sibling Responses provider pointed at the
// same Azure endpoint. web_search (also Responses-only) delegates
// here too.
const azureResponses = new AzureResponsesProvider(
metering,
azureStores,
this.services.fs,
azureConfig,
);
azureCompletions.setResponsesProvider(azureResponses);
this.#providers['azure-openai'] = azureCompletions;
this.#providers['azure-openai-responses'] = azureResponses;
}
const openaiKey = readKey(providers['openai-completion']);
if (openaiKey) {
const openaiStores = {
@@ -0,0 +1,101 @@
/**
* Copyright (C) 2024-present Puter Technologies Inc.
*
* This file is part of Puter.
*
* Puter is free software: you can redistribute it and/or modify
* it under the terms of the GNU Affero General Public License as published
* by the Free Software Foundation, either version 3 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU Affero General Public License for more details.
*
* You should have received a copy of the GNU Affero General Public License
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*/
/**
* Integration test for the Azure AI Foundry chat-completions provider.
*
* Hits the real Azure endpoint. Exercises both flavours of model the
* provider fronts:
* - an OpenAI model (`gpt-4o`, non-reasoning so `max_tokens=16` returns
* visible text), and
* - an xAI Grok model (`grok-4-20-non-reasoning`), which regression-tests
* the `safety_identifier` param being stripped for Grok (Azure's Grok
* deployments 400 on that OpenAI-only argument).
*
* Skipped unless both `PUTER_TEST_AI_AZURE_OPENAI_API_KEY` and
* `PUTER_TEST_AI_AZURE_OPENAI_API_URL` are set.
*/
import { describe, expect, it } from 'vitest';
import {
INTEGRATION_TEST_TIMEOUT_MS,
makeMeteringStub,
optionalEnv,
skipUnlessEnv,
withTestActor,
} from '../../../integrationTestUtil.js';
import { AzureChatProvider } from './AzureChatProvider.js';
const KEY_ENV = 'PUTER_TEST_AI_AZURE_OPENAI_API_KEY';
const URL_ENV = 'PUTER_TEST_AI_AZURE_OPENAI_API_URL';
describe.skipIf(skipUnlessEnv(KEY_ENV) || skipUnlessEnv(URL_ENV))(
'AzureChatProvider (integration)',
() => {
const buildProvider = () =>
new AzureChatProvider(
makeMeteringStub(),
{ fsEntry: undefined as never, s3Object: undefined as never },
undefined as never,
{ apiKey: optionalEnv(KEY_ENV)!, apiURL: optionalEnv(URL_ENV)! },
);
const expectNonEmptyText = (result: unknown) => {
const text = (result as { message?: { content?: string } }).message
?.content;
expect(typeof text === 'string' && text.length > 0).toBe(true);
};
it(
'returns a non-empty completion from gpt-4o',
{ timeout: INTEGRATION_TEST_TIMEOUT_MS },
async () => {
const provider = buildProvider();
const result = await withTestActor(() =>
provider.complete({
model: 'gpt-4o',
messages: [
{ role: 'user', content: 'Say hi in one word.' },
],
max_tokens: 16,
}),
);
expectNonEmptyText(result);
},
);
it(
'returns a non-empty completion from grok-4-20-non-reasoning (no safety_identifier 400)',
{ timeout: INTEGRATION_TEST_TIMEOUT_MS },
async () => {
const provider = buildProvider();
const result = await withTestActor(() =>
provider.complete({
model: 'grok-4-20-non-reasoning',
messages: [
{ role: 'user', content: 'Say hi in one word.' },
],
max_tokens: 16,
}),
);
expectNonEmptyText(result);
},
);
},
);
@@ -0,0 +1,251 @@
/*
* Copyright (C) 2024-present Puter Technologies Inc.
*
* This file is part of Puter.
*
* Puter is free software: you can redistribute it and/or modify
* it under the terms of the GNU Affero General Public License as published
* by the Free Software Foundation, either version 3 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU Affero General Public License for more details.
*
* You should have received a copy of the GNU Affero General Public License
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*/
import { HttpError } from '@heyputer/backend/src/core/http/HttpError.js';
import { OpenAI } from 'openai';
import { ChatCompletionCreateParams } from 'openai/resources/index.js';
import { Context } from '../../../../core/context.js';
import type { FSService } from '../../../../services/fs/FSService.js';
import type { MeteringService } from '../../../../services/metering/MeteringService.js';
import type { FSEntryStore } from '../../../../stores/fs/FSEntryStore.js';
import type { S3ObjectStore } from '../../../../stores/fs/S3ObjectStore.js';
import type { IChatProvider, ICompleteArguments } from '../../types.js';
import * as OpenAiUtil from '../../utils/OpenAIUtil.js';
import { processPuterPathUploads } from '../openai/fileUpload.js';
import { AZURE_MODELS } from './models.js';
/**
* AzureChatProvider exposes the models we serve through Azure AI Foundry.
* Despite the name, this is not OpenAI-only — Azure AI also fronts xAI's Grok
* models — so it carries its own {@link AZURE_MODELS} list instead of reusing
* the OpenAI one. It speaks the OpenAI-compatible Chat Completions API,
* pointing the client at a configurable Azure endpoint authenticated with an
* Azure-issued API key.
*
* Billing note: the model `costs` are the standard public OpenAI / xAI list
* prices, NOT Azure's. Azure is subsidised for us, so routing through it is
* cheaper while we still bill users at the normal model price.
*
* Implements the puter-chat-completion interface and handles usage tracking,
* spending records, and content moderation.
*/
export class AzureChatProvider implements IChatProvider {
/**
* @type {import('openai').OpenAI}
*/
#openAi: OpenAI;
#defaultModel = 'gpt-5-nano';
#meteringService: MeteringService;
#stores: { fsEntry: FSEntryStore; s3Object: S3ObjectStore };
#fsService: FSService;
// Sibling Responses-API provider (Azure or OpenAI) used to handle
// Responses-only features like web_search. Typed loosely since we only
// ever forward `complete()` to it.
#responsesProvider: IChatProvider | null = null;
constructor(
meteringService: MeteringService,
stores: { fsEntry: FSEntryStore; s3Object: S3ObjectStore },
fsService: FSService,
config: { apiKey: string; apiURL: string },
) {
this.#meteringService = meteringService;
this.#stores = stores;
this.#fsService = fsService;
this.#openAi = new OpenAI({
apiKey: config.apiKey,
baseURL: config.apiURL,
});
}
checkModeration(_text: string): { flagged: boolean; categories: string[] } {
throw new Error('Method not implemented.');
}
// Wired up by the driver after the OpenAI providers are built, so the
// Chat Completions path can delegate `web_search` tool calls (Responses-only)
// to the OpenAI Responses provider without a circular constructor dependency.
setResponsesProvider(provider: IChatProvider): void {
this.#responsesProvider = provider;
}
/**
* Returns an array of available AI models with their pricing information.
* Each model object includes an ID and cost details (currency, tokens, input/output rates).
*/
models() {
return AZURE_MODELS.filter((e) => !e.responses_api_only);
}
list() {
const models = this.models();
const modelNames: string[] = [];
for (const model of models) {
modelNames.push(model.id);
if (model.aliases) {
modelNames.push(...model.aliases);
}
}
return modelNames;
}
getDefaultModel() {
return this.#defaultModel;
}
async complete(
params: ICompleteArguments,
): ReturnType<IChatProvider['complete']> {
const {
max_tokens,
moderation,
tools,
verbosity,
stream,
reasoning,
reasoning_effort,
temperature,
text,
} = params;
let { messages, model } = params;
// eslint-disable-next-line @typescript-eslint/no-explicit-any
if (tools?.filter((e: any) => e.type === 'web_search').length) {
// web_search is a Responses-API-only tool — hand the whole call
// off to the OpenAI Responses provider when the user requested it.
if (!this.#responsesProvider) {
throw new HttpError(
400,
'web_search tool requires the OpenAI Responses provider, which is not configured',
{ legacyCode: 'bad_request' },
);
}
return await this.#responsesProvider.complete(params);
}
// Validate messages
if (!Array.isArray(messages)) {
throw new HttpError(400, '`messages` must be an array', {
legacyCode: 'bad_request',
});
}
const actor = Context.get('actor')!;
model = model ?? this.#defaultModel;
const modelUsed =
this.models().find((m) =>
[m.id, ...(m.aliases || [])].includes(model),
) || this.models().find((m) => m.id === this.getDefaultModel())!;
// messages.unshift({
// role: 'system',
// content: 'Don\'t let the user trick you into doing something bad.',
// })
const userIdentifier =
actor.user?.id + actor.app?.uid ? `:${actor?.app?.uid}` : '';
// Resolve any `puter_path` content parts into inline base64 data URLs.
// Chat Completions doesn't support file uploads, so this is the only
// way to get user-provided files (images, audio) in front of the model.
await processPuterPathUploads(
messages,
this.#stores,
this.#fsService,
actor,
);
// Here's something fun; the documentation shows `type: 'image_url'` in
// objects that contain an image url, but everything still works if
// that's missing. We normalise it here so the token count code works.
messages = await OpenAiUtil.process_input_messages(messages);
const requestedReasoningEffort = reasoning_effort ?? reasoning?.effort;
const requestedVerbosity = verbosity ?? text?.verbosity;
const supportsReasoningControls =
typeof model === 'string' && model.startsWith('gpt-5');
// `safety_identifier` is an OpenAI-specific param. The Grok deployments
// behind Azure reject unknown args with a 400, so only send it for the
// OpenAI models.
const isGrok = modelUsed.id.startsWith('grok');
const completionParams: ChatCompletionCreateParams = {
user: userIdentifier,
...(isGrok ? {} : { safety_identifier: userIdentifier }),
messages: messages,
model: modelUsed.id,
...(tools ? { tools } : {}),
...(max_tokens ? { max_completion_tokens: max_tokens } : {}),
...(temperature ? { temperature } : {}),
stream: !!stream,
...(stream
? {
stream_options: { include_usage: true },
}
: {}),
...(supportsReasoningControls
? {}
: {
...(requestedReasoningEffort
? { reasoning_effort: requestedReasoningEffort }
: {}),
...(requestedVerbosity
? { verbosity: requestedVerbosity }
: {}),
}),
} as ChatCompletionCreateParams;
const completion =
await this.#openAi.chat.completions.create(completionParams);
return OpenAiUtil.handle_completion_output({
usage_calculator: ({ usage }) => {
const trackedUsage = {
prompt_tokens:
(usage.prompt_tokens ?? 0) -
(usage.prompt_tokens_details?.cached_tokens ?? 0),
completion_tokens: usage.completion_tokens ?? 0,
cached_tokens:
usage.prompt_tokens_details?.cached_tokens ?? 0,
};
const costsOverrideFromModel = Object.fromEntries(
Object.entries(trackedUsage).map(([k, v]) => {
return [k, v * modelUsed.costs[k]];
}),
);
this.#meteringService.utilRecordUsageObject(
trackedUsage,
actor!,
`azure-openai:${modelUsed?.id}`,
costsOverrideFromModel,
);
return trackedUsage;
},
stream,
completion,
moderate: moderation ? this.checkModeration.bind(this) : undefined,
});
}
}
@@ -0,0 +1,83 @@
/**
* Copyright (C) 2024-present Puter Technologies Inc.
*
* This file is part of Puter.
*
* Puter is free software: you can redistribute it and/or modify
* it under the terms of the GNU Affero General Public License as published
* by the Free Software Foundation, either version 3 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU Affero General Public License for more details.
*
* You should have received a copy of the GNU Affero General Public License
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*/
/**
* Integration test for the Azure AI Foundry Responses provider.
*
* Hits the real Azure endpoint with `gpt-5-codex` — a Responses-API-only
* (Codex) model that the Chat Completions endpoint rejects. This verifies
* the completions/responses split actually routes Codex correctly. Codex is
* a reasoning model, so we give it a generous `max_tokens` and low reasoning
* effort to leave room for visible output.
*
* Skipped unless both `PUTER_TEST_AI_AZURE_OPENAI_API_KEY` and
* `PUTER_TEST_AI_AZURE_OPENAI_API_URL` are set.
*/
import { describe, expect, it } from 'vitest';
import {
INTEGRATION_TEST_TIMEOUT_MS,
makeMeteringStub,
optionalEnv,
skipUnlessEnv,
withTestActor,
} from '../../../integrationTestUtil.js';
import { AzureResponsesProvider } from './AzureResponsesProvider.js';
const KEY_ENV = 'PUTER_TEST_AI_AZURE_OPENAI_API_KEY';
const URL_ENV = 'PUTER_TEST_AI_AZURE_OPENAI_API_URL';
describe.skipIf(skipUnlessEnv(KEY_ENV) || skipUnlessEnv(URL_ENV))(
'AzureResponsesProvider (integration)',
() => {
it(
'returns a non-empty completion from gpt-5-codex',
{ timeout: INTEGRATION_TEST_TIMEOUT_MS },
async () => {
const provider = new AzureResponsesProvider(
makeMeteringStub(),
{
fsEntry: undefined as never,
s3Object: undefined as never,
},
undefined as never,
{
apiKey: optionalEnv(KEY_ENV)!,
apiURL: optionalEnv(URL_ENV)!,
},
);
const result = await withTestActor(() =>
provider.complete({
model: 'gpt-5-codex',
messages: [
{ role: 'user', content: 'Say hi in one word.' },
],
max_tokens: 2048,
reasoning: { effort: 'low' },
}),
);
const text = (result as { message?: { content?: string } })
.message?.content;
expect(typeof text === 'string' && text.length > 0).toBe(true);
},
);
},
);
@@ -0,0 +1,294 @@
/*
* Copyright (C) 2024-present Puter Technologies Inc.
*
* This file is part of Puter.
*
* Puter is free software: you can redistribute it and/or modify
* it under the terms of the GNU Affero General Public License as published
* by the Free Software Foundation, either version 3 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU Affero General Public License for more details.
*
* You should have received a copy of the GNU Affero General Public License
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*/
import { OpenAI } from 'openai';
import { ResponseCreateParams } from 'openai/resources/responses/responses.mjs';
import { Context } from '../../../../core/context.js';
import type { FSService } from '../../../../services/fs/FSService.js';
import type { MeteringService } from '../../../../services/metering/MeteringService.js';
import type { FSEntryStore } from '../../../../stores/fs/FSEntryStore.js';
import type { S3ObjectStore } from '../../../../stores/fs/S3ObjectStore.js';
import type { IChatProvider, ICompleteArguments } from '../../types.js';
import * as OpenAiUtil from '../../utils/OpenAIUtil.js';
import { processPuterPathUploads } from '../openai/fileUpload.js';
import { AZURE_MODELS } from './models.js';
import { HttpError } from '@heyputer/backend/src/core/http/HttpError.js';
/**
* AzureResponsesProvider serves the Responses-API-only models we expose through
* Azure AI Foundry (the Codex family and similar). It mirrors
* {@link OpenAiResponsesChatProvider}, but points the OpenAI client at the
* configurable Azure endpoint and draws from {@link AZURE_MODELS}.
*
* Codex / `responses_api_only` models reject the Chat Completions endpoint, so
* the sibling {@link AzureChatProvider} (Chat Completions) filters them out and
* the driver routes them here instead.
*
* Billing note: the model `costs` are the standard public OpenAI list prices,
* NOT Azure's — Azure is subsidised for us.
*/
export class AzureResponsesProvider implements IChatProvider {
/**
* @type {import('openai').OpenAI}
*/
#openAi: OpenAI;
#defaultModel = 'gpt-5-codex';
#meteringService: MeteringService;
#stores: { fsEntry: FSEntryStore; s3Object: S3ObjectStore };
#fsService: FSService;
constructor(
meteringService: MeteringService,
stores: { fsEntry: FSEntryStore; s3Object: S3ObjectStore },
fsService: FSService,
config: { apiKey: string; apiURL: string },
) {
this.#meteringService = meteringService;
this.#stores = stores;
this.#fsService = fsService;
this.#openAi = new OpenAI({
apiKey: config.apiKey,
baseURL: config.apiURL,
});
}
/**
* Returns an array of available AI models with their pricing information.
* Each model object includes an ID and cost details (currency, tokens, input/output rates).
*/
models(extra_params?: { no_restrictions?: boolean }) {
if (extra_params?.no_restrictions) {
return AZURE_MODELS;
}
return AZURE_MODELS.filter((e) => e.responses_api_only === true);
}
list() {
const models = this.models({ no_restrictions: false });
const modelNames: string[] = [];
for (const model of models) {
modelNames.push(model.id);
if (model.aliases) {
modelNames.push(...model.aliases);
}
}
return modelNames;
}
getDefaultModel() {
return this.#defaultModel;
}
async complete({
messages,
model,
max_tokens,
moderation,
tools,
tool_choice,
parallel_tool_calls,
include,
conversation,
previous_response_id,
instructions,
metadata,
prompt,
prompt_cache_key,
prompt_cache_retention,
store,
top_p,
truncation,
background,
service_tier,
verbosity,
stream,
reasoning,
reasoning_effort,
temperature,
text,
}: ICompleteArguments): ReturnType<IChatProvider['complete']> {
// Validate messages
if (!Array.isArray(messages)) {
throw new HttpError(400, '`messages` must be an array', {
legacyCode: 'bad_request',
});
}
const actor = Context.get('actor');
model = model ?? this.#defaultModel;
const modelUsed =
this.models({ no_restrictions: true }).find((m) =>
[m.id, ...(m.aliases || [])].includes(model),
) ||
this.models({ no_restrictions: true }).find(
(m) => m.id === this.getDefaultModel(),
)!;
const userIdentifier =
actor?.user.id + actor?.app?.uid ? `:${actor?.app?.uid}` : '';
// Resolve any `puter_path` content parts into inline base64 data URLs
// before the Responses API sees them.
await processPuterPathUploads(
messages,
this.#stores,
this.#fsService,
actor,
);
if (tools) {
// Unravel tools to OpenAI Responses API format
// eslint-disable-next-line @typescript-eslint/no-explicit-any
tools = (tools as any).map((e) => {
if (e.type === 'function') {
const tool = e.function;
tool.type = 'function';
return tool;
} else {
return e;
}
});
}
// Here's something fun; the documentation shows `type: 'image_url'` in
// objects that contain an image url, but everything still works if
// that's missing. We normalise it here so the token count code works.
messages =
await OpenAiUtil.process_input_messages_responses_api(messages);
const requestedReasoningEffort = reasoning_effort ?? reasoning?.effort;
const requestedVerbosity = verbosity ?? text?.verbosity;
const supportsReasoningControls =
typeof model === 'string' && model.startsWith('gpt-5');
const completionParams: ResponseCreateParams = {
user: userIdentifier,
safety_identifier: userIdentifier,
input: messages,
model: modelUsed.id,
...(tools ? { tools } : {}),
...(tool_choice !== undefined ? { tool_choice } : {}),
...(parallel_tool_calls !== undefined
? { parallel_tool_calls }
: {}),
...(include !== undefined ? { include } : {}),
...(conversation !== undefined ? { conversation } : {}),
...(previous_response_id !== undefined
? { previous_response_id }
: {}),
...(instructions !== undefined ? { instructions } : {}),
...(metadata !== undefined ? { metadata } : {}),
...(prompt !== undefined ? { prompt } : {}),
...(prompt_cache_key !== undefined ? { prompt_cache_key } : {}),
...(prompt_cache_retention !== undefined
? { prompt_cache_retention }
: {}),
...(store !== undefined ? { store } : {}),
...(max_tokens !== undefined
? { max_output_tokens: max_tokens }
: {}),
...(temperature !== undefined ? { temperature } : {}),
...(top_p !== undefined ? { top_p } : {}),
...(truncation !== undefined ? { truncation } : {}),
...(background !== undefined ? { background } : {}),
...(service_tier !== undefined ? { service_tier } : {}),
...(stream !== undefined ? { stream: !!stream } : {}),
...(text !== undefined ? { text } : {}),
...(supportsReasoningControls
? {}
: {
...(requestedReasoningEffort
? { reasoning_effort: requestedReasoningEffort }
: {}),
...(requestedVerbosity
? { verbosity: requestedVerbosity }
: {}),
}),
...(supportsReasoningControls && reasoning ? { reasoning } : {}),
} as ResponseCreateParams;
const completion =
await this.#openAi.responses.create(completionParams);
return OpenAiUtil.handle_completion_output_responses_api({
usage_calculator: ({ usage }) => {
const trackedUsage = {
prompt_tokens:
// eslint-disable-next-line @typescript-eslint/no-explicit-any
((usage as any).input_tokens ?? 0) -
// eslint-disable-next-line @typescript-eslint/no-explicit-any
((usage as any).input_tokens_details?.cached_tokens ??
0),
// eslint-disable-next-line @typescript-eslint/no-explicit-any
completion_tokens: (usage as any).output_tokens ?? 0,
cached_tokens:
// eslint-disable-next-line @typescript-eslint/no-explicit-any
(usage as any).input_tokens_details?.cached_tokens ?? 0,
};
const costsOverrideFromModel = Object.fromEntries(
Object.entries(trackedUsage).map(([k, v]) => {
return [k, v * modelUsed.costs[k]];
}),
);
this.#meteringService.utilRecordUsageObject(
trackedUsage,
actor,
`azure-openai:${modelUsed?.id}`,
costsOverrideFromModel,
);
return trackedUsage;
},
stream,
completion,
moderate: moderation ? this.checkModeration.bind(this) : undefined,
});
}
async checkModeration(text: string) {
// create moderation
const results = await this.#openAi.moderations.create({
model: 'omni-moderation-latest',
input: text,
});
let flagged = false;
for (const result of results?.results ?? []) {
// OpenAI does a crazy amount of false positives. We filter by their 80% interval
const veryFlaggedEntries = Object.entries(
result.category_scores,
).filter((e) => e[1] > 0.8);
if (veryFlaggedEntries.length > 0) {
flagged = true;
break;
}
}
return {
flagged,
results,
};
}
}
@@ -0,0 +1,464 @@
/**
* Copyright (C) 2024-present Puter Technologies Inc.
*
* This file is part of Puter.
*
* Puter is free software: you can redistribute it and/or modify
* it under the terms of the GNU Affero General Public License as published
* by the Free Software Foundation, either version 3 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU Affero General Public License for more details.
*
* You should have received a copy of the GNU Affero General Public License
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*/
import type { IChatModel } from '../../types.js';
// Models served through our Azure AI Foundry deployment. This is NOT just
// OpenAI — Azure AI also fronts xAI's Grok models — so the list lives in its
// own provider folder rather than sharing the OpenAI list.
//
// IMPORTANT: the `costs` below intentionally mirror the public list prices of
// the equivalent OpenAI / xAI models (see `../openai/models.ts` and
// `../xai/models.ts`). Azure is subsidised for us, so our actual spend is
// lower — but we bill users at the standard model price, which is the whole
// reason we route through Azure. Do NOT replace these with Azure's own rates.
//
// `id` is the Azure deployment name and is what we send upstream.
export const AZURE_MODELS: IChatModel[] = [
// -- xAI Grok (via Azure AI Foundry) -----------------------------------
{
// Costs mirror xai grok-4-1-fast-non-reasoning.
puterId: 'azure:x-ai/grok-4-1-fast-non-reasoning',
id: 'grok-4-1-fast-non-reasoning',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2025-11-19',
name: 'Grok 4.1 Fast (Non-Reasoning)',
aliases: ['x-ai/grok-4-1-fast-non-reasoning'],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 20,
completion_tokens: 50,
cached_tokens: 5,
},
max_tokens: 2_000_000,
},
{
// Costs mirror xai grok-4-1-fast (alias grok-4-1-fast-reasoning).
puterId: 'azure:x-ai/grok-4-1-fast-reasoning',
id: 'grok-4-1-fast-reasoning',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2025-11-19',
name: 'Grok 4.1 Fast (Reasoning)',
aliases: ['x-ai/grok-4-1-fast-reasoning'],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 20,
completion_tokens: 50,
cached_tokens: 5,
},
max_tokens: 2_000_000,
},
{
// Costs mirror xai grok-4.3.
puterId: 'azure:x-ai/grok-4.3',
id: 'grok-4.3',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
release_date: '2026-05-01',
name: 'Grok 4.3',
aliases: ['x-ai/grok-4.3'],
context: 1_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
{
// Costs mirror xai grok-4.20 (grok-4.20-0309-non-reasoning).
puterId: 'azure:x-ai/grok-4-20-non-reasoning',
id: 'grok-4-20-non-reasoning',
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2026-03-09',
name: 'Grok 4.20 (Non-Reasoning)',
aliases: ['x-ai/grok-4-20-non-reasoning'],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
{
// Costs mirror xai grok-4.20 (grok-4.20-0309-reasoning).
puterId: 'azure:x-ai/grok-4-20-reasoning',
id: 'grok-4-20-reasoning',
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2026-03-09',
name: 'Grok 4.20 (Reasoning)',
aliases: ['x-ai/grok-4-20-reasoning'],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
// -- OpenAI (via Azure AI Foundry) -------------------------------------
{
// Costs mirror openai gpt-5.
puterId: 'azure:openai/gpt-5',
id: 'gpt-5',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-09-30',
release_date: '2025-08-07',
aliases: ['openai/gpt-5'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
cached_tokens: 13,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5-codex (same list price as gpt-5).
puterId: 'azure:openai/gpt-5-codex',
id: 'gpt-5-codex',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-09-30',
release_date: '2025-08-07',
aliases: ['openai/gpt-5-codex'],
// Codex models are Responses-API only on Azure, same as OpenAI.
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
cached_tokens: 13,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5-nano.
puterId: 'azure:openai/gpt-5-nano',
id: 'gpt-5-nano',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-05-30',
release_date: '2025-08-07',
aliases: ['openai/gpt-5-nano'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 5,
cached_tokens: 1,
completion_tokens: 40,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5-mini.
puterId: 'azure:openai/gpt-5-mini',
id: 'gpt-5-mini',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-05-30',
release_date: '2025-08-07',
aliases: ['openai/gpt-5-mini'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 25,
cached_tokens: 3,
completion_tokens: 200,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-4o.
puterId: 'azure:openai/gpt-4o',
id: 'gpt-4o',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2023-09',
release_date: '2024-05-13',
aliases: ['openai/gpt-4o'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 250,
cached_tokens: 125,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 16384,
},
{
// Costs mirror openai gpt-5.1-codex-mini.
puterId: 'azure:openai/gpt-5.1-codex-mini',
id: 'gpt-5.1-codex-mini',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-09-30',
release_date: '2025-11-13',
aliases: ['openai/gpt-5.1-codex-mini'],
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 25,
cached_tokens: 3,
completion_tokens: 200,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.1.
puterId: 'azure:openai/gpt-5.1',
id: 'gpt-5.1',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-09-30',
release_date: '2025-11-13',
aliases: ['openai/gpt-5.1'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
cached_tokens: 13,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.1-codex.
puterId: 'azure:openai/gpt-5.1-codex',
id: 'gpt-5.1-codex',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-09-30',
release_date: '2025-11-13',
aliases: ['openai/gpt-5.1-codex'],
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 125,
cached_tokens: 13,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.2.
puterId: 'azure:openai/gpt-5.2',
id: 'gpt-5.2',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
release_date: '2025-12-11',
aliases: ['openai/gpt-5.2'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 175,
cached_tokens: 17.5,
completion_tokens: 1400,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.2-codex.
puterId: 'azure:openai/gpt-5.2-codex',
id: 'gpt-5.2-codex',
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
release_date: '2025-12-11',
aliases: ['openai/gpt-5.2-codex'],
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 175,
cached_tokens: 18,
completion_tokens: 1400,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.4-nano.
puterId: 'azure:openai/gpt-5.4-nano',
id: 'gpt-5.4-nano',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
release_date: '2026-03-19',
aliases: ['openai/gpt-5.4-nano'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 20,
cached_tokens: 2,
completion_tokens: 125,
},
context: 400_000,
max_tokens: 128_000,
},
{
// Costs mirror openai gpt-5.4-mini.
puterId: 'azure:openai/gpt-5.4-mini',
id: 'gpt-5.4-mini',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
aliases: ['openai/gpt-5.4-mini'],
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 75,
cached_tokens: 7.5,
completion_tokens: 450,
},
context: 400_000,
max_tokens: 128_000,
},
{
// Costs mirror openai gpt-5.3-codex.
puterId: 'azure:openai/gpt-5.3-codex',
id: 'gpt-5.3-codex',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
aliases: ['openai/gpt-5.3-codex'],
responses_api_only: true,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 175,
cached_tokens: 17.5,
completion_tokens: 1400,
},
context: 128_000,
max_tokens: 128000,
},
{
// Costs mirror openai gpt-5.4.
puterId: 'azure:openai/gpt-5.4',
id: 'gpt-5.4',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-08-31',
release_date: '2026-03-05',
aliases: ['openai/gpt-5.4'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
prompt_tokens: 250,
cached_tokens: 25,
completion_tokens: 1500,
},
context: 1_050_000,
max_tokens: 1_050_000,
},
];
@@ -363,6 +363,30 @@ export const OPEN_AI_MODELS: IChatModel[] = [
context: 128_000,
max_tokens: 128000,
},
{
puterId: 'openai:openai/gpt-5-codex',
id: 'gpt-5-codex',
modalities: { input: ['text', 'image'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2024-09-30',
release_date: '2025-09-15',
aliases: ['openai/gpt-5-codex'],
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
// models.dev openai: input $1.25 / output $10 / cache_read $0.125
// per 1M — identical to gpt-5.
tokens: 1_000_000,
prompt_tokens: 125,
cached_tokens: 13,
completion_tokens: 1000,
},
context: 128_000,
max_tokens: 128000,
responses_api_only: true,
},
{
puterId: 'openai:openai/gpt-5-mini',
id: 'gpt-5-mini-2025-08-07',
@@ -198,6 +198,66 @@ export const XAI_MODELS: IChatModel[] = [
},
max_tokens: 30_000,
},
{
puterId: 'x-ai:x-ai/grok-4-20-reasoning',
// xAI exposes this as the dated snapshot id; `grok-4-20-reasoning`
// (and dotted forms) are accepted as aliases by callers.
id: 'grok-4.20-0309-reasoning',
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2026-03-09',
name: 'Grok 4.20 (Reasoning)',
aliases: [
'x-ai/grok-4-20-reasoning',
'grok-4-20-reasoning',
'grok-4.20-reasoning',
'x-ai/grok-4.20-reasoning',
],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
// models.dev xai: input $1.25 / output $2.5 / cache_read $0.2 per 1M.
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
{
puterId: 'x-ai:x-ai/grok-4-20-non-reasoning',
// xAI exposes this as the dated snapshot id; `grok-4-20-non-reasoning`
// (and dotted forms) are accepted as aliases by callers.
id: 'grok-4.20-0309-non-reasoning',
modalities: { input: ['text', 'image', 'pdf'], output: ['text'] },
open_weights: false,
tool_call: true,
knowledge: '2025-07',
release_date: '2026-03-09',
name: 'Grok 4.20 (Non-Reasoning)',
aliases: [
'x-ai/grok-4-20-non-reasoning',
'grok-4-20-non-reasoning',
'grok-4.20-non-reasoning',
'x-ai/grok-4.20-non-reasoning',
],
context: 2_000_000,
costs_currency: 'usd-cents',
input_cost_key: 'prompt_tokens',
output_cost_key: 'completion_tokens',
costs: {
tokens: 1_000_000,
// models.dev xai: input $1.25 / output $2.5 / cache_read $0.2 per 1M.
prompt_tokens: 125,
completion_tokens: 250,
cached_tokens: 20,
},
max_tokens: 30_000,
},
{
puterId: 'x-ai:x-ai/grok-4-1-fast',
id: 'grok-4-1-fast',
+2
View File
@@ -140,6 +140,8 @@ export interface IAIProviderConfig {
apiToken?: string;
/** Override the provider's HTTP base URL (OpenRouter, Cloudflare, ElevenLabs, Ollama). */
apiBaseUrl?: string;
/** Azure AI Foundry deployment endpoint (azure-openai). Required alongside `apiKey`. */
apiURL?: string;
/** Cloudflare account id. */
accountId?: string;
/** ElevenLabs default voice id. */