返回源码地图

packages/ai/src/types.ts

v1.0.0 · a13d35a742c6 · 04:消息与provider compat契约;非全部模型类型审计

完整原文供逐行核对;页面收录不代表每行都经过人工语义审核。MIT 许可见 许可证。

1import type { TelemetryContext } from "@earendil-works/pi-telemetry";
2import type { AnthropicOptions } from "./api/anthropic-messages.ts";
3import type { AzureOpenAIResponsesOptions } from "./api/azure-openai-responses.ts";
4import type { BedrockOptions } from "./api/bedrock-converse-stream.ts";
5import type { GoogleOptions } from "./api/google-generative-ai.ts";
6import type { GoogleVertexOptions } from "./api/google-vertex.ts";
7import type { MistralOptions } from "./api/mistral-conversations.ts";
8import type { OpenAICodexResponsesOptions } from "./api/openai-codex-responses.ts";
9import type { OpenAICompletionsOptions } from "./api/openai-completions.ts";
10import type { OpenAIResponsesOptions } from "./api/openai-responses.ts";
11import type { PiMessagesOptions } from "./api/pi-messages.ts";
12import type { AssistantMessageDiagnostic } from "./utils/diagnostics.ts";
13import type { AssistantMessageEventStream } from "./utils/event-stream.ts";
14
15export type { AssistantMessageEventStream } from "./utils/event-stream.ts";
16
17export type KnownApi =
18 | "openai-completions"
19 | "mistral-conversations"
20 | "openai-responses"
21 | "azure-openai-responses"
22 | "openai-codex-responses"
23 | "anthropic-messages"
24 | "bedrock-converse-stream"
25 | "google-generative-ai"
26 | "google-vertex"
27 | "pi-messages";
28
29export type Api = KnownApi | (string & {});
30
31export type KnownImageApi = "openrouter-images";
32
33export type ImageApi = KnownImageApi | (string & {});
34
35export type KnownClassifierApi = "typesafe-system-one" | "cloudflare-workers-ai-system-one" | "llama-cpp-classify";
36
37export type ClassifierApi = KnownClassifierApi | (string & {});
38
39export type KnownProvider =
40 | "amazon-bedrock"
41 | "ant-ling"
42 | "anthropic"
43 | "google"
44 | "google-vertex"
45 | "openai"
46 | "azure-openai-responses"
47 | "openai-codex"
48 | "radius"
49 | "typesafe"
50 | "nvidia"
51 | "deepseek"
52 | "github-copilot"
53 | "xai"
54 | "groq"
55 | "cerebras"
56 | "openrouter"
57 | "vercel-ai-gateway"
58 | "zai"
59 | "zai-coding-cn"
60 | "mistral"
61 | "minimax"
62 | "minimax-cn"
63 | "moonshotai"
64 | "moonshotai-cn"
65 | "huggingface"
66 | "fireworks"
67 | "together"
68 | "baseten"
69 | "opencode"
70 | "opencode-go"
71 | "kimi-coding"
72 | "meta"
73 | "cloudflare-workers-ai"
74 | "cloudflare-ai-gateway"
75 | "qwen-token-plan"
76 | "qwen-token-plan-cn"
77 | "qwen-token-plan-individual"
78 | "xiaomi"
79 | "xiaomi-token-plan-cn"
80 | "xiaomi-token-plan-ams"
81 | "xiaomi-token-plan-sgp";
82export type ProviderId = KnownProvider | string;
83
84export type ToolChoice = "auto" | "none";
85export type ThinkingLevel = "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
86export type ModelThinkingLevel = "off" | ThinkingLevel;
87export type ThinkingLevelMap = Partial<Record<ModelThinkingLevel, string | null>>;
88export type ChatTemplateKwargValue =
89 | string
90 | number
91 | boolean
92 | null
93 | {
94 $var: "thinking.enabled" | "thinking.effort" | "thinking.budget";
95 omitWhenOff?: boolean;
96 };
97
98/** Top-level request field used to cap reasoning tokens on OpenAI-compatible servers. */
99export type ThinkingTokenBudgetField = "thinking_token_budget" | "thinking_budget" | "thinking_budget_tokens";
100
101/** Token budgets for each thinking level (token-based providers only) */
102export interface ThinkingBudgets {
103 minimal?: number;
104 low?: number;
105 medium?: number;
106 high?: number;
107}
108
109// Base options all providers share
110export type CacheRetention = "none" | "short" | "long";
111
112/**
113 * Best-effort prompt cache lifetime in seconds for each retention tier a request can ask for.
114 * A missing tier means the lifetime is unknown; pi does not warm such caches.
115 */
116export type ModelPromptCache = Partial<Record<Exclude<CacheRetention, "none">, number>>;
117
118export type Transport = "sse" | "websocket" | "websocket-cached" | "auto";
119
120/** Provider-scoped environment overrides. Values take precedence over process.env. */
121export type ProviderEnv = Record<string, string>;
122export type ProviderHeaders = Record<string, string | null>;
123export type FetchFunction = typeof globalThis.fetch;
124export type SessionAffinityFormat = "openai" | "openai-nosession" | "openrouter";
125
126export interface ProviderResponse {
127 status: number;
128 headers: Record<string, string>;
129}
130
131/** Authentication, HTTP transport, and lifecycle callbacks shared by provider requests. */
132export interface ProviderRequestOptions<TModel = Model<Api>> {
133 signal?: AbortSignal;
134 /** Explicit parent context for telemetry produced by this logical request. */
135 telemetryContext?: TelemetryContext;
136 apiKey?: string;
137 /**
138 * Optional fetch implementation for provider HTTP requests.
139 * Defaults to `globalThis.fetch`. Provider adapters that cannot inject a custom implementation may reject it.
140 * This does not affect WebSocket transports.
141 */
142 fetch?: FetchFunction;
143 /**
144 * Provider-scoped environment values. These take precedence over process.env for
145 * provider configuration such as regional settings, endpoint placeholders, and
146 * proxy variables.
147 */
148 env?: ProviderEnv;
149 /**
150 * Optional callback for inspecting or replacing provider payloads before sending.
151 * Return undefined to keep the payload unchanged.
152 */
153 onPayload?: (payload: unknown, model: TModel) => unknown | undefined | Promise<unknown | undefined>;
154 /**
155 * Optional callback invoked after an HTTP response is received.
156 */
157 onResponse?: (response: ProviderResponse, model: TModel) => void | Promise<void>;
158 /**
159 * Optional custom HTTP headers to include in API requests.
160 * Merged with provider defaults; caller values override default headers.
161 * On AWS Bedrock these are injected via a Smithy `build`-step middleware so
162 * they are covered by SigV4 signing; reserved headers (`x-amz-*`,
163 * `authorization`, `host`) are silently ignored to preserve SigV4 / bearer auth.
164 * A null value suppresses a provider/API default header with the same name.
165 */
166 headers?: ProviderHeaders;
167 /**
168 * HTTP request timeout in milliseconds for providers/SDKs that support it.
169 * For example, OpenAI and Anthropic SDK clients default to 10 minutes.
170 */
171 timeoutMs?: number;
172 /**
173 * Maximum retry attempts for providers/SDKs that support client-side retries.
174 * For example, OpenAI and Anthropic SDK clients default to 2.
175 */
176 maxRetries?: number;
177 /**
178 * Maximum delay in milliseconds to wait for a retry when the server requests a long wait.
179 * If the server's requested delay exceeds this value, the request fails immediately
180 * with an error containing the requested delay, allowing higher-level retry logic
181 * to handle it with user visibility.
182 * Default: 60000 (60 seconds). Set to 0 to disable the cap.
183 */
184 maxRetryDelayMs?: number;
185}
186
187export interface StreamOptions extends ProviderRequestOptions<Model<Api>> {
188 /**
189 * Optional callback invoked after an HTTP response is received and before
190 * its body stream is consumed.
191 */
192 onResponse?: (response: ProviderResponse, model: Model<Api>) => void | Promise<void>;
193 /**
194 * Optional observer for each parsed provider stream event before Pi normalization.
195 * Event data is adapter-owned and must be treated as read-only.
196 * Adapter support is explicit; unsupported adapters do not invoke it.
197 */
198 onProviderStreamEvent?: (data: unknown, model: Model<Api>) => void | Promise<void>;
199 temperature?: number;
200 /**
201 * Arbitrary sampling parameters merged into the request body as-is, after the named request
202 * fields, so keys here override them. Lets custom OpenAI-compatible servers (llama.cpp, vLLM,
203 * SGLang, ...) receive parameters pi does not model, e.g. `top_p`, `top_k`, `min_p`,
204 * `repetition_penalty`. Merged over `Model.samplingParams` per key. Only applied by
205 * OpenAI-compatible adapters (completions, responses, Azure responses); other APIs ignore it.
206 */
207 samplingParams?: Record<string, unknown>;
208 maxTokens?: number;
209 /**
210 * Preferred transport for providers that support multiple transports.
211 * Providers that do not support this option ignore it.
212 */
213 transport?: Transport;
214 /**
215 * Prompt cache retention preference. Providers map this to their supported values.
216 * Default: "short".
217 */
218 cacheRetention?: CacheRetention;
219 /**
220 * Optional session identifier for providers that support session-based caching.
221 * Providers can use this to enable prompt caching, request routing, or other
222 * session-aware features. Ignored by providers that don't support it.
223 */
224 sessionId?: string;
225 /**
226 * WebSocket connect timeout in milliseconds for providers that support
227 * WebSocket transports. This covers the connection/open handshake only;
228 * stream idleness after connection uses timeoutMs.
229 */
230 websocketConnectTimeoutMs?: number;
231 /**
232 * Optional metadata to include in API requests.
233 * Providers extract the fields they understand and ignore the rest.
234 * For example, Anthropic uses `user_id` for abuse tracking and rate limiting.
235 */
236 metadata?: Record<string, unknown>;
237}
238
239export type ProviderStreamOptions = StreamOptions & Record<string, unknown>;
240
241export interface DeferredFetchOptions extends ProviderRequestOptions<Model<Api>> {
242 /**
243 * Maximum provider long-poll duration in milliseconds.
244 * Defaults to 0, which performs one status check.
245 */
246 wait?: number;
247}
248
249/** Request options for best-effort deferred-response cancellation. */
250export type DeferredCancelOptions = ProviderRequestOptions<Model<Api>>;
251
252/**
253 * Maps known APIs to their full provider-specific stream option types.
254 * Type-only imports from API implementation modules are erased at emit, so
255 * this is tree-shake safe.
256 */
257export interface ApiOptionsMap {
258 "anthropic-messages": AnthropicOptions;
259 "openai-completions": OpenAICompletionsOptions;
260 "openai-responses": OpenAIResponsesOptions;
261 "openai-codex-responses": OpenAICodexResponsesOptions;
262 "azure-openai-responses": AzureOpenAIResponsesOptions;
263 "google-generative-ai": GoogleOptions;
264 "google-vertex": GoogleVertexOptions;
265 "mistral-conversations": MistralOptions;
266 "bedrock-converse-stream": BedrockOptions;
267 "pi-messages": PiMessagesOptions;
268}
269
270/**
271 * Full stream options for an API. Known APIs resolve to their concrete option
272 * type; custom API strings fall back to the generic shape.
273 */
274export type ApiStreamOptions<TApi extends Api> = TApi extends keyof ApiOptionsMap
275 ? ApiOptionsMap[TApi]
276 : StreamOptions & Record<string, unknown>;
277
278/**
279 * The uniform stream contract of an API implementation module: every module
280 * under `src/api/` exports `stream` and `streamSimple`; capable modules may also
281 * export deferred-response methods. Lazy wrappers (`lazyApi()`) and provider
282 * factories pass these around as values. This is the untyped dispatch shape;
283 * per-API option typing lives on the implementation modules themselves and on
284 * `Provider.stream()` via `ApiStreamOptions`.
285 */
286export interface ProviderStreams {
287 stream(model: Model<Api>, context: TranscriptContext, options?: StreamOptions): AssistantMessageEventStream;
288 streamSimple(
289 model: Model<Api>,
290 context: TranscriptContext,
291 options?: SimpleStreamOptions,
292 ): AssistantMessageEventStream;
293 fetchDeferred?(
294 model: Model<Api>,
295 handle: DeferredHandle,
296 options?: DeferredFetchOptions,
297 ): AssistantMessageEventStream;
298 cancelDeferred?(model: Model<Api>, handle: DeferredHandle, options?: DeferredCancelOptions): Promise<void>;
299}
300
301/**
302 * The uniform contract of an image-generation API implementation module:
303 * every image API module under `src/api/` exports exactly `generateImages`,
304 * so the module itself satisfies this interface. Lazy wrappers and
305 * `createProvider({ images })` pass these around as values.
306 */
307export interface ProviderImages {
308 generateImages(
309 model: ImageModel<ImageApi>,
310 context: ImagesContext,
311 options?: ImagesOptions,
312 ): Promise<AssistantImages>;
313}
314
315/** The uniform contract implemented by classifier API modules. */
316export interface ProviderClassifier {
317 classify(
318 model: ClassifierModel<ClassifierApi>,
319 context: ClassifierContext,
320 options?: ClassifierOptions,
321 ): Promise<ClassifierResult>;
322}
323
324export interface ClassifierOptions extends ProviderRequestOptions<ClassifierModel<ClassifierApi>> {
325 /**
326 * Divides the answer logits by this value before they are normalized into probabilities.
327 * Values above 1 soften the distribution; values below 1 sharpen it. Must be positive.
328 * APIs that cannot apply it ignore it.
329 */
330 temperature?: number;
331}
332
333export interface ImagesOptions extends ProviderRequestOptions<ImageModel<ImageApi>> {
334 /**
335 * Optional metadata to include in API requests.
336 * Providers extract the fields they understand and ignore the rest.
337 */
338 metadata?: Record<string, unknown>;
339}
340
341export type ProviderImagesOptions = ImagesOptions & Record<string, unknown>;
342
343export interface AnthropicAllowedFallbackModel {
344 provider: ProviderId;
345 model: string;
346 cost: ModelCost;
347}
348
349// Unified options with reasoning passed to streamSimple() and completeSimple()
350export interface SimpleStreamOptions extends StreamOptions {
351 /** Provider-neutral tool selection for simple requests. When omitted, adapters use provider-specific behavior. */
352 toolChoice?: ToolChoice;
353 reasoning?: ThinkingLevel;
354 /** Ask a capable provider to return a durable handle and continue the request asynchronously. */
355 deferred?: boolean | { window?: "15m" | "1h" | "24h" };
356 /** Custom token budgets for thinking levels (token-based providers only) */
357 thinkingBudgets?: ThinkingBudgets;
358}
359
360// Generic StreamFunction with typed options.
361//
362// Contract:
363// - Receives a normalized transcript: the system prompt and tools live in the
364// leading system message, never on the context itself.
365// - Must return an AssistantMessageEventStream.
366// - Direct streamSimple() calls may throw synchronously when request auth is
367// missing. Once a stream is returned, request/model/runtime failures should
368// be encoded in that stream.
369// - Error termination must produce an AssistantMessage with stopReason
370// "error" or "aborted" and errorMessage, emitted via the stream protocol.
371export type StreamFunction<TApi extends Api = Api, TOptions extends StreamOptions = StreamOptions> = (
372 model: Model<TApi>,
373 context: TranscriptContext,
374 options?: TOptions,
375) => AssistantMessageEventStream;
376
377export type ImagesFunction<TOptions extends ImagesOptions = ImagesOptions> = (
378 model: ImageModel<ImageApi>,
379 context: ImagesContext,
380 options?: TOptions,
381) => Promise<AssistantImages>;
382
383export type ClassifierFunction<TOptions extends ClassifierOptions = ClassifierOptions> = (
384 model: ClassifierModel<ClassifierApi>,
385 context: ClassifierContext,
386 options?: TOptions,
387) => Promise<ClassifierResult>;
388
389export interface TextSignatureV1 {
390 v: 1;
391 id: string;
392 phase?: "commentary" | "final_answer";
393}
394
395export interface TextContent {
396 type: "text";
397 text: string;
398 textSignature?: string; // e.g., for OpenAI responses, message metadata (legacy id string or TextSignatureV1 JSON)
399}
400
401export interface ThinkingContent {
402 type: "thinking";
403 thinking: string;
404 thinkingSignature?: string; // Provider-specific opaque or serialized reasoning replay data
405 /** When true, the thinking content was redacted by safety filters. The opaque
406 * encrypted payload is stored in `thinkingSignature` so it can be passed back
407 * to the API for multi-turn continuity. */
408 redacted?: boolean;
409}
410
411export interface ImageContent {
412 type: "image";
413 data: string; // base64 encoded image data
414 mimeType: string; // e.g., "image/jpeg", "image/png"
415}
416
417export interface ToolCall {
418 type: "toolCall";
419 id: string;
420 name: string;
421 arguments: JsonObject;
422 thoughtSignature?: string; // Google-specific: opaque signature for reusing thought context
423 /** OpenAI Responses namespace for calls to dynamically loaded or namespaced tools. */
424 namespace?: string;
425}
426
427export interface Usage {
428 input: number;
429 output: number;
430 cacheRead: number;
431 cacheWrite: number;
432 /** Subset of `cacheWrite` written with 1h retention. Only Anthropic reports this split. */
433 cacheWrite1h?: number;
434 /**
435 * Reasoning/thinking tokens, when the provider reports them. This is a subset of
436 * `output`: `output` already includes these tokens. Set to a number (possibly 0) by
437 * providers that expose a reasoning breakdown; left undefined by providers that don't.
438 */
439 reasoning?: number;
440 totalTokens: number;
441 cost: {
442 input: number;
443 output: number;
444 cacheRead: number;
445 cacheWrite: number;
446 total: number;
447 };
448}
449
450export type StopReason = "pending" | "stop" | "length" | "toolUse" | "error" | "aborted" | "deferred";
451
452export type JsonValue = null | boolean | number | string | readonly JsonValue[] | JsonObject;
453export type JsonObject = { [key: string]: JsonValue };
454
455type IsAny<T> = 0 extends 1 & T ? true : false;
456type IsExactlyJsonValue<T> = [T] extends [JsonValue] ? ([JsonValue] extends [T] ? true : false) : false;
457type IsJsonProperty<T> = IsAny<T> extends true
458 ? false
459 : unknown extends T
460 ? false
461 : [Exclude<T, undefined>] extends [never]
462 ? true
463 : IsJsonCompatible<Exclude<T, undefined>>;
464type InvalidJsonKeys<T extends object> = {
465 [TKey in keyof T]-?: TKey extends string | number ? (IsJsonProperty<T[TKey]> extends true ? never : TKey) : TKey;
466}[keyof T];
467type IsJsonCompatible<T> = IsAny<T> extends true
468 ? false
469 : unknown extends T
470 ? false
471 : IsExactlyJsonValue<T> extends true
472 ? true
473 : T extends null | boolean | number | string
474 ? true
475 : T extends undefined
476 ? false
477 : T extends readonly (infer TItem)[]
478 ? IsJsonCompatible<TItem>
479 : T extends (...args: never[]) => unknown
480 ? false
481 : T extends object
482 ? [InvalidJsonKeys<T>] extends [never]
483 ? true
484 : false
485 : false;
486
487/** The JSON representation of a typed in-memory value. Optional object properties remain optional. */
488export type JsonRepresentation<T> = IsAny<T> extends true
489 ? JsonValue
490 : unknown extends T
491 ? JsonValue
492 : [T] extends [JsonValue]
493 ? T
494 : T extends readonly unknown[]
495 ? { [TKey in keyof T]: JsonRepresentation<Exclude<T[TKey], undefined>> }
496 : T extends object
497 ? { [TKey in keyof T]: JsonRepresentation<Exclude<T[TKey], undefined>> }
498 : never;
499
500export interface DeferredHandle {
501 provider: string;
502 modelId: string;
503 api: string;
504 /** Provider token, such as a response id or batch id plus row id. */
505 id: string;
506 expiresAt?: number;
507 pollAfterMs?: number;
508 /** Provider conversion data required to reconstruct the final assistant message. */
509 data?: JsonValue;
510}
511
512/**
513 * System instructions and tool declarations at one point in the transcript.
514 *
515 * The leading system message is the system prompt. Later system messages change it:
516 * `content` adds instructions from that point on, `sections` replace or remove named
517 * prompt sections, and `toolsAdded`/`toolsRemoved` change the tool set. Replaying
518 * every system message in order yields the current prompt and tools. Providers that
519 * accept system messages mid-conversation send each one in place; other providers
520 * rebuild the leading system message from the replayed state.
521 */
522export interface SystemMessage {
523 role: "system";
524 /** Instruction text. On the leading message this is the base prompt; later, additional instructions. */
525 content: string | TextContent[];
526 /**
527 * Named, ordered prompt sections rendered verbatim after `content`. The leading message
528 * declares them; later messages replace sections by name, and `null` removes one. Keep
529 * each section self-delimiting (a tag, a heading) so the model can relate an update to
530 * the original. Avoid integer-like names; JSON objects reorder those.
531 */
532 sections?: Record<string, string | null>;
533 /** Complete definitions of tools that become available at this point. */
534 toolsAdded?: Tool[];
535 /** Tools that stop being available at this point. */
536 toolsRemoved?: ToolReference[];
537 timestamp: number; // Unix timestamp in milliseconds
538}
539
540export interface UserMessage {
541 role: "user";
542 content: string | (TextContent | ImageContent)[];
543 timestamp: number; // Unix timestamp in milliseconds
544}
545
546export interface AssistantMessage {
547 role: "assistant";
548 content: (TextContent | ThinkingContent | ToolCall)[];
549 api: Api;
550 provider: ProviderId;
551 model: string;
552 responseModel?: string; // Concrete model reported by the provider when different from the requested `model`
553 responseId?: string; // Provider-specific response/message identifier when the upstream API exposes one
554 /** Exact provider-native effort level used for this response. Absent for legacy or unmanaged responses. */
555 providerThinkingLevel?: string;
556 /** Pi thinking level the agent loop requested for this response. Absent outside the agent loop and for legacy responses. */
557 thinkingLevel?: ModelThinkingLevel;
558 diagnostics?: AssistantMessageDiagnostic[]; // Redacted provider/runtime diagnostics for failures and recoveries.
559 usage: Usage;
560 stopReason: StopReason;
561 deferred?: DeferredHandle;
562 errorMessage?: string;
563 rawStopReason?: string;
564 /**
565 * Provider indication of whether the model explicitly ended its turn.
566 * Preserved for debugging and does not currently affect agent control flow.
567 */
568 endTurn?: boolean;
569 timestamp: number; // Unix timestamp in milliseconds
570}
571
572/** A tool call that another tool made while it ran, for example from a codemode script. */
573export interface NestedToolCallRecord {
574 id: string;
575 name: string;
576 /** Omitted when over the size limits; `argumentsBytes` then gives their size. */
577 arguments?: JsonObject;
578 /** UTF-8 size of the arguments as JSON, set when `arguments` is omitted. */
579 argumentsBytes?: number;
580 /** `unfinished`: the call was still running when the calling tool finished. */
581 status: "ok" | "error" | "unfinished";
582 durationMs?: number;
583 /** Error text, truncated. */
584 error?: string;
585}
586
587/** Bounded record of the nested calls a tool made. Results are not recorded. */
588export interface NestedToolCalls {
589 calls: NestedToolCallRecord[];
590 /** False when calls were dropped, arguments omitted, or calls had not finished. */
591 complete: boolean;
592}
593
594export type ToolResultMessage<TDetails = JsonValue> = IsJsonCompatible<TDetails> extends true
595 ? {
596 role: "toolResult";
597 toolCallId: string;
598 toolName: string;
599 content: (TextContent | ImageContent)[]; // Supports text and images
600 details?: JsonRepresentation<TDetails>;
601 /** Usage from the tool execution itself, if available. Not part of main LLM context accounting. */
602 usage?: Usage;
603 /** Calls this tool made to other tools. Kept for the session record; not sent to the model. */
604 nestedCalls?: NestedToolCalls;
605 isError: boolean;
606 timestamp: number; // Unix timestamp in milliseconds
607 }
608 : never;
609
610export type Message = SystemMessage | UserMessage | AssistantMessage | ToolResultMessage;
611
612export type ImagesInputContent = TextContent | ImageContent;
613export type ImagesOutputContent = TextContent | ImageContent;
614
615export interface ImagesContext {
616 input: ImagesInputContent[];
617}
618
619export type ImagesStopReason = "stop" | "error" | "aborted";
620
621export interface AssistantImages {
622 api: ImageApi;
623 provider: ProviderId;
624 model: string;
625 output: ImagesOutputContent[];
626 responseId?: string;
627 usage?: Usage;
628 stopReason: ImagesStopReason;
629 errorMessage?: string;
630 timestamp: number; // Unix timestamp in milliseconds
631}
632
633export interface ClassifierChoiceQuestion {
634 type: "choice";
635 instructions: string;
636 criteria: Record<string, string>;
637}
638
639export interface ClassifierScoreQuestion {
640 type: "score";
641 instructions: string;
642 criteria: string[];
643}
644
645export interface ClassifierBoolQuestion {
646 type: "bool";
647 instructions: string;
648 criteria: { true: string; false: string };
649}
650
651export type ClassifierQuestion = ClassifierChoiceQuestion | ClassifierScoreQuestion | ClassifierBoolQuestion;
652
653export interface ClassifierContext {
654 state: JsonObject;
655 questions: Record<string, ClassifierQuestion>;
656}
657
658export interface ClassifierChoiceAnswer {
659 type: "choice";
660 choice: string;
661 probabilities: Record<string, number>;
662 confidence: number;
663}
664
665export interface ClassifierScoreAnswer {
666 type: "score";
667 score: number;
668 confidence: number;
669}
670
671export interface ClassifierBoolAnswer {
672 type: "bool";
673 probability: number;
674}
675
676export type ClassifierAnswer = ClassifierChoiceAnswer | ClassifierScoreAnswer | ClassifierBoolAnswer;
677export type ClassifierStopReason = "stop" | "error" | "aborted";
678
679export interface ClassifierResult {
680 api: ClassifierApi;
681 provider: ProviderId;
682 model: string;
683 answers: Record<string, ClassifierAnswer>;
684 /** Token usage and its cost at the model's catalog price, when the service reports token counts. */
685 usage?: Usage;
686 stopReason: ClassifierStopReason;
687 errorMessage?: string;
688 timestamp: number; // Unix timestamp in milliseconds
689}
690
691import type { TSchema } from "typebox";
692
693/** OpenAI grammar variants for constrained sampling. */
694export type GrammarFormat = "openai_lark" | "openai_regex";
695
696export type GrammarVariants = Partial<Record<GrammarFormat, string>>;
697
698/**
699 * Optional provider-side constrained sampling configs for a tool.
700 *
701 * The `json_schema` value roughly maps to the concept of `strict` in APIs which is
702 * implemented as json-schema constrained sampling by APIs. Grammar variants let
703 * callers provide provider-specific encodings of the same intended language.
704 */
705export type ConstrainedSamplingConfig =
706 | {
707 type: "json_schema";
708 strict: "prefer" | "require";
709 }
710 | {
711 type: "grammar";
712 variants: GrammarVariants;
713 };
714
715export interface Tool<TParameters extends TSchema = TSchema> {
716 name: string;
717 description: string;
718 parameters: TParameters;
719 constrainedSampling?: false | ConstrainedSamplingConfig;
720}
721
722export interface ToolReference {
723 name: string;
724}
725
726/**
727 * Request input accepted by the public stream entry points (`Models.stream()`,
728 * `streamSimple()`, ...). `systemPrompt` and `tools` are shorthand for a leading
729 * system message; `normalizeContext()` folds them into one before the request
730 * reaches a provider.
731 */
732export interface Context {
733 systemPrompt?: string;
734 messages: Message[];
735 tools?: Tool[];
736}
737
738declare const transcriptContextBrand: unique symbol;
739
740/**
741 * Normalized request context passed to providers and API implementations. The
742 * prompt and tool declarations are carried by the transcript's system messages.
743 * Only `normalizeContext()` produces this type, so a raw `Context` cannot reach
744 * provider code by accident.
745 */
746export type TranscriptContext = {
747 messages: Message[];
748 readonly [transcriptContextBrand]: true;
749};
750
751/**
752 * Event protocol for AssistantMessageEventStream.
753 *
754 * Successful streams emit `start` before partial updates and terminate with
755 * `done`. A stream may terminate directly with `error` when request setup fails
756 * before generation starts; after `start`, failures also terminate with `error`.
757 * Direct `streamSimple()` calls throw synchronously when request auth is missing.
758 * Updates and `done` must never appear before `start`.
759 *
760 * `partial` is the shared live response-so-far helper, not an event-time
761 * snapshot. Text and thinking blocks are empty when their `*_start` event is
762 * emitted and grow only through their corresponding `*_delta` events until the
763 * authoritative `*_end`. Redacted thinking may be complete at start and emit no
764 * deltas. Tool-call arguments at `toolcall_start` are provider-specific;
765 * `toolcall_delta` carries subsequent JSON updates.
766 */
767export type AssistantMessageEvent =
768 | { type: "start"; partial: AssistantMessage }
769 | { type: "text_start"; contentIndex: number; partial: AssistantMessage }
770 | { type: "text_delta"; contentIndex: number; delta: string; partial: AssistantMessage }
771 | { type: "text_end"; contentIndex: number; content: string; partial: AssistantMessage }
772 | { type: "thinking_start"; contentIndex: number; partial: AssistantMessage }
773 | { type: "thinking_delta"; contentIndex: number; delta: string; partial: AssistantMessage }
774 | { type: "thinking_end"; contentIndex: number; content: string; partial: AssistantMessage }
775 | { type: "toolcall_start"; contentIndex: number; partial: AssistantMessage }
776 | { type: "toolcall_delta"; contentIndex: number; delta: string; partial: AssistantMessage }
777 | { type: "toolcall_end"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage }
778 | {
779 type: "done";
780 reason: Extract<StopReason, "stop" | "length" | "toolUse" | "deferred">;
781 message: AssistantMessage;
782 }
783 | { type: "error"; reason: Extract<StopReason, "aborted" | "error">; error: AssistantMessage };
784
785/**
786 * Compatibility settings for OpenAI-compatible completions APIs.
787 * Use this to override URL-based auto-detection for custom providers.
788 */
789export interface OpenAICompletionsCompat {
790 /** Whether the provider supports the `store` field. Default: auto-detected from URL. */
791 supportsStore?: boolean;
792 /** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */
793 supportsDeveloperRole?: boolean;
794 /** Whether the provider supports `reasoning_effort`. Default: auto-detected from URL. */
795 supportsReasoningEffort?: boolean;
796 /** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */
797 supportsUsageInStreaming?: boolean;
798 /** Whether streamed responses include `finish_reason`. When false, pi infers `stop` or `toolUse` when the stream ends. Default: true. */
799 supportsFinishReason?: boolean;
800 /** Which field to use for max tokens. Default: auto-detected from URL. */
801 maxTokensField?: "max_completion_tokens" | "max_tokens";
802 /** Whether tool results require the `name` field. Default: auto-detected from URL. */
803 requiresToolResultName?: boolean;
804 /** Whether a user message after tool results requires an assistant message in between. Default: auto-detected from URL. */
805 requiresAssistantAfterToolResult?: boolean;
806 /** Whether thinking blocks must be converted to text blocks with <thinking> delimiters. Default: auto-detected from URL. */
807 requiresThinkingAsText?: boolean;
808 /** Whether all replayed assistant messages must include an empty reasoning_content field when reasoning is enabled. Default: auto-detected from URL. */
809 requiresReasoningContentOnAssistantMessages?: boolean;
810 /** Format for reasoning/thinking parameter. "openai" uses reasoning_effort, "openrouter" uses reasoning: { effort }, "deepseek" uses thinking: { type } plus reasoning_effort when supported, "together" uses reasoning: { enabled } plus reasoning_effort when supported, "baseten" uses configurable chat_template_args plus reasoning_effort when supported, "zai" uses thinking: { type }, "qwen" uses top-level enable_thinking: boolean, "qwen-chat-template" uses chat_template_kwargs.enable_thinking and preserve_thinking, "chat-template" uses configurable chat_template_kwargs, "string-thinking" uses top-level thinking: string, and "ant-ling" uses reasoning: { effort } only when the mapped effort is non-null. Default: "openai". */
811 thinkingFormat?:
812 | "openai"
813 | "openrouter"
814 | "deepseek"
815 | "together"
816 | "baseten"
817 | "zai"
818 | "qwen"
819 | "chat-template"
820 | "qwen-chat-template"
821 | "string-thinking"
822 | "ant-ling";
823 /** Kwargs to send as `chat_template_kwargs` when `thinkingFormat` is `chat-template`. Use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for pi-controlled thinking values. */
824 chatTemplateKwargs?: Record<string, ChatTemplateKwargValue>;
825 /** Arguments to send as `chat_template_args` when `thinkingFormat` is `baseten`. Use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for pi-controlled thinking values. */
826 chatTemplateArgs?: Record<string, ChatTemplateKwargValue>;
827 /** OpenRouter-compatible routing preferences sent as the `provider` request field. */
828 openRouterRouting?: OpenRouterRouting;
829 /** Vercel AI Gateway routing preferences. Only used when baseUrl points to Vercel AI Gateway. */
830 vercelGatewayRouting?: VercelGatewayRouting;
831 /** Whether z.ai supports top-level `tool_stream: true` for streaming tool call deltas. Default: false. */
832 zaiToolStream?: boolean;
833 /**
834 * Top-level request field used to cap reasoning tokens from `thinkingBudgets`.
835 * Reasoning and the answer share `max_tokens` on these endpoints, so without a budget a
836 * reasoning-heavy turn can consume the whole response and emit no answer.
837 * `"thinking_token_budget"` is vLLM, `"thinking_budget"` is Qwen/DashScope/SGLang,
838 * `"thinking_budget_tokens"` is llama.cpp. Off by default; not set on the generated catalog.
839 */
840 thinkingTokenBudgetField?: ThinkingTokenBudgetField;
841 /** Alias for `thinkingTokenBudgetField: "thinking_token_budget"` (vLLM). Prefer `thinkingTokenBudgetField`. Default: false. */
842 supportsThinkingTokenBudget?: boolean;
843 /** Whether the provider supports OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */
844 supportsOpenAIGrammarTools?: boolean;
845 /** Whether the exact model accepts system or developer messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false; the generated model catalog enables it for verified models. */
846 supportsMidConvoSystemMessages?: boolean;
847 /** Whether system messages can introduce additional tools mid-conversation. Requires `supportsMidConvoSystemMessages`. Default: false; the generated model catalog enables it for capable models. */
848 supportsMidConvoToolAdditions?: boolean;
849 /** Whether the provider supports the `strict` field in tool definitions. Default: false; generated capable models enable it explicitly. */
850 supportsStrictMode?: boolean;
851 /** Cache control convention for prompt caching. "anthropic" applies Anthropic-style `cache_control` markers to the system prompt, last tool definition, and last user, assistant, or tool-result text content. */
852 cacheControlFormat?: "anthropic";
853 /** Whether to send session-affinity data from `options.sessionId`. Default: true for OpenRouter endpoints, false otherwise. */
854 sendSessionAffinityHeaders?: boolean;
855 /** Session-affinity header format: `openai` sends `session_id`, `x-client-request-id`, and `x-session-affinity`; `openai-nosession` sends `x-client-request-id` and `x-session-affinity`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */
856 sessionAffinityFormat?: SessionAffinityFormat;
857 /** Whether the provider supports long prompt cache retention (`prompt_cache_retention: "24h"` or Anthropic-style `cache_control.ttl: "1h"`, depending on format). Default: true. */
858 supportsLongCacheRetention?: boolean;
859 /**
860 * vLLM scheduler priority sent as the top-level `priority` request field (lower values are
861 * handled earlier; server default 0). Only meaningful when vLLM runs with
862 * `--scheduling-policy priority`; useful for keeping background/batch work from stalling
863 * interactive sessions. Off by default; not set on the generated catalog.
864 */
865 vllmPriority?: number;
866}
867
868/** Compatibility settings for OpenAI Responses APIs. */
869export interface OpenAIResponsesCompat {
870 /** Whether the provider supports the `developer` role (vs `system`). Default: true. */
871 supportsDeveloperRole?: boolean;
872 /** Whether the exact model accepts developer or system messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false; the generated model catalog enables it for verified models. */
873 supportsMidConvoSystemMessages?: boolean;
874 /** Session-affinity header format: `openai` sends `session_id` and `x-client-request-id`; `openai-nosession` sends `x-client-request-id`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */
875 sessionAffinityFormat?: SessionAffinityFormat;
876 /** Whether the provider supports long prompt cache retention. This uses `prompt_cache_options.ttl: "30m"` on GPT-5.6+ and `prompt_cache_retention: "24h"` on earlier models. Default: true. */
877 supportsLongCacheRetention?: boolean;
878 /** Whether the provider supports strict JSON-schema function tools. Defaults are API-specific; generated OpenAI models enable it explicitly. */
879 supportsStrictMode?: boolean;
880 /** Whether to emit OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */
881 supportsOpenAIGrammarTools?: boolean;
882 /** Whether the model supports message-anchored `additional_tools` input items. Default: false. */
883 supportsAdditionalTools?: boolean;
884 /** Whether the model supports client-executed tool search for transcript-anchored additions. Default: false. */
885 supportsToolSearch?: boolean;
886 /** Whether the model accepts `prompt_cache_options` (OpenAI GPT-5.6+ prompt caching). Older OpenAI models reject the parameter. Default: false. */
887 supportsExplicitPromptCacheMode?: boolean;
888 /** Whether the provider accepts the `max_output_tokens` parameter. Some Codex-protocol gateways reject it. Default: true. */
889 supportsMaxOutputTokens?: boolean;
890}
891
892/** Compatibility settings for Anthropic Messages-compatible APIs. */
893export interface AnthropicMessagesCompat {
894 /**
895 * Whether the provider accepts per-tool `eager_input_streaming`.
896 * When false, the Anthropic provider omits `tools[].eager_input_streaming`
897 * and sends the legacy `fine-grained-tool-streaming-2025-05-14` beta header
898 * for tool-enabled requests.
899 * Default: true.
900 */
901 supportsEagerToolInputStreaming?: boolean;
902 /** Whether the provider supports Anthropic long cache retention (`cache_control.ttl: "1h"`). Default: true. */
903 supportsLongCacheRetention?: boolean;
904 /**
905 * Whether to send the `x-session-affinity` header from `options.sessionId`
906 * when caching is enabled. Required for providers like Fireworks that use
907 * session affinity for prompt cache routing (requests to the same replica
908 * maximize cache hits).
909 * Default: false.
910 */
911 sendSessionAffinityHeaders?: boolean;
912 /** Session-affinity format. `"openrouter"` sends `x-session-id`; when unset, sends `x-session-affinity`. */
913 sessionAffinityFormat?: "openrouter";
914 /**
915 * Whether the provider supports Anthropic-style `cache_control` markers on
916 * tool definitions. When false, `cache_control` is omitted from tool params.
917 * Some Anthropic-compatible providers (e.g., Fireworks) do not support this
918 * field on tools and may reject or ignore it.
919 * Default: true.
920 */
921 supportsCacheControlOnTools?: boolean;
922 /**
923 * Whether the model accepts the Anthropic `temperature` request field.
924 * Claude Opus 4.7+ rejects non-default temperature values.
925 * Default: true.
926 */
927 supportsTemperature?: boolean;
928 /**
929 * Whether to force adaptive thinking (`thinking.type: "adaptive"` plus
930 * `output_config.effort`) regardless of the model id. Built-in models that
931 * require adaptive thinking set this in generated metadata. Custom
932 * Anthropic-compatible providers can set this to `true` for any model whose
933 * upstream requires the adaptive format. Set to `false` to
934 * opt out on overridden built-in models.
935 * Default: false.
936 */
937 forceAdaptiveThinking?: boolean;
938 /** Whether to replay empty thinking signatures as `signature: ""` instead of converting thinking to text. Default: false. */
939 allowEmptySignature?: boolean;
940 /** Whether the provider supports Anthropic strict tool schemas. Default: false; generated Anthropic models enable it explicitly. */
941 supportsStrictTools?: boolean;
942 /** Whether the exact model transport supports effort-only system messages and thinking binding controls. Default: false. */
943 supportsMidConvoEffort?: boolean;
944 /** Whether the exact model accepts system-role messages inside the conversation. When false, later system messages are folded into the top-level system prompt. Default: false. */
945 supportsMidConvoSystemMessages?: boolean;
946 /** Whether the exact model accepts mid-conversation `tool_addition` and `tool_removal` blocks. Requires `supportsMidConvoSystemMessages`. Default: false. */
947 supportsMidConvoToolChanges?: boolean;
948 /**
949 * Models Anthropic accepts in `fallbacks` for server-side refusal fallback,
950 * with local pricing metadata for returned fallback responses. When absent or
951 * empty, callers must omit `fallbacks`; Anthropic rejects the field for models
952 * with no permitted fallback targets.
953 */
954 allowedFallbackModels?: AnthropicAllowedFallbackModel[];
955}
956
957/** Compatibility settings for Amazon Bedrock models. */
958export interface BedrockCompat {
959 /** Whether the model supports Bedrock strict tool schemas. Default: false. */
960 supportsStrictMode?: boolean;
961}
962
963/** Compatibility settings for the Mistral chat API. */
964export interface MistralConversationsCompat {
965 /** Whether the exact model accepts system messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false. */
966 supportsMidConvoSystemMessages?: boolean;
967}
968
969/**
970 * OpenRouter provider routing preferences.
971 * Controls which upstream providers OpenRouter routes requests to.
972 * Sent as the `provider` field in the OpenRouter API request body.
973 * @see https://openrouter.ai/docs/guides/routing/provider-selection
974 */
975export interface OpenRouterRouting {
976 /** Whether to allow backup providers to serve requests. Default: true. */
977 allow_fallbacks?: boolean;
978 /** Whether to filter providers to only those that support all parameters in the request. Default: false. */
979 require_parameters?: boolean;
980 /** Data collection setting. "allow" (default): allow providers that may store/train on data. "deny": only use providers that don't collect user data. */
981 data_collection?: "deny" | "allow";
982 /** Whether to restrict routing to only ZDR (Zero Data Retention) endpoints. */
983 zdr?: boolean;
984 /** Whether to restrict routing to only models that allow text distillation. */
985 enforce_distillable_text?: boolean;
986 /** An ordered list of provider names/slugs to try in sequence, falling back to the next if unavailable. */
987 order?: string[];
988 /** List of provider names/slugs to exclusively allow for this request. */
989 only?: string[];
990 /** List of provider names/slugs to skip for this request. */
991 ignore?: string[];
992 /** A list of quantization levels to filter providers by (e.g., ["fp16", "bf16", "fp8", "fp6", "int8", "int4", "fp4", "fp32"]). */
993 quantizations?: string[];
994 /** Sorting strategy. Can be a string (e.g., "price", "throughput", "latency") or an object with `by` and `partition`. */
995 sort?:
996 | string
997 | {
998 /** The sorting metric: "price", "throughput", "latency". */
999 by?: string;
1000 /** Partitioning strategy: "model" (default) or "none". */
1001 partition?: string | null;
1002 };
1003 /** Maximum price per million tokens (USD). */
1004 max_price?: {
1005 /** Price per million prompt tokens. */
1006 prompt?: number | string;
1007 /** Price per million completion tokens. */
1008 completion?: number | string;
1009 /** Price per image. */
1010 image?: number | string;
1011 /** Price per audio unit. */
1012 audio?: number | string;
1013 /** Price per request. */
1014 request?: number | string;
1015 };
1016 /** Preferred minimum throughput (tokens/second). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */
1017 preferred_min_throughput?:
1018 | number
1019 | {
1020 /** Minimum tokens/second at the 50th percentile. */
1021 p50?: number;
1022 /** Minimum tokens/second at the 75th percentile. */
1023 p75?: number;
1024 /** Minimum tokens/second at the 90th percentile. */
1025 p90?: number;
1026 /** Minimum tokens/second at the 99th percentile. */
1027 p99?: number;
1028 };
1029 /** Preferred maximum latency (seconds). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */
1030 preferred_max_latency?:
1031 | number
1032 | {
1033 /** Maximum latency in seconds at the 50th percentile. */
1034 p50?: number;
1035 /** Maximum latency in seconds at the 75th percentile. */
1036 p75?: number;
1037 /** Maximum latency in seconds at the 90th percentile. */
1038 p90?: number;
1039 /** Maximum latency in seconds at the 99th percentile. */
1040 p99?: number;
1041 };
1042}
1043
1044/**
1045 * Vercel AI Gateway routing preferences.
1046 * Controls which upstream providers the gateway routes requests to.
1047 * @see https://vercel.com/docs/ai-gateway/models-and-providers/provider-options
1048 */
1049export interface VercelGatewayRouting {
1050 /** List of provider slugs to exclusively use for this request (e.g., ["bedrock", "anthropic"]). */
1051 only?: string[];
1052 /** List of provider slugs to try in order (e.g., ["anthropic", "openai"]). */
1053 order?: string[];
1054}
1055
1056export interface ModelCostRates {
1057 input: number; // $/million tokens
1058 output: number; // $/million tokens
1059 cacheRead: number; // $/million tokens
1060 cacheWrite: number; // $/million tokens
1061}
1062
1063export interface ModelCostTier extends ModelCostRates {
1064 /** Use this tier for requests whose total input usage exceeds this token count. */
1065 inputTokensAbove: number;
1066}
1067
1068export interface ModelCost extends ModelCostRates {
1069 /** Request-wide pricing tiers. The highest matching input threshold applies to the full request. */
1070 tiers?: ModelCostTier[];
1071}
1072
1073export interface ModelImageResizeOptions {
1074 maxWidth?: number;
1075 maxHeight?: number;
1076 /** Maximum base64-encoded payload size in bytes. */
1077 maxBytes?: number;
1078 jpegQuality?: number;
1079}
1080
1081export interface ModelImageInputLimits {
1082 /** Cache-safe resize profile applied before a new image enters conversation history. */
1083 resize?: ModelImageResizeOptions;
1084 /** Maximum images accepted in one provider message. */
1085 maxPerMessage?: number;
1086 /** Maximum images accepted across one provider request. */
1087 maxPerRequest?: number;
1088}
1089
1090export interface ModelInputLimits {
1091 /** Maximum serialized provider request size in bytes. */
1092 maxRequestBytes?: number;
1093 images?: ModelImageInputLimits;
1094}
1095
1096/** Fields shared by every catalog entry, regardless of what you can do with it. */
1097export interface BaseModel<TApi extends string> {
1098 id: string;
1099 name: string;
1100 api: TApi;
1101 provider: ProviderId;
1102 baseUrl: string;
1103 input: ("text" | "image")[];
1104 /** Provider input limits and cache-safe preprocessing metadata. */
1105 inputLimits?: ModelInputLimits;
1106 cost: ModelCost;
1107 headers?: Record<string, string>;
1108}
1109
1110/** Chat model: usable with `stream()` and friends. */
1111export interface Model<TApi extends Api> extends BaseModel<TApi> {
1112 /**
1113 * Optional: chat is the default model type, so models without `type` are chat
1114 * models. Narrow mixed model lists with `isModelType()` instead of comparing
1115 * `type` directly.
1116 */
1117 type?: "chat";
1118 reasoning: boolean;
1119 /**
1120 * Maps pi thinking levels to provider/model-specific values.
1121 * Missing keys use provider defaults. null marks a level as unsupported.
1122 */
1123 thinkingLevelMap?: ThinkingLevelMap;
1124 /** Prompt cache lifetimes per retention tier. Unset when the provider's cache behavior is unknown. */
1125 promptCache?: ModelPromptCache;
1126 contextWindow: number;
1127 maxTokens: number;
1128 /** Default sampling parameters for this model. See {@link StreamOptions.samplingParams}; per-request keys override these. */
1129 samplingParams?: Record<string, unknown>;
1130 /** Compatibility overrides for OpenAI-compatible APIs. If not set, auto-detected from baseUrl. */
1131 compat?: TApi extends "openai-completions"
1132 ? OpenAICompletionsCompat
1133 : TApi extends "openai-responses" | "azure-openai-responses" | "openai-codex-responses"
1134 ? OpenAIResponsesCompat
1135 : TApi extends "anthropic-messages"
1136 ? AnthropicMessagesCompat
1137 : TApi extends "bedrock-converse-stream"
1138 ? BedrockCompat
1139 : TApi extends "mistral-conversations"
1140 ? MistralConversationsCompat
1141 : never;
1142}
1143
1144/** Image-generation model: usable with `generateImages()` only. */
1145export interface ImageModel<TApi extends ImageApi> extends BaseModel<TApi> {
1146 type: "image";
1147 /** Output modalities. Always includes `"image"`; `"text"` means the model can also return text blocks. */
1148 output: ("text" | "image")[];
1149}
1150
1151/** Structured classifier model: usable with `classify()` only. */
1152export interface ClassifierModel<TApi extends ClassifierApi> extends BaseModel<TApi> {
1153 type: "classifier";
1154 contextWindow: number;
1155}
1156
1157/** Model shape for each model type. */
1158export interface ModelTypeMap {
1159 chat: Model<Api>;
1160 image: ImageModel<ImageApi>;
1161 classifier: ClassifierModel<ClassifierApi>;
1162}
1163
1164/** What a catalog entry is for. Decides which `Models` operation accepts it. */
1165export type ModelType = keyof ModelTypeMap;
1166
1167/** Anything a provider can list. Narrow with `isModelType()`. */
1168export type AnyModel = ModelTypeMap[ModelType];