1
import type { TelemetryContext } from "@earendil-works/pi-telemetry";2
import type { AnthropicOptions } from "./api/anthropic-messages.ts";3
import type { AzureOpenAIResponsesOptions } from "./api/azure-openai-responses.ts";4
import type { BedrockOptions } from "./api/bedrock-converse-stream.ts";5
import type { GoogleOptions } from "./api/google-generative-ai.ts";6
import type { GoogleVertexOptions } from "./api/google-vertex.ts";7
import type { MistralOptions } from "./api/mistral-conversations.ts";8
import type { OpenAICodexResponsesOptions } from "./api/openai-codex-responses.ts";9
import type { OpenAICompletionsOptions } from "./api/openai-completions.ts";10
import type { OpenAIResponsesOptions } from "./api/openai-responses.ts";11
import type { PiMessagesOptions } from "./api/pi-messages.ts";12
import type { AssistantMessageDiagnostic } from "./utils/diagnostics.ts";13
import type { AssistantMessageEventStream } from "./utils/event-stream.ts";15
export type { AssistantMessageEventStream } from "./utils/event-stream.ts";17
export type KnownApi =18
| "openai-completions"19
| "mistral-conversations"20
| "openai-responses"21
| "azure-openai-responses"22
| "openai-codex-responses"23
| "anthropic-messages"24
| "bedrock-converse-stream"25
| "google-generative-ai"26
| "google-vertex"27
| "pi-messages";29
export type Api = KnownApi | (string & {});31
export type KnownImageApi = "openrouter-images";33
export type ImageApi = KnownImageApi | (string & {});35
export type KnownClassifierApi = "typesafe-system-one" | "cloudflare-workers-ai-system-one" | "llama-cpp-classify";37
export type ClassifierApi = KnownClassifierApi | (string & {});39
export type KnownProvider =40
| "amazon-bedrock"41
| "ant-ling"42
| "anthropic"43
| "google"44
| "google-vertex"45
| "openai"46
| "azure-openai-responses"47
| "openai-codex"48
| "radius"49
| "typesafe"50
| "nvidia"51
| "deepseek"52
| "github-copilot"53
| "xai"54
| "groq"55
| "cerebras"56
| "openrouter"57
| "vercel-ai-gateway"58
| "zai"59
| "zai-coding-cn"60
| "mistral"61
| "minimax"62
| "minimax-cn"63
| "moonshotai"64
| "moonshotai-cn"65
| "huggingface"66
| "fireworks"67
| "together"68
| "baseten"69
| "opencode"70
| "opencode-go"71
| "kimi-coding"72
| "meta"73
| "cloudflare-workers-ai"74
| "cloudflare-ai-gateway"75
| "qwen-token-plan"76
| "qwen-token-plan-cn"77
| "qwen-token-plan-individual"78
| "xiaomi"79
| "xiaomi-token-plan-cn"80
| "xiaomi-token-plan-ams"81
| "xiaomi-token-plan-sgp";82
export type ProviderId = KnownProvider | string;84
export type ToolChoice = "auto" | "none";85
export type ThinkingLevel = "minimal" | "low" | "medium" | "high" | "xhigh" | "max";86
export type ModelThinkingLevel = "off" | ThinkingLevel;87
export type ThinkingLevelMap = Partial<Record<ModelThinkingLevel, string | null>>;88
export type ChatTemplateKwargValue =89
| string90
| number91
| boolean92
| null93
| {94
$var: "thinking.enabled" | "thinking.effort" | "thinking.budget";95
omitWhenOff?: boolean;96
};98
/** Top-level request field used to cap reasoning tokens on OpenAI-compatible servers. */99
export type ThinkingTokenBudgetField = "thinking_token_budget" | "thinking_budget" | "thinking_budget_tokens";101
/** Token budgets for each thinking level (token-based providers only) */102
export interface ThinkingBudgets {103
minimal?: number;104
low?: number;105
medium?: number;106
high?: number;107
}109
// Base options all providers share110
export type CacheRetention = "none" | "short" | "long";112
/**113
* Best-effort prompt cache lifetime in seconds for each retention tier a request can ask for.114
* A missing tier means the lifetime is unknown; pi does not warm such caches.115
*/116
export type ModelPromptCache = Partial<Record<Exclude<CacheRetention, "none">, number>>;118
export type Transport = "sse" | "websocket" | "websocket-cached" | "auto";120
/** Provider-scoped environment overrides. Values take precedence over process.env. */121
export type ProviderEnv = Record<string, string>;122
export type ProviderHeaders = Record<string, string | null>;123
export type FetchFunction = typeof globalThis.fetch;124
export type SessionAffinityFormat = "openai" | "openai-nosession" | "openrouter";126
export interface ProviderResponse {127
status: number;128
headers: Record<string, string>;129
}131
/** Authentication, HTTP transport, and lifecycle callbacks shared by provider requests. */132
export interface ProviderRequestOptions<TModel = Model<Api>> {133
signal?: AbortSignal;134
/** Explicit parent context for telemetry produced by this logical request. */135
telemetryContext?: TelemetryContext;136
apiKey?: string;137
/**138
* Optional fetch implementation for provider HTTP requests.139
* Defaults to `globalThis.fetch`. Provider adapters that cannot inject a custom implementation may reject it.140
* This does not affect WebSocket transports.141
*/142
fetch?: FetchFunction;143
/**144
* Provider-scoped environment values. These take precedence over process.env for145
* provider configuration such as regional settings, endpoint placeholders, and146
* proxy variables.147
*/148
env?: ProviderEnv;149
/**150
* Optional callback for inspecting or replacing provider payloads before sending.151
* Return undefined to keep the payload unchanged.152
*/153
onPayload?: (payload: unknown, model: TModel) => unknown | undefined | Promise<unknown | undefined>;154
/**155
* Optional callback invoked after an HTTP response is received.156
*/157
onResponse?: (response: ProviderResponse, model: TModel) => void | Promise<void>;158
/**159
* Optional custom HTTP headers to include in API requests.160
* Merged with provider defaults; caller values override default headers.161
* On AWS Bedrock these are injected via a Smithy `build`-step middleware so162
* they are covered by SigV4 signing; reserved headers (`x-amz-*`,163
* `authorization`, `host`) are silently ignored to preserve SigV4 / bearer auth.164
* A null value suppresses a provider/API default header with the same name.165
*/166
headers?: ProviderHeaders;167
/**168
* HTTP request timeout in milliseconds for providers/SDKs that support it.169
* For example, OpenAI and Anthropic SDK clients default to 10 minutes.170
*/171
timeoutMs?: number;172
/**173
* Maximum retry attempts for providers/SDKs that support client-side retries.174
* For example, OpenAI and Anthropic SDK clients default to 2.175
*/176
maxRetries?: number;177
/**178
* Maximum delay in milliseconds to wait for a retry when the server requests a long wait.179
* If the server's requested delay exceeds this value, the request fails immediately180
* with an error containing the requested delay, allowing higher-level retry logic181
* to handle it with user visibility.182
* Default: 60000 (60 seconds). Set to 0 to disable the cap.183
*/184
maxRetryDelayMs?: number;185
}187
export interface StreamOptions extends ProviderRequestOptions<Model<Api>> {188
/**189
* Optional callback invoked after an HTTP response is received and before190
* its body stream is consumed.191
*/192
onResponse?: (response: ProviderResponse, model: Model<Api>) => void | Promise<void>;193
/**194
* Optional observer for each parsed provider stream event before Pi normalization.195
* Event data is adapter-owned and must be treated as read-only.196
* Adapter support is explicit; unsupported adapters do not invoke it.197
*/198
onProviderStreamEvent?: (data: unknown, model: Model<Api>) => void | Promise<void>;199
temperature?: number;200
/**201
* Arbitrary sampling parameters merged into the request body as-is, after the named request202
* fields, so keys here override them. Lets custom OpenAI-compatible servers (llama.cpp, vLLM,203
* SGLang, ...) receive parameters pi does not model, e.g. `top_p`, `top_k`, `min_p`,204
* `repetition_penalty`. Merged over `Model.samplingParams` per key. Only applied by205
* OpenAI-compatible adapters (completions, responses, Azure responses); other APIs ignore it.206
*/207
samplingParams?: Record<string, unknown>;208
maxTokens?: number;209
/**210
* Preferred transport for providers that support multiple transports.211
* Providers that do not support this option ignore it.212
*/213
transport?: Transport;214
/**215
* Prompt cache retention preference. Providers map this to their supported values.216
* Default: "short".217
*/218
cacheRetention?: CacheRetention;219
/**220
* Optional session identifier for providers that support session-based caching.221
* Providers can use this to enable prompt caching, request routing, or other222
* session-aware features. Ignored by providers that don't support it.223
*/224
sessionId?: string;225
/**226
* WebSocket connect timeout in milliseconds for providers that support227
* WebSocket transports. This covers the connection/open handshake only;228
* stream idleness after connection uses timeoutMs.229
*/230
websocketConnectTimeoutMs?: number;231
/**232
* Optional metadata to include in API requests.233
* Providers extract the fields they understand and ignore the rest.234
* For example, Anthropic uses `user_id` for abuse tracking and rate limiting.235
*/236
metadata?: Record<string, unknown>;237
}239
export type ProviderStreamOptions = StreamOptions & Record<string, unknown>;241
export interface DeferredFetchOptions extends ProviderRequestOptions<Model<Api>> {242
/**243
* Maximum provider long-poll duration in milliseconds.244
* Defaults to 0, which performs one status check.245
*/246
wait?: number;247
}249
/** Request options for best-effort deferred-response cancellation. */250
export type DeferredCancelOptions = ProviderRequestOptions<Model<Api>>;252
/**253
* Maps known APIs to their full provider-specific stream option types.254
* Type-only imports from API implementation modules are erased at emit, so255
* this is tree-shake safe.256
*/257
export interface ApiOptionsMap {258
"anthropic-messages": AnthropicOptions;259
"openai-completions": OpenAICompletionsOptions;260
"openai-responses": OpenAIResponsesOptions;261
"openai-codex-responses": OpenAICodexResponsesOptions;262
"azure-openai-responses": AzureOpenAIResponsesOptions;263
"google-generative-ai": GoogleOptions;264
"google-vertex": GoogleVertexOptions;265
"mistral-conversations": MistralOptions;266
"bedrock-converse-stream": BedrockOptions;267
"pi-messages": PiMessagesOptions;268
}270
/**271
* Full stream options for an API. Known APIs resolve to their concrete option272
* type; custom API strings fall back to the generic shape.273
*/274
export type ApiStreamOptions<TApi extends Api> = TApi extends keyof ApiOptionsMap275
? ApiOptionsMap[TApi]276
: StreamOptions & Record<string, unknown>;278
/**279
* The uniform stream contract of an API implementation module: every module280
* under `src/api/` exports `stream` and `streamSimple`; capable modules may also281
* export deferred-response methods. Lazy wrappers (`lazyApi()`) and provider282
* factories pass these around as values. This is the untyped dispatch shape;283
* per-API option typing lives on the implementation modules themselves and on284
* `Provider.stream()` via `ApiStreamOptions`.285
*/286
export interface ProviderStreams {287
stream(model: Model<Api>, context: TranscriptContext, options?: StreamOptions): AssistantMessageEventStream;288
streamSimple(289
model: Model<Api>,290
context: TranscriptContext,291
options?: SimpleStreamOptions,292
): AssistantMessageEventStream;293
fetchDeferred?(294
model: Model<Api>,295
handle: DeferredHandle,296
options?: DeferredFetchOptions,297
): AssistantMessageEventStream;298
cancelDeferred?(model: Model<Api>, handle: DeferredHandle, options?: DeferredCancelOptions): Promise<void>;299
}301
/**302
* The uniform contract of an image-generation API implementation module:303
* every image API module under `src/api/` exports exactly `generateImages`,304
* so the module itself satisfies this interface. Lazy wrappers and305
* `createProvider({ images })` pass these around as values.306
*/307
export interface ProviderImages {308
generateImages(309
model: ImageModel<ImageApi>,310
context: ImagesContext,311
options?: ImagesOptions,312
): Promise<AssistantImages>;313
}315
/** The uniform contract implemented by classifier API modules. */316
export interface ProviderClassifier {317
classify(318
model: ClassifierModel<ClassifierApi>,319
context: ClassifierContext,320
options?: ClassifierOptions,321
): Promise<ClassifierResult>;322
}324
export interface ClassifierOptions extends ProviderRequestOptions<ClassifierModel<ClassifierApi>> {325
/**326
* Divides the answer logits by this value before they are normalized into probabilities.327
* Values above 1 soften the distribution; values below 1 sharpen it. Must be positive.328
* APIs that cannot apply it ignore it.329
*/330
temperature?: number;331
}333
export interface ImagesOptions extends ProviderRequestOptions<ImageModel<ImageApi>> {334
/**335
* Optional metadata to include in API requests.336
* Providers extract the fields they understand and ignore the rest.337
*/338
metadata?: Record<string, unknown>;339
}341
export type ProviderImagesOptions = ImagesOptions & Record<string, unknown>;343
export interface AnthropicAllowedFallbackModel {344
provider: ProviderId;345
model: string;346
cost: ModelCost;347
}349
// Unified options with reasoning passed to streamSimple() and completeSimple()350
export interface SimpleStreamOptions extends StreamOptions {351
/** Provider-neutral tool selection for simple requests. When omitted, adapters use provider-specific behavior. */352
toolChoice?: ToolChoice;353
reasoning?: ThinkingLevel;354
/** Ask a capable provider to return a durable handle and continue the request asynchronously. */355
deferred?: boolean | { window?: "15m" | "1h" | "24h" };356
/** Custom token budgets for thinking levels (token-based providers only) */357
thinkingBudgets?: ThinkingBudgets;358
}360
// Generic StreamFunction with typed options.361
//362
// Contract:363
// - Receives a normalized transcript: the system prompt and tools live in the364
// leading system message, never on the context itself.365
// - Must return an AssistantMessageEventStream.366
// - Direct streamSimple() calls may throw synchronously when request auth is367
// missing. Once a stream is returned, request/model/runtime failures should368
// be encoded in that stream.369
// - Error termination must produce an AssistantMessage with stopReason370
// "error" or "aborted" and errorMessage, emitted via the stream protocol.371
export type StreamFunction<TApi extends Api = Api, TOptions extends StreamOptions = StreamOptions> = (372
model: Model<TApi>,373
context: TranscriptContext,374
options?: TOptions,375
) => AssistantMessageEventStream;377
export type ImagesFunction<TOptions extends ImagesOptions = ImagesOptions> = (378
model: ImageModel<ImageApi>,379
context: ImagesContext,380
options?: TOptions,381
) => Promise<AssistantImages>;383
export type ClassifierFunction<TOptions extends ClassifierOptions = ClassifierOptions> = (384
model: ClassifierModel<ClassifierApi>,385
context: ClassifierContext,386
options?: TOptions,387
) => Promise<ClassifierResult>;389
export interface TextSignatureV1 {390
v: 1;391
id: string;392
phase?: "commentary" | "final_answer";393
}395
export interface TextContent {396
type: "text";397
text: string;398
textSignature?: string; // e.g., for OpenAI responses, message metadata (legacy id string or TextSignatureV1 JSON)399
}401
export interface ThinkingContent {402
type: "thinking";403
thinking: string;404
thinkingSignature?: string; // Provider-specific opaque or serialized reasoning replay data405
/** When true, the thinking content was redacted by safety filters. The opaque406
* encrypted payload is stored in `thinkingSignature` so it can be passed back407
* to the API for multi-turn continuity. */408
redacted?: boolean;409
}411
export interface ImageContent {412
type: "image";413
data: string; // base64 encoded image data414
mimeType: string; // e.g., "image/jpeg", "image/png"415
}417
export interface ToolCall {418
type: "toolCall";419
id: string;420
name: string;421
arguments: JsonObject;422
thoughtSignature?: string; // Google-specific: opaque signature for reusing thought context423
/** OpenAI Responses namespace for calls to dynamically loaded or namespaced tools. */424
namespace?: string;425
}427
export interface Usage {428
input: number;429
output: number;430
cacheRead: number;431
cacheWrite: number;432
/** Subset of `cacheWrite` written with 1h retention. Only Anthropic reports this split. */433
cacheWrite1h?: number;434
/**435
* Reasoning/thinking tokens, when the provider reports them. This is a subset of436
* `output`: `output` already includes these tokens. Set to a number (possibly 0) by437
* providers that expose a reasoning breakdown; left undefined by providers that don't.438
*/439
reasoning?: number;440
totalTokens: number;441
cost: {442
input: number;443
output: number;444
cacheRead: number;445
cacheWrite: number;446
total: number;447
};448
}450
export type StopReason = "pending" | "stop" | "length" | "toolUse" | "error" | "aborted" | "deferred";452
export type JsonValue = null | boolean | number | string | readonly JsonValue[] | JsonObject;453
export type JsonObject = { [key: string]: JsonValue };455
type IsAny<T> = 0 extends 1 & T ? true : false;456
type IsExactlyJsonValue<T> = [T] extends [JsonValue] ? ([JsonValue] extends [T] ? true : false) : false;457
type IsJsonProperty<T> = IsAny<T> extends true458
? false459
: unknown extends T460
? false461
: [Exclude<T, undefined>] extends [never]462
? true463
: IsJsonCompatible<Exclude<T, undefined>>;464
type InvalidJsonKeys<T extends object> = {465
[TKey in keyof T]-?: TKey extends string | number ? (IsJsonProperty<T[TKey]> extends true ? never : TKey) : TKey;466
}[keyof T];467
type IsJsonCompatible<T> = IsAny<T> extends true468
? false469
: unknown extends T470
? false471
: IsExactlyJsonValue<T> extends true472
? true473
: T extends null | boolean | number | string474
? true475
: T extends undefined476
? false477
: T extends readonly (infer TItem)[]478
? IsJsonCompatible<TItem>479
: T extends (...args: never[]) => unknown480
? false481
: T extends object482
? [InvalidJsonKeys<T>] extends [never]483
? true484
: false485
: false;487
/** The JSON representation of a typed in-memory value. Optional object properties remain optional. */488
export type JsonRepresentation<T> = IsAny<T> extends true489
? JsonValue490
: unknown extends T491
? JsonValue492
: [T] extends [JsonValue]493
? T494
: T extends readonly unknown[]495
? { [TKey in keyof T]: JsonRepresentation<Exclude<T[TKey], undefined>> }496
: T extends object497
? { [TKey in keyof T]: JsonRepresentation<Exclude<T[TKey], undefined>> }498
: never;500
export interface DeferredHandle {501
provider: string;502
modelId: string;503
api: string;504
/** Provider token, such as a response id or batch id plus row id. */505
id: string;506
expiresAt?: number;507
pollAfterMs?: number;508
/** Provider conversion data required to reconstruct the final assistant message. */509
data?: JsonValue;510
}512
/**513
* System instructions and tool declarations at one point in the transcript.514
*515
* The leading system message is the system prompt. Later system messages change it:516
* `content` adds instructions from that point on, `sections` replace or remove named517
* prompt sections, and `toolsAdded`/`toolsRemoved` change the tool set. Replaying518
* every system message in order yields the current prompt and tools. Providers that519
* accept system messages mid-conversation send each one in place; other providers520
* rebuild the leading system message from the replayed state.521
*/522
export interface SystemMessage {523
role: "system";524
/** Instruction text. On the leading message this is the base prompt; later, additional instructions. */525
content: string | TextContent[];526
/**527
* Named, ordered prompt sections rendered verbatim after `content`. The leading message528
* declares them; later messages replace sections by name, and `null` removes one. Keep529
* each section self-delimiting (a tag, a heading) so the model can relate an update to530
* the original. Avoid integer-like names; JSON objects reorder those.531
*/532
sections?: Record<string, string | null>;533
/** Complete definitions of tools that become available at this point. */534
toolsAdded?: Tool[];535
/** Tools that stop being available at this point. */536
toolsRemoved?: ToolReference[];537
timestamp: number; // Unix timestamp in milliseconds538
}540
export interface UserMessage {541
role: "user";542
content: string | (TextContent | ImageContent)[];543
timestamp: number; // Unix timestamp in milliseconds544
}546
export interface AssistantMessage {547
role: "assistant";548
content: (TextContent | ThinkingContent | ToolCall)[];549
api: Api;550
provider: ProviderId;551
model: string;552
responseModel?: string; // Concrete model reported by the provider when different from the requested `model`553
responseId?: string; // Provider-specific response/message identifier when the upstream API exposes one554
/** Exact provider-native effort level used for this response. Absent for legacy or unmanaged responses. */555
providerThinkingLevel?: string;556
/** Pi thinking level the agent loop requested for this response. Absent outside the agent loop and for legacy responses. */557
thinkingLevel?: ModelThinkingLevel;558
diagnostics?: AssistantMessageDiagnostic[]; // Redacted provider/runtime diagnostics for failures and recoveries.559
usage: Usage;560
stopReason: StopReason;561
deferred?: DeferredHandle;562
errorMessage?: string;563
rawStopReason?: string;564
/**565
* Provider indication of whether the model explicitly ended its turn.566
* Preserved for debugging and does not currently affect agent control flow.567
*/568
endTurn?: boolean;569
timestamp: number; // Unix timestamp in milliseconds570
}572
/** A tool call that another tool made while it ran, for example from a codemode script. */573
export interface NestedToolCallRecord {574
id: string;575
name: string;576
/** Omitted when over the size limits; `argumentsBytes` then gives their size. */577
arguments?: JsonObject;578
/** UTF-8 size of the arguments as JSON, set when `arguments` is omitted. */579
argumentsBytes?: number;580
/** `unfinished`: the call was still running when the calling tool finished. */581
status: "ok" | "error" | "unfinished";582
durationMs?: number;583
/** Error text, truncated. */584
error?: string;585
}587
/** Bounded record of the nested calls a tool made. Results are not recorded. */588
export interface NestedToolCalls {589
calls: NestedToolCallRecord[];590
/** False when calls were dropped, arguments omitted, or calls had not finished. */591
complete: boolean;592
}594
export type ToolResultMessage<TDetails = JsonValue> = IsJsonCompatible<TDetails> extends true595
? {596
role: "toolResult";597
toolCallId: string;598
toolName: string;599
content: (TextContent | ImageContent)[]; // Supports text and images600
details?: JsonRepresentation<TDetails>;601
/** Usage from the tool execution itself, if available. Not part of main LLM context accounting. */602
usage?: Usage;603
/** Calls this tool made to other tools. Kept for the session record; not sent to the model. */604
nestedCalls?: NestedToolCalls;605
isError: boolean;606
timestamp: number; // Unix timestamp in milliseconds607
}608
: never;610
export type Message = SystemMessage | UserMessage | AssistantMessage | ToolResultMessage;612
export type ImagesInputContent = TextContent | ImageContent;613
export type ImagesOutputContent = TextContent | ImageContent;615
export interface ImagesContext {616
input: ImagesInputContent[];617
}619
export type ImagesStopReason = "stop" | "error" | "aborted";621
export interface AssistantImages {622
api: ImageApi;623
provider: ProviderId;624
model: string;625
output: ImagesOutputContent[];626
responseId?: string;627
usage?: Usage;628
stopReason: ImagesStopReason;629
errorMessage?: string;630
timestamp: number; // Unix timestamp in milliseconds631
}633
export interface ClassifierChoiceQuestion {634
type: "choice";635
instructions: string;636
criteria: Record<string, string>;637
}639
export interface ClassifierScoreQuestion {640
type: "score";641
instructions: string;642
criteria: string[];643
}645
export interface ClassifierBoolQuestion {646
type: "bool";647
instructions: string;648
criteria: { true: string; false: string };649
}651
export type ClassifierQuestion = ClassifierChoiceQuestion | ClassifierScoreQuestion | ClassifierBoolQuestion;653
export interface ClassifierContext {654
state: JsonObject;655
questions: Record<string, ClassifierQuestion>;656
}658
export interface ClassifierChoiceAnswer {659
type: "choice";660
choice: string;661
probabilities: Record<string, number>;662
confidence: number;663
}665
export interface ClassifierScoreAnswer {666
type: "score";667
score: number;668
confidence: number;669
}671
export interface ClassifierBoolAnswer {672
type: "bool";673
probability: number;674
}676
export type ClassifierAnswer = ClassifierChoiceAnswer | ClassifierScoreAnswer | ClassifierBoolAnswer;677
export type ClassifierStopReason = "stop" | "error" | "aborted";679
export interface ClassifierResult {680
api: ClassifierApi;681
provider: ProviderId;682
model: string;683
answers: Record<string, ClassifierAnswer>;684
/** Token usage and its cost at the model's catalog price, when the service reports token counts. */685
usage?: Usage;686
stopReason: ClassifierStopReason;687
errorMessage?: string;688
timestamp: number; // Unix timestamp in milliseconds689
}691
import type { TSchema } from "typebox";693
/** OpenAI grammar variants for constrained sampling. */694
export type GrammarFormat = "openai_lark" | "openai_regex";696
export type GrammarVariants = Partial<Record<GrammarFormat, string>>;698
/**699
* Optional provider-side constrained sampling configs for a tool.700
*701
* The `json_schema` value roughly maps to the concept of `strict` in APIs which is702
* implemented as json-schema constrained sampling by APIs. Grammar variants let703
* callers provide provider-specific encodings of the same intended language.704
*/705
export type ConstrainedSamplingConfig =706
| {707
type: "json_schema";708
strict: "prefer" | "require";709
}710
| {711
type: "grammar";712
variants: GrammarVariants;713
};715
export interface Tool<TParameters extends TSchema = TSchema> {716
name: string;717
description: string;718
parameters: TParameters;719
constrainedSampling?: false | ConstrainedSamplingConfig;720
}722
export interface ToolReference {723
name: string;724
}726
/**727
* Request input accepted by the public stream entry points (`Models.stream()`,728
* `streamSimple()`, ...). `systemPrompt` and `tools` are shorthand for a leading729
* system message; `normalizeContext()` folds them into one before the request730
* reaches a provider.731
*/732
export interface Context {733
systemPrompt?: string;734
messages: Message[];735
tools?: Tool[];736
}738
declare const transcriptContextBrand: unique symbol;740
/**741
* Normalized request context passed to providers and API implementations. The742
* prompt and tool declarations are carried by the transcript's system messages.743
* Only `normalizeContext()` produces this type, so a raw `Context` cannot reach744
* provider code by accident.745
*/746
export type TranscriptContext = {747
messages: Message[];748
readonly [transcriptContextBrand]: true;749
};751
/**752
* Event protocol for AssistantMessageEventStream.753
*754
* Successful streams emit `start` before partial updates and terminate with755
* `done`. A stream may terminate directly with `error` when request setup fails756
* before generation starts; after `start`, failures also terminate with `error`.757
* Direct `streamSimple()` calls throw synchronously when request auth is missing.758
* Updates and `done` must never appear before `start`.759
*760
* `partial` is the shared live response-so-far helper, not an event-time761
* snapshot. Text and thinking blocks are empty when their `*_start` event is762
* emitted and grow only through their corresponding `*_delta` events until the763
* authoritative `*_end`. Redacted thinking may be complete at start and emit no764
* deltas. Tool-call arguments at `toolcall_start` are provider-specific;765
* `toolcall_delta` carries subsequent JSON updates.766
*/767
export type AssistantMessageEvent =768
| { type: "start"; partial: AssistantMessage }769
| { type: "text_start"; contentIndex: number; partial: AssistantMessage }770
| { type: "text_delta"; contentIndex: number; delta: string; partial: AssistantMessage }771
| { type: "text_end"; contentIndex: number; content: string; partial: AssistantMessage }772
| { type: "thinking_start"; contentIndex: number; partial: AssistantMessage }773
| { type: "thinking_delta"; contentIndex: number; delta: string; partial: AssistantMessage }774
| { type: "thinking_end"; contentIndex: number; content: string; partial: AssistantMessage }775
| { type: "toolcall_start"; contentIndex: number; partial: AssistantMessage }776
| { type: "toolcall_delta"; contentIndex: number; delta: string; partial: AssistantMessage }777
| { type: "toolcall_end"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage }778
| {779
type: "done";780
reason: Extract<StopReason, "stop" | "length" | "toolUse" | "deferred">;781
message: AssistantMessage;782
}783
| { type: "error"; reason: Extract<StopReason, "aborted" | "error">; error: AssistantMessage };785
/**786
* Compatibility settings for OpenAI-compatible completions APIs.787
* Use this to override URL-based auto-detection for custom providers.788
*/789
export interface OpenAICompletionsCompat {790
/** Whether the provider supports the `store` field. Default: auto-detected from URL. */791
supportsStore?: boolean;792
/** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */793
supportsDeveloperRole?: boolean;794
/** Whether the provider supports `reasoning_effort`. Default: auto-detected from URL. */795
supportsReasoningEffort?: boolean;796
/** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */797
supportsUsageInStreaming?: boolean;798
/** Whether streamed responses include `finish_reason`. When false, pi infers `stop` or `toolUse` when the stream ends. Default: true. */799
supportsFinishReason?: boolean;800
/** Which field to use for max tokens. Default: auto-detected from URL. */801
maxTokensField?: "max_completion_tokens" | "max_tokens";802
/** Whether tool results require the `name` field. Default: auto-detected from URL. */803
requiresToolResultName?: boolean;804
/** Whether a user message after tool results requires an assistant message in between. Default: auto-detected from URL. */805
requiresAssistantAfterToolResult?: boolean;806
/** Whether thinking blocks must be converted to text blocks with <thinking> delimiters. Default: auto-detected from URL. */807
requiresThinkingAsText?: boolean;808
/** Whether all replayed assistant messages must include an empty reasoning_content field when reasoning is enabled. Default: auto-detected from URL. */809
requiresReasoningContentOnAssistantMessages?: boolean;810
/** Format for reasoning/thinking parameter. "openai" uses reasoning_effort, "openrouter" uses reasoning: { effort }, "deepseek" uses thinking: { type } plus reasoning_effort when supported, "together" uses reasoning: { enabled } plus reasoning_effort when supported, "baseten" uses configurable chat_template_args plus reasoning_effort when supported, "zai" uses thinking: { type }, "qwen" uses top-level enable_thinking: boolean, "qwen-chat-template" uses chat_template_kwargs.enable_thinking and preserve_thinking, "chat-template" uses configurable chat_template_kwargs, "string-thinking" uses top-level thinking: string, and "ant-ling" uses reasoning: { effort } only when the mapped effort is non-null. Default: "openai". */811
thinkingFormat?:812
| "openai"813
| "openrouter"814
| "deepseek"815
| "together"816
| "baseten"817
| "zai"818
| "qwen"819
| "chat-template"820
| "qwen-chat-template"821
| "string-thinking"822
| "ant-ling";823
/** Kwargs to send as `chat_template_kwargs` when `thinkingFormat` is `chat-template`. Use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for pi-controlled thinking values. */824
chatTemplateKwargs?: Record<string, ChatTemplateKwargValue>;825
/** Arguments to send as `chat_template_args` when `thinkingFormat` is `baseten`. Use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for pi-controlled thinking values. */826
chatTemplateArgs?: Record<string, ChatTemplateKwargValue>;827
/** OpenRouter-compatible routing preferences sent as the `provider` request field. */828
openRouterRouting?: OpenRouterRouting;829
/** Vercel AI Gateway routing preferences. Only used when baseUrl points to Vercel AI Gateway. */830
vercelGatewayRouting?: VercelGatewayRouting;831
/** Whether z.ai supports top-level `tool_stream: true` for streaming tool call deltas. Default: false. */832
zaiToolStream?: boolean;833
/**834
* Top-level request field used to cap reasoning tokens from `thinkingBudgets`.835
* Reasoning and the answer share `max_tokens` on these endpoints, so without a budget a836
* reasoning-heavy turn can consume the whole response and emit no answer.837
* `"thinking_token_budget"` is vLLM, `"thinking_budget"` is Qwen/DashScope/SGLang,838
* `"thinking_budget_tokens"` is llama.cpp. Off by default; not set on the generated catalog.839
*/840
thinkingTokenBudgetField?: ThinkingTokenBudgetField;841
/** Alias for `thinkingTokenBudgetField: "thinking_token_budget"` (vLLM). Prefer `thinkingTokenBudgetField`. Default: false. */842
supportsThinkingTokenBudget?: boolean;843
/** Whether the provider supports OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */844
supportsOpenAIGrammarTools?: boolean;845
/** Whether the exact model accepts system or developer messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false; the generated model catalog enables it for verified models. */846
supportsMidConvoSystemMessages?: boolean;847
/** Whether system messages can introduce additional tools mid-conversation. Requires `supportsMidConvoSystemMessages`. Default: false; the generated model catalog enables it for capable models. */848
supportsMidConvoToolAdditions?: boolean;849
/** Whether the provider supports the `strict` field in tool definitions. Default: false; generated capable models enable it explicitly. */850
supportsStrictMode?: boolean;851
/** Cache control convention for prompt caching. "anthropic" applies Anthropic-style `cache_control` markers to the system prompt, last tool definition, and last user, assistant, or tool-result text content. */852
cacheControlFormat?: "anthropic";853
/** Whether to send session-affinity data from `options.sessionId`. Default: true for OpenRouter endpoints, false otherwise. */854
sendSessionAffinityHeaders?: boolean;855
/** Session-affinity header format: `openai` sends `session_id`, `x-client-request-id`, and `x-session-affinity`; `openai-nosession` sends `x-client-request-id` and `x-session-affinity`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */856
sessionAffinityFormat?: SessionAffinityFormat;857
/** Whether the provider supports long prompt cache retention (`prompt_cache_retention: "24h"` or Anthropic-style `cache_control.ttl: "1h"`, depending on format). Default: true. */858
supportsLongCacheRetention?: boolean;859
/**860
* vLLM scheduler priority sent as the top-level `priority` request field (lower values are861
* handled earlier; server default 0). Only meaningful when vLLM runs with862
* `--scheduling-policy priority`; useful for keeping background/batch work from stalling863
* interactive sessions. Off by default; not set on the generated catalog.864
*/865
vllmPriority?: number;866
}868
/** Compatibility settings for OpenAI Responses APIs. */869
export interface OpenAIResponsesCompat {870
/** Whether the provider supports the `developer` role (vs `system`). Default: true. */871
supportsDeveloperRole?: boolean;872
/** Whether the exact model accepts developer or system messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false; the generated model catalog enables it for verified models. */873
supportsMidConvoSystemMessages?: boolean;874
/** Session-affinity header format: `openai` sends `session_id` and `x-client-request-id`; `openai-nosession` sends `x-client-request-id`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */875
sessionAffinityFormat?: SessionAffinityFormat;876
/** Whether the provider supports long prompt cache retention. This uses `prompt_cache_options.ttl: "30m"` on GPT-5.6+ and `prompt_cache_retention: "24h"` on earlier models. Default: true. */877
supportsLongCacheRetention?: boolean;878
/** Whether the provider supports strict JSON-schema function tools. Defaults are API-specific; generated OpenAI models enable it explicitly. */879
supportsStrictMode?: boolean;880
/** Whether to emit OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */881
supportsOpenAIGrammarTools?: boolean;882
/** Whether the model supports message-anchored `additional_tools` input items. Default: false. */883
supportsAdditionalTools?: boolean;884
/** Whether the model supports client-executed tool search for transcript-anchored additions. Default: false. */885
supportsToolSearch?: boolean;886
/** Whether the model accepts `prompt_cache_options` (OpenAI GPT-5.6+ prompt caching). Older OpenAI models reject the parameter. Default: false. */887
supportsExplicitPromptCacheMode?: boolean;888
/** Whether the provider accepts the `max_output_tokens` parameter. Some Codex-protocol gateways reject it. Default: true. */889
supportsMaxOutputTokens?: boolean;890
}892
/** Compatibility settings for Anthropic Messages-compatible APIs. */893
export interface AnthropicMessagesCompat {894
/**895
* Whether the provider accepts per-tool `eager_input_streaming`.896
* When false, the Anthropic provider omits `tools[].eager_input_streaming`897
* and sends the legacy `fine-grained-tool-streaming-2025-05-14` beta header898
* for tool-enabled requests.899
* Default: true.900
*/901
supportsEagerToolInputStreaming?: boolean;902
/** Whether the provider supports Anthropic long cache retention (`cache_control.ttl: "1h"`). Default: true. */903
supportsLongCacheRetention?: boolean;904
/**905
* Whether to send the `x-session-affinity` header from `options.sessionId`906
* when caching is enabled. Required for providers like Fireworks that use907
* session affinity for prompt cache routing (requests to the same replica908
* maximize cache hits).909
* Default: false.910
*/911
sendSessionAffinityHeaders?: boolean;912
/** Session-affinity format. `"openrouter"` sends `x-session-id`; when unset, sends `x-session-affinity`. */913
sessionAffinityFormat?: "openrouter";914
/**915
* Whether the provider supports Anthropic-style `cache_control` markers on916
* tool definitions. When false, `cache_control` is omitted from tool params.917
* Some Anthropic-compatible providers (e.g., Fireworks) do not support this918
* field on tools and may reject or ignore it.919
* Default: true.920
*/921
supportsCacheControlOnTools?: boolean;922
/**923
* Whether the model accepts the Anthropic `temperature` request field.924
* Claude Opus 4.7+ rejects non-default temperature values.925
* Default: true.926
*/927
supportsTemperature?: boolean;928
/**929
* Whether to force adaptive thinking (`thinking.type: "adaptive"` plus930
* `output_config.effort`) regardless of the model id. Built-in models that931
* require adaptive thinking set this in generated metadata. Custom932
* Anthropic-compatible providers can set this to `true` for any model whose933
* upstream requires the adaptive format. Set to `false` to934
* opt out on overridden built-in models.935
* Default: false.936
*/937
forceAdaptiveThinking?: boolean;938
/** Whether to replay empty thinking signatures as `signature: ""` instead of converting thinking to text. Default: false. */939
allowEmptySignature?: boolean;940
/** Whether the provider supports Anthropic strict tool schemas. Default: false; generated Anthropic models enable it explicitly. */941
supportsStrictTools?: boolean;942
/** Whether the exact model transport supports effort-only system messages and thinking binding controls. Default: false. */943
supportsMidConvoEffort?: boolean;944
/** Whether the exact model accepts system-role messages inside the conversation. When false, later system messages are folded into the top-level system prompt. Default: false. */945
supportsMidConvoSystemMessages?: boolean;946
/** Whether the exact model accepts mid-conversation `tool_addition` and `tool_removal` blocks. Requires `supportsMidConvoSystemMessages`. Default: false. */947
supportsMidConvoToolChanges?: boolean;948
/**949
* Models Anthropic accepts in `fallbacks` for server-side refusal fallback,950
* with local pricing metadata for returned fallback responses. When absent or951
* empty, callers must omit `fallbacks`; Anthropic rejects the field for models952
* with no permitted fallback targets.953
*/954
allowedFallbackModels?: AnthropicAllowedFallbackModel[];955
}957
/** Compatibility settings for Amazon Bedrock models. */958
export interface BedrockCompat {959
/** Whether the model supports Bedrock strict tool schemas. Default: false. */960
supportsStrictMode?: boolean;961
}963
/** Compatibility settings for the Mistral chat API. */964
export interface MistralConversationsCompat {965
/** Whether the exact model accepts system messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false. */966
supportsMidConvoSystemMessages?: boolean;967
}969
/**970
* OpenRouter provider routing preferences.971
* Controls which upstream providers OpenRouter routes requests to.972
* Sent as the `provider` field in the OpenRouter API request body.973
* @see https://openrouter.ai/docs/guides/routing/provider-selection974
*/975
export interface OpenRouterRouting {976
/** Whether to allow backup providers to serve requests. Default: true. */977
allow_fallbacks?: boolean;978
/** Whether to filter providers to only those that support all parameters in the request. Default: false. */979
require_parameters?: boolean;980
/** Data collection setting. "allow" (default): allow providers that may store/train on data. "deny": only use providers that don't collect user data. */981
data_collection?: "deny" | "allow";982
/** Whether to restrict routing to only ZDR (Zero Data Retention) endpoints. */983
zdr?: boolean;984
/** Whether to restrict routing to only models that allow text distillation. */985
enforce_distillable_text?: boolean;986
/** An ordered list of provider names/slugs to try in sequence, falling back to the next if unavailable. */987
order?: string[];988
/** List of provider names/slugs to exclusively allow for this request. */989
only?: string[];990
/** List of provider names/slugs to skip for this request. */991
ignore?: string[];992
/** A list of quantization levels to filter providers by (e.g., ["fp16", "bf16", "fp8", "fp6", "int8", "int4", "fp4", "fp32"]). */993
quantizations?: string[];994
/** Sorting strategy. Can be a string (e.g., "price", "throughput", "latency") or an object with `by` and `partition`. */995
sort?:996
| string997
| {998
/** The sorting metric: "price", "throughput", "latency". */999
by?: string;1000
/** Partitioning strategy: "model" (default) or "none". */1001
partition?: string | null;1002
};1003
/** Maximum price per million tokens (USD). */1004
max_price?: {1005
/** Price per million prompt tokens. */1006
prompt?: number | string;1007
/** Price per million completion tokens. */1008
completion?: number | string;1009
/** Price per image. */1010
image?: number | string;1011
/** Price per audio unit. */1012
audio?: number | string;1013
/** Price per request. */1014
request?: number | string;1015
};1016
/** Preferred minimum throughput (tokens/second). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */1017
preferred_min_throughput?:1018
| number1019
| {1020
/** Minimum tokens/second at the 50th percentile. */1021
p50?: number;1022
/** Minimum tokens/second at the 75th percentile. */1023
p75?: number;1024
/** Minimum tokens/second at the 90th percentile. */1025
p90?: number;1026
/** Minimum tokens/second at the 99th percentile. */1027
p99?: number;1028
};1029
/** Preferred maximum latency (seconds). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */1030
preferred_max_latency?:1031
| number1032
| {1033
/** Maximum latency in seconds at the 50th percentile. */1034
p50?: number;1035
/** Maximum latency in seconds at the 75th percentile. */1036
p75?: number;1037
/** Maximum latency in seconds at the 90th percentile. */1038
p90?: number;1039
/** Maximum latency in seconds at the 99th percentile. */1040
p99?: number;1041
};1042
}1044
/**1045
* Vercel AI Gateway routing preferences.1046
* Controls which upstream providers the gateway routes requests to.1047
* @see https://vercel.com/docs/ai-gateway/models-and-providers/provider-options1048
*/1049
export interface VercelGatewayRouting {1050
/** List of provider slugs to exclusively use for this request (e.g., ["bedrock", "anthropic"]). */1051
only?: string[];1052
/** List of provider slugs to try in order (e.g., ["anthropic", "openai"]). */1053
order?: string[];1054
}1056
export interface ModelCostRates {1057
input: number; // $/million tokens1058
output: number; // $/million tokens1059
cacheRead: number; // $/million tokens1060
cacheWrite: number; // $/million tokens1061
}1063
export interface ModelCostTier extends ModelCostRates {1064
/** Use this tier for requests whose total input usage exceeds this token count. */1065
inputTokensAbove: number;1066
}1068
export interface ModelCost extends ModelCostRates {1069
/** Request-wide pricing tiers. The highest matching input threshold applies to the full request. */1070
tiers?: ModelCostTier[];1071
}1073
export interface ModelImageResizeOptions {1074
maxWidth?: number;1075
maxHeight?: number;1076
/** Maximum base64-encoded payload size in bytes. */1077
maxBytes?: number;1078
jpegQuality?: number;1079
}1081
export interface ModelImageInputLimits {1082
/** Cache-safe resize profile applied before a new image enters conversation history. */1083
resize?: ModelImageResizeOptions;1084
/** Maximum images accepted in one provider message. */1085
maxPerMessage?: number;1086
/** Maximum images accepted across one provider request. */1087
maxPerRequest?: number;1088
}1090
export interface ModelInputLimits {1091
/** Maximum serialized provider request size in bytes. */1092
maxRequestBytes?: number;1093
images?: ModelImageInputLimits;1094
}1096
/** Fields shared by every catalog entry, regardless of what you can do with it. */1097
export interface BaseModel<TApi extends string> {1098
id: string;1099
name: string;1100
api: TApi;1101
provider: ProviderId;1102
baseUrl: string;1103
input: ("text" | "image")[];1104
/** Provider input limits and cache-safe preprocessing metadata. */1105
inputLimits?: ModelInputLimits;1106
cost: ModelCost;1107
headers?: Record<string, string>;1108
}1110
/** Chat model: usable with `stream()` and friends. */1111
export interface Model<TApi extends Api> extends BaseModel<TApi> {1112
/**1113
* Optional: chat is the default model type, so models without `type` are chat1114
* models. Narrow mixed model lists with `isModelType()` instead of comparing1115
* `type` directly.1116
*/1117
type?: "chat";1118
reasoning: boolean;1119
/**1120
* Maps pi thinking levels to provider/model-specific values.1121
* Missing keys use provider defaults. null marks a level as unsupported.1122
*/1123
thinkingLevelMap?: ThinkingLevelMap;1124
/** Prompt cache lifetimes per retention tier. Unset when the provider's cache behavior is unknown. */1125
promptCache?: ModelPromptCache;1126
contextWindow: number;1127
maxTokens: number;1128
/** Default sampling parameters for this model. See {@link StreamOptions.samplingParams}; per-request keys override these. */1129
samplingParams?: Record<string, unknown>;1130
/** Compatibility overrides for OpenAI-compatible APIs. If not set, auto-detected from baseUrl. */1131
compat?: TApi extends "openai-completions"1132
? OpenAICompletionsCompat1133
: TApi extends "openai-responses" | "azure-openai-responses" | "openai-codex-responses"1134
? OpenAIResponsesCompat1135
: TApi extends "anthropic-messages"1136
? AnthropicMessagesCompat1137
: TApi extends "bedrock-converse-stream"1138
? BedrockCompat1139
: TApi extends "mistral-conversations"1140
? MistralConversationsCompat1141
: never;1142
}1144
/** Image-generation model: usable with `generateImages()` only. */1145
export interface ImageModel<TApi extends ImageApi> extends BaseModel<TApi> {1146
type: "image";1147
/** Output modalities. Always includes `"image"`; `"text"` means the model can also return text blocks. */1148
output: ("text" | "image")[];1149
}1151
/** Structured classifier model: usable with `classify()` only. */1152
export interface ClassifierModel<TApi extends ClassifierApi> extends BaseModel<TApi> {1153
type: "classifier";1154
contextWindow: number;1155
}1157
/** Model shape for each model type. */1158
export interface ModelTypeMap {1159
chat: Model<Api>;1160
image: ImageModel<ImageApi>;1161
classifier: ClassifierModel<ClassifierApi>;1162
}1164
/** What a catalog entry is for. Decides which `Models` operation accepts it. */1165
export type ModelType = keyof ModelTypeMap;1167
/** Anything a provider can list. Narrow with `isModelType()`. */1168
export type AnyModel = ModelTypeMap[ModelType];