diff --git a/config/assertion-safety-baseline.txt b/config/assertion-safety-baseline.txt index 4afca275ff69..c569beaab794 100644 --- a/config/assertion-safety-baseline.txt +++ b/config/assertion-safety-baseline.txt @@ -523,7 +523,7 @@ extensions/google/oauth-token-shared.ts 1 extensions/google/onboard.ts 3 extensions/google/provider-registration.ts 2 extensions/google/realtime-voice-provider.ts 6 -extensions/google/transport-stream.ts 15 +extensions/google/transport-stream.ts 13 extensions/google/vertex-adc.ts 6 extensions/google/video-generation-provider.ts 13 extensions/googlechat/src/accounts.ts 1 @@ -1456,10 +1456,10 @@ packages/ai/src/providers/anthropic-server-fallback.ts 3 packages/ai/src/providers/anthropic-thinking-replay.ts 2 packages/ai/src/providers/anthropic-tool-projection.ts 2 packages/ai/src/providers/anthropic-usage.ts 3 -packages/ai/src/providers/anthropic.ts 12 +packages/ai/src/providers/anthropic.ts 7 packages/ai/src/providers/azure-openai-responses.ts 1 packages/ai/src/providers/clean-for-gemini.ts 15 -packages/ai/src/providers/google-shared.ts 5 +packages/ai/src/providers/google-shared.ts 3 packages/ai/src/providers/mistral.ts 6 packages/ai/src/providers/openai-chatgpt-responses-protocol.ts 1 packages/ai/src/providers/openai-chatgpt-responses.ts 27 @@ -1478,7 +1478,7 @@ packages/ai/src/providers/tool-schema-json-projection.ts 2 packages/ai/src/stream.ts 1 packages/ai/src/transports/anthropic-compaction-replay.ts 2 packages/ai/src/transports/anthropic-payload-policy.ts 7 -packages/ai/src/transports/anthropic-transport-stream.ts 15 +packages/ai/src/transports/anthropic-transport-stream.ts 5 packages/ai/src/transports/openai-compatible-conversation-turn.ts 2 packages/ai/src/transports/openai-completions-compat.ts 2 packages/ai/src/transports/openai-completions-params.ts 9 diff --git a/config/max-lines-baseline.txt b/config/max-lines-baseline.txt index 7991836efd87..05d4370a060d 100644 --- a/config/max-lines-baseline.txt +++ b/config/max-lines-baseline.txt @@ -272,7 +272,6 @@ packages/agent-core/src/harness/compaction/compaction.ts packages/ai/src/providers/agent-tools-parameter-schema.ts packages/ai/src/providers/anthropic.test.ts packages/ai/src/providers/anthropic.ts -packages/ai/src/providers/google-shared.ts packages/ai/src/providers/mistral.ts packages/ai/src/providers/openai-chatgpt-responses.ts packages/ai/src/providers/openai-completions.test.ts diff --git a/docs/plugins/sdk-provider-plugins.md b/docs/plugins/sdk-provider-plugins.md index 1d51046afdb0..2cf21cc27220 100644 --- a/docs/plugins/sdk-provider-plugins.md +++ b/docs/plugins/sdk-provider-plugins.md @@ -663,6 +663,7 @@ catalog, API-key auth, and dynamic model resolution. - `openclaw/plugin-sdk/provider-model-shared` - `ProviderReplayFamily`, `buildProviderReplayFamilyHooks(...)`, and the raw replay builders (`buildOpenAICompatibleReplayPolicy`, `buildAnthropicReplayPolicyForModel`, `buildGoogleGeminiReplayPolicy`, `buildHybridAnthropicOrOpenAIReplayPolicy`). Also exports Gemini replay helpers (`sanitizeGoogleGeminiReplayHistory`, `resolveTaggedReasoningOutputMode`) and endpoint/model helpers (`resolveProviderEndpoint`, `normalizeProviderId`, `normalizeGooglePreviewModelId`). - `openclaw/plugin-sdk/provider-stream` - `ProviderStreamFamily`, `buildProviderStreamFamilyHooks(...)`, `composeProviderStreamWrappers(...)`, plus the shared OpenAI/Codex wrappers (`createOpenAIAttributionHeadersWrapper`, `createOpenAIFastModeWrapper`, `createOpenAIServiceTierWrapper`, `createOpenAIResponsesContextManagementWrapper`, `createCodexNativeWebSearchWrapper`), DeepSeek V4 OpenAI-compatible wrapper (`createDeepSeekV4OpenAICompatibleThinkingWrapper`), Anthropic Messages thinking prefill cleanup (`createAnthropicThinkingPrefillPayloadWrapper`), plain-text tool-call compat (`createPlainTextToolCallCompatWrapper`), and shared proxy/provider wrappers (`createOpenRouterWrapper`, `createToolStreamWrapper`, `createMinimaxFastModeWrapper`). - `openclaw/plugin-sdk/provider-stream-shared` - lightweight payload and event wrappers for hot provider paths, including `createOpenAICompatibleCompletionsThinkingOffWrapper`, `createPayloadPatchStreamWrapper`, `createPlainTextToolCallCompatWrapper`, `normalizeOpenAICompatibleReasoningPayload(...)`, and `setQwenChatTemplateThinking(...)`. + - `openclaw/plugin-sdk/provider-transport-runtime` - native Google wire helpers: `projectGoogleMessages(...)`, `convertGoogleTools(...)`, `requiresGoogleToolCallId(...)`, and `consumeGoogleGenerateContentStream(...)`. Prepare and normalize transcript routes before projection. Use `replay: "managed"` and stream `profile: "managed"` for managed SSE; the direct SDK uses `replay: "signed-parts"` and the default stream profile to preserve individual signed parts. Transport owners retain authentication, retries, HTTP cancellation, and trusted video admission; the reducer emits events and usage, and throws failures for the caller to finalize. - `openclaw/plugin-sdk/provider-tools` - `ProviderToolCompatFamily`, `buildProviderToolCompatFamilyHooks("deepseek" | "gemini" | "openai")`, and underlying provider schema helpers. For Gemini-family providers, keep the reasoning-output mode aligned with diff --git a/extensions/google/transport-stream.ts b/extensions/google/transport-stream.ts index f806e172824a..4cdc780c9b3f 100644 --- a/extensions/google/transport-stream.ts +++ b/extensions/google/transport-stream.ts @@ -1,7 +1,6 @@ // Google plugin module implements transport stream behavior. import type { StreamFn } from "openclaw/plugin-sdk/agent-core"; import { - calculateCost, getEnvApiKey, resolveProviderContext, type AssistantMessage, @@ -24,23 +23,21 @@ import { providerOperationRetryConfig, resolveProviderRequestHeaders, } from "openclaw/plugin-sdk/provider-http"; -import { notifyLlmRequestActivity } from "openclaw/plugin-sdk/provider-stream-shared"; import { buildGuardedModelFetch, - coerceTransportToolCallArguments, + consumeGoogleGenerateContentStream, + projectGoogleMessages, + requiresGoogleToolCallId, + convertGoogleTools, + type GoogleStreamChunk as GoogleSseChunk, createEmptyTransportUsage, createWritableTransportEventStream, - describeToolResultMediaPlaceholder, - extractToolResultText, failTransportStream, - finalizeTransportStream, mergeTransportHeaders, notifyProviderHttpResponse, sanitizeTransportPayloadText, - sortPromptCacheToolsByName, stripSystemPromptCacheBoundary, transformTransportMessages, - type WritableTransportStream, } from "openclaw/plugin-sdk/provider-transport-runtime"; import { isRecord, @@ -126,158 +123,15 @@ const GOOGLE_SSE_EVENT_BOUNDARY_RE = /(?:\r\n|\r(?!\n)|\n){2}/u; const GOOGLE_VERTEX_MODEL_RESOURCE_PREFIX = /^(?:projects\/[^/]+\/locations\/[^/]+\/)?publishers\/google\/models\//u; -type GoogleTransportContentBlock = - | { type: "text"; text: string; textSignature?: string } - | { type: "thinking"; thinking: string; thinkingSignature?: string } - | { - type: "toolCall"; - id: string; - name: string; - arguments: Record; - thoughtSignature?: string; - }; - -type MutableAssistantOutput = Omit & { - content: Array; - api: CanonicalGoogleTransportApi; -}; +type MutableAssistantOutput = AssistantMessage & { api: CanonicalGoogleTransportApi }; const GOOGLE_VERTEX_DEFAULT_API_VERSION = "v1"; -type GoogleSseChunk = { - responseId?: string; - modelVersion?: string; - promptFeedback?: { - blockReason?: string; - blockReasonMessage?: string; - }; - candidates?: Array<{ - content?: { - parts?: Array<{ - text?: string; - thought?: boolean; - thoughtSignature?: string; - functionCall?: { - id?: string; - name?: string; - args?: Record; - }; - }>; - }; - finishReason?: string; - finishMessage?: string; - }>; - usageMetadata?: { - promptTokenCount?: number; - cachedContentTokenCount?: number; - candidatesTokenCount?: number; - thoughtsTokenCount?: number; - toolUsePromptTokenCount?: number; - totalTokenCount?: number; - }; -}; - let toolCallCounter = 0; -const GEMINI_THOUGHT_SIGNATURE_VALIDATOR_SKIP = "skip_thought_signature_validator"; - -function requiresToolCallId(modelId: string): boolean { - return modelId.startsWith("claude-") || modelId.startsWith("gpt-oss-"); -} - function requiresToolCallThoughtSignature(modelId: string): boolean { return isGoogleGemini3ProModel(modelId) || isGoogleGemini3FlashModel(modelId); } -function supportsMultimodalFunctionResponse(modelId: string): boolean { - const match = normalizeLowercaseStringOrEmpty(modelId).match(/(?:^|\/)gemini(?:-live)?-(\d+)/); - if (!match) { - return true; - } - return Number.parseInt(match[1] ?? "", 10) >= 3; -} - -function retainThoughtSignature(existing: string | undefined, incoming: string | undefined) { - if (typeof incoming === "string" && incoming.length > 0) { - return incoming; - } - return existing; -} - -function stableStringifyGoogleToolCallValue(value: unknown): string { - if (Array.isArray(value)) { - return `[${value.map((item) => stableStringifyGoogleToolCallValue(item)).join(",")}]`; - } - if (value && typeof value === "object") { - const record = value as Record; - return `{${Object.keys(record) - .toSorted() - .map((key) => `${JSON.stringify(key)}:${stableStringifyGoogleToolCallValue(record[key])}`) - .join(",")}}`; - } - return JSON.stringify(value); -} - -function isJsonLikeThoughtSignature(value: string): boolean { - const trimmed = value.trim(); - return ( - trimmed.startsWith("{") || - trimmed.startsWith("[") || - trimmed.includes('":') || - trimmed.includes('","') || - trimmed.includes('"type"') - ); -} - -const GEMINI_THOUGHT_SIGNATURE_ELLIPSIS_RE = /[\u2026]|\.\.\./; -const GEMINI_THOUGHT_SIGNATURE_BASE64_RE = - /^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/; - -function hasGeminiThoughtSignatureTruncationFootprint(value: string): boolean { - return GEMINI_THOUGHT_SIGNATURE_ELLIPSIS_RE.test(value); -} - -function isGeminiThoughtSignaturePayload(value: string): boolean { - return GEMINI_THOUGHT_SIGNATURE_BASE64_RE.test(value) && value.length > 0; -} - -function sanitizeGeminiThoughtSignature(thoughtSignature: string | undefined): string | undefined { - if (typeof thoughtSignature !== "string") { - return undefined; - } - const trimmed = thoughtSignature.trim(); - if (!trimmed) { - return undefined; - } - if (isJsonLikeThoughtSignature(trimmed)) { - return undefined; - } - const lowered = normalizeLowercaseStringOrEmpty(trimmed); - if ( - lowered === "reasoning" || - lowered === normalizeLowercaseStringOrEmpty(GEMINI_THOUGHT_SIGNATURE_VALIDATOR_SKIP) - ) { - return undefined; - } - if (hasGeminiThoughtSignatureTruncationFootprint(trimmed)) { - return undefined; - } - if (!isGeminiThoughtSignaturePayload(trimmed)) { - return undefined; - } - return trimmed; -} - -function isSameGoogleTransportRoute( - source: { api?: string; provider?: string; model?: string }, - model: GoogleTransportModel, -): boolean { - return ( - source.provider === model.provider && - normalizeGoogleTransportRouteApi(source.api) === normalizeGoogleTransportRouteApi(model.api) && - source.model === model.id - ); -} - function normalizeGoogleTransportRouteApi( api: string | undefined, ): CanonicalGoogleTransportApi | undefined { @@ -324,18 +178,6 @@ function normalizeGoogleTransportMessageRoutes(messages: Context["messages"]): C }); } -function toolCallThoughtSignatureReplayKey(block: { - id: string; - name: string; - arguments: unknown; -}): string { - return [ - block.id, - block.name, - stableStringifyGoogleToolCallValue(coerceTransportToolCallArguments(block.arguments)), - ].join("\u0000"); -} - function mapToolChoice( choice: GoogleTransportOptions["toolChoice"], ): { mode: "AUTO" | "NONE" | "ANY"; allowedFunctionNames?: string[] } | undefined { @@ -566,208 +408,23 @@ function convertGoogleMessages( context: Context | ProviderContext, videoSlots?: GoogleVideoSlots, ) { - const contents: Array> = []; - const replayToolCallThoughtSignatures = new Map(); - const sameRouteToolCallIds = new Set(); - const shouldReplayToolCallThoughtSignature = requiresToolCallThoughtSignature(model.id); const routeModel = normalizeGoogleTransportModelRoute(model); - const transformedMessages = transformTransportMessages( - normalizeGoogleTransportMessageRoutes(context.messages as Context["messages"]), - canonicalGoogleModel(routeModel), - (id) => (requiresToolCallId(model.id) ? normalizeToolCallId(id) : id), - { - preserveCrossModelToolCallThoughtSignature: requiresToolCallThoughtSignature(model.id), + return projectGoogleMessages({ + model: routeModel, + messages: transformTransportMessages( + normalizeGoogleTransportMessageRoutes(context.messages as Context["messages"]), + canonicalGoogleModel(routeModel), + (id) => (requiresGoogleToolCallId(model.id) ? normalizeToolCallId(id) : id), + { preserveCrossModelToolCallThoughtSignature: requiresToolCallThoughtSignature(model.id) }, + ) as ProviderContext["messages"], + replay: "managed", + requiresToolCallSignature: requiresToolCallThoughtSignature(model.id), + videoPart: (video) => { + const placeholder = { text: GOOGLE_VIDEO_SLOT_OMISSION }; + videoSlots?.set(placeholder, video); + return placeholder; }, - ) as ProviderContext["messages"]; - // Parallel calls need one immediate function-response turn. Gemini < 3 images cannot - // live inside functionResponse, so hold them until the consecutive result run ends. - const pendingToolResultImageTurns: Array> = []; - let activeToolResultParts: Array> | undefined; - const flushToolResultRun = (): void => { - contents.push(...pendingToolResultImageTurns); - pendingToolResultImageTurns.length = 0; - activeToolResultParts = undefined; - }; - - for (const msg of transformedMessages) { - if (msg.role !== "toolResult") { - flushToolResultRun(); - } - if (msg.role === "user") { - if (typeof msg.content === "string") { - contents.push({ - role: "user", - parts: [{ text: sanitizeTransportPayloadText(msg.content) || " " }], - }); - continue; - } - const parts = msg.content - .map((item) => { - if (item.type === "text") { - return { text: sanitizeTransportPayloadText(item.text) || " " }; - } - if (item.type === "image") { - return { inlineData: { mimeType: item.mimeType, data: item.data } }; - } - const placeholder = { text: GOOGLE_VIDEO_SLOT_OMISSION }; - videoSlots?.set(placeholder, item); - return placeholder; - }) - .filter((item) => model.input.includes("image") || !("inlineData" in item)); - if (parts.length === 0) { - parts.push({ text: " " }); - } - contents.push({ role: "user", parts }); - continue; - } - - if (msg.role === "assistant") { - const isSameRoute = isSameGoogleTransportRoute(msg, model); - const parts: Array> = []; - const nextReplayToolCallThoughtSignatures = new Map(); - for (const block of msg.content) { - if (block.type === "text") { - if (!block.text.trim()) { - continue; - } - const sanitizedTextSignature = isSameRoute - ? sanitizeGeminiThoughtSignature(block.textSignature) - : undefined; - parts.push({ - text: sanitizeTransportPayloadText(block.text), - ...(sanitizedTextSignature ? { thoughtSignature: sanitizedTextSignature } : {}), - }); - continue; - } - if (block.type === "thinking") { - if (!block.thinking.trim()) { - continue; - } - if (isSameRoute) { - const sanitizedThinkingSignature = sanitizeGeminiThoughtSignature( - block.thinkingSignature, - ); - parts.push({ - thought: true, - text: sanitizeTransportPayloadText(block.thinking), - ...(sanitizedThinkingSignature - ? { thoughtSignature: sanitizedThinkingSignature } - : {}), - }); - } else { - parts.push({ text: sanitizeTransportPayloadText(block.thinking) }); - } - continue; - } - if (block.type === "toolCall") { - if (isSameRoute) { - sameRouteToolCallIds.add(block.id); - } - const replayKey = toolCallThoughtSignatureReplayKey(block); - const replayedThoughtSignature = - shouldReplayToolCallThoughtSignature && isSameRoute - ? replayToolCallThoughtSignatures.get(replayKey) - : undefined; - // Use a block's own same-route signature first; otherwise fall back - // to a same-route replayed value from already-converted context. - // Never replay signatures from foreign providers — Gemini requires - // its own signatures returned exactly as issued. - const ownSignature = isSameRoute - ? sanitizeGeminiThoughtSignature(block.thoughtSignature) - : undefined; - if (ownSignature) { - nextReplayToolCallThoughtSignatures.set(replayKey, ownSignature); - } - const thoughtSignature = - ownSignature ?? - replayedThoughtSignature ?? - (shouldReplayToolCallThoughtSignature - ? GEMINI_THOUGHT_SIGNATURE_VALIDATOR_SKIP - : undefined); - parts.push({ - functionCall: { - name: block.name, - args: coerceTransportToolCallArguments(block.arguments), - ...(isSameRoute || requiresToolCallId(model.id) ? { id: block.id } : {}), - }, - ...(thoughtSignature ? { thoughtSignature } : {}), - }); - } - } - for (const [key, signature] of nextReplayToolCallThoughtSignatures) { - replayToolCallThoughtSignatures.set(key, signature); - } - if (parts.length > 0) { - contents.push({ role: "model", parts }); - } - continue; - } - - if (msg.role === "toolResult") { - const textResult = extractToolResultText(msg.content); - const imageContent = model.input.includes("image") - ? msg.content.filter( - (item): item is Extract<(typeof msg.content)[number], { type: "image" }> => - item.type === "image" && describeToolResultMediaPlaceholder([item]) !== undefined, - ) - : []; - const mediaPlaceholder = describeToolResultMediaPlaceholder(msg.content); - const responseValue = textResult - ? sanitizeTransportPayloadText(textResult) - : (mediaPlaceholder ?? ""); - const imageParts = imageContent.map((imageBlock) => ({ - inlineData: { - mimeType: imageBlock.mimeType, - data: imageBlock.data, - }, - })); - const modelSupportsMultimodalFunctionResponse = supportsMultimodalFunctionResponse(model.id); - const functionResponse = { - functionResponse: { - name: msg.toolName, - response: msg.isError ? { error: responseValue } : { output: responseValue }, - ...(modelSupportsMultimodalFunctionResponse && imageParts.length > 0 - ? { parts: imageParts } - : {}), - ...(sameRouteToolCallIds.has(msg.toolCallId) || requiresToolCallId(model.id) - ? { id: msg.toolCallId } - : {}), - }, - }; - if (activeToolResultParts) { - activeToolResultParts.push(functionResponse); - } else { - activeToolResultParts = [functionResponse]; - contents.push({ role: "user", parts: activeToolResultParts }); - } - if (imageParts.length > 0 && !modelSupportsMultimodalFunctionResponse) { - pendingToolResultImageTurns.push({ - role: "user", - parts: [{ text: "Tool result image:" }, ...imageParts], - }); - } - } - } - flushToolResultRun(); - if (contents.length === 0) { - contents.push({ role: "user", parts: [{ text: " " }] }); - } - return contents; -} - -function convertGoogleTools(tools: NonNullable) { - if (tools.length === 0) { - return undefined; - } - return [ - { - functionDeclarations: sortPromptCacheToolsByName(tools).map((tool) => ({ - name: tool.name, - description: tool.description, - parametersJsonSchema: tool.parameters, - })), - }, - ]; + }); } export function buildGoogleGenerativeAiParams( @@ -1404,66 +1061,6 @@ async function* parseGoogleSseChunks( } } -function updateUsage( - output: MutableAssistantOutput, - model: GoogleTransportModel, - chunk: GoogleSseChunk, - knownUsage: NonNullable, -): void { - if (!chunk.usageMetadata) { - return; - } - for (const field of Object.keys(knownUsage) as Array) { - const value = chunk.usageMetadata[field]; - if (typeof value === "number") { - knownUsage[field] = value; - } - } - const promptTokens = knownUsage.promptTokenCount ?? 0; - const cacheRead = knownUsage.cachedContentTokenCount ?? 0; - const toolUsePromptTokens = knownUsage.toolUsePromptTokenCount ?? 0; - const outputTokens = - (knownUsage.candidatesTokenCount ?? 0) + (knownUsage.thoughtsTokenCount ?? 0); - output.usage = { - input: Math.max(0, promptTokens - cacheRead) + toolUsePromptTokens, - output: outputTokens, - cacheRead, - cacheWrite: 0, - totalTokens: - chunk.usageMetadata.totalTokenCount ?? promptTokens + outputTokens + toolUsePromptTokens, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }; - calculateCost(canonicalGoogleModel(model), output.usage); -} - -function pushTextBlockEnd( - stream: WritableTransportStream, - output: MutableAssistantOutput, - blockIndex: number, -) { - const block = output.content[blockIndex]; - if (!block) { - return; - } - if (block.type === "thinking") { - stream.push({ - type: "thinking_end", - contentIndex: blockIndex, - content: block.thinking, - partial: output, - }); - return; - } - if (block.type === "text") { - stream.push({ - type: "text_end", - contentIndex: blockIndex, - content: block.text, - partial: output, - }); - } -} - function createGoogleTransportStreamFn(kind: CanonicalGoogleTransportApi): StreamFn { return (rawModel, context, rawOptions) => { const model = rawModel as GoogleTransportModel; @@ -1530,21 +1127,6 @@ function createGoogleTransportStreamFn(kind: CanonicalGoogleTransportApi): Strea execute: openSse, }) : await openSse(apiKey); - stream.push({ type: "start", partial: output }); - let currentBlockIndex = -1; - let sawTerminalReason = false; - let terminalGenerationError: Error | undefined; - const knownUsage: NonNullable = { - promptTokenCount: 0, - cachedContentTokenCount: 0, - toolUsePromptTokenCount: 0, - candidatesTokenCount: 0, - thoughtsTokenCount: 0, - }; - const toolCallBlocksById = new Map< - string, - Extract - >(); const chunks = sse.firstChunk === undefined ? sse.chunks @@ -1552,187 +1134,19 @@ function createGoogleTransportStreamFn(kind: CanonicalGoogleTransportApi): Strea yield firstChunk; yield* sse.chunks; })(sse.firstChunk); - for await (const chunk of chunks) { - notifyLlmRequestActivity(options?.signal); - output.responseId ||= chunk.responseId; - const responseModel = normalizeOptionalString(chunk.modelVersion); - if ( - responseModel && - resolveGoogleModelPath(model.id.replace(GOOGLE_VERTEX_MODEL_RESOURCE_PREFIX, "")) !== - resolveGoogleModelPath(responseModel.replace(GOOGLE_VERTEX_MODEL_RESOURCE_PREFIX, "")) - ) { - output.responseModel ||= responseModel; - } - updateUsage(output, model, chunk, knownUsage); - const candidate = chunk.candidates?.[0]; - const promptFeedback = chunk.promptFeedback; - if (!candidate && promptFeedback) { - const blockReason = - normalizeOptionalString(promptFeedback.blockReason) ?? "PROMPT_BLOCKED"; - const blockMessage = normalizeOptionalString(promptFeedback.blockReasonMessage); - const message = `Google prompt blocked (${blockReason})${blockMessage ? `: ${blockMessage}` : ""}`; - throw Object.assign(new Error(message), { - code: blockReason, - type: "google_prompt_blocked", - }); - } - if (candidate?.content?.parts) { - for (const part of candidate.content.parts) { - const hasThoughtSignature = - typeof part.thoughtSignature === "string" && part.thoughtSignature.length > 0; - const rawText = part.text; - const hasText = typeof rawText === "string"; - const partText = typeof rawText === "string" ? rawText : ""; - if (hasText || (hasThoughtSignature && !part.functionCall)) { - if (hasThoughtSignature && !hasText && part.thought !== true) { - const latestBlock = output.content[output.content.length - 1]; - if (latestBlock?.type === "toolCall") { - latestBlock.thoughtSignature = retainThoughtSignature( - latestBlock.thoughtSignature, - part.thoughtSignature, - ); - continue; - } - } - const isThinking = part.thought === true || !hasText; - const currentBlock = output.content[currentBlockIndex]; - if ( - currentBlockIndex < 0 || - !currentBlock || - (isThinking && currentBlock.type !== "thinking") || - (!isThinking && currentBlock.type !== "text") - ) { - if (currentBlockIndex >= 0) { - pushTextBlockEnd(stream, output, currentBlockIndex); - } - if (isThinking) { - output.content.push({ type: "thinking", thinking: "" }); - currentBlockIndex = output.content.length - 1; - stream.push({ - type: "thinking_start", - contentIndex: currentBlockIndex, - partial: output, - }); - } else { - output.content.push({ type: "text", text: "" }); - currentBlockIndex = output.content.length - 1; - stream.push({ - type: "text_start", - contentIndex: currentBlockIndex, - partial: output, - }); - } - } - const activeBlock = output.content[currentBlockIndex]; - if (activeBlock?.type === "thinking") { - activeBlock.thinking += partText; - activeBlock.thinkingSignature = retainThoughtSignature( - activeBlock.thinkingSignature, - part.thoughtSignature, - ); - stream.push({ - type: "thinking_delta", - contentIndex: currentBlockIndex, - delta: partText, - partial: output, - }); - } else if (activeBlock?.type === "text") { - activeBlock.text += partText; - activeBlock.textSignature = retainThoughtSignature( - activeBlock.textSignature, - part.thoughtSignature, - ); - stream.push({ - type: "text_delta", - contentIndex: currentBlockIndex, - delta: partText, - partial: output, - }); - } - } - if (part.functionCall) { - if (currentBlockIndex >= 0) { - pushTextBlockEnd(stream, output, currentBlockIndex); - currentBlockIndex = -1; - } - const providedId = part.functionCall.id; - const existingToolCall = - typeof providedId === "string" ? toolCallBlocksById.get(providedId) : undefined; - const isDuplicate = existingToolCall !== undefined; - const toolCallId = - providedId && !isDuplicate - ? providedId - : `${part.functionCall.name || "tool"}_${Date.now()}_${++toolCallCounter}`; - const toolCall: GoogleTransportContentBlock = { - type: "toolCall", - id: toolCallId, - name: part.functionCall.name || "", - arguments: part.functionCall.args ?? {}, - ...(part.thoughtSignature ? { thoughtSignature: part.thoughtSignature } : {}), - }; - output.content.push(toolCall); - if (!toolCallBlocksById.has(toolCall.id)) { - toolCallBlocksById.set(toolCall.id, toolCall); - } - const blockIndex = output.content.length - 1; - stream.push({ - type: "toolcall_start", - contentIndex: blockIndex, - partial: output, - }); - stream.push({ - type: "toolcall_delta", - contentIndex: blockIndex, - delta: JSON.stringify(toolCall.arguments), - partial: output, - }); - stream.push({ - type: "toolcall_end", - contentIndex: blockIndex, - toolCall, - partial: output, - }); - } - } - } - if ( - typeof candidate?.finishReason === "string" && - candidate.finishReason !== "FINISH_REASON_UNSPECIFIED" - ) { - sawTerminalReason = true; - output.stopReason = mapStopReasonString(candidate.finishReason); - if (output.stopReason === "error") { - const finishMessage = normalizeOptionalString(candidate.finishMessage); - terminalGenerationError = Object.assign( - new Error( - `Google generation stopped (${candidate.finishReason})${finishMessage ? `: ${finishMessage}` : ""}`, - ), - { code: candidate.finishReason, type: "google_generation_failed" }, - ); - } - // MAX_TOKENS can leave a complete-looking partial call. Only a normal - // Google stop may promote parsed calls into an executable tool-use turn. - if ( - output.stopReason === "stop" && - output.content.some((block) => block.type === "toolCall") - ) { - output.stopReason = "toolUse"; - } - } - } - if (currentBlockIndex >= 0) { - pushTextBlockEnd(stream, output, currentBlockIndex); - } - if (terminalGenerationError && !options?.signal?.aborted) { - throw terminalGenerationError; - } - if (!sawTerminalReason && !options?.signal?.aborted) { - throw Object.assign(new Error("Google stream ended before a terminal finish reason"), { - code: "STREAM_INCOMPLETE", - type: "google_incomplete_stream", - }); - } - finalizeTransportStream({ stream, output, signal: options?.signal }); + await consumeGoogleGenerateContentStream({ + chunks, + model: canonicalModel, + output, + stream, + signal: options?.signal, + nextToolCallId: (name) => `${name || "tool"}_${Date.now()}_${++toolCallCounter}`, + // Managed SSE has always accumulated text deltas; the SDK preserves signed Parts. + profile: "managed", + normalizeModelId: (id) => + resolveGoogleModelPath(id.replace(GOOGLE_VERTEX_MODEL_RESOURCE_PREFIX, "")), + resolveStopReason: mapStopReasonString, + }); } catch (error) { failTransportStream({ stream, output, signal: options?.signal, error }); } diff --git a/packages/ai/src/provider-transport-parity.test.ts b/packages/ai/src/provider-transport-parity.test.ts index 1b0650e8d2a8..4c42a23cdb99 100644 --- a/packages/ai/src/provider-transport-parity.test.ts +++ b/packages/ai/src/provider-transport-parity.test.ts @@ -526,6 +526,72 @@ describe("provider and transport observable parity fixtures", () => { }, ); + it.each([ + { name: "malformed seeded input", input: "{", providerError: true }, + { + name: "encoded object input", + input: '{"query":"seed"}', + providerArguments: { query: "seed" }, + }, + { + name: "streamed arguments superseding a malformed seed", + input: "{", + delta: '{"query":"streamed"}', + providerArguments: { query: "streamed" }, + }, + ])("preserves Anthropic terminal tool validation for $name", async (fixture) => { + const events = [ + { + type: "message_start", + message: { + id: "msg_seeded", + model: anthropicModel.id, + usage: { input_tokens: 1, output_tokens: 0 }, + }, + }, + { + type: "content_block_start", + index: 0, + content_block: { type: "tool_use", id: "call_seed", name: "lookup", input: fixture.input }, + }, + ...(fixture.delta + ? [ + { + type: "content_block_delta", + index: 0, + delta: { type: "input_json_delta", partial_json: fixture.delta }, + }, + ] + : []), + { type: "content_block_stop", index: 0 }, + { type: "message_delta", delta: { stop_reason: "tool_use" } }, + { type: "message_stop" }, + ]; + for (const implementation of ["provider", "transport"] as const) { + const result = await runAnthropic(implementation, "success", events); + if (implementation === "provider" && fixture.providerError) { + expect(result.terminal.stopReason).toBe("error"); + expect(result.errorFields.errorMessage).toContain("malformed JSON arguments"); + expect(result.terminal.content).toEqual([]); + expect(result.eventTrace).not.toContainEqual( + expect.objectContaining({ type: "toolcall_end" }), + ); + } else { + expect(result.terminal).toMatchObject({ + stopReason: "toolUse", + content: [ + { + type: "toolCall", + id: "call_seed", + arguments: + implementation === "provider" || fixture.delta ? fixture.providerArguments : {}, + }, + ], + }); + } + } + }); + it("marks content interrupted by native reasoning as commentary", async () => { for (const implementation of ["provider", "transport"] as const) { for (const chunks of [openAiInterleavedReasoningChunks, openAiCoalescedReasoningChunks]) { diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 1defc3950158..628caa6e3da5 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -3,48 +3,36 @@ import Anthropic from "@anthropic-ai/sdk"; import { Stream } from "@anthropic-ai/sdk/core/streaming.js"; import type { CacheControlEphemeral, - ContentBlockParam, MessageCreateParamsStreaming, MessageParam, RawMessageStreamEvent, TextBlockParam, } from "@anthropic-ai/sdk/resources/messages.js"; -import { appendAssistantThinking } from "@openclaw/llm-core/event-stream"; -import { isRecord } from "@openclaw/normalization-core/record-coerce"; import { getEnvApiKey } from "../env-api-keys.js"; import { getAiTransportHost, resolveAiTransportHeaderSentinels } from "../host.js"; -import { - createAnthropicInlineImageBudget, - normalizeAnthropicInlineContent, - resolveAnthropicImageMediaType, - type AnthropicInlineImageBudget, -} from "../internal/anthropic-inline-images.js"; -import { calculateCost } from "../model-utils.js"; -import type { - AnthropicContextManagementOptions, - AnthropicOptions, - AnthropicThinkingDisplay, -} from "../provider-options.js"; +import type { AnthropicContextManagementOptions, AnthropicOptions } from "../provider-options.js"; import { transformProviderMessages as transformMessages } from "../provider-transcript-transform.js"; import { buildAnthropicReplayPlan, - createCompactionCapture, isAnthropicReplayRejection, suppressAnthropicCompaction, - type AnthropicCompactionBlock, } from "../transports/anthropic-compaction-replay.js"; +import { + convertAnthropicMessages, + convertAnthropicTools, + buildAnthropicGenerationParams, +} from "../transports/anthropic-messages.js"; import { applyAnthropicCacheControlToMessages, applyAnthropicContextManagementToRequest, isDirectAnthropicModel, - logAnthropicContextEdits, resolveAnthropicContextManagementBetaHeader, } from "../transports/anthropic-payload-policy.js"; +import { consumeAnthropicStream } from "../transports/anthropic-stream-reducer.js"; import { assignTransportErrorDetails, - finalizeTerminalToolCallArguments, + finalizeTransportStream, notifyProviderHttpResponse, - transportAbortError, } from "../transports/transport-stream-shared.js"; import { MALFORMED_STREAMING_FRAGMENT_ERROR_MESSAGE } from "../transports/transport-utils.js"; import type { @@ -54,23 +42,13 @@ import type { AssistantMessageEvent, CacheRetention, Context, - Message, Model, SimpleStreamOptions, StreamFunction, - TextContent, - ThinkingContent, - Tool, - ToolCall, } from "../types.js"; import { createDeferredEventBuffer } from "../utils/deferred-event-buffer.js"; import { AssistantMessageEventStream } from "../utils/event-stream.js"; -import { - createToolArgumentPreviewSchedule, - parseJsonWithRepair, - parseStreamingJson, - type ToolArgumentPreviewSchedule, -} from "../utils/json-parse.js"; +import { parseJsonWithRepair } from "../utils/json-parse.js"; import { notifyLlmRequestActivity } from "../utils/llm-request-activity.js"; import { sanitizeSurrogates } from "../utils/sanitize-unicode.js"; import { @@ -86,46 +64,24 @@ import { applyClaudeRequestContract, ANTHROPIC_CLAUDE_CODE_BILLING_SYSTEM_BLOCK, ANTHROPIC_CLAUDE_CODE_VERSION, - mapAnthropicStopReason, prepareClaudeNoPrefillRequestContext, resolveAnthropicThinkingEffort, resolveClaudeOpus5ModelIdentity, resolveClaudeSonnet5ModelIdentity, requiresClaudeAdaptiveThinking, supportsClaudeAdaptiveThinking, - supportsClaudeNativeXhighEffort, usesClaudeFable5MessagesContract, usesClaudeStreamingRefusalContract, } from "./anthropic-model-contract.js"; -import { applyAnthropicRefusal } from "./anthropic-refusal.js"; import { ANTHROPIC_SERVER_SIDE_FALLBACK_BETA, ANTHROPIC_SERVER_SIDE_FALLBACKS, - applyAnthropicFallbackBoundary, - readAnthropicFallbackBoundary, - resolveAnthropicFallbackServingModelCost, } from "./anthropic-server-fallback.js"; -import { - ANTHROPIC_OMITTED_REASONING_TEXT, - applyAnthropicThinkingBindingControls, - findActiveAnthropicToolTurnAssistantIndex, - logAnthropicThinkingDrops, - readAnthropicInputTransformations, -} from "./anthropic-thinking-replay.js"; +import { applyAnthropicThinkingBindingControls } from "./anthropic-thinking-replay.js"; import { normalizeAnthropicToolCallId, - normalizeAnthropicToolChoice, - projectAnthropicTools, - reconcileAnthropicToolChoice, - resolveOriginalAnthropicToolName, - toClaudeCodeToolName, type AnthropicToolProjection, } from "./anthropic-tool-projection.js"; -import { - applyAnthropicMessageDeltaUsage, - applyAnthropicMessageStartUsage, - type AnthropicPromptUsageSnapshot, -} from "./anthropic-usage.js"; import { resolveCacheRetention } from "./cache-retention.js"; import { resolveCloudflareBaseUrl } from "./cloudflare.js"; import { buildCopilotDynamicHeaders, hasCopilotVisionInput } from "./github-copilot-headers.js"; @@ -134,15 +90,8 @@ import { buildBaseOptions, clampMaxTokensToModel, } from "./simple-options.js"; -import { - describeToolResultMediaPlaceholder, - extractToolResultBlockText, - extractToolResultText, - isImageWithMediaPayload, -} from "./tool-result-text.js"; const ANTHROPIC_CACHE_CONTROL_LIMIT = 4; -const EMPTY_ERROR_TOOL_RESULT_TEXT = "[tool error with no output]"; type AnthropicCompactionOptions = AnthropicOptions & { authProfileId?: string; @@ -164,93 +113,6 @@ function getCacheControl( }; } -/** - * Convert content blocks to Anthropic API format - */ -async function convertContentBlocks( - content: readonly unknown[], - isError: boolean, - imageBudget: AnthropicInlineImageBudget, -): Promise< - | string - | Array< - | { type: "text"; text: string } - | { - type: "image"; - source: { - type: "base64"; - media_type: "image/jpeg" | "image/png" | "image/gif" | "image/webp"; - data: string; - }; - } - > -> { - const text = extractToolResultText(content); - const mediaPlaceholder = describeToolResultMediaPlaceholder(content); - const hasImages = content.some(isImageWithMediaPayload); - - if (!hasImages) { - const sanitized = sanitizeSurrogates(text); - return sanitized.trim().length > 0 - ? sanitized - : (mediaPlaceholder ?? (isError ? EMPTY_ERROR_TOOL_RESULT_TEXT : "")); - } - - const blocks: Array< - | { type: "text"; text: string } - | { - type: "image"; - source: { - type: "base64"; - media_type: "image/jpeg" | "image/png" | "image/gif" | "image/webp"; - data: string; - }; - } - > = []; - let hasTextBlock = false; - - for (const block of content) { - if (!block || typeof block !== "object") { - continue; - } - const record = block as Record; - const blockText = extractToolResultBlockText(block); - if (blockText) { - blocks.push({ type: "text" as const, text: sanitizeSurrogates(blockText) }); - hasTextBlock = true; - } - if (!isImageWithMediaPayload(record)) { - continue; - } - const [normalizedImage] = await normalizeAnthropicInlineContent( - [ - { - type: "image" as const, - data: typeof record.data === "string" ? record.data : "", - mimeType: typeof record.mimeType === "string" ? record.mimeType : "image/jpeg", - }, - ], - imageBudget, - ); - if (normalizedImage?.type !== "image") { - continue; - } - blocks.push({ - type: "image" as const, - source: { - type: "base64" as const, - media_type: resolveAnthropicImageMediaType(normalizedImage.mimeType), - data: normalizedImage.data, - }, - }); - } - if (!hasTextBlock) { - blocks.unshift({ type: "text" as const, text: mediaPlaceholder ?? "(see attached image)" }); - } - - return blocks; -} - export type { AnthropicEffort, AnthropicOptions, @@ -299,15 +161,12 @@ const ANTHROPIC_MESSAGE_EVENTS: ReadonlySet = new Set([ async function* iterateAnthropicEvents( response: Response, - requireMessageStop = false, signal?: AbortSignal, ): AsyncGenerator { if (!response.body) { throw new Error("Attempted to iterate over an Anthropic response with no body"); } - let sawMessageEnd = false; - for await (const sse of Stream.rawEvents(response)) { if (sse.event === "error") { throw new Error(sse.data); @@ -320,9 +179,6 @@ async function* iterateAnthropicEvents( try { const event = parseJsonWithRepair(sse.data) as RawMessageStreamEvent; - if (event.type === "message_stop") { - sawMessageEnd = true; - } yield event; } catch (error) { // Frame payloads carry model output, so surface the shared malformed-fragment @@ -333,10 +189,6 @@ async function* iterateAnthropicEvents( throw error; } } - - if (requireMessageStop && !sawMessageEnd) { - throw new Error("Anthropic stream ended before message_stop"); - } } export const streamAnthropic: StreamFunction<"anthropic-messages", AnthropicCompactionOptions> = ( @@ -371,13 +223,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages", AnthropicComp const refusalBuffer = usesClaudeStreamingRefusalContract(model) ? createDeferredEventBuffer(stream) : undefined; - const eventSink = refusalBuffer ?? stream; - // Fallback-served turns bill at the serving model's rates; a boundary - // swaps this to the fallback model's cost table. - let costModel = model; - let messageStartPromptUsage: AnthropicPromptUsageSnapshot | undefined; let usedCompactionReplay = false; - let inputTransformations: unknown[] | undefined; try { let client: Anthropic; @@ -459,285 +305,18 @@ export const streamAnthropic: StreamFunction<"anthropic-messages", AnthropicComp .asResponse(); await notifyProviderHttpResponse({ options: requestOptions, response, model }); - type Block = (ThinkingContent | TextContent | (ToolCall & { partialJson?: string })) & { - index: number; - }; - const blocks = output.content as Block[]; - const blockIndexes = new Map(); - // Preview schedules are per active tool call; WeakMap keys die with the block. - const toolArgumentPreviewSchedules = new WeakMap< - Extract, - ToolArgumentPreviewSchedule - >(); - const sealedToolCalls: Array<{ - block: Extract; - contentIndex: number; - }> = []; - const compactionCapture = createCompactionCapture(output, model, requestOptions); - const requireMessageStop = refusalBuffer !== undefined || isDirectAnthropicModel(model); - - for await (const event of iterateAnthropicEvents( - response, - requireMessageStop, - requestOptions?.signal, - )) { - // A serving-model fallback replaces the initial snapshot; report only once at completion. - inputTransformations = readAnthropicInputTransformations(event) ?? inputTransformations; - if (event.type === "message_start") { - output.responseId = event.message.id; - output.responseModel = event.message.model; - messageStartPromptUsage = applyAnthropicMessageStartUsage( - output.usage, - event.message.usage, - ); - calculateCost(costModel, output.usage); - // Defer start until after message_start so that pre-stream SSE errors - // (e.g. invalid thinking signatures) arrive before any non-error event - // is yielded, keeping yieldedOutput=false in pumpStreamWithRecovery - // and allowing the thinking-block recovery retry to fire. - eventSink.push({ type: "start", partial: output }); - } else if (event.type === "content_block_start") { - const rawContentBlock = isRecord(event.content_block) ? event.content_block : undefined; - if ( - requestOptions?.anthropicServerCompaction === true && - compactionCapture.begin(event.index, rawContentBlock, output.content.length) - ) { - continue; - } - const fallbackBoundary = refusalBuffer - ? readAnthropicFallbackBoundary(event.content_block) - : null; - if (fallbackBoundary) { - // Server-side fallback boundary: pre-boundary thinking/tool blocks - // must not replay or execute, and the buffered preview events - // reference them, so rebuild the deferred timeline from the - // surviving text prefix the fallback model continued from. - refusalBuffer?.discard(); - sealedToolCalls.length = 0; - blockIndexes.clear(); - applyAnthropicFallbackBoundary({ - output, - boundary: fallbackBoundary, - provider: model.provider, - }); - // Fallback-only iteration partials stay outside the serving-model - // estimate. Compaction responses are the exception: usage policy - // aggregates their complete billed iteration list. - costModel = { - ...model, - cost: resolveAnthropicFallbackServingModelCost({ - requestedModelId: model.id, - servingModelId: fallbackBoundary.toModel, - requestedCost: model.cost, - }), - }; - calculateCost(costModel, output.usage); - eventSink.push({ type: "start", partial: output }); - for (const [i, block] of blocks.entries()) { - if (block.type !== "text") { - continue; - } - delete (block as Partial).index; - eventSink.push({ type: "text_start", contentIndex: i, partial: output }); - if (block.text) { - eventSink.push({ - type: "text_delta", - contentIndex: i, - delta: block.text, - partial: output, - }); - } - eventSink.push({ - type: "text_end", - contentIndex: i, - content: block.text, - partial: output, - }); - } - } else if (event.content_block.type === "text") { - const block: Block = { - type: "text", - text: "", - index: event.index, - }; - output.content.push(block); - blockIndexes.set(event.index, output.content.length - 1); - eventSink.push({ - type: "text_start", - contentIndex: output.content.length - 1, - partial: output, - }); - } else if (event.content_block.type === "thinking") { - const block: Block = { - type: "thinking", - thinking: "", - thinkingSignature: "", - index: event.index, - }; - output.content.push(block); - blockIndexes.set(event.index, output.content.length - 1); - eventSink.push({ - type: "thinking_start", - contentIndex: output.content.length - 1, - partial: output, - }); - } else if (event.content_block.type === "redacted_thinking") { - const block: Block = { - type: "thinking", - thinking: "[Reasoning redacted]", - thinkingSignature: event.content_block.data, - redacted: true, - index: event.index, - }; - output.content.push(block); - blockIndexes.set(event.index, output.content.length - 1); - eventSink.push({ - type: "thinking_start", - contentIndex: output.content.length - 1, - partial: output, - }); - } else if (event.content_block.type === "tool_use") { - const block: Block = { - type: "toolCall", - id: event.content_block.id, - name: isOAuth - ? resolveOriginalAnthropicToolName(event.content_block.name, toolProjection) - : event.content_block.name, - arguments: (event.content_block.input as Record) ?? {}, - partialJson: "", - index: event.index, - }; - output.content.push(block); - blockIndexes.set(event.index, output.content.length - 1); - toolArgumentPreviewSchedules.set(block, createToolArgumentPreviewSchedule()); - eventSink.push({ - type: "toolcall_start", - contentIndex: output.content.length - 1, - partial: output, - }); - } - } else if (event.type === "content_block_delta") { - const rawDelta = isRecord(event.delta) ? event.delta : undefined; - if (compactionCapture.delta(event.index, rawDelta)) { - continue; - } else if (event.delta.type === "text_delta") { - const index = blockIndexes.get(event.index); - const block = index === undefined ? undefined : blocks[index]; - if (index !== undefined && block?.type === "text") { - block.text += event.delta.text; - eventSink.push({ - type: "text_delta", - contentIndex: index, - delta: event.delta.text, - partial: output, - }); - } - } else if (event.delta.type === "thinking_delta") { - const index = blockIndexes.get(event.index); - const block = index === undefined ? undefined : blocks[index]; - if (index !== undefined && block?.type === "thinking") { - appendAssistantThinking(block, event.delta.thinking); - eventSink.push({ - type: "thinking_delta", - contentIndex: index, - delta: event.delta.thinking, - partial: output, - }); - } - } else if (event.delta.type === "input_json_delta") { - const index = blockIndexes.get(event.index); - const block = index === undefined ? undefined : blocks[index]; - if (index !== undefined && block?.type === "toolCall") { - block.partialJson = (block.partialJson ?? "") + event.delta.partial_json; - // Preview refresh is scheduled geometrically; the terminal - // finalize re-parses the full buffer authoritatively either way. - if (toolArgumentPreviewSchedules.get(block)?.(block.partialJson.length)) { - block.arguments = parseStreamingJson(block.partialJson); - } - eventSink.push({ - type: "toolcall_delta", - contentIndex: index, - delta: event.delta.partial_json, - partial: output, - }); - } - } else if (event.delta.type === "signature_delta") { - const index = blockIndexes.get(event.index); - const block = index === undefined ? undefined : blocks[index]; - if (index !== undefined && block?.type === "thinking") { - block.thinkingSignature = block.thinkingSignature || ""; - block.thinkingSignature += event.delta.signature; - } - } - } else if (event.type === "content_block_stop") { - if (compactionCapture.complete(event.index)) { - continue; - } - const index = blockIndexes.get(event.index); - const block = index === undefined ? undefined : blocks[index]; - if (index !== undefined && block) { - blockIndexes.delete(event.index); - delete (block as Partial).index; - if (block.type === "text") { - eventSink.push({ - type: "text_end", - contentIndex: index, - content: block.text, - partial: output, - }); - } else if (block.type === "thinking") { - eventSink.push({ - type: "thinking_end", - contentIndex: index, - content: block.thinking, - partial: output, - }); - } else if (block.type === "toolCall") { - sealedToolCalls.push({ block, contentIndex: index }); - } - } - } else if (event.type === "message_delta") { - logAnthropicContextEdits(event); - if (event.delta.stop_reason) { - if (event.delta.stop_reason === "refusal") { - applyAnthropicRefusal(output, event.delta.stop_details, model.provider); - } else { - output.stopReason = mapAnthropicStopReason(event.delta.stop_reason); - } - } - applyAnthropicMessageDeltaUsage(output.usage, event.usage, messageStartPromptUsage); - calculateCost(costModel, output.usage); - } - } - - if (requestOptions?.signal?.aborted) { - throw transportAbortError(requestOptions.signal); - } - - if (output.stopReason === "aborted" || output.stopReason === "error") { - throw new Error(output.errorMessage ?? "An unknown error occurred"); - } - if ([...blockIndexes.values()].some((index) => blocks[index]?.type === "toolCall")) { - throw new Error("Provider completed stream with an incomplete tool call"); - } - finalizeTerminalToolCallArguments( - sealedToolCalls.map(({ block }) => block), - (block) => - block.partialJson && block.partialJson.length > 0 ? block.partialJson : block.arguments, - ); - for (const sealed of sealedToolCalls) { - delete sealed.block.partialJson; - eventSink.push({ - type: "toolcall_end", - contentIndex: sealed.contentIndex, - toolCall: sealed.block, - partial: output, - }); - } - - refusalBuffer?.flush(); - stream.push({ type: "done", reason: output.stopReason, message: output }); - stream.end(); + await consumeAnthropicStream({ + events: iterateAnthropicEvents(response, requestOptions?.signal), + model, + options: requestOptions ?? {}, + output, + stream, + refusalBuffer, + isOAuthToken: isOAuth, + toolProjection, + profile: "provider", + }); + finalizeTransportStream({ stream, output }); } catch (error) { const terminal = assignTransportErrorDetails(output, error, requestOptions?.signal); output.content = output.content.filter((block) => block.type !== "toolCall"); @@ -755,8 +334,6 @@ export const streamAnthropic: StreamFunction<"anthropic-messages", AnthropicComp } stream.push({ type: "error", reason: terminal.stopReason, error: output }); stream.end(); - } finally { - logAnthropicThinkingDrops(inputTransformations); } })(); @@ -1080,7 +657,7 @@ async function buildParams( const system = buildAnthropicSystemBlocks(context.systemPrompt, isOAuthTokenResult, cacheControl); const compat = getAnthropicCompat(model); const convertedTools = context.tools - ? convertTools( + ? convertAnthropicTools( context.tools, isOAuthTokenResult, compat.supportsEagerToolInputStreaming, @@ -1102,20 +679,32 @@ async function buildParams( }); const params: MessageCreateParamsStreaming = { model: model.id, - messages: await convertMessages( - replayPlan.messages, + // The SDK's stable message union omits compaction blocks accepted by its beta endpoint. + messages: (await convertAnthropicMessages( + transformMessages(replayPlan.messages, model, normalizeAnthropicToolCallId), model, isOAuthTokenResult, - cacheControl, - messageCacheControlLimit, - replayThinkingEnabled, - compat.allowEmptySignature, - replayPlan.compaction, - ), + { + profile: "provider", + allowEmptySignature: compat.allowEmptySignature, + compaction: replayPlan.compaction, + replayThinkingEnabled, + }, + )) as MessageParam[], max_tokens: options?.maxTokens ?? model.maxTokens, stream: true, }; + if (cacheControl) { + // Anthropic-family carriers are append-only, so they are stable cache anchors too. + applyAnthropicCacheControlToMessages( + params.messages, + cacheControl, + messageCacheControlLimit, + new Set(), + ); + } + if (system) { params.system = system; } @@ -1127,255 +716,14 @@ async function buildParams( (params as { fallbacks?: "default" }).fallbacks = ANTHROPIC_SERVER_SIDE_FALLBACKS; } - // Thinking and post-4.6 Claude models reject custom temperature values. - if ( - options?.temperature !== undefined && - !options?.thinkingEnabled && - !supportsClaudeNativeXhighEffort(model) - ) { - params.temperature = options.temperature; - } - - if (options?.stop !== undefined && options.stop.length > 0) { - params.stop_sequences = options.stop; - } - - if (tools && tools.length > 0) { - params.tools = tools; - } - - // Configure thinking mode: always-on adaptive (Fable 5 and Mythos 5), - // adaptive (Opus 4.6+ and Sonnet 4.6), - // budget-based (older models), or explicitly disabled. - if (mandatoryAdaptiveThinking || model.reasoning || supportsClaudeAdaptiveThinking(model)) { - if (mandatoryAdaptiveThinking || options?.thinkingEnabled) { - // Default to "summarized" so Opus 4.7+ and Mythos Preview behave like - // older Claude 4 models (whose API default is also "summarized"). - const display: AnthropicThinkingDisplay = options?.thinkingDisplay ?? "summarized"; - if (supportsClaudeAdaptiveThinking(model)) { - // Adaptive thinking: Claude decides when and how much to think. - params.thinking = { type: "adaptive", display }; - const effort = options?.effort ?? (mandatoryAdaptiveThinking ? "high" : undefined); - if (effort) { - params.output_config = { effort }; - } - } else { - // Budget-based thinking for older models. - params.thinking = { - type: "enabled", - budget_tokens: options?.thinkingBudgetTokens ?? ANTHROPIC_MIN_THINKING_BUDGET_TOKENS, - display, - }; - } - } else if (options?.thinkingEnabled === false) { - params.thinking = { type: "disabled" }; - } - } - - if (options?.metadata) { - const userId = options.metadata.user_id; - if (typeof userId === "string") { - params.metadata = { user_id: userId }; - } - } - - if (options?.toolChoice) { - const normalizedToolChoice = normalizeAnthropicToolChoice( - replayThinkingEnabled, - options.toolChoice, - ); - const projectedToolChoice = toolProjection - ? reconcileAnthropicToolChoice(normalizedToolChoice, toolProjection) - : normalizedToolChoice; - if (projectedToolChoice) { - params.tool_choice = projectedToolChoice; - } - } + Object.assign( + params, + buildAnthropicGenerationParams({ model, options, tools, toolProjection, profile: "provider" }), + ); return { params, toolProjection, usedCompactionReplay: replayPlan.compaction !== undefined }; } -async function convertMessages( - messages: Message[], - model: Model<"anthropic-messages">, - isOAuthTokenValue: boolean, - cacheControl?: CacheControlEphemeral, - messageCacheControlLimit = 4, - replayThinkingEnabled = true, - allowEmptySignature = false, - compaction?: AnthropicCompactionBlock, -): Promise { - const params: MessageParam[] = []; - const imageBudget = createAnthropicInlineImageBudget(); - - // Transform messages for cross-provider compatibility - const transformedMessages = transformMessages(messages, model, normalizeAnthropicToolCallId); - const activeToolTurnAssistantIndex = replayThinkingEnabled - ? -1 - : findActiveAnthropicToolTurnAssistantIndex(transformedMessages); - - for (let i = 0; i < transformedMessages.length; i++) { - const msg = transformedMessages[i]; - if (!msg) { - continue; - } - - if (msg.role === "user") { - if (typeof msg.content === "string") { - if (msg.content.trim().length > 0) { - params.push({ - role: "user", - content: sanitizeSurrogates(msg.content), - }); - } - } else { - const normalizedContent = await normalizeAnthropicInlineContent(msg.content, imageBudget); - const blocks: ContentBlockParam[] = normalizedContent.map((item) => { - if (item.type === "text") { - return { - type: "text", - text: sanitizeSurrogates(item.text), - }; - } - return { - type: "image", - source: { - type: "base64", - media_type: resolveAnthropicImageMediaType(item.mimeType), - data: item.data, - }, - }; - }); - const filteredBlocks = blocks.filter((b) => { - if (b.type === "text") { - return b.text.trim().length > 0; - } - return true; - }); - if (filteredBlocks.length === 0) { - continue; - } - params.push({ - role: "user", - content: filteredBlocks, - }); - } - } else if (msg.role === "assistant") { - const blocks: ContentBlockParam[] = - i === 0 && compaction ? ([compaction] as unknown as ContentBlockParam[]) : []; - let omittedThinking = false; - - for (const block of msg.content) { - if (block.type === "text") { - if (block.text.trim().length === 0) { - continue; - } - blocks.push({ - type: "text", - text: sanitizeSurrogates(block.text), - }); - } else if (block.type === "thinking") { - if (!replayThinkingEnabled && i !== activeToolTurnAssistantIndex) { - omittedThinking = true; - continue; - } - // Redacted thinking: pass the opaque payload back as redacted_thinking - if (block.redacted) { - if (!block.thinkingSignature) { - throw new Error("redacted thinking block is missing its opaque signature"); - } - blocks.push({ - type: "redacted_thinking", - data: block.thinkingSignature, - }); - continue; - } - const thinkingSignature = block.thinkingSignature?.trim(); - const hasNativeThinkingSignature = - Boolean(thinkingSignature) && thinkingSignature !== "reasoning_content"; - if (block.thinking.trim().length === 0 && !hasNativeThinkingSignature) { - continue; - } - // If thinking signature is missing/empty (e.g., from aborted stream), - // convert to plain text block without tags to avoid API rejection - // and prevent Claude from mimicking the tags in responses - if (!thinkingSignature && !allowEmptySignature) { - blocks.push({ - type: "text", - text: sanitizeSurrogates(block.thinking), - }); - } else { - // OpenAI-compatible reasoning markers are field names, not native - // Anthropic replay signatures; sending them bricks persisted replays. - if (thinkingSignature === "reasoning_content") { - continue; - } - blocks.push({ - type: "thinking", - thinking: block.thinking, - signature: thinkingSignature ?? "", - }); - } - } else if (block.type === "toolCall") { - blocks.push({ - type: "tool_use", - id: block.id, - name: isOAuthTokenValue ? toClaudeCodeToolName(block.name) : block.name, - input: block.arguments ?? {}, - }); - } - } - if (blocks.length === 0 && omittedThinking) { - blocks.push({ type: "text", text: ANTHROPIC_OMITTED_REASONING_TEXT }); - } - if (blocks.length === 0) { - continue; - } - params.push({ - role: "assistant", - content: blocks, - }); - } else if (msg.role === "toolResult") { - // Collect all consecutive toolResult messages, needed for z.ai Anthropic endpoint - const toolResults: ContentBlockParam[] = []; - toolResults.push({ - type: "tool_result", - tool_use_id: msg.toolCallId, - content: await convertContentBlocks(msg.content, msg.isError, imageBudget), - is_error: msg.isError, - }); - - let j = i + 1; - while (j < transformedMessages.length) { - const nextMsg = transformedMessages.at(j); - if (nextMsg?.role !== "toolResult") { - break; - } - toolResults.push({ - type: "tool_result", - tool_use_id: nextMsg.toolCallId, - content: await convertContentBlocks(nextMsg.content, nextMsg.isError, imageBudget), - is_error: nextMsg.isError, - }); - j++; - } - - i = j - 1; - params.push({ - role: "user", - content: toolResults, - }); - } - } - - if (cacheControl) { - // Anthropic-family carriers are append-only, so they are stable cache anchors too. - applyAnthropicCacheControlToMessages(params, cacheControl, messageCacheControlLimit, new Set()); - } - - return params; -} - function buildAnthropicSystemBlocks( systemPrompt: string | undefined, isOAuthTokenResult: boolean, @@ -1458,37 +806,4 @@ function shouldUseFineGrainedToolStreamingBeta( ); } -function convertTools( - tools: Tool[], - isOAuthTokenLocal: boolean, - supportsEagerToolInputStreaming: boolean, - cacheControl?: CacheControlEphemeral, -): { - projection: AnthropicToolProjection; - tools: Anthropic.Messages.Tool[]; -} { - const projection = projectAnthropicTools(tools, (name) => - isOAuthTokenLocal ? toClaudeCodeToolName(name) : name, - ); - const convertedTools: Anthropic.Messages.Tool[] = []; - for (const [index, tool] of projection.tools.entries()) { - const convertedTool: Anthropic.Messages.Tool = { - name: tool.wireName, - description: tool.description, - input_schema: tool.inputSchema, - }; - if (supportsEagerToolInputStreaming) { - convertedTool.eager_input_streaming = true; - } - if (cacheControl && index === projection.tools.length - 1) { - convertedTool.cache_control = cacheControl; - } - convertedTools.push(convertedTool); - } - return { - projection, - tools: convertedTools, - }; -} - /* oxlint-disable max-lines -- TODO: split this grandfathered oversized file. */ diff --git a/packages/ai/src/providers/google-messages.ts b/packages/ai/src/providers/google-messages.ts new file mode 100644 index 000000000000..3f775775fc6d --- /dev/null +++ b/packages/ai/src/providers/google-messages.ts @@ -0,0 +1,317 @@ +import type { Part } from "@google/genai"; +import type { ProviderContext, ProviderModel, VideoContent } from "../provider-types.js"; +import { + coerceTransportToolCallArguments, + sanitizeTransportPayloadText, +} from "../transports/transport-stream-shared.js"; +import type { Tool } from "../types.js"; +import { sortPromptCacheToolsByName } from "../utils/prompt-cache-stability.js"; +import { sanitizeSurrogates } from "../utils/sanitize-unicode.js"; +import { + describeToolResultMediaPlaceholder, + extractToolResultText, + isImageWithMediaPayload, +} from "./tool-result-text.js"; + +type GoogleContentPart = Part & Record; +type GoogleContent = { role: string; parts: GoogleContentPart[] }; +const GEMINI_THOUGHT_SIGNATURE_VALIDATOR_SKIP = "skip_thought_signature_validator"; + +// SDK history preserves bytes; managed replay also trims and rejects malformed padding. +function resolveThoughtSignature(value: string | undefined): string | undefined { + return value && value.length % 4 === 0 && /^[A-Za-z0-9+/]+={0,2}$/.test(value) + ? value + : undefined; +} + +function stableStringifyGoogleToolCallValue(value: unknown): string { + if (Array.isArray(value)) { + return `[${value.map((item) => stableStringifyGoogleToolCallValue(item)).join(",")}]`; + } + if (value && typeof value === "object") { + return `{${Object.keys(value) + .toSorted() + .map( + (key) => + `${JSON.stringify(key)}:${stableStringifyGoogleToolCallValue(Reflect.get(value, key))}`, + ) + .join(",")}}`; + } + return JSON.stringify(value); +} + +function sanitizeGeminiThoughtSignature(value: string | undefined): string | undefined { + if (typeof value !== "string") { + return undefined; + } + const trimmed = value.trim(); + return trimmed && /^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/.test(trimmed) + ? trimmed + : undefined; +} + +function toolCallThoughtSignatureReplayKey(block: { + id: string; + name: string; + arguments: unknown; +}): string { + return [ + block.id, + block.name, + stableStringifyGoogleToolCallValue(coerceTransportToolCallArguments(block.arguments)), + ].join("\u0000"); +} + +export function requiresGoogleToolCallId(modelId: string): boolean { + return modelId.startsWith("claude-") || modelId.startsWith("gpt-oss-"); +} + +function getGeminiMajorVersion(modelId: string): number | undefined { + const match = modelId.toLowerCase().match(/(?:^|\/)gemini(?:-live)?-(\d+)/); + if (!match) { + return undefined; + } + const majorVersion = match.at(1); + return majorVersion === undefined ? undefined : Number.parseInt(majorVersion, 10); +} + +function supportsMultimodalFunctionResponse(modelId: string): boolean { + const geminiMajorVersion = getGeminiMajorVersion(modelId); + if (geminiMajorVersion !== undefined) { + return geminiMajorVersion >= 3; + } + return true; +} + +/** Project a prepared transcript; route repair and trusted video admission remain caller-owned. */ +export function projectGoogleMessages(params: { + model: Pick; + messages: ProviderContext["messages"]; + replay: "signed-parts" | "managed"; + requiresToolCallSignature: boolean; + videoPart?: (video: VideoContent) => GoogleContentPart; +}): GoogleContent[] { + const { model, messages: transformedMessages } = params; + const managed = params.replay === "managed"; + const sanitizeText = managed ? sanitizeTransportPayloadText : sanitizeSurrogates; + const signature = managed ? sanitizeGeminiThoughtSignature : resolveThoughtSignature; + const replaySignatures = new Map(); + const contents: GoogleContent[] = []; + // Parallel calls need one immediate function-response turn. Gemini < 3 images cannot + // live inside functionResponse, so hold them until the consecutive result run ends. + const pendingToolResultImageTurns: GoogleContent[] = []; + const sameRouteToolCallIds = new Set(); + let activeToolResultParts: GoogleContentPart[] | undefined; + const flushToolResultRun = (): void => { + contents.push(...pendingToolResultImageTurns); + pendingToolResultImageTurns.length = 0; + activeToolResultParts = undefined; + }; + + for (const msg of transformedMessages) { + if (msg.role !== "toolResult") { + flushToolResultRun(); + } + if (msg.role === "user") { + if (typeof msg.content === "string") { + contents.push({ + role: "user", + parts: [{ text: sanitizeText(msg.content) || " " }], + }); + } else { + const parts: GoogleContentPart[] = msg.content.map((item) => { + if (item.type === "text") { + return { text: sanitizeText(item.text) || " " }; + } + if (managed && item.type === "video") { + return ( + params.videoPart?.(item) ?? { text: "(video omitted: native video slot unavailable)" } + ); + } + return { + inlineData: { + mimeType: item.mimeType, + data: item.data, + }, + }; + }); + const visibleParts = + managed && !model.input.includes("image") + ? parts.filter((part) => !part.inlineData) + : parts; + if (visibleParts.length === 0) { + visibleParts.push({ text: " " }); + } + contents.push({ + role: "user", + parts: visibleParts, + }); + } + } else if (msg.role === "assistant") { + const parts: GoogleContentPart[] = []; + let sawFunctionCall = false; + const nextReplaySignatures = new Map(); + const isSameProviderAndModel = + msg.provider === model.provider && msg.api === model.api && msg.model === model.id; + + for (const block of msg.content) { + if (block.type === "text") { + const thoughtSignature = isSameProviderAndModel + ? signature(block.textSignature) + : undefined; + if ((!block.text || block.text.trim() === "") && (managed || !thoughtSignature)) { + continue; + } + parts.push({ + text: sanitizeText(block.text), + ...(thoughtSignature && { thoughtSignature }), + }); + } else if (block.type === "thinking") { + const thoughtSignature = isSameProviderAndModel + ? signature(block.thinkingSignature) + : undefined; + if ((!block.thinking || block.thinking.trim() === "") && (managed || !thoughtSignature)) { + continue; + } + if (isSameProviderAndModel) { + parts.push({ + thought: true, + text: sanitizeText(block.thinking), + ...(thoughtSignature && { thoughtSignature }), + }); + } else { + parts.push({ + text: sanitizeText(block.thinking), + }); + } + } else if (block.type === "toolCall") { + if (isSameProviderAndModel && (managed || model.provider !== "google-gemini-cli")) { + sameRouteToolCallIds.add(block.id); + } + const args = coerceTransportToolCallArguments(block.arguments); + const ownSignature = isSameProviderAndModel + ? signature(block.thoughtSignature) + : undefined; + const replayKey = managed ? toolCallThoughtSignatureReplayKey(block) : ""; + if (managed && ownSignature) { + nextReplaySignatures.set(replayKey, ownSignature); + } + const thoughtSignature = + ownSignature ?? + (managed && params.requiresToolCallSignature && isSameProviderAndModel + ? replaySignatures.get(replayKey) + : undefined) ?? + (params.requiresToolCallSignature && (managed || !sawFunctionCall) + ? GEMINI_THOUGHT_SIGNATURE_VALIDATOR_SKIP + : undefined); + sawFunctionCall = true; + const part: GoogleContentPart = { + functionCall: { + name: block.name, + args, + ...((managed ? isSameProviderAndModel : sameRouteToolCallIds.has(block.id)) || + requiresGoogleToolCallId(model.id) + ? { id: block.id } + : {}), + }, + ...(thoughtSignature && { thoughtSignature }), + }; + parts.push(part); + } + } + + for (const [key, value] of nextReplaySignatures) { + replaySignatures.set(key, value); + } + if (parts.length === 0) { + continue; + } + contents.push({ + role: "model", + parts, + }); + } else if (msg.role === "toolResult") { + const textResult = extractToolResultText(msg.content); + const imageContent = model.input.includes("image") + ? msg.content.filter((item): item is Extract => + managed + ? item.type === "image" && describeToolResultMediaPlaceholder([item]) !== undefined + : isImageWithMediaPayload(item), + ) + : []; + + const hasText = textResult.length > 0; + const hasImages = imageContent.length > 0; + const mediaPlaceholder = describeToolResultMediaPlaceholder(msg.content); + + const modelSupportsMultimodalFunctionResponse = supportsMultimodalFunctionResponse(model.id); + + // Use "output" key for success, "error" key for errors as per SDK documentation + const responseValue = hasText ? sanitizeText(textResult) : (mediaPlaceholder ?? ""); + + const imageParts: GoogleContentPart[] = imageContent.map((imageBlock) => ({ + inlineData: { + mimeType: imageBlock.mimeType, + data: imageBlock.data, + }, + })); + + const includeId = + sameRouteToolCallIds.has(msg.toolCallId) || requiresGoogleToolCallId(model.id); + const functionResponsePart: GoogleContentPart = { + functionResponse: { + name: msg.toolName, + response: msg.isError ? { error: responseValue } : { output: responseValue }, + ...(hasImages && modelSupportsMultimodalFunctionResponse && { parts: imageParts }), + ...(includeId ? { id: msg.toolCallId } : {}), + }, + }; + + // Cloud Code Assist API requires all function responses to be in a single user turn. + if (activeToolResultParts) { + activeToolResultParts.push(functionResponsePart); + } else { + activeToolResultParts = [functionResponsePart]; + contents.push({ + role: "user", + parts: activeToolResultParts, + }); + } + + // For Gemini < 3, add images in a separate user message + if (hasImages && !modelSupportsMultimodalFunctionResponse) { + pendingToolResultImageTurns.push({ + role: "user", + parts: [{ text: "Tool result image:" }, ...imageParts], + }); + } + } + } + + flushToolResultRun(); + if (contents.length === 0) { + contents.push({ role: "user", parts: [{ text: " " }] }); + } + return contents; +} + +/** + * Convert tools to Gemini function declarations format. + * @internal Directly tested provider implementation detail. + */ +export function convertGoogleTools( + tools: Tool[], +): { functionDeclarations: Record[] }[] | undefined { + if (tools.length === 0) { + return undefined; + } + return [ + { + functionDeclarations: sortPromptCacheToolsByName(tools).map((tool) => ({ + name: tool.name, + description: tool.description, + parametersJsonSchema: tool.parameters, + })), + }, + ]; +} diff --git a/packages/ai/src/providers/google-shared.convert.test.ts b/packages/ai/src/providers/google-shared.convert.test.ts index 3b1ac80b9ec1..bfa15a024886 100644 --- a/packages/ai/src/providers/google-shared.convert.test.ts +++ b/packages/ai/src/providers/google-shared.convert.test.ts @@ -1,9 +1,10 @@ import { expectDefined } from "@openclaw/normalization-core"; import { describe, expect, it } from "vitest"; import type { Context, Tool } from "../types.js"; -import { convertMessages, convertTools } from "./google-shared.js"; +import { convertGoogleTools } from "./google-messages.js"; import { assertRecord, + convertMessages, expectConvertedRoles, getFirstToolParameters, makeGeminiCliAssistantMessage, @@ -25,8 +26,8 @@ describe("google-shared convertTools", () => { { name: "alpha", description: "First", parameters: { type: "object" } }, ] as Tool[]; - expect(convertTools(tools)).toEqual(convertTools(tools.toReversed())); - expect(convertTools(tools)?.[0]?.functionDeclarations.map((tool) => tool.name)).toEqual([ + expect(convertGoogleTools(tools)).toEqual(convertGoogleTools(tools.toReversed())); + expect(convertGoogleTools(tools)?.[0]?.functionDeclarations.map((tool) => tool.name)).toEqual([ "alpha", "zeta", ]); @@ -46,7 +47,7 @@ describe("google-shared convertTools", () => { }, ] as unknown as Tool[]; - const converted = convertTools(tools); + const converted = convertGoogleTools(tools); const params = getFirstToolParameters( converted as Parameters[0], ); @@ -90,7 +91,7 @@ describe("google-shared convertTools", () => { }, ] as unknown as Tool[]; - const converted = convertTools(tools); + const converted = convertGoogleTools(tools); const params = getFirstToolParameters( converted as Parameters[0], ); @@ -133,7 +134,7 @@ describe("google-shared convertTools", () => { }, ] as unknown as Tool[]; - const converted = convertTools(tools); + const converted = convertGoogleTools(tools); const params = getFirstToolParameters( converted as Parameters[0], ); diff --git a/packages/ai/src/providers/google-shared.parallel-image-repro.test.ts b/packages/ai/src/providers/google-shared.parallel-image-repro.test.ts index 5baf401191f9..bbca31a4d980 100644 --- a/packages/ai/src/providers/google-shared.parallel-image-repro.test.ts +++ b/packages/ai/src/providers/google-shared.parallel-image-repro.test.ts @@ -2,8 +2,7 @@ import type { Part } from "@google/genai"; import { expectDefined } from "@openclaw/normalization-core"; import { describe, expect, it } from "vitest"; import type { Context, Model } from "../types.js"; -import { convertMessages } from "./google-shared.js"; -import { makeGoogleAssistantMessage } from "./google-shared.test-helpers.js"; +import { convertMessages, makeGoogleAssistantMessage } from "./google-shared.test-helpers.js"; const convertMessagesForTest = convertMessages as unknown as ( model: Model<"google-generative-ai">, diff --git a/packages/ai/src/providers/google-shared.test-helpers.ts b/packages/ai/src/providers/google-shared.test-helpers.ts index 5868ed23cd6a..d88c1a31bd32 100644 --- a/packages/ai/src/providers/google-shared.test-helpers.ts +++ b/packages/ai/src/providers/google-shared.test-helpers.ts @@ -1,6 +1,14 @@ // Google provider test helpers assert converted message and stream payloads. import { expect } from "vitest"; import type { Model } from "../types.js"; +import { buildGoogleGenerateContentParams } from "./google-shared.js"; + +export function convertMessages( + model: Parameters[0], + context: Parameters[1], +) { + return buildGoogleGenerateContentParams(model, context).contents; +} function makeZeroUsageSnapshot() { return { diff --git a/packages/ai/src/providers/google-shared.test.ts b/packages/ai/src/providers/google-shared.test.ts index 0e02b8143b5c..ce4ebaf9bd83 100644 --- a/packages/ai/src/providers/google-shared.test.ts +++ b/packages/ai/src/providers/google-shared.test.ts @@ -17,11 +17,11 @@ import { SYSTEM_PROMPT_CACHE_BOUNDARY } from "../utils/system-prompt-cache-bound import { buildGoogleGenerateContentParams, buildGoogleSimpleThinking, - convertMessages, - consumeGoogleGenerateContentStream, createGoogleAssistantOutput, runGoogleGenerateContentLifecycle, } from "./google-shared.js"; +import { convertMessages } from "./google-shared.test-helpers.js"; +import { consumeGoogleGenerateContentStream } from "./google-stream.js"; const model: Model<"google-generative-ai"> = { id: "gemini-test", diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index c50105b618ab..aa177610c1ef 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -1,24 +1,20 @@ import { type Content, - FinishReason, FunctionCallingConfigMode, type GenerateContentConfig, type GenerateContentParameters, type GenerateContentResponse, - type Part, type ThinkingConfig, ThinkingLevel, } from "@google/genai"; -import { appendAssistantThinking } from "@openclaw/llm-core/event-stream"; /** * Shared utilities for Google Generative AI and Google Vertex providers. */ -import { calculateCost, clampThinkingLevel } from "../model-utils.js"; +import { clampThinkingLevel } from "../model-utils.js"; import { transformProviderMessages as transformMessages } from "../provider-transcript-transform.js"; import { googleFlashSupportsMinimalThinking } from "../transports/google-thinking-level.js"; import { assignTransportErrorDetails, - coerceTransportToolCallArguments, notifyProviderStreamOpened, transportAbortError, } from "../transports/transport-stream-shared.js"; @@ -28,32 +24,22 @@ import type { Context, Model, SimpleStreamOptions, - StopReason, - TextContent, ThinkingBudgets, - ThinkingContent, ThinkingLevel as AgentThinkingLevel, - Tool, - ToolCall, StreamOptions, } from "../types.js"; import type { AssistantMessageEventStream } from "../utils/event-stream.js"; -import { notifyLlmRequestActivity } from "../utils/llm-request-activity.js"; -import { sortPromptCacheToolsByName } from "../utils/prompt-cache-stability.js"; import { sanitizeSurrogates } from "../utils/sanitize-unicode.js"; import { stripSystemPromptCacheBoundary } from "../utils/system-prompt-cache-boundary.js"; import { - describeToolResultMediaPlaceholder, - extractToolResultText, - isImageWithMediaPayload, -} from "./tool-result-text.js"; + projectGoogleMessages, + requiresGoogleToolCallId, + convertGoogleTools, +} from "./google-messages.js"; +import { consumeGoogleGenerateContentStream } from "./google-stream.js"; type GoogleApiType = "google-generative-ai" | "google-vertex"; -// Google-owned SDK resource spellings identify the same model; other publishers do not. -const GOOGLE_MODEL_RESOURCE_PREFIX = - /^(?:(?:projects\/[^/]+\/locations\/[^/]+\/)?publishers\/google\/models\/|google\/|models\/)/u; - type GoogleThinkingLevel = `${ThinkingLevel}`; type GoogleToolChoice = "auto" | "none" | "any"; @@ -79,314 +65,17 @@ type GoogleGenerateContentClient = { type ClampedGoogleThinkingLevel = Exclude; -/** - * Determines whether a streamed Gemini `Part` should be treated as "thinking". - * - * Protocol note (Gemini / Vertex AI thought signatures): - * - `thought: true` is the definitive marker for thinking content (thought summaries). - * - `thoughtSignature` is an encrypted representation of the model's internal thought process - * used to preserve reasoning context across multi-turn interactions. - * - `thoughtSignature` can appear on ANY part type (text, functionCall, etc.) - it does NOT - * indicate the part itself is thinking content. - * - For non-functionCall responses, the signature appears on the last part for context replay. - * - When persisting/replaying model outputs, signature-bearing parts must be preserved as-is; - * do not merge/move signatures across parts. - * - * See: https://ai.google.dev/gemini-api/docs/thought-signatures - */ -function isThinkingPart(part: Pick): boolean { - return part.thought === true; -} - -/** - * Retain thought signatures during streaming. - * - * Some backends only send `thoughtSignature` on the first delta for a given part/block; later deltas may omit it. - * This helper preserves the last non-empty signature for the current block. - * - * Note: this does NOT merge or move signatures across distinct response parts. It only prevents - * a signature from being overwritten with `undefined` within the same streamed block. - * @internal Directly tested provider implementation detail. - */ -function retainThoughtSignature( - existing: string | undefined, - incoming: string | undefined, -): string | undefined { - if (typeof incoming === "string" && incoming.length > 0) { - return incoming; - } - return existing; -} - -// Thought signatures must be base64 for Google APIs (TYPE_BYTES). -const base64SignaturePattern = /^[A-Za-z0-9+/]+={0,2}$/; - -function isValidThoughtSignature(signature: string | undefined): boolean { - if (!signature) { - return false; - } - if (signature.length % 4 !== 0) { - return false; - } - return base64SignaturePattern.test(signature); -} - -/** - * Only keep signatures from the same provider/model and with valid base64. - */ -function resolveThoughtSignature( - isSameProviderAndModel: boolean, - signature: string | undefined, -): string | undefined { - return isSameProviderAndModel && isValidThoughtSignature(signature) ? signature : undefined; -} - -/** - * Models via Google APIs that require explicit tool call IDs in function calls/responses. - * @internal Directly tested provider implementation detail. - */ -function requiresToolCallId(modelId: string): boolean { - return modelId.startsWith("claude-") || modelId.startsWith("gpt-oss-"); -} - -function getGeminiMajorVersion(modelId: string): number | undefined { - const match = modelId.toLowerCase().match(/(?:^|\/)gemini(?:-live)?-(\d+)/); - if (!match) { - return undefined; - } - const majorVersion = match.at(1); - return majorVersion === undefined ? undefined : Number.parseInt(majorVersion, 10); -} - -function supportsMultimodalFunctionResponse(modelId: string): boolean { - const geminiMajorVersion = getGeminiMajorVersion(modelId); - if (geminiMajorVersion !== undefined) { - return geminiMajorVersion >= 3; - } - return true; -} - -/** - * Convert internal messages to Gemini Content[] format. - * @internal Directly tested provider implementation detail. - */ -export function convertMessages( - model: Model, - context: Context, -): Content[] { - const contents: Content[] = []; - const normalizeToolCallId = (id: string): string => { - if (!requiresToolCallId(model.id)) { - return id; - } - return id.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64); - }; - - const transformedMessages = transformMessages(context.messages, model, normalizeToolCallId); - const requiresToolCallThoughtSignature = - model.provider !== "google-gemini-cli" && - (isGemini3ProModel(model) || isGemini3FlashModel(model)); - // Parallel calls need one immediate function-response turn. Gemini < 3 images cannot - // live inside functionResponse, so hold them until the consecutive result run ends. - const pendingToolResultImageTurns: Content[] = []; - const sameRouteToolCallIds = new Set(); - let activeToolResultParts: Part[] | undefined; - const flushToolResultRun = (): void => { - contents.push(...pendingToolResultImageTurns); - pendingToolResultImageTurns.length = 0; - activeToolResultParts = undefined; - }; - - for (const msg of transformedMessages) { - if (msg.role !== "toolResult") { - flushToolResultRun(); - } - if (msg.role === "user") { - if (typeof msg.content === "string") { - contents.push({ - role: "user", - parts: [{ text: sanitizeSurrogates(msg.content) || " " }], - }); - } else { - const parts: Part[] = msg.content.map((item) => { - if (item.type === "text") { - return { text: sanitizeSurrogates(item.text) || " " }; - } - return { - inlineData: { - mimeType: item.mimeType, - data: item.data, - }, - }; - }); - if (parts.length === 0) { - parts.push({ text: " " }); - } - contents.push({ - role: "user", - parts, - }); - } - } else if (msg.role === "assistant") { - const parts: Part[] = []; - let sawFunctionCall = false; - // Check if message is from same provider and model - only then keep thinking blocks - const isSameProviderAndModel = - msg.provider === model.provider && msg.api === model.api && msg.model === model.id; - - for (const block of msg.content) { - if (block.type === "text") { - const thoughtSignature = resolveThoughtSignature( - isSameProviderAndModel, - block.textSignature, - ); - if ((!block.text || block.text.trim() === "") && !thoughtSignature) { - continue; - } - parts.push({ - text: sanitizeSurrogates(block.text), - ...(thoughtSignature && { thoughtSignature }), - }); - } else if (block.type === "thinking") { - const thoughtSignature = resolveThoughtSignature( - isSameProviderAndModel, - block.thinkingSignature, - ); - if ((!block.thinking || block.thinking.trim() === "") && !thoughtSignature) { - continue; - } - // Only keep as thinking block if same provider AND same model - // Otherwise convert to plain text (no tags to avoid model mimicking them) - if (isSameProviderAndModel) { - parts.push({ - thought: true, - text: sanitizeSurrogates(block.thinking), - ...(thoughtSignature && { thoughtSignature }), - }); - } else { - parts.push({ - text: sanitizeSurrogates(block.thinking), - }); - } - } else if (block.type === "toolCall") { - if (isSameProviderAndModel && model.provider !== "google-gemini-cli") { - sameRouteToolCallIds.add(block.id); - } - const args = coerceTransportToolCallArguments(block.arguments); - const ownSignature = resolveThoughtSignature( - isSameProviderAndModel, - block.thoughtSignature, - ); - const thoughtSignature = - ownSignature ?? - (!sawFunctionCall && requiresToolCallThoughtSignature - ? "skip_thought_signature_validator" - : undefined); - sawFunctionCall = true; - const part: Part = { - functionCall: { - name: block.name, - args, - ...(sameRouteToolCallIds.has(block.id) || requiresToolCallId(model.id) - ? { id: block.id } - : {}), - }, - ...(thoughtSignature && { thoughtSignature }), - }; - parts.push(part); - } - } - - if (parts.length === 0) { - continue; - } - contents.push({ - role: "model", - parts, - }); - } else if (msg.role === "toolResult") { - // Extract text and image content - const textResult = extractToolResultText(msg.content); - const imageContent = model.input.includes("image") - ? msg.content.filter(isImageWithMediaPayload) - : []; - - const hasText = textResult.length > 0; - const hasImages = imageContent.length > 0; - const mediaPlaceholder = describeToolResultMediaPlaceholder(msg.content); - - // Gemini 3+ models support multimodal function responses with images nested inside - // functionResponse.parts. Claude and other non-Gemini models behind Cloud Code Assist / - // Gemini < 3 still needs a separate user image turn. - const modelSupportsMultimodalFunctionResponse = supportsMultimodalFunctionResponse(model.id); - - // Use "output" key for success, "error" key for errors as per SDK documentation - const responseValue = hasText ? sanitizeSurrogates(textResult) : (mediaPlaceholder ?? ""); - - const imageParts: Part[] = imageContent.map((imageBlock) => ({ - inlineData: { - mimeType: imageBlock.mimeType, - data: imageBlock.data, - }, - })); - - const includeId = sameRouteToolCallIds.has(msg.toolCallId) || requiresToolCallId(model.id); - const functionResponsePart: Part = { - functionResponse: { - name: msg.toolName, - response: msg.isError ? { error: responseValue } : { output: responseValue }, - ...(hasImages && modelSupportsMultimodalFunctionResponse && { parts: imageParts }), - ...(includeId ? { id: msg.toolCallId } : {}), - }, - }; - - // Cloud Code Assist API requires all function responses to be in a single user turn. - if (activeToolResultParts) { - activeToolResultParts.push(functionResponsePart); - } else { - activeToolResultParts = [functionResponsePart]; - contents.push({ - role: "user", - parts: activeToolResultParts, - }); - } - - // For Gemini < 3, add images in a separate user message - if (hasImages && !modelSupportsMultimodalFunctionResponse) { - pendingToolResultImageTurns.push({ - role: "user", - parts: [{ text: "Tool result image:" }, ...imageParts], - }); - } - } - } - - flushToolResultRun(); - if (contents.length === 0) { - contents.push({ role: "user", parts: [{ text: " " }] }); - } - return contents; -} - -/** - * Convert tools to Gemini function declarations format. - * @internal Directly tested provider implementation detail. - */ -export function convertTools( - tools: Tool[], -): { functionDeclarations: Record[] }[] | undefined { - if (tools.length === 0) { - return undefined; - } - return [ - { - functionDeclarations: sortPromptCacheToolsByName(tools).map((tool) => ({ - name: tool.name, - description: tool.description, - parametersJsonSchema: tool.parameters, - })), - }, - ]; +function convertMessages(model: Model, context: Context): Content[] { + return projectGoogleMessages({ + model, + messages: transformMessages(context.messages, model, (id) => + requiresGoogleToolCallId(model.id) ? id.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64) : id, + ), + replay: "signed-parts", + requiresToolCallSignature: + model.provider !== "google-gemini-cli" && + (isGemini3ProModel(model) || isGemini3FlashModel(model)), + }); } /** @@ -484,7 +173,7 @@ export function buildGoogleGenerateContentParams( model: Model, context: Context, options: GoogleProviderOptions = {}, -): GenerateContentParameters { +): Omit & { contents: Content[] } { const contents = convertMessages(model, context); const generationConfig: GenerateContentConfig = {}; @@ -503,7 +192,7 @@ export function buildGoogleGenerateContentParams( ...(context.systemPrompt && { systemInstruction: sanitizeSurrogates(stripSystemPromptCacheBoundary(context.systemPrompt)), }), - ...(context.tools && context.tools.length > 0 && { tools: convertTools(context.tools) }), + ...(context.tools && context.tools.length > 0 && { tools: convertGoogleTools(context.tools) }), }; if (context.tools && context.tools.length > 0 && options.toolChoice) { @@ -718,333 +407,3 @@ function getGoogleBudget( return -1; } - -/** - * Map Gemini FinishReason to our StopReason. - * @internal Directly tested provider implementation detail. - */ -function mapStopReason(reason: FinishReason): StopReason { - switch (reason) { - case FinishReason.STOP: - return "stop"; - case FinishReason.MAX_TOKENS: - return "length"; - case FinishReason.BLOCKLIST: - case FinishReason.PROHIBITED_CONTENT: - case FinishReason.SPII: - case FinishReason.SAFETY: - case FinishReason.IMAGE_SAFETY: - case FinishReason.IMAGE_PROHIBITED_CONTENT: - case FinishReason.IMAGE_RECITATION: - case FinishReason.IMAGE_OTHER: - case FinishReason.RECITATION: - case FinishReason.FINISH_REASON_UNSPECIFIED: - case FinishReason.OTHER: - case FinishReason.LANGUAGE: - case FinishReason.MALFORMED_FUNCTION_CALL: - case FinishReason.TOO_MANY_TOOL_CALLS: - case FinishReason.UNEXPECTED_TOOL_CALL: - case FinishReason.NO_IMAGE: - return "error"; - default: { - const exhaustive: never = reason; - throw new Error(`Unhandled stop reason: ${String(exhaustive)}`); - } - } -} - -/** @internal Directly tested provider implementation detail. */ -export async function consumeGoogleGenerateContentStream(params: { - chunks: AsyncIterable; - model: Model; - output: AssistantMessage; - stream: AssistantMessageEventStream; - signal?: AbortSignal; - nextToolCallId: (name: string | undefined) => string; -}): Promise { - params.stream.push({ type: "start", partial: params.output }); - let currentBlock: TextContent | ThinkingContent | null = null; - const blocks = params.output.content; - let sawTerminalReason = false; - let terminalGenerationError: (Error & { code: string; type: string }) | undefined; - const knownUsage = { - promptTokenCount: 0, - cachedContentTokenCount: 0, - toolUsePromptTokenCount: 0, - candidatesTokenCount: 0, - thoughtsTokenCount: 0, - }; - const toolCallIds = new Set(); - for (const block of blocks) { - if (block.type === "toolCall") { - toolCallIds.add(block.id); - } - } - const blockIndex = () => blocks.length - 1; - - const endCurrentBlock = () => { - if (!currentBlock) { - return; - } - if (currentBlock.type === "text") { - params.stream.push({ - type: "text_end", - contentIndex: blockIndex(), - content: currentBlock.text, - partial: params.output, - }); - } else { - params.stream.push({ - type: "thinking_end", - contentIndex: blockIndex(), - content: currentBlock.thinking, - partial: params.output, - }); - } - currentBlock = null; - }; - - for await (const chunk of params.chunks) { - notifyLlmRequestActivity(params.signal); - params.output.responseId ||= chunk.responseId; - const responseModel = chunk.modelVersion?.trim(); - if ( - responseModel && - params.model.id.replace(GOOGLE_MODEL_RESOURCE_PREFIX, "") !== - responseModel.replace(GOOGLE_MODEL_RESOURCE_PREFIX, "") - ) { - params.output.responseModel ||= responseModel; - } - if (chunk.usageMetadata) { - for (const field of Object.keys(knownUsage) as Array) { - const value = chunk.usageMetadata[field]; - if (typeof value === "number") { - knownUsage[field] = value; - } - } - const promptTokens = knownUsage.promptTokenCount; - const cacheRead = knownUsage.cachedContentTokenCount; - const toolUsePromptTokens = knownUsage.toolUsePromptTokenCount; - const outputTokens = knownUsage.candidatesTokenCount + knownUsage.thoughtsTokenCount; - params.output.usage = { - input: Math.max(0, promptTokens - cacheRead) + toolUsePromptTokens, - output: outputTokens, - cacheRead, - cacheWrite: 0, - totalTokens: - chunk.usageMetadata.totalTokenCount ?? promptTokens + outputTokens + toolUsePromptTokens, - cost: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - total: 0, - }, - }; - calculateCost(params.model, params.output.usage); - } - const candidate = chunk.candidates?.[0]; - const promptFeedback = chunk.promptFeedback; - if (!candidate && promptFeedback) { - const blockReason = promptFeedback.blockReason ?? "PROMPT_BLOCKED"; - const blockMessage = promptFeedback.blockReasonMessage?.trim(); - params.output.errorCode = blockReason; - params.output.errorType = "google_prompt_blocked"; - throw new Error( - `Google prompt blocked (${blockReason})${blockMessage ? `: ${blockMessage}` : ""}`, - ); - } - if (candidate?.content?.parts) { - for (const [partIndex, part] of candidate.content.parts.entries()) { - const text = part.text; - const hasText = typeof text === "string"; - const hasThoughtSignature = - typeof part.thoughtSignature === "string" && part.thoughtSignature.length > 0; - const signatureOnly = - hasThoughtSignature && - (!hasText || text.length === 0) && - Object.keys(part).every( - (key) => key === "thought" || key === "thoughtSignature" || key === "text", - ); - if (signatureOnly) { - if (!hasText && part.thought !== true) { - const latestBlock = blocks.at(-1); - if ( - partIndex === 0 && - latestBlock?.type === "toolCall" && - !latestBlock.thoughtSignature - ) { - latestBlock.thoughtSignature = retainThoughtSignature( - latestBlock.thoughtSignature, - part.thoughtSignature, - ); - continue; - } - } - // Empty signed Parts have their own wire identity; merging moves an opaque signature. - endCurrentBlock(); - } - - if (hasText || signatureOnly) { - if (currentBlock && (hasThoughtSignature || partIndex > 0)) { - const currentSignature = - currentBlock.type === "thinking" - ? currentBlock.thinkingSignature - : currentBlock.textSignature; - const currentText = - currentBlock.type === "thinking" ? currentBlock.thinking : currentBlock.text; - if ( - currentText.length > 0 && - (currentSignature !== part.thoughtSignature || - (partIndex > 0 && (currentSignature || hasThoughtSignature))) - ) { - endCurrentBlock(); - } - } - const isThinking = isThinkingPart(part); - if ( - !currentBlock || - (isThinking && currentBlock.type !== "thinking") || - (!isThinking && currentBlock.type !== "text") - ) { - endCurrentBlock(); - if (isThinking) { - currentBlock = { type: "thinking", thinking: "", thinkingSignature: undefined }; - params.output.content.push(currentBlock); - params.stream.push({ - type: "thinking_start", - contentIndex: blockIndex(), - partial: params.output, - }); - } else { - currentBlock = { type: "text", text: "" }; - params.output.content.push(currentBlock); - params.stream.push({ - type: "text_start", - contentIndex: blockIndex(), - partial: params.output, - }); - } - } - const delta = hasText ? text : ""; - if (currentBlock.type === "thinking") { - appendAssistantThinking(currentBlock, delta); - currentBlock.thinkingSignature = retainThoughtSignature( - currentBlock.thinkingSignature, - part.thoughtSignature, - ); - params.stream.push({ - type: "thinking_delta", - contentIndex: blockIndex(), - delta, - partial: params.output, - }); - } else { - currentBlock.text += delta; - currentBlock.textSignature = retainThoughtSignature( - currentBlock.textSignature, - part.thoughtSignature, - ); - params.stream.push({ - type: "text_delta", - contentIndex: blockIndex(), - delta, - partial: params.output, - }); - } - if (signatureOnly) { - endCurrentBlock(); - } - } - - if (part.functionCall) { - endCurrentBlock(); - const providedId = part.functionCall.id; - const needsNewId = !providedId || toolCallIds.has(providedId); - const toolCall: ToolCall = { - type: "toolCall", - id: needsNewId ? params.nextToolCallId(part.functionCall.name) : providedId, - name: part.functionCall.name || "", - arguments: (part.functionCall.args as Record) ?? {}, - ...(part.thoughtSignature && { thoughtSignature: part.thoughtSignature }), - }; - - params.output.content.push(toolCall); - toolCallIds.add(toolCall.id); - params.stream.push({ - type: "toolcall_start", - contentIndex: blockIndex(), - partial: params.output, - }); - params.stream.push({ - type: "toolcall_delta", - contentIndex: blockIndex(), - delta: JSON.stringify(toolCall.arguments), - partial: params.output, - }); - params.stream.push({ - type: "toolcall_end", - contentIndex: blockIndex(), - toolCall, - partial: params.output, - }); - } - } - } - - if ( - candidate?.finishReason && - candidate.finishReason !== FinishReason.FINISH_REASON_UNSPECIFIED - ) { - sawTerminalReason = true; - params.output.stopReason = mapStopReason(candidate.finishReason); - if (params.output.stopReason === "error") { - const finishMessage = candidate.finishMessage?.trim(); - terminalGenerationError = Object.assign( - new Error( - `Google generation stopped (${candidate.finishReason})${finishMessage ? `: ${finishMessage}` : ""}`, - ), - { code: candidate.finishReason, type: "google_generation_failed" }, - ); - } - // MAX_TOKENS can leave a complete-looking partial call. Only a normal - // Google stop may promote parsed calls into an executable tool-use turn. - if ( - params.output.stopReason === "stop" && - params.output.content.some((block) => block.type === "toolCall") - ) { - params.output.stopReason = "toolUse"; - } - } - } - - endCurrentBlock(); - - if (params.signal?.aborted) { - throw transportAbortError(params.signal); - } - - if (terminalGenerationError) { - params.output.errorCode = terminalGenerationError.code; - params.output.errorType = terminalGenerationError.type; - throw terminalGenerationError; - } - - if (!sawTerminalReason) { - params.output.errorCode = "STREAM_INCOMPLETE"; - params.output.errorType = "google_incomplete_stream"; - throw new Error("Google stream ended before a terminal finish reason"); - } - - if (params.output.stopReason === "aborted" || params.output.stopReason === "error") { - throw new Error("An unknown error occurred"); - } - - params.stream.push({ - type: "done", - reason: params.output.stopReason, - message: params.output, - }); - params.stream.end(); -} -/* oxlint-disable max-lines -- TODO: split this grandfathered oversized file. */ diff --git a/packages/ai/src/providers/google-stream.ts b/packages/ai/src/providers/google-stream.ts new file mode 100644 index 000000000000..1dd2eec9e668 --- /dev/null +++ b/packages/ai/src/providers/google-stream.ts @@ -0,0 +1,422 @@ +import type { FinishReason } from "@google/genai"; +import { appendAssistantThinking } from "@openclaw/llm-core/event-stream"; +import { calculateCost } from "../model-utils.js"; +import { + transportAbortError, + type WritableTransportStream, +} from "../transports/transport-stream-shared.js"; +import type { + AssistantMessage, + Model, + StopReason, + TextContent, + ThinkingContent, + ToolCall, +} from "../types.js"; +import { notifyLlmRequestActivity } from "../utils/llm-request-activity.js"; + +// Only Google-owned resource spellings identify the same model. +const GOOGLE_MODEL_RESOURCE_PREFIX = + /^(?:(?:projects\/[^/]+\/locations\/[^/]+\/)?publishers\/google\/models\/|google\/|models\/)/u; + +export type GoogleStreamChunk = { + responseId?: string; + modelVersion?: string; + promptFeedback?: { + blockReason?: string; + blockReasonMessage?: string; + }; + candidates?: Array<{ + content?: { + parts?: Array<{ + text?: string; + thought?: boolean; + thoughtSignature?: string; + functionCall?: { + id?: string; + name?: string; + args?: Record; + }; + }>; + }; + finishReason?: string; + finishMessage?: string; + }>; + usageMetadata?: { + promptTokenCount?: number; + cachedContentTokenCount?: number; + candidatesTokenCount?: number; + thoughtsTokenCount?: number; + toolUsePromptTokenCount?: number; + totalTokenCount?: number; + }; +}; + +function retainThoughtSignature( + existing: string | undefined, + incoming: string | undefined, +): string | undefined { + if (typeof incoming === "string" && incoming.length > 0) { + return incoming; + } + return existing; +} + +const stopReasons = new Map( + Object.entries({ + STOP: "stop", + MAX_TOKENS: "length", + BLOCKLIST: "error", + PROHIBITED_CONTENT: "error", + SPII: "error", + SAFETY: "error", + IMAGE_SAFETY: "error", + IMAGE_PROHIBITED_CONTENT: "error", + IMAGE_RECITATION: "error", + IMAGE_OTHER: "error", + RECITATION: "error", + FINISH_REASON_UNSPECIFIED: "error", + OTHER: "error", + LANGUAGE: "error", + MALFORMED_FUNCTION_CALL: "error", + TOO_MANY_TOOL_CALLS: "error", + UNEXPECTED_TOOL_CALL: "error", + NO_IMAGE: "error", + } satisfies Record), +); + +function mapStopReason(reason: string): StopReason { + const mapped = stopReasons.get(reason); + if (!mapped) { + throw new Error(`Unhandled stop reason: ${reason}`); + } + return mapped; +} + +/** @internal Directly tested provider implementation detail. */ +export async function consumeGoogleGenerateContentStream(params: { + chunks: AsyncIterable; + model: Model; + output: AssistantMessage; + stream: WritableTransportStream; + signal?: AbortSignal; + nextToolCallId: (name: string | undefined) => string; + // Preserve the shipped SDK signed-Part and managed SSE delta/error timing contracts. + profile?: "sdk" | "managed"; + normalizeModelId?: (id: string) => string; + resolveStopReason?: (reason: string) => StopReason; +}): Promise { + const preserveParts = params.profile !== "managed"; + const normalizeModelId = + params.normalizeModelId ?? ((id: string) => id.replace(GOOGLE_MODEL_RESOURCE_PREFIX, "")); + params.stream.push({ type: "start", partial: params.output }); + let currentBlock: TextContent | ThinkingContent | null = null; + const blocks = params.output.content; + let sawTerminalReason = false; + let terminalGenerationError: (Error & { code: string; type: string }) | undefined; + const knownUsage = { + promptTokenCount: 0, + cachedContentTokenCount: 0, + toolUsePromptTokenCount: 0, + candidatesTokenCount: 0, + thoughtsTokenCount: 0, + }; + const toolCallIds = new Set(); + for (const block of blocks) { + if (block.type === "toolCall") { + toolCallIds.add(block.id); + } + } + const blockIndex = () => blocks.length - 1; + + const endCurrentBlock = () => { + if (!currentBlock) { + return; + } + if (currentBlock.type === "text") { + params.stream.push({ + type: "text_end", + contentIndex: blockIndex(), + content: currentBlock.text, + partial: params.output, + }); + } else { + params.stream.push({ + type: "thinking_end", + contentIndex: blockIndex(), + content: currentBlock.thinking, + partial: params.output, + }); + } + currentBlock = null; + }; + + for await (const chunk of params.chunks) { + notifyLlmRequestActivity(params.signal); + params.output.responseId ||= chunk.responseId; + const responseModel = chunk.modelVersion?.trim(); + if (responseModel && normalizeModelId(params.model.id) !== normalizeModelId(responseModel)) { + params.output.responseModel ||= responseModel; + } + if (chunk.usageMetadata) { + for (const field of [ + "promptTokenCount", + "cachedContentTokenCount", + "toolUsePromptTokenCount", + "candidatesTokenCount", + "thoughtsTokenCount", + ] as const) { + const value = chunk.usageMetadata[field]; + if (typeof value === "number") { + knownUsage[field] = value; + } + } + const promptTokens = knownUsage.promptTokenCount; + const cacheRead = knownUsage.cachedContentTokenCount; + const toolUsePromptTokens = knownUsage.toolUsePromptTokenCount; + const outputTokens = knownUsage.candidatesTokenCount + knownUsage.thoughtsTokenCount; + params.output.usage = { + input: Math.max(0, promptTokens - cacheRead) + toolUsePromptTokens, + output: outputTokens, + cacheRead, + cacheWrite: 0, + totalTokens: + chunk.usageMetadata.totalTokenCount ?? promptTokens + outputTokens + toolUsePromptTokens, + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + total: 0, + }, + }; + calculateCost(params.model, params.output.usage); + } + const candidate = chunk.candidates?.[0]; + const promptFeedback = chunk.promptFeedback; + if (!candidate && promptFeedback) { + const blockReason = + (preserveParts + ? promptFeedback.blockReason + : promptFeedback.blockReason?.trim() || undefined) ?? "PROMPT_BLOCKED"; + const blockMessage = promptFeedback.blockReasonMessage?.trim(); + if (preserveParts) { + params.output.errorCode = blockReason; + params.output.errorType = "google_prompt_blocked"; + } + throw Object.assign( + new Error( + `Google prompt blocked (${blockReason})${blockMessage ? `: ${blockMessage}` : ""}`, + ), + { code: blockReason, type: "google_prompt_blocked" }, + ); + } + if (candidate?.content?.parts) { + for (const [partIndex, part] of candidate.content.parts.entries()) { + const text = part.text; + const hasText = typeof text === "string"; + const hasThoughtSignature = + typeof part.thoughtSignature === "string" && part.thoughtSignature.length > 0; + const signatureOnly = + preserveParts && + hasThoughtSignature && + (!hasText || text.length === 0) && + Object.keys(part).every( + (key) => key === "thought" || key === "thoughtSignature" || key === "text", + ); + if (signatureOnly || (!preserveParts && hasThoughtSignature && !part.functionCall)) { + if (!hasText && part.thought !== true) { + const latestBlock = blocks.at(-1); + if ( + latestBlock?.type === "toolCall" && + (!preserveParts || (partIndex === 0 && !latestBlock.thoughtSignature)) + ) { + latestBlock.thoughtSignature = retainThoughtSignature( + latestBlock.thoughtSignature, + part.thoughtSignature, + ); + continue; + } + } + // Empty signed Parts have their own wire identity; merging moves an opaque signature. + if (preserveParts) { + endCurrentBlock(); + } + } + + if ( + hasText || + signatureOnly || + (!preserveParts && hasThoughtSignature && !part.functionCall) + ) { + if (preserveParts && currentBlock && (hasThoughtSignature || partIndex > 0)) { + const currentSignature = + currentBlock.type === "thinking" + ? currentBlock.thinkingSignature + : currentBlock.textSignature; + const currentText = + currentBlock.type === "thinking" ? currentBlock.thinking : currentBlock.text; + if ( + currentText.length > 0 && + (currentSignature !== part.thoughtSignature || + (partIndex > 0 && (currentSignature || hasThoughtSignature))) + ) { + endCurrentBlock(); + } + } + const isThinking = part.thought === true || (!preserveParts && !hasText); + if ( + !currentBlock || + (isThinking && currentBlock.type !== "thinking") || + (!isThinking && currentBlock.type !== "text") + ) { + endCurrentBlock(); + if (isThinking) { + currentBlock = { + type: "thinking", + thinking: "", + ...(preserveParts ? { thinkingSignature: undefined } : {}), + }; + params.output.content.push(currentBlock); + params.stream.push({ + type: "thinking_start", + contentIndex: blockIndex(), + partial: params.output, + }); + } else { + currentBlock = { type: "text", text: "" }; + params.output.content.push(currentBlock); + params.stream.push({ + type: "text_start", + contentIndex: blockIndex(), + partial: params.output, + }); + } + } + const delta = hasText ? text : ""; + if (currentBlock.type === "thinking") { + appendAssistantThinking(currentBlock, delta); + currentBlock.thinkingSignature = retainThoughtSignature( + currentBlock.thinkingSignature, + part.thoughtSignature, + ); + params.stream.push({ + type: "thinking_delta", + contentIndex: blockIndex(), + delta, + partial: params.output, + }); + } else { + currentBlock.text += delta; + currentBlock.textSignature = retainThoughtSignature( + currentBlock.textSignature, + part.thoughtSignature, + ); + params.stream.push({ + type: "text_delta", + contentIndex: blockIndex(), + delta, + partial: params.output, + }); + } + if (signatureOnly) { + endCurrentBlock(); + } + } + + if (part.functionCall) { + endCurrentBlock(); + const providedId = part.functionCall.id; + const needsNewId = !providedId || toolCallIds.has(providedId); + const toolCall: ToolCall = { + type: "toolCall", + id: needsNewId ? params.nextToolCallId(part.functionCall.name) : providedId, + name: part.functionCall.name || "", + arguments: part.functionCall.args ?? {}, + ...(part.thoughtSignature && { thoughtSignature: part.thoughtSignature }), + }; + + params.output.content.push(toolCall); + toolCallIds.add(toolCall.id); + params.stream.push({ + type: "toolcall_start", + contentIndex: blockIndex(), + partial: params.output, + }); + params.stream.push({ + type: "toolcall_delta", + contentIndex: blockIndex(), + delta: JSON.stringify(toolCall.arguments), + partial: params.output, + }); + params.stream.push({ + type: "toolcall_end", + contentIndex: blockIndex(), + toolCall, + partial: params.output, + }); + } + } + } + + if (candidate?.finishReason && candidate.finishReason !== "FINISH_REASON_UNSPECIFIED") { + sawTerminalReason = true; + params.output.stopReason = (params.resolveStopReason ?? mapStopReason)( + candidate.finishReason, + ); + if (params.output.stopReason === "error") { + const finishMessage = candidate.finishMessage?.trim(); + terminalGenerationError = Object.assign( + new Error( + `Google generation stopped (${candidate.finishReason})${finishMessage ? `: ${finishMessage}` : ""}`, + ), + { code: candidate.finishReason, type: "google_generation_failed" }, + ); + } + // MAX_TOKENS can leave a complete-looking partial call. Only a normal + // Google stop may promote parsed calls into an executable tool-use turn. + if ( + params.output.stopReason === "stop" && + params.output.content.some((block) => block.type === "toolCall") + ) { + params.output.stopReason = "toolUse"; + } + } + } + + endCurrentBlock(); + + if (params.signal?.aborted) { + throw transportAbortError(params.signal); + } + + if (terminalGenerationError) { + if (preserveParts) { + params.output.errorCode = terminalGenerationError.code; + params.output.errorType = terminalGenerationError.type; + } + throw terminalGenerationError; + } + + if (!sawTerminalReason) { + if (preserveParts) { + params.output.errorCode = "STREAM_INCOMPLETE"; + params.output.errorType = "google_incomplete_stream"; + } + throw Object.assign(new Error("Google stream ended before a terminal finish reason"), { + code: "STREAM_INCOMPLETE", + type: "google_incomplete_stream", + }); + } + + if (params.output.stopReason === "aborted" || params.output.stopReason === "error") { + throw new Error("An unknown error occurred"); + } + + params.stream.push({ + type: "done", + reason: params.output.stopReason, + message: params.output, + }); + params.stream.end(); +} diff --git a/packages/ai/src/transports.ts b/packages/ai/src/transports.ts index 665bd0d2d26d..e97ce05adb67 100644 --- a/packages/ai/src/transports.ts +++ b/packages/ai/src/transports.ts @@ -27,3 +27,12 @@ export { isCodeModeModelVisibleToolName, MALFORMED_STREAMING_FRAGMENT_ERROR_MESSAGE, } from "./transports/transport-utils.js"; +export { + consumeGoogleGenerateContentStream, + type GoogleStreamChunk, +} from "./providers/google-stream.js"; +export { + convertGoogleTools, + projectGoogleMessages, + requiresGoogleToolCallId, +} from "./providers/google-messages.js"; diff --git a/packages/ai/src/transports/anthropic-messages.ts b/packages/ai/src/transports/anthropic-messages.ts new file mode 100644 index 000000000000..199bc290390b --- /dev/null +++ b/packages/ai/src/transports/anthropic-messages.ts @@ -0,0 +1,475 @@ +import type { + CacheControlEphemeral, + ContentBlockParam, + MessageCreateParamsStreaming, + Tool as AnthropicTool, + ImageBlockParam, + TextBlockParam, + ToolResultBlockParam, +} from "@anthropic-ai/sdk/resources/messages.js"; +import type { Context, Model, Tool } from "@openclaw/llm-core"; +import { asOptionalObjectRecord } from "@openclaw/normalization-core/record-coerce"; +import { + createAnthropicInlineImageBudget, + normalizeAnthropicInlineContent, + resolveAnthropicImageMediaType, + type AnthropicInlineImageBudget, +} from "../internal/anthropic-inline-images.js"; +import type { AnthropicOptions, AnthropicThinkingDisplay } from "../provider-options.js"; +import { + requiresClaudeAdaptiveThinking, + supportsClaudeAdaptiveThinking, + supportsClaudeNativeXhighEffort, +} from "../providers/anthropic-model-contract.js"; +import { + ANTHROPIC_OMITTED_REASONING_TEXT, + findActiveAnthropicToolTurnAssistantIndex, +} from "../providers/anthropic-thinking-replay.js"; +import { + toClaudeCodeToolName, + normalizeAnthropicToolChoice, + reconcileAnthropicToolChoice, + projectAnthropicTools, + type AnthropicToolProjection, +} from "../providers/anthropic-tool-projection.js"; +import { + describeToolResultMediaPlaceholder, + extractToolResultBlockText, + extractToolResultText, + isImageWithMediaPayload, +} from "../providers/tool-result-text.js"; +import type { AnthropicCompactionBlock } from "./anthropic-compaction-replay.js"; +import { + coerceTransportToolCallArguments, + sanitizeNonEmptyTransportPayloadText, + sanitizeTransportPayloadText, +} from "./transport-stream-shared.js"; + +type AnthropicReplayBlock = + | ContentBlockParam + | AnthropicCompactionBlock + | { + type: "redacted_thinking"; + data?: string; + }; + +type AnthropicWireMessage = { + role: "user" | "assistant"; + content: string | AnthropicReplayBlock[]; + reasoning_content?: string; +}; + +const NON_VISION_USER_IMAGE_PLACEHOLDER = "(image omitted: model does not support images)"; + +async function convertContentBlocks( + content: readonly unknown[], + model: { input: readonly string[] }, + imageBudget: AnthropicInlineImageBudget, + profile: "provider" | "transport", + isError: boolean, +) { + const text = extractToolResultText(content); + const mediaPlaceholder = describeToolResultMediaPlaceholder(content); + const hasImages = + (profile === "provider" || model.input.includes("image")) && + content.some(isImageWithMediaPayload); + if (!hasImages) { + return sanitizeNonEmptyTransportPayloadText( + text, + mediaPlaceholder ?? + (profile === "transport" ? "(no output)" : isError ? "[tool error with no output]" : ""), + ); + } + const blocks: Array = []; + let hasTextBlock = false; + for (const block of content) { + const record = asOptionalObjectRecord(block); + if (!record) { + continue; + } + const blockText = extractToolResultBlockText(block); + if (blockText) { + blocks.push({ type: "text", text: sanitizeTransportPayloadText(blockText) }); + hasTextBlock = true; + } + if (!isImageWithMediaPayload(record)) { + continue; + } + const [normalizedImage] = await normalizeAnthropicInlineContent( + [ + { + type: "image" as const, + data: typeof record.data === "string" ? record.data : "", + mimeType: + typeof record.mimeType === "string" + ? record.mimeType + : profile === "provider" + ? "image/jpeg" + : "image/png", + }, + ], + imageBudget, + ); + if (normalizedImage?.type !== "image") { + continue; + } + blocks.push({ + type: "image" as const, + source: { + type: "base64", + media_type: resolveAnthropicImageMediaType(normalizedImage.mimeType), + data: normalizedImage.data, + }, + }); + } + if (!hasTextBlock) { + blocks.unshift({ type: "text", text: mediaPlaceholder ?? "(see attached image)" }); + } + return blocks; +} + +export async function convertAnthropicMessages( + transformedMessages: Context["messages"], + model: Model<"anthropic-messages">, + isOAuthToken: boolean, + options: { + allowReasoningContentReplay?: boolean; + compaction?: AnthropicCompactionBlock; + replayThinkingEnabled?: boolean; + allowEmptySignature?: boolean; + profile: "provider" | "transport"; + }, +): Promise { + const params: AnthropicWireMessage[] = []; + const imageBudget = createAnthropicInlineImageBudget(); + const allowReasoningContentReplay = options.allowReasoningContentReplay === true; + const replayThinkingEnabled = options.replayThinkingEnabled !== false; + const managed = options.profile === "transport"; + const activeToolTurnAssistantIndex = replayThinkingEnabled + ? -1 + : findActiveAnthropicToolTurnAssistantIndex(transformedMessages); + for (let i = 0; i < transformedMessages.length; i += 1) { + const msg = transformedMessages[i]; + if (!msg) { + continue; + } + if (msg.role === "user") { + if (typeof msg.content === "string") { + if (msg.content.trim().length > 0) { + const userParam: AnthropicWireMessage = { + role: "user", + content: sanitizeTransportPayloadText(msg.content), + }; + params.push(userParam); + } + continue; + } + const normalizedContent = + !managed || model.input.includes("image") + ? await normalizeAnthropicInlineContent(msg.content, imageBudget) + : msg.content.map((item) => + item.type === "image" + ? { type: "text" as const, text: NON_VISION_USER_IMAGE_PLACEHOLDER } + : item, + ); + const blocks: Array = normalizedContent.map((item) => + item.type === "text" + ? { + type: "text", + text: sanitizeTransportPayloadText(item.text), + } + : { + type: "image", + source: { + type: "base64", + media_type: resolveAnthropicImageMediaType(item.mimeType), + data: item.data, + }, + }, + ); + let filteredBlocks = + !managed || model.input.includes("image") + ? blocks + : blocks.filter((block) => block.type !== "image"); + filteredBlocks = filteredBlocks.filter( + (block) => block.type !== "text" || block.text.trim().length > 0, + ); + if (filteredBlocks.length === 0) { + continue; + } + const userParam: AnthropicWireMessage = { + role: "user", + content: filteredBlocks, + }; + params.push(userParam); + continue; + } + if (msg.role === "assistant") { + const blocks: AnthropicReplayBlock[] = + i === 0 && options.compaction ? [options.compaction] : []; + const reasoningContent: string[] = []; + let omittedThinking = false; + for (const block of msg.content) { + if (block.type === "text") { + if (block.text.trim().length > 0) { + blocks.push({ + type: "text", + text: sanitizeTransportPayloadText(block.text), + }); + } + continue; + } + if (block.type === "thinking") { + const thinkingSignature = block.thinkingSignature?.trim(); + const isReasoningContent = thinkingSignature === "reasoning_content"; + if ( + !replayThinkingEnabled && + i !== activeToolTurnAssistantIndex && + (!managed || !isReasoningContent) + ) { + omittedThinking = true; + continue; + } + if (block.redacted) { + if (!managed && !block.thinkingSignature) { + throw new Error("redacted thinking block is missing its opaque signature"); + } + blocks.push({ + type: "redacted_thinking", + data: block.thinkingSignature, + }); + continue; + } + const hasNativeThinkingSignature = Boolean(thinkingSignature) && !isReasoningContent; + if (block.thinking.trim().length === 0 && !hasNativeThinkingSignature) { + continue; + } + if (!thinkingSignature && !options.allowEmptySignature) { + blocks.push({ + type: "text", + text: sanitizeTransportPayloadText(block.thinking), + }); + } else { + const thinking = + thinkingSignature === "reasoning_content" + ? sanitizeTransportPayloadText(block.thinking) + : block.thinking; + if (thinkingSignature === "reasoning_content") { + if (allowReasoningContentReplay) { + blocks.push({ + type: "thinking", + thinking, + signature: thinkingSignature ?? "", + }); + reasoningContent.push(thinking); + } + continue; + } + blocks.push({ + type: "thinking", + thinking, + signature: thinkingSignature ?? "", + }); + } + continue; + } + if (block.type === "toolCall") { + blocks.push({ + type: "tool_use", + id: block.id, + name: isOAuthToken ? toClaudeCodeToolName(block.name) : block.name, + input: managed + ? coerceTransportToolCallArguments(block.arguments) + : (block.arguments ?? {}), + }); + } + } + if (blocks.length === 0 && omittedThinking) { + blocks.push({ type: "text", text: ANTHROPIC_OMITTED_REASONING_TEXT }); + } + if (blocks.length > 0) { + const assistantMsg: AnthropicWireMessage = { role: "assistant", content: blocks }; + if (reasoningContent.length > 0) { + assistantMsg.reasoning_content = reasoningContent.join("\n"); + } else if (allowReasoningContentReplay) { + blocks.unshift({ + type: "thinking", + thinking: "", + signature: "reasoning_content", + }); + } + params.push(assistantMsg); + } + continue; + } + if (msg.role === "toolResult") { + const toolResult = msg; + const toolResults: ToolResultBlockParam[] = [ + { + type: "tool_result", + tool_use_id: toolResult.toolCallId, + content: await convertContentBlocks( + toolResult.content, + model, + imageBudget, + options.profile, + toolResult.isError, + ), + is_error: toolResult.isError, + }, + ]; + let j = i + 1; + while (j < transformedMessages.length) { + const nextMsg = transformedMessages.at(j); + if (nextMsg?.role !== "toolResult") { + break; + } + toolResults.push({ + type: "tool_result", + tool_use_id: nextMsg.toolCallId, + content: await convertContentBlocks( + nextMsg.content, + model, + imageBudget, + options.profile, + nextMsg.isError, + ), + is_error: nextMsg.isError, + }); + j += 1; + } + i = j - 1; + params.push({ + role: "user", + content: toolResults, + }); + } + } + return params; +} + +/** Shared generation contract, after each entry point resolves its defaults and tool policy. */ +export function buildAnthropicGenerationParams({ + model, + options, + tools, + toolProjection, + profile, +}: { + model: Model<"anthropic-messages">; + options?: AnthropicOptions; + tools?: AnthropicTool[]; + toolProjection?: AnthropicToolProjection; + profile: "provider" | "transport"; +}) { + const params: Pick< + MessageCreateParamsStreaming, + | "temperature" + | "stop_sequences" + | "tools" + | "thinking" + | "output_config" + | "metadata" + | "tool_choice" + > = {}; + const mandatoryAdaptiveThinking = requiresClaudeAdaptiveThinking(model); + // Thinking and post-4.6 Claude models reject custom temperature values. + if ( + options?.temperature !== undefined && + !options?.thinkingEnabled && + !supportsClaudeNativeXhighEffort(model) + ) { + params.temperature = options.temperature; + } + + if (options?.stop !== undefined && options.stop.length > 0) { + params.stop_sequences = options.stop; + } + + if (tools && tools.length > 0) { + params.tools = tools; + } + + // Configure thinking mode: always-on adaptive (Fable 5 and Mythos 5), + // adaptive (Opus 4.6+ and Sonnet 4.6), + // budget-based (older models), or explicitly disabled. + if (mandatoryAdaptiveThinking || model.reasoning || supportsClaudeAdaptiveThinking(model)) { + if (mandatoryAdaptiveThinking || options?.thinkingEnabled) { + // Default to "summarized" so Opus 4.7+ and Mythos Preview behave like + // older Claude 4 models (whose API default is also "summarized"). + const display: AnthropicThinkingDisplay = options?.thinkingDisplay ?? "summarized"; + if (supportsClaudeAdaptiveThinking(model)) { + // Adaptive thinking: Claude decides when and how much to think. + params.thinking = { type: "adaptive", display }; + const effort = options?.effort ?? (mandatoryAdaptiveThinking ? "high" : undefined); + if (effort) { + params.output_config = { effort }; + } + } else { + // Budget-based thinking for older models. + params.thinking = { + type: "enabled", + budget_tokens: options?.thinkingBudgetTokens ?? 1024, + ...(profile === "provider" ? { display } : {}), + }; + } + } else if (options?.thinkingEnabled === false) { + params.thinking = { type: "disabled" }; + } + } + + if (options?.metadata) { + const userId = options.metadata.user_id; + if (typeof userId === "string") { + params.metadata = { user_id: userId }; + } + } + + if (options?.toolChoice) { + const normalizedToolChoice = normalizeAnthropicToolChoice( + mandatoryAdaptiveThinking || options?.thinkingEnabled === true, + options.toolChoice, + ); + const projectedToolChoice = toolProjection + ? reconcileAnthropicToolChoice(normalizedToolChoice, toolProjection) + : normalizedToolChoice; + if (projectedToolChoice) { + params.tool_choice = projectedToolChoice; + } + } + + return params; +} + +export function convertAnthropicTools( + tools: Tool[], + isOAuthTokenLocal: boolean, + supportsEagerToolInputStreaming = false, + cacheControl?: CacheControlEphemeral, +): { + projection: AnthropicToolProjection; + tools: AnthropicTool[]; +} { + const projection = projectAnthropicTools(tools, (name) => + isOAuthTokenLocal ? toClaudeCodeToolName(name) : name, + ); + const convertedTools: AnthropicTool[] = []; + for (const [index, tool] of projection.tools.entries()) { + const convertedTool: AnthropicTool = { + name: tool.wireName, + description: tool.description, + input_schema: tool.inputSchema, + }; + if (supportsEagerToolInputStreaming) { + convertedTool.eager_input_streaming = true; + } + if (cacheControl && index === projection.tools.length - 1) { + convertedTool.cache_control = cacheControl; + } + convertedTools.push(convertedTool); + } + return { + projection, + tools: convertedTools, + }; +} diff --git a/packages/ai/src/transports/anthropic-stream-reducer.ts b/packages/ai/src/transports/anthropic-stream-reducer.ts new file mode 100644 index 000000000000..65bcf1c425e3 --- /dev/null +++ b/packages/ai/src/transports/anthropic-stream-reducer.ts @@ -0,0 +1,663 @@ +import type { AssistantMessage, AssistantMessageEvent, Model } from "@openclaw/llm-core"; +import { appendAssistantThinking } from "@openclaw/llm-core/event-stream"; +import { + asRecord, + asOptionalObjectRecord, + readStringField, +} from "@openclaw/normalization-core/record-coerce"; +import { calculateCost } from "../model-utils.js"; +import type { AnthropicOptions } from "../provider-options.js"; +import { mapAnthropicStopReason } from "../providers/anthropic-model-contract.js"; +import { applyAnthropicRefusal } from "../providers/anthropic-refusal.js"; +import { + applyAnthropicFallbackBoundary, + readAnthropicFallbackBoundary, + resolveAnthropicFallbackServingModelCost, +} from "../providers/anthropic-server-fallback.js"; +import { + logAnthropicThinkingDrops, + readAnthropicInputTransformations, +} from "../providers/anthropic-thinking-replay.js"; +import { + resolveOriginalAnthropicToolName, + type AnthropicToolProjection, +} from "../providers/anthropic-tool-projection.js"; +import { + applyAnthropicMessageDeltaUsage, + applyAnthropicMessageStartUsage, + type AnthropicPromptUsageSnapshot, +} from "../providers/anthropic-usage.js"; +import { tagPendingCommentaryText } from "../utils/assistant-text-phase.js"; +import { createDeferredEventBuffer } from "../utils/deferred-event-buffer.js"; +import { + createToolArgumentPreviewSchedule, + parseStreamingJson, + type ToolArgumentPreviewSchedule, +} from "../utils/json-parse.js"; +import { notifyLlmRequestActivity } from "../utils/llm-request-activity.js"; +import { createCompactionCapture } from "./anthropic-compaction-replay.js"; +import { isDirectAnthropicModel, logAnthropicContextEdits } from "./anthropic-payload-policy.js"; +import { resolveProviderEndpoint } from "./host-policy.js"; +import { parseJsonObjectPreservingUnsafeIntegers } from "./json-unsafe-integers.js"; +import { + coerceTransportToolCallArguments, + finalizeTerminalToolCallArguments, + sanitizeTransportPayloadText, + transportAbortError, + type WritableTransportStream, +} from "./transport-stream-shared.js"; + +export type AnthropicStreamBlock = AssistantMessage["content"][number] & { + index?: number; + partialJson?: string; +}; + +/** One Messages protocol reducer; entry points retain their established preview/replay contracts. */ +export async function consumeAnthropicStream(params: { + events: AsyncIterable | Iterable; + model: Model<"anthropic-messages">; + options: AnthropicOptions & { authProfileId?: string }; + output: AssistantMessage; + stream: WritableTransportStream; + refusalBuffer?: ReturnType>; + isOAuthToken: boolean; + toolProjection?: AnthropicToolProjection; + profile: "provider" | "transport"; +}): Promise { + const { model, options, output, stream, refusalBuffer, isOAuthToken, toolProjection } = params; + const managed = params.profile === "transport"; + const eventSink = refusalBuffer ?? stream; + let costModel = model; + let messageStartPromptUsage: AnthropicPromptUsageSnapshot | undefined; + let inputTransformations: unknown[] | undefined; + const anthropicStream = params.events; + try { + const blocks: AnthropicStreamBlock[] = output.content; + const blockIndexes = new Map(); + // Preview schedules are per active tool call; WeakMap keys die with the block. + const toolArgumentPreviewSchedules = new WeakMap< + Extract, + ToolArgumentPreviewSchedule + >(); + const seededToolArguments = new WeakMap(); + const sealedToolCalls: Array<{ + block: Extract; + contentIndex: number; + }> = []; + const compactionCapture = createCompactionCapture(output, model, options); + // Signature deltas are opaque and only complete at content_block_stop. + // Keep partial bytes out of output so interrupted streams cannot poison replay. + const pendingThinkingSignatures = new Map(); + const allowReasoningContentReplay = + managed && resolveProviderEndpoint(model).endpointClass === "xiaomi-native"; + const reasoningContentThinkingBlocks = new Map(); + const reasoningContentTextBlocks = new Map(); + let sawMessageStop = false; + const pendingTextEnds: Array> = []; + // Hold text_end until tool-boundary classification is known. + const flushPendingTextEnds = () => { + for (const event of pendingTextEnds) { + eventSink.push(event); + } + pendingTextEnds.length = 0; + }; + const emitTextEnd = (event: Extract) => { + if (managed) { + pendingTextEnds.push(event); + } else { + eventSink.push(event); + } + }; + const eventIndexKey = (eventIndex: unknown) => + typeof eventIndex === "number" ? eventIndex : -1; + const appendReasoningContentThinkingDelta = ( + eventIndex: unknown, + rawText: unknown, + ): boolean => { + if (typeof rawText !== "string") { + return false; + } + const text = sanitizeTransportPayloadText(rawText); + if (text.length === 0) { + return false; + } + const key = eventIndexKey(eventIndex); + let contentIndex = reasoningContentThinkingBlocks.get(key); + let block = contentIndex === undefined ? undefined : blocks[contentIndex]; + if (!block || block.type !== "thinking") { + block = { type: "thinking", thinking: "", thinkingSignature: "reasoning_content" }; + output.content.push(block); + contentIndex = output.content.length - 1; + reasoningContentThinkingBlocks.set(key, contentIndex); + eventSink.push({ + type: "thinking_start", + contentIndex, + partial: output, + }); + } + if (contentIndex === undefined) { + return false; + } + appendAssistantThinking(block, text); + block.thinkingSignature = "reasoning_content"; + eventSink.push({ + type: "thinking_delta", + contentIndex, + delta: text, + partial: output, + }); + return true; + }; + const appendReasoningContentTextDelta = (eventIndex: unknown, rawText: unknown): boolean => { + if (typeof rawText !== "string") { + return false; + } + const text = sanitizeTransportPayloadText(rawText); + if (text.length === 0) { + return false; + } + const key = eventIndexKey(eventIndex); + let contentIndex = reasoningContentTextBlocks.get(key); + let block = contentIndex === undefined ? undefined : blocks[contentIndex]; + if (!block || block.type !== "text") { + block = { type: "text", text: "" }; + output.content.push(block); + contentIndex = output.content.length - 1; + reasoningContentTextBlocks.set(key, contentIndex); + eventSink.push({ + type: "text_start", + contentIndex, + partial: output, + }); + } + if (contentIndex === undefined) { + return false; + } + block.text += text; + eventSink.push({ + type: "text_delta", + contentIndex, + delta: text, + partial: output, + }); + return true; + }; + const finishReasoningContentSidecars = (eventIndex: unknown) => { + const key = eventIndexKey(eventIndex); + const thinkingContentIndex = reasoningContentThinkingBlocks.get(key); + if (thinkingContentIndex !== undefined) { + reasoningContentThinkingBlocks.delete(key); + const block = output.content[thinkingContentIndex]; + if (block?.type === "thinking") { + eventSink.push({ + type: "thinking_end", + contentIndex: thinkingContentIndex, + content: block.thinking, + partial: output, + }); + } + } + const textContentIndex = reasoningContentTextBlocks.get(key); + if (textContentIndex === undefined) { + return; + } + reasoningContentTextBlocks.delete(key); + const block = output.content[textContentIndex]; + if (block?.type === "text") { + eventSink.push({ + type: "text_end", + contentIndex: textContentIndex, + content: block.text, + partial: output, + }); + } + }; + for await (const rawEvent of anthropicStream) { + const event = asRecord(rawEvent); + // A serving-model fallback replaces the initial snapshot; report only once at completion. + inputTransformations = readAnthropicInputTransformations(event) ?? inputTransformations; + if (managed) { + notifyLlmRequestActivity(options.signal); + } + if (event.type === "error") { + const error = asOptionalObjectRecord(event.error); + throw new Error(readStringField(error, "message") || "Anthropic Messages stream failed"); + } + if (event.type === "message_start") { + const message = asOptionalObjectRecord(event.message); + const usage = asRecord(message?.usage); + output.responseId = typeof message?.id === "string" ? message.id : undefined; + output.responseModel = typeof message?.model === "string" ? message.model : undefined; + messageStartPromptUsage = applyAnthropicMessageStartUsage(output.usage, usage); + calculateCost(costModel, output.usage); + // Defer start until after message_start so that pre-stream SSE errors + // (e.g. invalid thinking signatures) arrive before any non-error event + // is yielded, keeping yieldedOutput=false in pumpStreamWithRecovery + // and allowing the thinking-block recovery retry to fire. + eventSink.push({ type: "start", partial: output }); + continue; + } + if (event.type === "message_stop") { + sawMessageStop = true; + continue; + } + if (event.type === "content_block_start") { + const contentBlock = asOptionalObjectRecord(event.content_block); + const index = typeof event.index === "number" ? event.index : -1; + if ( + options.anthropicServerCompaction === true && + compactionCapture.begin(index, contentBlock, output.content.length) + ) { + continue; + } + const fallbackBoundary = refusalBuffer ? readAnthropicFallbackBoundary(contentBlock) : null; + if (fallbackBoundary) { + // Server-side fallback boundary: pre-boundary thinking/tool + // blocks must not replay or execute, and the buffered preview + // events reference them, so rebuild the deferred timeline from + // the surviving text prefix the fallback model continued from. + refusalBuffer?.discard(); + sealedToolCalls.length = 0; + pendingTextEnds.length = 0; + blockIndexes.clear(); + pendingThinkingSignatures.clear(); + applyAnthropicFallbackBoundary({ + output, + boundary: fallbackBoundary, + provider: model.provider, + }); + // Fallback-only iteration partials stay outside the serving-model + // estimate. Compaction responses are the exception: usage policy + // aggregates their complete billed iteration list. + costModel = { + ...model, + cost: resolveAnthropicFallbackServingModelCost({ + requestedModelId: model.id, + servingModelId: fallbackBoundary.toModel, + requestedCost: model.cost, + }), + }; + calculateCost(costModel, output.usage); + eventSink.push({ type: "start", partial: output }); + for (const [i, block] of blocks.entries()) { + if (block.type !== "text") { + continue; + } + delete block.index; + eventSink.push({ + type: "text_start", + contentIndex: i, + partial: output, + }); + if (block.text) { + eventSink.push({ + type: "text_delta", + contentIndex: i, + delta: block.text, + partial: output, + }); + } + emitTextEnd({ + type: "text_end", + contentIndex: i, + content: block.text, + partial: output, + }); + } + continue; + } + pendingThinkingSignatures.delete(index); + if (contentBlock?.type === "text") { + const text = + managed && typeof contentBlock.text === "string" + ? sanitizeTransportPayloadText(contentBlock.text) + : ""; + const block: AnthropicStreamBlock = { type: "text", text, index }; + output.content.push(block); + const contentIndex = output.content.length - 1; + blockIndexes.set(index, contentIndex); + eventSink.push({ + type: "text_start", + contentIndex, + partial: output, + }); + if (text.length > 0) { + eventSink.push({ + type: "text_delta", + contentIndex, + delta: text, + partial: output, + }); + } + continue; + } + if (contentBlock?.type === "thinking") { + const thinking = + managed && typeof contentBlock.thinking === "string" ? contentBlock.thinking : ""; + const block: AnthropicStreamBlock = { + type: "thinking", + thinking, + thinkingSignature: + managed && typeof contentBlock.signature === "string" ? contentBlock.signature : "", + index, + }; + output.content.push(block); + const contentIndex = output.content.length - 1; + blockIndexes.set(index, contentIndex); + eventSink.push({ + type: "thinking_start", + contentIndex, + partial: output, + }); + if (thinking.length > 0) { + eventSink.push({ + type: "thinking_delta", + contentIndex, + delta: thinking, + partial: output, + }); + } + continue; + } + if (contentBlock?.type === "redacted_thinking") { + const block: AnthropicStreamBlock = { + type: "thinking", + thinking: "[Reasoning redacted]", + thinkingSignature: typeof contentBlock.data === "string" ? contentBlock.data : "", + redacted: true, + index, + }; + output.content.push(block); + blockIndexes.set(index, output.content.length - 1); + eventSink.push({ + type: "thinking_start", + contentIndex: output.content.length - 1, + partial: output, + }); + continue; + } + if (contentBlock?.type === "tool_use") { + if (managed) { + tagPendingCommentaryText(output.content); + } + flushPendingTextEnds(); + const block: AnthropicStreamBlock = { + type: "toolCall", + id: typeof contentBlock.id === "string" ? contentBlock.id : "", + name: + typeof contentBlock.name === "string" + ? isOAuthToken + ? resolveOriginalAnthropicToolName(contentBlock.name, toolProjection) + : contentBlock.name + : "", + arguments: asRecord(contentBlock.input), + partialJson: "", + index, + }; + output.content.push(block); + blockIndexes.set(index, output.content.length - 1); + // Standalone callers may supply encoded input; terminal validation owns its shape. + seededToolArguments.set(block, managed ? block.arguments : (contentBlock.input ?? {})); + toolArgumentPreviewSchedules.set(block, createToolArgumentPreviewSchedule()); + eventSink.push({ + type: "toolcall_start", + contentIndex: output.content.length - 1, + partial: output, + }); + } + continue; + } + if (event.type === "content_block_delta") { + const delta = asOptionalObjectRecord(event.delta); + const eventIndex = typeof event.index === "number" ? event.index : undefined; + if (eventIndex !== undefined && compactionCapture.delta(eventIndex, delta)) { + continue; + } + let index = eventIndex === undefined ? undefined : blockIndexes.get(eventIndex); + let block = index === undefined ? undefined : blocks[index]; + if (allowReasoningContentReplay) { + const appendedThinking = appendReasoningContentThinkingDelta( + event.index, + delta?.reasoning_content, + ); + const hasNativeAnthropicDelta = + (delta?.type === "text_delta" && typeof delta.text === "string") || + (delta?.type === "thinking_delta" && typeof delta.thinking === "string") || + (delta?.type === "input_json_delta" && typeof delta.partial_json === "string") || + (delta?.type === "signature_delta" && typeof delta.signature === "string"); + let appendedContent = false; + if ( + !hasNativeAnthropicDelta && + typeof delta?.content === "string" && + delta.content.length > 0 + ) { + const text = sanitizeTransportPayloadText(delta.content); + if (text.length > 0) { + if (block?.type === "text" && index !== undefined) { + block.text += text; + eventSink.push({ + type: "text_delta", + contentIndex: index, + delta: text, + partial: output, + }); + appendedContent = true; + } else { + appendedContent = appendReasoningContentTextDelta(event.index, text); + } + } + } + if ((appendedThinking || appendedContent) && !hasNativeAnthropicDelta) { + continue; + } + } + if (managed && !block && delta?.type === "text_delta" && typeof delta.text === "string") { + const recoveredIndex = typeof event.index === "number" ? event.index : blocks.length; + block = { type: "text", text: "", index: recoveredIndex }; + output.content.push(block); + index = output.content.length - 1; + if (typeof event.index === "number") { + blockIndexes.set(event.index, index); + } + eventSink.push({ + type: "text_start", + contentIndex: index, + partial: output, + }); + } + if (index === undefined) { + continue; + } + if ( + block?.type === "text" && + delta?.type === "text_delta" && + typeof delta.text === "string" + ) { + block.text += delta.text; + eventSink.push({ + type: "text_delta", + contentIndex: index, + delta: delta.text, + partial: output, + }); + continue; + } + if ( + block?.type === "thinking" && + delta?.type === "thinking_delta" && + typeof delta.thinking === "string" + ) { + appendAssistantThinking(block, delta.thinking); + eventSink.push({ + type: "thinking_delta", + contentIndex: index, + delta: delta.thinking, + partial: output, + }); + continue; + } + if ( + block?.type === "toolCall" && + delta?.type === "input_json_delta" && + typeof delta.partial_json === "string" + ) { + const partialJson = `${block.partialJson ?? ""}${delta.partial_json}`; + block.partialJson = partialJson; + // Preview refresh is scheduled geometrically; content_block_stop + // re-parses the full buffer authoritatively either way. + if (toolArgumentPreviewSchedules.get(block)?.(partialJson.length)) { + block.arguments = managed + ? coerceTransportToolCallArguments( + parseJsonObjectPreservingUnsafeIntegers(partialJson) ?? + parseStreamingJson(partialJson), + ) + : parseStreamingJson(partialJson); + } + eventSink.push({ + type: "toolcall_delta", + contentIndex: index, + delta: delta.partial_json, + partial: output, + }); + continue; + } + if ( + block?.type === "thinking" && + delta?.type === "signature_delta" && + typeof delta.signature === "string" + ) { + if (!managed) { + block.thinkingSignature = (block.thinkingSignature || "") + delta.signature; + continue; + } + const signatureIndex = eventIndexKey(event.index); + const pendingSignature = pendingThinkingSignatures.get(signatureIndex); + if (pendingSignature === undefined) { + block.thinkingSignature = ""; + pendingThinkingSignatures.set(signatureIndex, delta.signature); + } else { + pendingThinkingSignatures.set(signatureIndex, pendingSignature + delta.signature); + } + } + continue; + } + if (event.type === "content_block_stop") { + const eventIndex = typeof event.index === "number" ? event.index : undefined; + if (eventIndex !== undefined && compactionCapture.complete(eventIndex)) { + continue; + } + const pendingSignature = + eventIndex === undefined ? undefined : pendingThinkingSignatures.get(eventIndex); + if (eventIndex !== undefined) { + pendingThinkingSignatures.delete(eventIndex); + } + const index = eventIndex === undefined ? undefined : blockIndexes.get(eventIndex); + const block = index === undefined ? undefined : blocks[index]; + if (eventIndex === undefined || index === undefined || !block) { + finishReasoningContentSidecars(event.index); + continue; + } + blockIndexes.delete(eventIndex); + delete block.index; + if (block.type === "text") { + emitTextEnd({ + type: "text_end", + contentIndex: index, + content: block.text, + partial: output, + }); + finishReasoningContentSidecars(event.index); + continue; + } + if (block.type === "thinking") { + if (pendingSignature !== undefined) { + block.thinkingSignature = pendingSignature; + } + eventSink.push({ + type: "thinking_end", + contentIndex: index, + content: block.thinking, + partial: output, + }); + finishReasoningContentSidecars(event.index); + continue; + } + if (block.type === "toolCall") { + sealedToolCalls.push({ block, contentIndex: index }); + finishReasoningContentSidecars(event.index); + } + continue; + } + if (event.type === "message_delta") { + logAnthropicContextEdits(event); + const delta = asOptionalObjectRecord(event.delta); + const usage = asOptionalObjectRecord(event.usage); + if (typeof delta?.stop_reason === "string" && delta.stop_reason) { + if (delta.stop_reason === "refusal") { + applyAnthropicRefusal(output, delta.stop_details, model.provider); + } else { + output.stopReason = mapAnthropicStopReason(delta.stop_reason); + } + } + applyAnthropicMessageDeltaUsage(output.usage, usage, messageStartPromptUsage); + calculateCost(costModel, output.usage); + // Gate on the turn CONTAINING a tool call, not the provider's stop_reason + // label: Bedrock/Vertex-proxied routes (e.g. pioneer) report "end_turn" on + // tool-using turns. No-op for direct Anthropic (already "toolUse" here). + if ( + managed && + (output.stopReason === "toolUse" || + output.content.some((block) => block.type === "toolCall")) + ) { + tagPendingCommentaryText(output.content); + } + flushPendingTextEnds(); + } + } + // Anthropic completes every SSE response with message_stop. Compatible + // proxy providers are not held to that first-party transport contract. + if ((isDirectAnthropicModel(model) || (!managed && refusalBuffer)) && !sawMessageStop) { + throw new Error("Anthropic stream ended before message_stop"); + } + if (options.signal?.aborted) { + throw transportAbortError(options.signal); + } + if (output.stopReason === "aborted" || output.stopReason === "error") { + throw new Error(output.errorMessage ?? "An unknown error occurred"); + } + if ([...blockIndexes.values()].some((index) => blocks[index]?.type === "toolCall")) { + throw new Error("Provider completed stream with an incomplete tool call"); + } + finalizeTerminalToolCallArguments( + sealedToolCalls.map(({ block }) => block), + (block) => + block.partialJson && block.partialJson.length > 0 + ? block.partialJson + : seededToolArguments.get(block), + ); + for (const sealed of sealedToolCalls) { + delete sealed.block.partialJson; + eventSink.push({ + type: "toolcall_end", + contentIndex: sealed.contentIndex, + toolCall: sealed.block, + partial: output, + }); + } + refusalBuffer?.flush(); + // Backstop: streaming tags commentary at the tool-boundary above, but + // replay/non-streaming assembly may reach here with tool calls untagged. + // Idempotent, so it never double-tags the streaming path. Gate on the turn + // containing a tool call (not stop_reason) so proxied Bedrock/Vertex routes + // that mislabel tool turns as "end_turn" still tag their narration. + if ( + managed && + (output.stopReason === "toolUse" || output.content.some((block) => block.type === "toolCall")) + ) { + tagPendingCommentaryText(output.content); + } + flushPendingTextEnds(); + } finally { + logAnthropicThinkingDrops(inputTransformations); + } +} diff --git a/packages/ai/src/transports/anthropic-transport-stream.ts b/packages/ai/src/transports/anthropic-transport-stream.ts index 6613e6acbb9d..55eab84295f4 100644 --- a/packages/ai/src/transports/anthropic-transport-stream.ts +++ b/packages/ai/src/transports/anthropic-transport-stream.ts @@ -2,14 +2,10 @@ import type { AssistantMessage, AssistantMessageEvent, Context, - ImageContent, Model, SimpleStreamOptions, StreamFn, - TextContent, - ToolCall, } from "@openclaw/llm-core"; -import { appendAssistantThinking } from "@openclaw/llm-core/event-stream"; import { toErrorObject } from "@openclaw/normalization-core/error-coercion"; /** * Native Anthropic Messages streaming transport. @@ -19,14 +15,7 @@ import { toErrorObject } from "@openclaw/normalization-core/error-coercion"; import { normalizeLowercaseStringOrEmpty } from "@openclaw/normalization-core/string-coerce"; import { getEnvApiKey } from "../env-api-keys.js"; import { getAiTransportHost } from "../host.js"; -import { - createAnthropicInlineImageBudget, - normalizeAnthropicInlineContent, - resolveAnthropicImageMediaType, - type AnthropicInlineImageBudget, -} from "../internal/anthropic-inline-images.js"; -import { calculateCost } from "../model-utils.js"; -import type { AnthropicOptions, AnthropicThinkingDisplay } from "../provider-options.js"; +import type { AnthropicOptions } from "../provider-options.js"; import { isAnthropicOAuthApiKey, omitFoundryBearerCredentialHeaders, @@ -37,95 +26,58 @@ import { ANTHROPIC_CLAUDE_CODE_BILLING_SYSTEM_BLOCK, ANTHROPIC_CLAUDE_CODE_VERSION, defaultsClaudeAdaptiveThinking, - mapAnthropicStopReason, prepareClaudeNoPrefillRequestContext, requiresClaudeAdaptiveThinking, resolveAnthropicThinkingEffort, resolveClaudeOpus5ModelIdentity, resolveClaudeSonnet5ModelIdentity, supportsClaudeAdaptiveThinking, - supportsClaudeNativeXhighEffort, usesClaudeFable5MessagesContract, usesClaudeStreamingRefusalContract, } from "../providers/anthropic-model-contract.js"; -import { applyAnthropicRefusal } from "../providers/anthropic-refusal.js"; import { ANTHROPIC_SERVER_SIDE_FALLBACK_BETA, ANTHROPIC_SERVER_SIDE_FALLBACKS, - applyAnthropicFallbackBoundary, - readAnthropicFallbackBoundary, - resolveAnthropicFallbackServingModelCost, } from "../providers/anthropic-server-fallback.js"; -import { - ANTHROPIC_OMITTED_REASONING_TEXT, - applyAnthropicThinkingBindingControls, - findActiveAnthropicToolTurnAssistantIndex, - logAnthropicThinkingDrops, - readAnthropicInputTransformations, -} from "../providers/anthropic-thinking-replay.js"; +import { applyAnthropicThinkingBindingControls } from "../providers/anthropic-thinking-replay.js"; import { normalizeAnthropicToolCallId, - normalizeAnthropicToolChoice, - projectAnthropicTools, - reconcileAnthropicToolChoice, - resolveOriginalAnthropicToolName, - toClaudeCodeToolName, type AnthropicToolProjection, } from "../providers/anthropic-tool-projection.js"; -import { - applyAnthropicMessageDeltaUsage, - applyAnthropicMessageStartUsage, - type AnthropicPromptUsageSnapshot, -} from "../providers/anthropic-usage.js"; import { adjustMaxTokensForThinking } from "../providers/simple-options.js"; -import { - describeToolResultMediaPlaceholder, - extractToolResultBlockText, - extractToolResultText, - isImageWithMediaPayload, -} from "../providers/tool-result-text.js"; -import { tagPendingCommentaryText } from "../utils/assistant-text-phase.js"; import { createDeferredEventBuffer } from "../utils/deferred-event-buffer.js"; -import { - createToolArgumentPreviewSchedule, - parseStreamingJson, - type ToolArgumentPreviewSchedule, -} from "../utils/json-parse.js"; -import { notifyLlmRequestActivity } from "../utils/llm-request-activity.js"; import { buildAnthropicReplayPlan, - createCompactionCapture, isAnthropicReplayRejection, suppressAnthropicCompaction, - type AnthropicCompactionBlock, } from "./anthropic-compaction-replay.js"; +import { + convertAnthropicMessages, + convertAnthropicTools, + buildAnthropicGenerationParams, +} from "./anthropic-messages.js"; import { applyAnthropicPayloadPolicyToParams, applyAnthropicContextManagementToRequest, isDirectAnthropicModel, - logAnthropicContextEdits, resolveAnthropicContextManagementBetaHeader, resolveAnthropicPayloadPolicy, } from "./anthropic-payload-policy.js"; +import { consumeAnthropicStream, type AnthropicStreamBlock } from "./anthropic-stream-reducer.js"; import { buildGuardedModelFetch, resolveProviderEndpoint, transformTransportMessages, } from "./host-policy.js"; -import { parseJsonObjectPreservingUnsafeIntegers } from "./json-unsafe-integers.js"; import { - coerceTransportToolCallArguments, copyProviderAcceptanceObserver, createEmptyTransportUsage, createWritableTransportEventStream, failTransportStream, - finalizeTerminalToolCallArguments, finalizeTransportStream, mergeTransportHeaders, notifyProviderHttpResponse, - sanitizeNonEmptyTransportPayloadText, sanitizeTransportPayloadText, - transportAbortError, } from "./transport-stream-shared.js"; import { createAbortError as createNamedAbortError, @@ -171,29 +123,6 @@ function resolveAnthropicRequestModelId(model: AnthropicTransportModel): string return model.id; } -type TransportContentBlock = - | { type: "text"; text: string; index?: number; textSignature?: string } - | { - type: "thinking"; - thinking: string; - thinkingSignature: string; - redacted?: boolean; - index?: number; - } - | ({ - type: "toolCall"; - id: string; - name: string; - arguments: Record; - partialJson?: string; - index?: number; - } & ToolCall); - -type MutableAssistantOutput = Omit & { - content: Array; - api: "anthropic-messages"; -}; - const EMPTY_ANTHROPIC_MESSAGES_FALLBACK_TEXT = "."; function resolvePositiveAnthropicTokenLimit(value: unknown): number | undefined { @@ -271,300 +200,12 @@ function buildAnthropicBetaHeader( : betaFeatures.join(","); } -const NON_VISION_USER_IMAGE_PLACEHOLDER = "(image omitted: model does not support images)"; - -async function convertContentBlocks( - content: readonly unknown[], - model: { input: readonly string[] }, - imageBudget: AnthropicInlineImageBudget, -) { - const text = extractToolResultText(content); - const mediaPlaceholder = describeToolResultMediaPlaceholder(content); - const hasImages = model.input.includes("image") && content.some(isImageWithMediaPayload); - if (!hasImages) { - return sanitizeNonEmptyTransportPayloadText(text, mediaPlaceholder ?? "(no output)"); - } - const blocks: Array< - | { type: "text"; text: string } - | { - type: "image"; - source: { type: "base64"; media_type: string; data: string }; - } - > = []; - let hasTextBlock = false; - for (const block of content) { - if (!block || typeof block !== "object") { - continue; - } - const record = block as Record; - const blockText = extractToolResultBlockText(block); - if (blockText) { - blocks.push({ type: "text", text: sanitizeTransportPayloadText(blockText) }); - hasTextBlock = true; - } - if (!isImageWithMediaPayload(record)) { - continue; - } - const [normalizedImage] = await normalizeAnthropicInlineContent( - [ - { - type: "image" as const, - data: typeof record.data === "string" ? record.data : "", - mimeType: typeof record.mimeType === "string" ? record.mimeType : "image/png", - }, - ], - imageBudget, - ); - if (normalizedImage?.type !== "image") { - continue; - } - blocks.push({ - type: "image" as const, - source: { - type: "base64", - media_type: resolveAnthropicImageMediaType(normalizedImage.mimeType), - data: normalizedImage.data, - }, - }); - } - if (!hasTextBlock) { - blocks.unshift({ type: "text", text: mediaPlaceholder ?? "(see attached image)" }); - } - return blocks; -} - -async function convertAnthropicMessages( - messages: Context["messages"], - model: AnthropicTransportModel, - isOAuthToken: boolean, - options: { - allowReasoningContentReplay?: boolean; - compaction?: AnthropicCompactionBlock; - replayThinkingEnabled?: boolean; - }, -): Promise>> { - const params: Array> = []; - const imageBudget = createAnthropicInlineImageBudget(); - const allowReasoningContentReplay = options.allowReasoningContentReplay === true; - const replayThinkingEnabled = options.replayThinkingEnabled !== false; - const transformedMessages = transformTransportMessages( - messages, - model, - normalizeAnthropicToolCallId, - ); - const activeToolTurnAssistantIndex = replayThinkingEnabled - ? -1 - : findActiveAnthropicToolTurnAssistantIndex(transformedMessages); - for (let i = 0; i < transformedMessages.length; i += 1) { - const msg = transformedMessages[i]; - if (!msg) { - continue; - } - if (msg.role === "user") { - if (typeof msg.content === "string") { - if (msg.content.trim().length > 0) { - const userParam = { - role: "user", - content: sanitizeTransportPayloadText(msg.content), - }; - params.push(userParam); - } - continue; - } - const normalizedContent = model.input.includes("image") - ? await normalizeAnthropicInlineContent( - msg.content as readonly (TextContent | ImageContent)[], - imageBudget, - ) - : msg.content.map((item) => - item.type === "image" - ? { type: "text" as const, text: NON_VISION_USER_IMAGE_PLACEHOLDER } - : item, - ); - const blocks: Array< - | { type: "text"; text: string } - | { - type: "image"; - source: { type: "base64"; media_type: string; data: string }; - } - > = normalizedContent.map((item) => - item.type === "text" - ? { - type: "text", - text: sanitizeTransportPayloadText(item.text), - } - : { - type: "image", - source: { - type: "base64", - media_type: resolveAnthropicImageMediaType(item.mimeType), - data: item.data, - }, - }, - ); - let filteredBlocks = model.input.includes("image") - ? blocks - : blocks.filter((block) => block.type !== "image"); - filteredBlocks = filteredBlocks.filter( - (block) => block.type !== "text" || block.text.trim().length > 0, - ); - if (filteredBlocks.length === 0) { - continue; - } - const userParam = { - role: "user", - content: filteredBlocks, - }; - params.push(userParam); - continue; - } - if (msg.role === "assistant") { - const blocks: Array> = - i === 0 && options.compaction ? [options.compaction] : []; - const reasoningContent: string[] = []; - let omittedThinking = false; - for (const block of msg.content) { - if (block.type === "text") { - if (block.text.trim().length > 0) { - blocks.push({ - type: "text", - text: sanitizeTransportPayloadText(block.text), - }); - } - continue; - } - if (block.type === "thinking") { - const thinkingSignature = block.thinkingSignature?.trim(); - const isReasoningContent = thinkingSignature === "reasoning_content"; - if (!replayThinkingEnabled && i !== activeToolTurnAssistantIndex && !isReasoningContent) { - omittedThinking = true; - continue; - } - if (block.redacted) { - blocks.push({ - type: "redacted_thinking", - data: block.thinkingSignature, - }); - continue; - } - const hasNativeThinkingSignature = Boolean(thinkingSignature) && !isReasoningContent; - if (block.thinking.trim().length === 0 && !hasNativeThinkingSignature) { - continue; - } - if (!thinkingSignature) { - blocks.push({ - type: "text", - text: sanitizeTransportPayloadText(block.thinking), - }); - } else { - const thinking = - thinkingSignature === "reasoning_content" - ? sanitizeTransportPayloadText(block.thinking) - : block.thinking; - if (thinkingSignature === "reasoning_content") { - if (allowReasoningContentReplay) { - blocks.push({ - type: "thinking", - thinking, - signature: thinkingSignature, - }); - reasoningContent.push(thinking); - } - continue; - } - blocks.push({ - type: "thinking", - thinking, - signature: thinkingSignature, - }); - } - continue; - } - if (block.type === "toolCall") { - blocks.push({ - type: "tool_use", - id: block.id, - name: isOAuthToken ? toClaudeCodeToolName(block.name) : block.name, - input: coerceTransportToolCallArguments(block.arguments), - }); - } - } - if (blocks.length === 0 && omittedThinking) { - blocks.push({ type: "text", text: ANTHROPIC_OMITTED_REASONING_TEXT }); - } - if (blocks.length > 0) { - const assistantMsg: Record = { role: "assistant", content: blocks }; - if (reasoningContent.length > 0) { - assistantMsg.reasoning_content = reasoningContent.join("\n"); - } else if (allowReasoningContentReplay) { - blocks.unshift({ - type: "thinking", - thinking: "", - signature: "reasoning_content", - }); - } - params.push(assistantMsg); - } - continue; - } - if (msg.role === "toolResult") { - const toolResult = msg; - const toolResults: Array> = [ - { - type: "tool_result", - tool_use_id: toolResult.toolCallId, - content: await convertContentBlocks(toolResult.content, model, imageBudget), - is_error: toolResult.isError, - }, - ]; - let j = i + 1; - while (j < transformedMessages.length) { - const nextMsg = transformedMessages.at(j); - if (nextMsg?.role !== "toolResult") { - break; - } - toolResults.push({ - type: "tool_result", - tool_use_id: nextMsg.toolCallId, - content: await convertContentBlocks(nextMsg.content, model, imageBudget), - is_error: nextMsg.isError, - }); - j += 1; - } - i = j - 1; - params.push({ - role: "user", - content: toolResults, - }); - } - } - return params; -} - function ensureNonEmptyAnthropicMessages(messages: Array>) { return messages.length > 0 ? messages : [{ role: "user", content: EMPTY_ANTHROPIC_MESSAGES_FALLBACK_TEXT }]; } -function convertAnthropicTools(tools: Context["tools"], isOAuthToken: boolean) { - const projection = projectAnthropicTools(tools ?? [], (name) => - isOAuthToken ? toClaudeCodeToolName(name) : name, - ); - const converted = projection.tools.map((tool) => ({ - name: tool.wireName, - description: tool.description, - input_schema: tool.inputSchema, - })); - return { projection, tools: converted }; -} - -function parseAnthropicToolCallArguments(inputJson: string): Record { - return coerceTransportToolCallArguments( - parseJsonObjectPreservingUnsafeIntegers(inputJson) ?? parseStreamingJson(inputJson), - ); -} - const DEFAULT_ANTHROPIC_BASE_URL = "https://api.anthropic.com"; /** Resolve the effective Anthropic API base URL from model or environment. */ @@ -942,11 +583,17 @@ async function buildAnthropicParams( authProfileId: options?.authProfileId, sessionId: options?.sessionId, }); - const messages = await convertAnthropicMessages(replayPlan.messages, model, isOAuthToken, { - allowReasoningContentReplay: supportsReasoningContentReplay(model), - compaction: replayPlan.compaction, - replayThinkingEnabled, - }); + const messages = await convertAnthropicMessages( + transformTransportMessages(replayPlan.messages, model, normalizeAnthropicToolCallId), + model, + isOAuthToken, + { + profile: "transport", + allowReasoningContentReplay: supportsReasoningContentReplay(model), + compaction: replayPlan.compaction, + replayThinkingEnabled, + }, + ); const params: Record = { model: resolveAnthropicRequestModelId(model), messages: ensureNonEmptyAnthropicMessages(messages), @@ -987,63 +634,20 @@ async function buildAnthropicParams( }, ]; } - if ( - options?.temperature !== undefined && - !options.thinkingEnabled && - !supportsClaudeNativeXhighEffort(model) - ) { - params.temperature = options.temperature; - } - if (options?.stop !== undefined && options.stop.length > 0) { - params.stop_sequences = options.stop; - } - let toolProjection: AnthropicToolProjection | undefined; - if (context.tools) { - const convertedTools = convertAnthropicTools(context.tools, isOAuthToken); - toolProjection = convertedTools.projection; - if (convertedTools.tools.length > 0) { - params.tools = convertedTools.tools; - } - } - if (mandatoryAdaptiveThinking || model.reasoning || supportsClaudeAdaptiveThinking(model)) { - if (mandatoryAdaptiveThinking || options?.thinkingEnabled) { - if (supportsClaudeAdaptiveThinking(model)) { - // Default display to "summarized" so Opus 4.7+/Fable 5 return a thinking - // summary like older Claude 4 models — mirrors the provider path - // (llm/providers/anthropic.ts). Without it the adaptive request omits the - // summary and only an encrypted signature comes back, so the 🧠 lane is - // blank (the live agent transport previously sent this for opus-4-8). - const display: AnthropicThinkingDisplay = options?.thinkingDisplay ?? "summarized"; - params.thinking = { type: "adaptive", display }; - const effort = options?.effort ?? (mandatoryAdaptiveThinking ? "high" : undefined); - if (effort) { - params.output_config = { effort }; - } - } else { - params.thinking = { - type: "enabled", - budget_tokens: options?.thinkingBudgetTokens ?? 1024, - }; - } - } else if (options?.thinkingEnabled === false) { - params.thinking = { type: "disabled" }; - } - } - if (options?.metadata && typeof options.metadata.user_id === "string") { - params.metadata = { user_id: options.metadata.user_id }; - } - if (options?.toolChoice) { - const normalizedToolChoice = normalizeAnthropicToolChoice( - replayThinkingEnabled, - options.toolChoice, - ); - const projectedToolChoice = toolProjection - ? reconcileAnthropicToolChoice(normalizedToolChoice, toolProjection) - : normalizedToolChoice; - if (projectedToolChoice) { - params.tool_choice = projectedToolChoice; - } - } + const convertedTools = context.tools + ? convertAnthropicTools(context.tools, isOAuthToken) + : undefined; + const toolProjection = convertedTools?.projection; + Object.assign( + params, + buildAnthropicGenerationParams({ + model, + options, + tools: convertedTools?.tools, + toolProjection, + profile: "transport", + }), + ); // Anthropic-family carriers are append-only, so they are stable cache anchors too. applyAnthropicPayloadPolicyToParams(params, payloadPolicy, new Set()); return { params, toolProjection, usedCompactionReplay: replayPlan.compaction !== undefined }; @@ -1134,7 +738,7 @@ export function createAnthropicMessagesTransportStreamFn(): StreamFn { const options = rawOptions as AnthropicTransportOptions | undefined; const { eventStream, stream } = createWritableTransportEventStream(); void (async () => { - const output: MutableAssistantOutput = { + const output: AssistantMessage = { role: "assistant", content: [], api: "anthropic-messages", @@ -1149,13 +753,7 @@ export function createAnthropicMessagesTransportStreamFn(): StreamFn { const refusalBuffer = usesClaudeStreamingRefusalContract(model) ? createDeferredEventBuffer(stream) : undefined; - const eventSink = refusalBuffer ?? stream; - // Fallback-served turns bill at the serving model's rates; a boundary - // swaps this to the fallback model's cost table. - let costModel = model; - let messageStartPromptUsage: AnthropicPromptUsageSnapshot | undefined; let usedCompactionReplay = false; - let inputTransformations: unknown[] | undefined; try { const apiKey = options?.apiKey ?? getEnvApiKey(model.provider) ?? ""; if (!apiKey) { @@ -1205,581 +803,17 @@ export function createAnthropicMessagesTransportStreamFn(): StreamFn { const detail = await readAnthropicMessagesErrorBodySnippet(response); throw new Error(formatAnthropicMessagesHttpError(response, detail)); } - const blocks = output.content; - const blockIndexes = new Map(); - // Preview schedules are per active tool call; WeakMap keys die with the block. - const toolArgumentPreviewSchedules = new WeakMap< - Extract, - ToolArgumentPreviewSchedule - >(); - const sealedToolCalls: Array<{ - block: Extract; - contentIndex: number; - }> = []; - const compactionCapture = createCompactionCapture(output, model, transportOptions); - // Signature deltas are opaque and only complete at content_block_stop. - // Keep partial bytes out of output so interrupted streams cannot poison replay. - const pendingThinkingSignatures = new Map(); - const allowReasoningContentReplay = supportsReasoningContentReplay(model); - const reasoningContentThinkingBlocks = new Map(); - const reasoningContentTextBlocks = new Map(); - let sawMessageStop = false; - const pendingTextEnds: Array[0]> = []; - // Hold text_end until tool-boundary classification is known. - const flushPendingTextEnds = () => { - for (const event of pendingTextEnds) { - eventSink.push(event); - } - pendingTextEnds.length = 0; - }; - const eventIndexKey = (eventIndex: unknown) => - typeof eventIndex === "number" ? eventIndex : -1; - const appendReasoningContentThinkingDelta = ( - eventIndex: unknown, - rawText: unknown, - ): boolean => { - if (typeof rawText !== "string") { - return false; - } - const text = sanitizeTransportPayloadText(rawText); - if (text.length === 0) { - return false; - } - const key = eventIndexKey(eventIndex); - let contentIndex = reasoningContentThinkingBlocks.get(key); - let block = - contentIndex === undefined - ? undefined - : (output.content[contentIndex] as TransportContentBlock | undefined); - if (!block || block.type !== "thinking") { - block = { type: "thinking", thinking: "", thinkingSignature: "reasoning_content" }; - output.content.push(block); - contentIndex = output.content.length - 1; - reasoningContentThinkingBlocks.set(key, contentIndex); - eventSink.push({ - type: "thinking_start", - contentIndex, - partial: output, - }); - } - if (contentIndex === undefined) { - return false; - } - appendAssistantThinking(block, text); - block.thinkingSignature = "reasoning_content"; - eventSink.push({ - type: "thinking_delta", - contentIndex, - delta: text, - partial: output, - }); - return true; - }; - const appendReasoningContentTextDelta = ( - eventIndex: unknown, - rawText: unknown, - ): boolean => { - if (typeof rawText !== "string") { - return false; - } - const text = sanitizeTransportPayloadText(rawText); - if (text.length === 0) { - return false; - } - const key = eventIndexKey(eventIndex); - let contentIndex = reasoningContentTextBlocks.get(key); - let block = - contentIndex === undefined - ? undefined - : (output.content[contentIndex] as TransportContentBlock | undefined); - if (!block || block.type !== "text") { - block = { type: "text", text: "" }; - output.content.push(block); - contentIndex = output.content.length - 1; - reasoningContentTextBlocks.set(key, contentIndex); - eventSink.push({ - type: "text_start", - contentIndex, - partial: output, - }); - } - if (contentIndex === undefined) { - return false; - } - block.text += text; - eventSink.push({ - type: "text_delta", - contentIndex, - delta: text, - partial: output, - }); - return true; - }; - const finishReasoningContentSidecars = (eventIndex: unknown) => { - const key = eventIndexKey(eventIndex); - const thinkingContentIndex = reasoningContentThinkingBlocks.get(key); - if (thinkingContentIndex !== undefined) { - reasoningContentThinkingBlocks.delete(key); - const block = output.content[thinkingContentIndex]; - if (block?.type === "thinking") { - eventSink.push({ - type: "thinking_end", - contentIndex: thinkingContentIndex, - content: block.thinking, - partial: output, - }); - } - } - const textContentIndex = reasoningContentTextBlocks.get(key); - if (textContentIndex === undefined) { - return; - } - reasoningContentTextBlocks.delete(key); - const block = output.content[textContentIndex]; - if (block?.type === "text") { - eventSink.push({ - type: "text_end", - contentIndex: textContentIndex, - content: block.text, - partial: output, - }); - } - }; - for await (const event of anthropicStream) { - // A serving-model fallback replaces the initial snapshot; report only once at completion. - inputTransformations = readAnthropicInputTransformations(event) ?? inputTransformations; - notifyLlmRequestActivity(transportOptions.signal); - if (event.type === "error") { - const error = event.error as { message?: string } | undefined; - throw new Error(error?.message || "Anthropic Messages stream failed"); - } - if (event.type === "message_start") { - const message = event.message as - | { id?: string; model?: string; usage?: Record } - | undefined; - const usage = message?.usage ?? {}; - output.responseId = typeof message?.id === "string" ? message.id : undefined; - output.responseModel = typeof message?.model === "string" ? message.model : undefined; - messageStartPromptUsage = applyAnthropicMessageStartUsage(output.usage, usage); - calculateCost(costModel, output.usage); - // Defer start until after message_start so that pre-stream SSE errors - // (e.g. invalid thinking signatures) arrive before any non-error event - // is yielded, keeping yieldedOutput=false in pumpStreamWithRecovery - // and allowing the thinking-block recovery retry to fire. - eventSink.push({ type: "start", partial: output }); - continue; - } - if (event.type === "message_stop") { - sawMessageStop = true; - continue; - } - if (event.type === "content_block_start") { - const contentBlock = event.content_block as Record | undefined; - const index = typeof event.index === "number" ? event.index : -1; - if ( - transportOptions.anthropicServerCompaction === true && - compactionCapture.begin(index, contentBlock, output.content.length) - ) { - continue; - } - const fallbackBoundary = refusalBuffer - ? readAnthropicFallbackBoundary(contentBlock) - : null; - if (fallbackBoundary) { - // Server-side fallback boundary: pre-boundary thinking/tool - // blocks must not replay or execute, and the buffered preview - // events reference them, so rebuild the deferred timeline from - // the surviving text prefix the fallback model continued from. - refusalBuffer?.discard(); - sealedToolCalls.length = 0; - pendingTextEnds.length = 0; - blockIndexes.clear(); - pendingThinkingSignatures.clear(); - applyAnthropicFallbackBoundary({ - output, - boundary: fallbackBoundary, - provider: model.provider, - }); - // Fallback-only iteration partials stay outside the serving-model - // estimate. Compaction responses are the exception: usage policy - // aggregates their complete billed iteration list. - costModel = { - ...model, - cost: resolveAnthropicFallbackServingModelCost({ - requestedModelId: model.id, - servingModelId: fallbackBoundary.toModel, - requestedCost: model.cost, - }), - }; - calculateCost(costModel, output.usage); - eventSink.push({ type: "start", partial: output }); - for (const [i, block] of output.content.entries()) { - if (block.type !== "text") { - continue; - } - delete block.index; - eventSink.push({ - type: "text_start", - contentIndex: i, - partial: output, - }); - if (block.text) { - eventSink.push({ - type: "text_delta", - contentIndex: i, - delta: block.text, - partial: output, - }); - } - pendingTextEnds.push({ - type: "text_end", - contentIndex: i, - content: block.text, - partial: output, - }); - } - continue; - } - pendingThinkingSignatures.delete(index); - if (contentBlock?.type === "text") { - const text = - typeof contentBlock.text === "string" - ? sanitizeTransportPayloadText(contentBlock.text) - : ""; - const block: TransportContentBlock = { type: "text", text, index }; - output.content.push(block); - const contentIndex = output.content.length - 1; - blockIndexes.set(index, contentIndex); - eventSink.push({ - type: "text_start", - contentIndex, - partial: output, - }); - if (text.length > 0) { - eventSink.push({ - type: "text_delta", - contentIndex, - delta: text, - partial: output, - }); - } - continue; - } - if (contentBlock?.type === "thinking") { - const thinking = - typeof contentBlock.thinking === "string" ? contentBlock.thinking : ""; - const block: TransportContentBlock = { - type: "thinking", - thinking, - thinkingSignature: - typeof contentBlock.signature === "string" ? contentBlock.signature : "", - index, - }; - output.content.push(block); - const contentIndex = output.content.length - 1; - blockIndexes.set(index, contentIndex); - eventSink.push({ - type: "thinking_start", - contentIndex, - partial: output, - }); - if (thinking.length > 0) { - eventSink.push({ - type: "thinking_delta", - contentIndex, - delta: thinking, - partial: output, - }); - } - continue; - } - if (contentBlock?.type === "redacted_thinking") { - const block: TransportContentBlock = { - type: "thinking", - thinking: "[Reasoning redacted]", - thinkingSignature: typeof contentBlock.data === "string" ? contentBlock.data : "", - redacted: true, - index, - }; - output.content.push(block); - blockIndexes.set(index, output.content.length - 1); - eventSink.push({ - type: "thinking_start", - contentIndex: output.content.length - 1, - partial: output, - }); - continue; - } - if (contentBlock?.type === "tool_use") { - tagPendingCommentaryText(output.content); - flushPendingTextEnds(); - const block: TransportContentBlock = { - type: "toolCall", - id: typeof contentBlock.id === "string" ? contentBlock.id : "", - name: - typeof contentBlock.name === "string" - ? isOAuthToken - ? resolveOriginalAnthropicToolName(contentBlock.name, toolProjection) - : contentBlock.name - : "", - arguments: - contentBlock.input && typeof contentBlock.input === "object" - ? (contentBlock.input as Record) - : {}, - partialJson: "", - index, - }; - output.content.push(block); - blockIndexes.set(index, output.content.length - 1); - toolArgumentPreviewSchedules.set(block, createToolArgumentPreviewSchedule()); - eventSink.push({ - type: "toolcall_start", - contentIndex: output.content.length - 1, - partial: output, - }); - } - continue; - } - if (event.type === "content_block_delta") { - const delta = event.delta as Record | undefined; - const eventIndex = typeof event.index === "number" ? event.index : undefined; - if (eventIndex !== undefined && compactionCapture.delta(eventIndex, delta)) { - continue; - } - let index = eventIndex === undefined ? undefined : blockIndexes.get(eventIndex); - let block = index === undefined ? undefined : blocks[index]; - if (allowReasoningContentReplay) { - const appendedThinking = appendReasoningContentThinkingDelta( - event.index, - delta?.reasoning_content, - ); - const hasNativeAnthropicDelta = - (delta?.type === "text_delta" && typeof delta.text === "string") || - (delta?.type === "thinking_delta" && typeof delta.thinking === "string") || - (delta?.type === "input_json_delta" && typeof delta.partial_json === "string") || - (delta?.type === "signature_delta" && typeof delta.signature === "string"); - let appendedContent = false; - if ( - !hasNativeAnthropicDelta && - typeof delta?.content === "string" && - delta.content.length > 0 - ) { - const text = sanitizeTransportPayloadText(delta.content); - if (text.length > 0) { - if (block?.type === "text" && index !== undefined) { - block.text += text; - eventSink.push({ - type: "text_delta", - contentIndex: index, - delta: text, - partial: output, - }); - appendedContent = true; - } else { - appendedContent = appendReasoningContentTextDelta(event.index, text); - } - } - } - if ((appendedThinking || appendedContent) && !hasNativeAnthropicDelta) { - continue; - } - } - if (!block && delta?.type === "text_delta" && typeof delta.text === "string") { - const recoveredIndex = typeof event.index === "number" ? event.index : blocks.length; - block = { type: "text", text: "", index: recoveredIndex }; - output.content.push(block); - index = output.content.length - 1; - if (typeof event.index === "number") { - blockIndexes.set(event.index, index); - } - eventSink.push({ - type: "text_start", - contentIndex: index, - partial: output, - }); - } - if (index === undefined) { - continue; - } - if ( - block?.type === "text" && - delta?.type === "text_delta" && - typeof delta.text === "string" - ) { - block.text += delta.text; - eventSink.push({ - type: "text_delta", - contentIndex: index, - delta: delta.text, - partial: output, - }); - continue; - } - if ( - block?.type === "thinking" && - delta?.type === "thinking_delta" && - typeof delta.thinking === "string" - ) { - appendAssistantThinking(block, delta.thinking); - eventSink.push({ - type: "thinking_delta", - contentIndex: index, - delta: delta.thinking, - partial: output, - }); - continue; - } - if ( - block?.type === "toolCall" && - delta?.type === "input_json_delta" && - typeof delta.partial_json === "string" - ) { - const partialJson = `${block.partialJson ?? ""}${delta.partial_json}`; - block.partialJson = partialJson; - // Preview refresh is scheduled geometrically; content_block_stop - // re-parses the full buffer authoritatively either way. - if (toolArgumentPreviewSchedules.get(block)?.(partialJson.length)) { - block.arguments = parseAnthropicToolCallArguments(partialJson); - } - eventSink.push({ - type: "toolcall_delta", - contentIndex: index, - delta: delta.partial_json, - partial: output, - }); - continue; - } - if ( - block?.type === "thinking" && - delta?.type === "signature_delta" && - typeof delta.signature === "string" - ) { - const signatureIndex = eventIndexKey(event.index); - const pendingSignature = pendingThinkingSignatures.get(signatureIndex); - if (pendingSignature === undefined) { - block.thinkingSignature = ""; - pendingThinkingSignatures.set(signatureIndex, delta.signature); - } else { - pendingThinkingSignatures.set(signatureIndex, pendingSignature + delta.signature); - } - } - continue; - } - if (event.type === "content_block_stop") { - const eventIndex = typeof event.index === "number" ? event.index : undefined; - if (eventIndex !== undefined && compactionCapture.complete(eventIndex)) { - continue; - } - const pendingSignature = - eventIndex === undefined ? undefined : pendingThinkingSignatures.get(eventIndex); - if (eventIndex !== undefined) { - pendingThinkingSignatures.delete(eventIndex); - } - const index = eventIndex === undefined ? undefined : blockIndexes.get(eventIndex); - const block = index === undefined ? undefined : blocks[index]; - if (eventIndex === undefined || index === undefined || !block) { - finishReasoningContentSidecars(event.index); - continue; - } - blockIndexes.delete(eventIndex); - delete block.index; - if (block.type === "text") { - pendingTextEnds.push({ - type: "text_end", - contentIndex: index, - content: block.text, - partial: output, - }); - finishReasoningContentSidecars(event.index); - continue; - } - if (block.type === "thinking") { - if (pendingSignature !== undefined) { - block.thinkingSignature = pendingSignature; - } - eventSink.push({ - type: "thinking_end", - contentIndex: index, - content: block.thinking, - partial: output, - }); - finishReasoningContentSidecars(event.index); - continue; - } - if (block.type === "toolCall") { - sealedToolCalls.push({ block, contentIndex: index }); - finishReasoningContentSidecars(event.index); - } - continue; - } - if (event.type === "message_delta") { - logAnthropicContextEdits(event); - const delta = event.delta as - | { stop_reason?: string; stop_details?: unknown } - | undefined; - const usage = event.usage as Record | undefined; - if (delta?.stop_reason) { - if (delta.stop_reason === "refusal") { - applyAnthropicRefusal(output, delta.stop_details, model.provider); - } else { - output.stopReason = mapAnthropicStopReason(delta.stop_reason); - } - } - applyAnthropicMessageDeltaUsage(output.usage, usage, messageStartPromptUsage); - calculateCost(costModel, output.usage); - // Gate on the turn CONTAINING a tool call, not the provider's stop_reason - // label: Bedrock/Vertex-proxied routes (e.g. pioneer) report "end_turn" on - // tool-using turns. No-op for direct Anthropic (already "toolUse" here). - if ( - output.stopReason === "toolUse" || - output.content.some((block) => block.type === "toolCall") - ) { - tagPendingCommentaryText(output.content); - } - flushPendingTextEnds(); - } - } - // Anthropic completes every SSE response with message_stop. Compatible - // proxy providers are not held to that first-party transport contract. - if (isDirectAnthropicModel(model) && !sawMessageStop) { - throw new Error("Anthropic stream ended before message_stop"); - } - if (transportOptions.signal?.aborted) { - throw transportAbortError(transportOptions.signal); - } - if (output.stopReason === "aborted" || output.stopReason === "error") { - throw new Error(output.errorMessage ?? "An unknown error occurred"); - } - if ([...blockIndexes.values()].some((index) => blocks[index]?.type === "toolCall")) { - throw new Error("Provider completed stream with an incomplete tool call"); - } - finalizeTerminalToolCallArguments( - sealedToolCalls.map(({ block }) => block), - (block) => - block.partialJson && block.partialJson.length > 0 ? block.partialJson : block.arguments, - ); - for (const sealed of sealedToolCalls) { - delete sealed.block.partialJson; - eventSink.push({ - type: "toolcall_end", - contentIndex: sealed.contentIndex, - toolCall: sealed.block, - partial: output, - }); - } - refusalBuffer?.flush(); - // Backstop: streaming tags commentary at the tool-boundary above, but - // replay/non-streaming assembly may reach here with tool calls untagged. - // Idempotent, so it never double-tags the streaming path. Gate on the turn - // containing a tool call (not stop_reason) so proxied Bedrock/Vertex routes - // that mislabel tool turns as "end_turn" still tag their narration. - if ( - output.stopReason === "toolUse" || - output.content.some((block) => block.type === "toolCall") - ) { - tagPendingCommentaryText(output.content); - } - flushPendingTextEnds(); + await consumeAnthropicStream({ + events: anthropicStream, + model, + options: transportOptions, + output, + stream, + refusalBuffer, + isOAuthToken, + toolProjection, + profile: "transport", + }); finalizeTransportStream({ stream, output }); } catch (error) { failTransportStream({ @@ -1798,12 +832,10 @@ export function createAnthropicMessagesTransportStreamFn(): StreamFn { suppressAnthropicCompaction(output, model, options); } for (const block of output.content) { - delete block.index; + delete (block as AnthropicStreamBlock).index; } }, }); - } finally { - logAnthropicThinkingDrops(inputTransformations); } })(); return eventStream; diff --git a/src/plugin-sdk/provider-transport-runtime.ts b/src/plugin-sdk/provider-transport-runtime.ts index 4d54b86b31ed..85b4a53da49b 100644 --- a/src/plugin-sdk/provider-transport-runtime.ts +++ b/src/plugin-sdk/provider-transport-runtime.ts @@ -17,6 +17,11 @@ export { } from "@openclaw/ai/internal/shared"; export { coerceTransportToolCallArguments, + consumeGoogleGenerateContentStream, + convertGoogleTools, + projectGoogleMessages, + requiresGoogleToolCallId, + type GoogleStreamChunk, copyProviderAcceptanceObserver, createEmptyTransportUsage, createWritableTransportEventStream,