diff --git a/packages/core/src/core/openaiContentGenerator/converter.test.ts b/packages/core/src/core/openaiContentGenerator/converter.test.ts index 453c476436..9c65b85320 100644 --- a/packages/core/src/core/openaiContentGenerator/converter.test.ts +++ b/packages/core/src/core/openaiContentGenerator/converter.test.ts @@ -544,6 +544,77 @@ describe('OpenAIContentConverter', () => { ); }); + it('rejects the recorded production unclosed content leak (issue #6666)', () => { + // Production capture shape (sanitized): a hybrid-thinking model + // skipped the reasoning channel entirely and streamed its thinking as + // literal text inside content — no reasoning_content on + // any chunk, no tool calls, and the tag is never closed before stop. + const stream = withStreamParser(); + stream.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; + const opening = converter.convertOpenAIChunkToGemini( + streamChunk('opening', { content: '\nThe user wants to query the compute resources for ' + + 'project space 10088. Let me check the available APIs.', + }), + stream, + ); + + expect(opening.candidates?.[0]?.content?.parts).toEqual([]); + expect(body.candidates?.[0]?.content?.parts).toEqual([]); + expect(() => finishStream(stream, 'stop')).toThrowError( + expect.objectContaining({ type: 'PROTOCOL_TAG_LEAK' }), + ); + }); + + it('holds a long confirmed opening tag until its closing tag arrives', () => { + const stream = withStreamParser(); + stream.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; + const text = `${'x'.repeat(200)}`; + + const opening = converter.convertOpenAIChunkToGemini( + streamChunk('long-balanced', { + content: `${'x'.repeat(200)}`, + }), + stream, + ); + const closing = converter.convertOpenAIChunkToGemini( + streamChunk('long-balanced', { content: '' }, 'stop'), + stream, + ); + + expect(opening.candidates?.[0]?.content?.parts).toEqual([]); + expect(closing.candidates?.[0]?.content?.parts).toEqual([{ text }]); + }); + + it('leaks the production shape without provider provenance', () => { + // Control for the test above: without contentOnlyThinkingTagLeaks the + // same stream passes through verbatim — the defense is provider-gated, + // so endpoints whose provider does not opt in remain exposed. + const stream = withStreamParser(); + const response = converter.convertOpenAIChunkToGemini( + streamChunk( + 'literal', + { + content: + '\nThe user wants to query the compute resources.', + }, + 'stop', + ), + stream, + ); + + expect(response.candidates?.[0]?.content?.parts).toEqual([ + { + text: '\nThe user wants to query the compute resources.', + }, + ]); + }); + it.each([ ['split literal block', ['literal']], ['empty block with a separate finish chunk', ['\n\n', '']], @@ -551,6 +622,7 @@ describe('OpenAIContentConverter', () => { 'two split valid blocks', ['\n\n', 'literal'], ], + ['long empty block', [`${' '.repeat(128)}`, '']], ])('preserves content-only %s', (_name, chunks) => { const stream = withStreamParser(); stream.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; @@ -570,24 +642,33 @@ describe('OpenAIContentConverter', () => { expect(parts.every((part) => part.thought !== true)).toBe(true); }); - it('releases a long undecided prefix before the stream finishes', () => { + it('releases a long unconfirmed prefix before the stream finishes', () => { const stream = withStreamParser(); stream.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; - const text = `${' '.repeat(257)}`; + const text = ` { + const stream = withStreamParser(); + stream.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; + const response = converter.convertOpenAIChunkToGemini( + streamChunk('unclosed', { + content: `${' '.repeat(128)}`, + }), stream, ); - expect(response.candidates?.[0]?.content?.parts).toEqual([{ text }]); - expect(continuation.candidates?.[0]?.content?.parts).toEqual([ - { text: 'literal' }, - ]); + expect(response.candidates?.[0]?.content?.parts).toEqual([]); + expect(() => finishStream(stream, 'stop')).toThrowError( + expect.objectContaining({ type: 'PROTOCOL_TAG_LEAK' }), + ); }); it('preserves a leak-shaped literal without provider provenance', () => { @@ -631,17 +712,38 @@ describe('OpenAIContentConverter', () => { expect(parts.map((part) => part.text).join('')).toBe(chunks.join('')); }); - it('fails closed when a suspicious prefix exceeds the buffer limit', () => { + it('rejects an unclosed outer block containing a balanced nested block', () => { + const stream = withStreamParser(); + stream.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; + + expect(() => + converter.convertOpenAIChunkToGemini( + streamChunk( + 'nested-unclosed', + { + content: 'innerouter text', + }, + 'stop', + ), + stream, + ), + ).toThrowError(expect.objectContaining({ type: 'PROTOCOL_TAG_LEAK' })); + }); + + it('fails closed for a long suspicious prefix at stream finish', () => { const stream = withStreamParser(); stream.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; const content = '9' + 'x'.repeat(257); - expect(() => - converter.convertOpenAIChunkToGemini( - streamChunk('long-leak', { content }), - stream, - ), - ).toThrowError(expect.objectContaining({ type: 'PROTOCOL_TAG_LEAK' })); + const response = converter.convertOpenAIChunkToGemini( + streamChunk('long-leak', { content }), + stream, + ); + + expect(response.candidates?.[0]?.content?.parts).toEqual([]); + expect(() => finishStream(stream, 'stop')).toThrowError( + expect.objectContaining({ type: 'PROTOCOL_TAG_LEAK' }), + ); }); it.each([ @@ -5079,6 +5181,7 @@ describe('OpenAIContentConverter', () => { it('should handle a single chunk delta with both reasoning_content and content simultaneously', () => { const ctx = withStreamParser(); + ctx.responseParsingOptions = { contentOnlyThinkingTagLeaks: true }; const part = converter.convertOpenAIChunkToGemini( { @@ -5215,7 +5318,10 @@ describe('OpenAIContentConverter', () => { }, ], } as unknown as OpenAI.Chat.ChatCompletion, - requestContext, + { + ...requestContext, + responseParsingOptions: { contentOnlyThinkingTagLeaks: true }, + }, ); expect(response.candidates?.[0]?.content?.parts).toEqual([ diff --git a/packages/core/src/core/openaiContentGenerator/converter.ts b/packages/core/src/core/openaiContentGenerator/converter.ts index 6585d0cc31..ffd8dfcfd6 100644 --- a/packages/core/src/core/openaiContentGenerator/converter.ts +++ b/packages/core/src/core/openaiContentGenerator/converter.ts @@ -1160,8 +1160,25 @@ function classifyContentOnlyThinkingTagPrefix( for (const closing of [false, true, false]) { const tagLength = consumeTag(rest, closing); if (tagLength === null) return 'pending'; - if (tagLength === undefined) return 'clean'; + if (tagLength === undefined) { + if (!closing) return 'clean'; + break; + } rest = rest.slice(tagLength).trimStart(); + if (closing && !rest) return 'clean'; + // An opening tag followed by ordinary text is only a legitimate literal + // if a closing tag still balances it later. Without one, the turn is an + // unclosed thinking block — the exact shape of the recorded production + // leaks (issue #6666). Hold it mid-stream (a closing tag may still + // arrive) and reject it once the stream finishes. Whitespace-only tails + // stay undecided: they may still resolve into a closing tag. + if ( + closing === false && + /\S/.test(rest) && + !/<\/think(?:ing)?\s*>/i.test(rest) + ) { + return streamFinished ? 'leaked' : 'pending'; + } } let depth = 1; @@ -1564,8 +1581,18 @@ export function convertOpenAIChunkToGemini( Boolean(choice.finish_reason) && !closingTagName && !/\S/.test(combinedCandidateText); + // The length cap releases undecided prefixes (e.g. a literal " MAX_THINKING_TAG_CANDIDATE_LENGTH); @@ -1591,6 +1618,7 @@ export function convertOpenAIChunkToGemini( requestContext.pendingThinkingTagCandidate = undefined; } else if (isPossibleTag) { if ( + !confirmedOpeningTagCandidate && !closingTagName && combinedCandidateText.trimStart().length > MAX_THINKING_TAG_CANDIDATE_LENGTH diff --git a/packages/core/src/core/openaiContentGenerator/provider/dashscope.ts b/packages/core/src/core/openaiContentGenerator/provider/dashscope.ts index 5f42ae9d09..d375990ba3 100644 --- a/packages/core/src/core/openaiContentGenerator/provider/dashscope.ts +++ b/packages/core/src/core/openaiContentGenerator/provider/dashscope.ts @@ -15,7 +15,6 @@ import type { ChatCompletionContentPartWithCache, ChatCompletionToolWithCache, } from './types.js'; -import type { OpenAIResponseParsingOptions } from '../responseParsingOptions.js'; import { buildRuntimeFetchOptions } from '../../../utils/runtimeFetchOptions.js'; import { createDebugLogger } from '../../../utils/debugLogger.js'; import { @@ -166,11 +165,6 @@ export class DashScopeOpenAICompatibleProvider extends DefaultOpenAICompatiblePr super(contentGeneratorConfig, cliConfig); } - getResponseParsingOptions(): OpenAIResponseParsingOptions { - // ponytail: DashScope-only fallback; remove after provider output stabilizes. - return { contentOnlyThinkingTagLeaks: true }; - } - /** * Determines whether to use the DashScope-compatible provider. * Covers the official regional hosts (DASHSCOPE_REGIONAL_HOSTS), diff --git a/packages/core/src/core/openaiContentGenerator/provider/default.test.ts b/packages/core/src/core/openaiContentGenerator/provider/default.test.ts index ab8ce51253..052b2cfa61 100644 --- a/packages/core/src/core/openaiContentGenerator/provider/default.test.ts +++ b/packages/core/src/core/openaiContentGenerator/provider/default.test.ts @@ -93,6 +93,14 @@ describe('DefaultOpenAICompatibleProvider', () => { }); }); + describe('getResponseParsingOptions', () => { + it('enables leak handling without treating balanced tags as protocol', () => { + expect(provider.getResponseParsingOptions()).toEqual({ + contentOnlyThinkingTagLeaks: true, + }); + }); + }); + describe('buildHeaders', () => { it('should build headers with User-Agent', () => { const headers = provider.buildHeaders(); diff --git a/packages/core/src/core/openaiContentGenerator/provider/default.ts b/packages/core/src/core/openaiContentGenerator/provider/default.ts index 19000fc59c..6835bd38a3 100644 --- a/packages/core/src/core/openaiContentGenerator/provider/default.ts +++ b/packages/core/src/core/openaiContentGenerator/provider/default.ts @@ -4,6 +4,7 @@ import type { Config } from '../../../config/config.js'; import type { ContentGeneratorConfig } from '../../contentGenerator.js'; import { DEFAULT_MAX_RETRIES, resolveRequestTimeout } from '../constants.js'; import type { OpenAICompatibleProvider } from './types.js'; +import type { OpenAIResponseParsingOptions } from '../responseParsingOptions.js'; import { buildRuntimeFetchOptions } from '../../../utils/runtimeFetchOptions.js'; import { tokenLimit, @@ -123,6 +124,15 @@ export class DefaultOpenAICompatibleProvider return {}; } + getResponseParsingOptions(): OpenAIResponseParsingOptions { + // Hybrid-thinking models occasionally bypass the reasoning channel and + // emit their thinking as literal / tags inside content + // (observed in production on qwen3-class models, issue #6666). + // Honored on the streaming path only; non-streaming responses are + // not classified. + return { contentOnlyThinkingTagLeaks: true }; + } + /** * Apply output token limit to a request's max_tokens parameter. * diff --git a/packages/core/src/core/openaiContentGenerator/provider/minimax.ts b/packages/core/src/core/openaiContentGenerator/provider/minimax.ts index 97008e1a45..49b0974792 100644 --- a/packages/core/src/core/openaiContentGenerator/provider/minimax.ts +++ b/packages/core/src/core/openaiContentGenerator/provider/minimax.ts @@ -38,7 +38,7 @@ export class MiniMaxOpenAICompatibleProvider extends DefaultOpenAICompatibleProv } } - getResponseParsingOptions(): OpenAIResponseParsingOptions { + override getResponseParsingOptions(): OpenAIResponseParsingOptions { return { taggedThinkingTags: true }; } }