diff --git a/.changeset/tidy-rivers-merge.md b/.changeset/tidy-rivers-merge.md new file mode 100644 index 0000000000..ea80a31d6b --- /dev/null +++ b/.changeset/tidy-rivers-merge.md @@ -0,0 +1,5 @@ +--- +"@moonshot-ai/kimi-code": patch +--- + +Fix streamed assistant replies rendering one token per line with OpenAI-compatible providers that pad empty reasoning fields into every chunk. diff --git a/packages/kosong/src/providers/openai-legacy.ts b/packages/kosong/src/providers/openai-legacy.ts index f40e59df7d..e3f3de595a 100644 --- a/packages/kosong/src/providers/openai-legacy.ts +++ b/packages/kosong/src/providers/openai-legacy.ts @@ -386,8 +386,17 @@ export class OpenAILegacyStreamedMessage implements StreamedMessage { // Reasoning content: honor the explicit key when set, otherwise scan the // de facto field set and remember the dialect for outbound echo. + // Skip empty strings on pure content messages: some gateways pad every + // content-phase chunk with an empty reasoning field, which would otherwise + // interleave empty think parts between text deltas and defeat sequential + // merging downstream. Tool-call messages keep the empty marker so the + // reasoning field is replayed on continuation, as some reasoning + // endpoints require. const reasoning = reasoningKeyDialect.observe(message); - if (reasoning !== undefined) { + if ( + reasoning !== undefined && + (reasoning.length > 0 || (message.tool_calls?.length ?? 0) > 0) + ) { yield { type: 'think', think: reasoning } satisfies StreamedMessagePart; } @@ -441,8 +450,18 @@ export class OpenAILegacyStreamedMessage implements StreamedMessage { // Reasoning content: honor the explicit key when set, otherwise scan // the de facto field set and remember the dialect for outbound echo. + // Skip empty strings on content-only chunks: some gateways pad every + // content-phase chunk with an empty reasoning field, which would + // otherwise interleave empty think parts between text deltas and + // defeat sequential merging. Tool-call chunks keep the empty marker + // so the reasoning field is replayed on continuation, as some + // reasoning endpoints require; adjacent empty think parts merge into + // one, so at most a single marker survives. const reasoning = reasoningKeyDialect.observe(delta); - if (reasoning !== undefined) { + if ( + reasoning !== undefined && + (reasoning.length > 0 || (delta.tool_calls?.length ?? 0) > 0) + ) { yield { type: 'think', think: reasoning } satisfies StreamedMessagePart; } diff --git a/packages/kosong/test/openai-legacy.test.ts b/packages/kosong/test/openai-legacy.test.ts index 3f5d8b5bf8..7718352b9f 100644 --- a/packages/kosong/test/openai-legacy.test.ts +++ b/packages/kosong/test/openai-legacy.test.ts @@ -1469,7 +1469,11 @@ describe('OpenAILegacyChatProvider', () => { ]); }); - it('yields an empty ThinkPart from an explicitly empty streaming reasoning field', async () => { + it('skips empty streaming reasoning fields instead of yielding empty ThinkParts', async () => { + // Some OpenAI-compatible gateways pad every content-phase chunk with an + // empty reasoning field. Yielding empty ThinkParts would interleave them + // between text deltas and defeat sequential merging (symptom downstream: + // one text part per token). const provider = new OpenAILegacyChatProvider({ model: 'deepseek-reasoner', apiKey: 'test-key', @@ -1488,7 +1492,126 @@ describe('OpenAILegacyChatProvider', () => { const parts: StreamedMessagePart[] = []; for await (const part of stream) parts.push(part); - expect(parts).toEqual([{ type: 'think', think: '' }]); + expect(parts).toEqual([]); + }); + + it('does not interleave empty ThinkParts between text deltas when chunks pad empty reasoning', async () => { + const provider = new OpenAILegacyChatProvider({ + model: 'some-reasoning-model', + apiKey: 'test-key', + stream: true, + }); + + async function* mockedStream(): AsyncIterable> { + yield { + id: 'c1', + choices: [{ index: 0, delta: { content: '', reasoning_content: 'think 1' } }], + }; + yield { + id: 'c1', + choices: [{ index: 0, delta: { content: 'Hello', reasoning_content: '' } }], + }; + yield { + id: 'c1', + choices: [{ index: 0, delta: { content: ' world', reasoning_content: '' } }], + }; + yield { id: 'c1', choices: [{ index: 0, delta: {}, finish_reason: 'stop' }] }; + } + + (provider as any)._client.chat.completions.create = vi + .fn() + .mockResolvedValue(mockedStream()); + + const stream = await provider.generate('', [], []); + const parts: StreamedMessagePart[] = []; + for await (const part of stream) parts.push(part); + + expect(parts).toEqual([ + { type: 'think', think: 'think 1' }, + { type: 'text', text: 'Hello' }, + { type: 'text', text: ' world' }, + ]); + }); + + it('assembles one merged text part through generate() when chunks pad empty reasoning', async () => { + // End-to-end invariant behind the per-token-line-breaks bug: after the + // stream is drained and merged, the reply must be a single text part. + const provider = new OpenAILegacyChatProvider({ + model: 'some-reasoning-model', + apiKey: 'test-key', + stream: true, + }); + + async function* mockedStream(): AsyncIterable> { + yield { + id: 'c1', + choices: [{ index: 0, delta: { content: '', reasoning_content: 'think 1' } }], + }; + yield { + id: 'c1', + choices: [{ index: 0, delta: { content: 'Hello', reasoning_content: '' } }], + }; + yield { + id: 'c1', + choices: [{ index: 0, delta: { content: ' world', reasoning_content: '' } }], + }; + yield { id: 'c1', choices: [{ index: 0, delta: {}, finish_reason: 'stop' }] }; + } + + (provider as any)._client.chat.completions.create = vi + .fn() + .mockResolvedValue(mockedStream()); + + const result = await generate( + provider, + '', + [], + [{ role: 'user', content: [{ type: 'text', text: 'hi' }], toolCalls: [] }], + ); + + expect(result.message.content).toEqual([ + { type: 'think', think: 'think 1' }, + { type: 'text', text: 'Hello world' }, + ]); + }); + + it('preserves the empty reasoning marker on streaming tool-call chunks', async () => { + // Reasoning endpoints may require the reasoning field to be replayed on + // tool-call continuation; the empty marker must survive so the outbound + // message keeps carrying the reasoning field. + const provider = new OpenAILegacyChatProvider({ + model: 'some-reasoning-model', + apiKey: 'test-key', + stream: true, + }); + + async function* mockedStream(): AsyncIterable> { + yield { + id: 'c1', + choices: [ + { + index: 0, + delta: { + reasoning_content: '', + tool_calls: [ + { id: 'call_1', index: 0, function: { name: 'foo', arguments: '{}' } }, + ], + }, + }, + ], + }; + yield { id: 'c1', choices: [{ index: 0, delta: {}, finish_reason: 'tool_calls' }] }; + } + + (provider as any)._client.chat.completions.create = vi + .fn() + .mockResolvedValue(mockedStream()); + + const stream = await provider.generate('', [], []); + const parts: StreamedMessagePart[] = []; + for await (const part of stream) parts.push(part); + + expect(parts.filter((p) => p.type === 'think')).toEqual([{ type: 'think', think: '' }]); }); it('treats blank reasoning_key as unset so defaults still apply', async () => { @@ -1838,7 +1961,7 @@ describe('OpenAILegacyChatProvider — non-stream response parsing', () => { ]); }); - it('yields an empty ThinkPart when the non-stream reasoning field is explicitly empty', async () => { + it('skips an explicitly empty non-stream reasoning field', async () => { const provider = new OpenAILegacyChatProvider({ model: 'deepseek-reasoner', apiKey: 'test-key', @@ -1855,7 +1978,33 @@ describe('OpenAILegacyChatProvider — non-stream response parsing', () => { }), ); - expect(parts).toEqual([{ type: 'think', think: '' }]); + expect(parts).toEqual([]); + }); + + it('preserves the empty reasoning marker on a non-stream tool-call response', async () => { + // Reasoning endpoints may require the reasoning field to be replayed on + // tool-call continuation; the empty marker must survive so the outbound + // message keeps carrying the reasoning field. + const provider = new OpenAILegacyChatProvider({ + model: 'deepseek-reasoner', + apiKey: 'test-key', + stream: false, + reasoningKey: 'reasoning_content', + }); + + const parts = await collectFromMockedResponse( + provider, + makeNonStreamResponse({ + role: 'assistant', + content: null, + reasoning_content: '', + tool_calls: [ + { id: 'call_1', type: 'function', function: { name: 'foo', arguments: '{}' } }, + ], + }), + ); + + expect(parts.filter((p) => p.type === 'think')).toEqual([{ type: 'think', think: '' }]); }); it('non-stream response yields ToolCall parts when tool_calls present', async () => {