From 70c5ecced6ffac1161d7f5a5dd23dc7dbcec8cdd Mon Sep 17 00:00:00 2001 From: Hoan HL Date: Fri, 18 Sep 2026 10:18:08 +0700 Subject: [PATCH 1/2] fix: keep pre-tool assistant text in tool-call history MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When a model emitted text before calling a tool, every adapter dropped it: the appended assistant turn went on the wire with content null (or with only tool_use/functionCall blocks). On the post-tool re-prompt the model had no record of having spoken, so it said the same sentence again — and since that turn is streamed to TTS, the caller hears it twice. appendAssistantToolCall now takes the pre-tool text and each adapter carries it in its native shape: OpenAI content alongside tool_calls, Anthropic and Bedrock a text block before the tool_use blocks, Gemini a text part before the functionCall part. Blank text is omitted rather than sent as an empty block, which Anthropic and Bedrock reject. The argument is optional, so existing callers keep compiling and get the old behaviour until they pass it. Contract tests [22] and [23] cover both paths for every adapter. Co-Authored-By: Claude Opus 5 (1M context) --- package-lock.json | 4 +- package.json | 2 +- src/adapters/_template/adapter.ts | 7 +++- src/adapters/anthropic/adapter.ts | 22 ++++++---- src/adapters/azure-openai/adapter.ts | 3 +- src/adapters/bedrock/adapter.ts | 23 ++++++---- src/adapters/google/_streaming.ts | 20 +++++---- src/adapters/google/adapter.ts | 3 +- src/adapters/openai/_streaming.ts | 6 ++- src/adapters/openai/adapter.ts | 3 +- src/adapters/vertex-gemini/adapter.ts | 3 +- src/adapters/vertex-openai/adapter.ts | 3 +- src/test-kit/fake-adapter.ts | 5 ++- src/test-kit/runner.ts | 60 +++++++++++++++++++++++++++ src/types.ts | 7 ++++ 15 files changed, 136 insertions(+), 35 deletions(-) diff --git a/package-lock.json b/package-lock.json index 98242f5..2dbdf67 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@jambonz/llm", - "version": "0.6.2", + "version": "0.6.3", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@jambonz/llm", - "version": "0.6.2", + "version": "0.6.3", "license": "MIT", "dependencies": { "@anthropic-ai/sdk": "^0.91.0", diff --git a/package.json b/package.json index 73edade..707d94d 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@jambonz/llm", - "version": "0.6.2", + "version": "0.6.3", "description": "Voice-oriented LLM adapter library — a uniform interface for streaming chat completions from multiple LLM vendors, with first-class support for tool calls, AbortSignal interruption, and lossless history round-tripping.", "type": "module", "main": "./dist/index.cjs", diff --git a/src/adapters/_template/adapter.ts b/src/adapters/_template/adapter.ts index 3478d3c..474b0aa 100644 --- a/src/adapters/_template/adapter.ts +++ b/src/adapters/_template/adapter.ts @@ -61,6 +61,7 @@ export class TemplateAdapter implements LlmAdapter { appendAssistantToolCall( _history: Message[], _toolCalls: ReadonlyArray, + _assistantText?: string, ): Message[] { // TODO: append an assistant turn containing the given tool calls to // history, in the vendor's native shape. This turn must precede the @@ -72,10 +73,12 @@ export class TemplateAdapter implements LlmAdapter { // ...history, // { // role: 'assistant', - // content: '', + // content: assistantText ?? '', // vendorRaw: { // role: 'assistant', - // content: null, + // // text the model emitted before the call — carry it, or the + // // model repeats itself on the post-tool turn + // content: assistantText || null, // tool_calls: toolCalls.map((tc) => ({ // id: tc.id, // type: 'function', diff --git a/src/adapters/anthropic/adapter.ts b/src/adapters/anthropic/adapter.ts index 518dc92..801f42c 100644 --- a/src/adapters/anthropic/adapter.ts +++ b/src/adapters/anthropic/adapter.ts @@ -280,21 +280,29 @@ export class AnthropicAdapter implements LlmAdapter { appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { + // A text block may precede tool_use blocks in the same assistant turn. + // Anthropic rejects an empty text block, so it is only included when the + // model actually said something before calling the tool. + const text = assistantText?.trim() ? assistantText : null; const wireMessage = { role: 'assistant', - content: toolCalls.map((tc) => ({ - type: 'tool_use', - id: tc.id, - name: tc.name, - input: tc.arguments ?? {}, - })), + content: [ + ...(text ? [{ type: 'text', text }] : []), + ...toolCalls.map((tc) => ({ + type: 'tool_use', + id: tc.id, + name: tc.name, + input: tc.arguments ?? {}, + })), + ], }; return [ ...history, { role: 'assistant', - content: '', + content: text ?? '', vendorRaw: wireMessage, }, ]; diff --git a/src/adapters/azure-openai/adapter.ts b/src/adapters/azure-openai/adapter.ts index 1776651..19eedac 100644 --- a/src/adapters/azure-openai/adapter.ts +++ b/src/adapters/azure-openai/adapter.ts @@ -105,8 +105,9 @@ export class AzureOpenAIAdapter implements LlmAdapter { appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { - return appendOpenAIAssistantToolCall(history, toolCalls); + return appendOpenAIAssistantToolCall(history, toolCalls, assistantText); } appendToolResult(history: Message[], toolCallId: string, result: unknown): Message[] { diff --git a/src/adapters/bedrock/adapter.ts b/src/adapters/bedrock/adapter.ts index da26775..bf19749 100644 --- a/src/adapters/bedrock/adapter.ts +++ b/src/adapters/bedrock/adapter.ts @@ -298,22 +298,29 @@ export class BedrockAdapter implements LlmAdapter { appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { + // Converse allows a text block before toolUse blocks in the same turn; + // an empty one is rejected, so it is only included when non-blank. + const text = assistantText?.trim() ? assistantText : null; const wireMessage = { role: 'assistant', - content: toolCalls.map((tc) => ({ - toolUse: { - toolUseId: tc.id, - name: tc.name, - input: tc.arguments ?? {}, - }, - })), + content: [ + ...(text ? [{ text }] : []), + ...toolCalls.map((tc) => ({ + toolUse: { + toolUseId: tc.id, + name: tc.name, + input: tc.arguments ?? {}, + }, + })), + ], }; return [ ...history, { role: 'assistant', - content: '', + content: text ?? '', vendorRaw: wireMessage, }, ]; diff --git a/src/adapters/google/_streaming.ts b/src/adapters/google/_streaming.ts index d572879..f86df5b 100644 --- a/src/adapters/google/_streaming.ts +++ b/src/adapters/google/_streaming.ts @@ -182,21 +182,27 @@ export async function* streamFromGemini( export function appendGeminiAssistantToolCall( history: Message[], toolCalls: ReadonlyArray>, + assistantText?: string, ): Message[] { + // A text part may precede functionCall parts in the same model turn. + const text = assistantText?.trim() ? assistantText : null; const wireMessage = { role: 'model', - parts: toolCalls.map((tc) => ({ - functionCall: { - name: parseToolName(tc.id), - args: tc.arguments ?? {}, - }, - })), + parts: [ + ...(text ? [{ text }] : []), + ...toolCalls.map((tc) => ({ + functionCall: { + name: parseToolName(tc.id), + args: tc.arguments ?? {}, + }, + })), + ], }; return [ ...history, { role: 'assistant', - content: '', + content: text ?? '', vendorRaw: wireMessage, }, ]; diff --git a/src/adapters/google/adapter.ts b/src/adapters/google/adapter.ts index 41e25b2..55f9eb5 100644 --- a/src/adapters/google/adapter.ts +++ b/src/adapters/google/adapter.ts @@ -51,8 +51,9 @@ export class GoogleAdapter implements LlmAdapter { appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { - return appendGeminiAssistantToolCall(history, toolCalls); + return appendGeminiAssistantToolCall(history, toolCalls, assistantText); } appendToolResult(history: Message[], toolCallId: string, result: unknown): Message[] { diff --git a/src/adapters/openai/_streaming.ts b/src/adapters/openai/_streaming.ts index d704060..8563775 100644 --- a/src/adapters/openai/_streaming.ts +++ b/src/adapters/openai/_streaming.ts @@ -332,10 +332,12 @@ export async function* streamFromOpenAI( export function appendOpenAIAssistantToolCall( history: Message[], toolCalls: ReadonlyArray>, + assistantText?: string, ): Message[] { + const text = assistantText?.trim() ? assistantText : null; const wireMessage = { role: 'assistant', - content: null, + content: text, tool_calls: toolCalls.map((tc) => ({ id: tc.id, type: 'function', @@ -351,7 +353,7 @@ export function appendOpenAIAssistantToolCall( ...history, { role: 'assistant', - content: '', + content: text ?? '', vendorRaw: wireMessage, }, ]; diff --git a/src/adapters/openai/adapter.ts b/src/adapters/openai/adapter.ts index 180c849..b307802 100644 --- a/src/adapters/openai/adapter.ts +++ b/src/adapters/openai/adapter.ts @@ -70,8 +70,9 @@ export class OpenAIAdapter implements LlmAdapter { appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { - return appendOpenAIAssistantToolCall(history, toolCalls); + return appendOpenAIAssistantToolCall(history, toolCalls, assistantText); } appendToolResult(history: Message[], toolCallId: string, result: unknown): Message[] { diff --git a/src/adapters/vertex-gemini/adapter.ts b/src/adapters/vertex-gemini/adapter.ts index 62fa461..465b4bf 100644 --- a/src/adapters/vertex-gemini/adapter.ts +++ b/src/adapters/vertex-gemini/adapter.ts @@ -67,8 +67,9 @@ export class VertexGeminiAdapter implements LlmAdapter appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { - return appendGeminiAssistantToolCall(history, toolCalls); + return appendGeminiAssistantToolCall(history, toolCalls, assistantText); } appendToolResult(history: Message[], toolCallId: string, result: unknown): Message[] { diff --git a/src/adapters/vertex-openai/adapter.ts b/src/adapters/vertex-openai/adapter.ts index 1168c44..88a6be5 100644 --- a/src/adapters/vertex-openai/adapter.ts +++ b/src/adapters/vertex-openai/adapter.ts @@ -109,8 +109,9 @@ export class VertexOpenAIAdapter implements LlmAdapter appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { - return appendOpenAIAssistantToolCall(history, toolCalls); + return appendOpenAIAssistantToolCall(history, toolCalls, assistantText); } appendToolResult(history: Message[], toolCallId: string, result: unknown): Message[] { diff --git a/src/test-kit/fake-adapter.ts b/src/test-kit/fake-adapter.ts index 915175a..c504bbd 100644 --- a/src/test-kit/fake-adapter.ts +++ b/src/test-kit/fake-adapter.ts @@ -92,13 +92,16 @@ export class FakeAdapter implements LlmAdapter { appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[] { + const text = assistantText?.trim() ? assistantText : null; return [ ...history, { role: 'assistant', - content: '', + content: text ?? '', vendorRaw: { + ...(text ? { text } : {}), role: 'assistant', toolCalls: toolCalls.map((tc) => ({ id: tc.id, diff --git a/src/test-kit/runner.ts b/src/test-kit/runner.ts index 5dfb575..8909b73 100644 --- a/src/test-kit/runner.ts +++ b/src/test-kit/runner.ts @@ -416,6 +416,66 @@ export function runContractTests(harness: ContractHarness): void { expect(appended.role).toBe('assistant'); expect(appended.vendorRaw).toBeDefined(); }); + + it('[22] appendAssistantToolCall carries pre-tool assistant text onto the wire', async () => { + await harness.mockScenario('tool-call'); + const adapter = await init(harness); + const first = await drainStream(adapter, { + model: harness.toolCapableModel, + messages: [{ role: 'user', content: 'go' }], + tools: [ + { name: 'test_tool', description: 'x', parameters: { type: 'object' } }, + ], + }); + const tc = first.find( + (e): e is Extract => e.type === 'toolCall', + ); + expect(tc).toBeDefined(); + + // Text the model emitted before calling the tool. Dropping it makes the + // model believe it never spoke, so it repeats itself after the tool + // result — the caller hears the same sentence twice. + const preToolText = 'Checking that now.'; + const history = adapter.appendAssistantToolCall( + [{ role: 'user', content: 'go' }], + [tc!], + preToolText, + ); + const appended = history[history.length - 1]!; + expect(appended.content).toBe(preToolText); + // Vendors encode it differently (a string, a text block, a text part), + // so assert on the serialized wire message rather than a fixed shape. + expect(JSON.stringify(appended.vendorRaw)).toContain(preToolText); + }); + + it('[23] appendAssistantToolCall omits empty pre-tool text from the wire', async () => { + await harness.mockScenario('tool-call'); + const adapter = await init(harness); + const first = await drainStream(adapter, { + model: harness.toolCapableModel, + messages: [{ role: 'user', content: 'go' }], + tools: [ + { name: 'test_tool', description: 'x', parameters: { type: 'object' } }, + ], + }); + const tc = first.find( + (e): e is Extract => e.type === 'toolCall', + ); + expect(tc).toBeDefined(); + + // Anthropic and Bedrock reject an empty text block, so a blank string + // must not produce one. Omitting the argument entirely is the same case. + for (const blank of [undefined, '', ' ']) { + const history = adapter.appendAssistantToolCall( + [{ role: 'user', content: 'go' }], + [tc!], + blank, + ); + const appended = history[history.length - 1]!; + expect(appended.content).toBe(''); + expect(JSON.stringify(appended.vendorRaw)).not.toContain('"text":""'); + } + }); }); } diff --git a/src/types.ts b/src/types.ts index 291d7a5..61be971 100644 --- a/src/types.ts +++ b/src/types.ts @@ -408,10 +408,17 @@ export interface LlmAdapter { * content:[{type:'tool_use',...}]}`; Google: `{role:'model', * parts:[{functionCall,...}]}`; etc.). The returned `Message` carries the * wire shape in `vendorRaw` and is round-trippable through `stream()`. + * + * `assistantText` is any text the model emitted BEFORE the tool calls in the + * same turn. Vendors allow it alongside the call (OpenAI: `content`; + * Anthropic and Google: a text block/part preceding the call), and dropping + * it makes the model believe it never spoke — on the post-tool turn it + * repeats itself. Adapters MUST carry it onto the wire when non-empty. */ appendAssistantToolCall( history: Message[], toolCalls: ReadonlyArray, + assistantText?: string, ): Message[]; /** From b34f21f153ac5a1f8bf65160146aa000833bd219 Mon Sep 17 00:00:00 2001 From: Hoan HL Date: Fri, 18 Sep 2026 11:18:01 +0700 Subject: [PATCH 2/2] chore: leave the version at 0.6.2 Releasing is a separate step owned by someone else; this branch should not carry a version bump. Co-Authored-By: Claude Opus 5 (1M context) --- package-lock.json | 4 ++-- package.json | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/package-lock.json b/package-lock.json index 2dbdf67..98242f5 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@jambonz/llm", - "version": "0.6.3", + "version": "0.6.2", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@jambonz/llm", - "version": "0.6.3", + "version": "0.6.2", "license": "MIT", "dependencies": { "@anthropic-ai/sdk": "^0.91.0", diff --git a/package.json b/package.json index 707d94d..73edade 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@jambonz/llm", - "version": "0.6.3", + "version": "0.6.2", "description": "Voice-oriented LLM adapter library — a uniform interface for streaming chat completions from multiple LLM vendors, with first-class support for tool calls, AbortSignal interruption, and lossless history round-tripping.", "type": "module", "main": "./dist/index.cjs",