diff --git a/.changeset/model-message-reasoning-content.md b/.changeset/model-message-reasoning-content.md new file mode 100644 index 000000000..af8a2af57 --- /dev/null +++ b/.changeset/model-message-reasoning-content.md @@ -0,0 +1,5 @@ +--- +"@truefoundry/trueforge-core": patch +--- + +Persist exact streamed reasoning_content on model.message session events (omit from thread context). diff --git a/packages/trueforge-core/src/core/llm/LLMTypes.ts b/packages/trueforge-core/src/core/llm/LLMTypes.ts index 065aea70e..66e880aa8 100644 --- a/packages/trueforge-core/src/core/llm/LLMTypes.ts +++ b/packages/trueforge-core/src/core/llm/LLMTypes.ts @@ -86,7 +86,7 @@ export const RawAssistantMessageSchema = ChatCompletionAssistantMessageParamSche .extend({ tool_calls: z.array(RawToolCallSchema).optional(), thinking_blocks: z.array(ThinkingBlockUnionSchema).optional(), - /** Plain-text thinking content streamed incrementally for frontend display; redundant with thinking_blocks[].thinking. */ + /** Plain-text thinking for frontend display; exact concat of streamed deltas when set on the assembled message. */ reasoning_content: z.string().optional(), /** Source of the message: which provider/model sent it (`provider_type/provider_name/model_name`). */ source: z.string().optional(), @@ -120,7 +120,7 @@ export const ExtendedChunkDeltaSchema = ChatCompletionChunkDeltaSchema.omit({ to tool_calls: z.array(ExtendedChunkDeltaToolCallSchema).optional(), /** Structured thinking blocks from the gateway; accumulated into complete blocks (with signatures) for multi-turn replay. */ thinking_blocks: z.array(ThinkingBlockUnionSchema).optional(), - /** Plain-text thinking content streamed incrementally for frontend display; not stored — redundant with thinking_blocks[].thinking. */ + /** Plain-text thinking fragment for frontend display; assembled message stores the full concat separately. */ reasoning_content: z.string().optional(), }) .openapi('ExtendedChunkDelta'); diff --git a/packages/trueforge-core/src/core/llm/VercelAILLM.ts b/packages/trueforge-core/src/core/llm/VercelAILLM.ts index da7e1a3b6..65148cc08 100644 --- a/packages/trueforge-core/src/core/llm/VercelAILLM.ts +++ b/packages/trueforge-core/src/core/llm/VercelAILLM.ts @@ -1128,6 +1128,7 @@ export async function* mapStreamToChunks({ const toolCallStates = new Map(); let nextToolIndex = 0; let accumulatedText = ''; + let accumulatedReasoning = ''; const accumulatedThinking: ThinkingBlock[] = []; const thinkingByReasoningItem = new Map(); let currentThinkingBlock: ThinkingBlock | null = null; @@ -1188,6 +1189,7 @@ export async function* mapStreamToChunks({ // signature, breaking replay while the text still streams out. currentThinkingBlock ??= openThinkingBlock(part.providerMetadata); currentThinkingBlock.thinking += part.text; + accumulatedReasoning += part.text; applyReasoningSignature({ block: currentThinkingBlock, providerMetadata: part.providerMetadata }); yield { ...makeBase(), @@ -1349,6 +1351,7 @@ export async function* mapStreamToChunks({ const output: RawAssistantMessage = { role: 'assistant', content: accumulatedText || null, + ...(accumulatedReasoning.length > 0 ? { reasoning_content: accumulatedReasoning } : {}), ...(accumulatedThinking.length > 0 ? { thinking_blocks: accumulatedThinking } : {}), ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}), }; diff --git a/packages/trueforge-core/src/core/runtime/AgentThread.ts b/packages/trueforge-core/src/core/runtime/AgentThread.ts index 3e2c6785b..6185bb779 100644 --- a/packages/trueforge-core/src/core/runtime/AgentThread.ts +++ b/packages/trueforge-core/src/core/runtime/AgentThread.ts @@ -1077,6 +1077,9 @@ export class AgentThread { // Hence, we resolve the underlying tool to get the tool information. resolveUnderlyingTool: true, }); + // Display-only; keep off thread context so replay uses thinking_blocks alone. + const { reasoning_content: _omitReasoning, ...assistantMessageForContext } = assistantMessage; + void _omitReasoning; const finishReason = result.value.finish_reason; const agentAssistantMessage = buildModelMessageEvent({ assistantMessage: await enrichAssistantMessage({ @@ -1115,7 +1118,7 @@ export class AgentThread { } yield* this.appendToContext({ - context: [assistantMessage], + context: [assistantMessageForContext], output: [agentAssistantMessage], currentContextUsage: currentContextUsageFromCompletion(result.value.usage), usage: result.value.usage, diff --git a/packages/trueforge-core/tests/core/llm/VercelAILLM.stream.test.ts b/packages/trueforge-core/tests/core/llm/VercelAILLM.stream.test.ts index bb15d4aed..6435bfcdd 100644 --- a/packages/trueforge-core/tests/core/llm/VercelAILLM.stream.test.ts +++ b/packages/trueforge-core/tests/core/llm/VercelAILLM.stream.test.ts @@ -269,6 +269,7 @@ describe('mapStreamToChunks', () => { expect(reasoningChunks).toHaveLength(1); expect(reasoningChunks[0]?.choices[0]?.delta.reasoning_content).toBe('step one'); expect(final.output.thinking_blocks).toEqual([{ type: 'thinking', thinking: 'step one' }]); + expect(final.output.reasoning_content).toBe('step one'); }); it('attaches signature from providerMetadata.*.reasoningEncryptedContent on reasoning-end (OpenAI)', async () => { diff --git a/packages/trueforge-core/tests/core/runtime/modelMessageReasoningContent.test.ts b/packages/trueforge-core/tests/core/runtime/modelMessageReasoningContent.test.ts new file mode 100644 index 000000000..4010f813d --- /dev/null +++ b/packages/trueforge-core/tests/core/runtime/modelMessageReasoningContent.test.ts @@ -0,0 +1,95 @@ +import type { ILLM } from '../../../src/core/llm/ILLM'; +import type { ExtendedChatCompletionChunk, RawAssistantMessageWithUsage } from '../../../src/core/llm/LLMTypes'; +import { getEmptyUsage } from '../../../src/core/llm/LLMTypes'; +import { AgentThread } from '../../../src/core/runtime/AgentThread'; +import { InternalEventType, type AgentThreadAppendContext } from '../../../src/core/runtime/AgentThread.types'; +import { NOOP_AGENT_TRACING } from '../../../src/core/tracing/NoopAgentTracing'; +import '../harnessMocks'; +import { makeSilentLogger } from '../harnessMocks'; + +const silentLogger = makeSilentLogger(); + +// eslint-disable-next-line @typescript-eslint/require-await -- async generator fixture, not awaiting I/O +async function* reasoningAnswerStream(): AsyncGenerator< + ExtendedChatCompletionChunk, + RawAssistantMessageWithUsage, + unknown +> { + yield { + id: 'chunk-1', + object: 'chat.completion.chunk', + created: 0, + model: 'test-model', + choices: [ + { + index: 0, + delta: { reasoning_content: 'step one' }, + finish_reason: null, + logprobs: null, + }, + ], + }; + yield { + id: 'chunk-2', + object: 'chat.completion.chunk', + created: 0, + model: 'test-model', + choices: [ + { + index: 0, + delta: { content: 'hello', role: 'assistant' }, + finish_reason: 'stop', + logprobs: null, + }, + ], + }; + + return { + output: { + role: 'assistant', + content: 'hello', + reasoning_content: 'step one', + thinking_blocks: [{ type: 'thinking', thinking: 'step one' }], + }, + usage: getEmptyUsage(), + finish_reason: 'stop', + }; +} + +describe('AgentThread model.message reasoning_content', () => { + it('persists reasoning_content on the event and omits it from context', async () => { + const modelClient: ILLM = { + create: jest.fn().mockImplementation(() => reasoningAnswerStream()), + createNonStream: jest.fn(), + }; + + const thread = new AgentThread({ + threadId: 'main', + title: 'Main', + tracing: NOOP_AGENT_TRACING, + logger: silentLogger, + definition: { + modelClient, + instruction: 'test', + toolSets: [], + }, + context: [{ role: 'user', content: 'hi' }], + }); + + let append: AgentThreadAppendContext | undefined; + for await (const event of thread.execute({ signal: new AbortController().signal })) { + if (event.type === InternalEventType.AGENT_CONTEXT_APPEND) { + append = event; + } + } + + expect(append?.context[0]).not.toHaveProperty('reasoning_content'); + expect(append?.context[0]).toMatchObject({ + thinking_blocks: [{ type: 'thinking', thinking: 'step one' }], + }); + + const modelMessage = append?.output.find(e => e.type === 'model.message'); + expect(modelMessage).toMatchObject({ reasoning_content: 'step one' }); + expect(modelMessage).not.toHaveProperty('thinking_blocks'); + }); +});