// BUG-1053: the user-visible consultation answer was written without the chart. // // The natal loop's step after `run-jyotish-consultation` saw the evidence, but // its text was drained and a second `agent.stream` (compose) wrote the answer // from history + the question only. These tests drive the real personal Agent // (`getJyotishAgent`, real skill binding, real calculation tool, real // prepareStep) over a prompt-recording fake model, with the calculation fed by // the golden engine capture in fixtures/ (AGENTS §7.4). What each model call // was shown is the assertion: the call that writes the answer must have the // tool result in its prompt, and there must be no second, blind call. import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; import test from "node:test"; import { getJyotishAgent } from "../src/mastra/index.ts"; import { createNdjsonParser } from "../src/lib/consultation-agent-events.ts"; import { consultationContinueMessages, natalAnswerShapeInstruction, REPORT_HEADING, } from "../src/lib/consultation-thinking-plan.ts"; import { streamAgentResponse } from "../src/lib/stream-agent-response.ts"; import { PASS4_RETRY_HINT } from "../src/lib/timing-output-guard.ts"; import { AGENT_MAX_STEPS, AGENT_TIMEOUT_MS, CONSULTATION_ANSWER_TIMEOUT_MS, consultationNatalPrepareStep, consultationStepBudgetReceipt, createConsultationAgentContext, createConsultationRunClock, createConsultationRuntimeState, publicConsultationRuntimeSteps, } from "../src/mastra/consultation-tools.ts"; const golden = JSON.parse(readFileSync( new URL("./fixtures/consultation-workflow-report-blocked-repairs-golden.json", import.meta.url), "utf8", )) as { smoke_birth: Record; themes: Record> }; const careerWorkflow = golden.themes.career!; // A fact only the calculation carries: the golden chart's Moon longitude. const GOLDEN_MOON_DEGREE = "3.2259"; // Fictional capture identity from the fixture itself, not a real person. const serverChart = { name: "测试", toolInput: { year: 1993, month: 6, day: 15, hour: 10, minute: 30, city: "Handan", lat: 36.42, lon: 114.21, tz: 8, ayanamsa: "raman" as const, declared_accuracy: "15min" as const, time_source: "family_vague", }, truth: { birthDate: "1993-06-15", reportedBirthTime: "10:30", activeBirthTime: null, selectedTimeKind: "reported" as const, birthTimeSource: "reported", birthTimeStatus: "reported", placeLabel: "Handan", placeCodes: { countryCode: "CN", provinceCode: null, cityCode: null, districtCode: null }, placeId: null, placeType: "city", placeProvider: "profile", latitude: 36.42, longitude: 114.21, timezoneId: "Asia/Shanghai", timezoneSource: "profile", timezoneOffset: 8, }, }; const PRE_TOOL = "我先排一下盘,稍等。"; const BETWEEN_TOOLS = "我再核对一次计算结果。"; const OPENER = "你这盘外面看着稳,底下其实一直在换跑道:表面求安定,底下要自己说了算。"; // 原值: 夹具里有 `## ${REPORT_HEADING.question}` 一节 // 新值: 该节删除,夹具从口语开场直接到 `## ${REPORT_HEADING.support}` // 原因: TASK-consult-answer-the-question-20260927 D8 / BUG-1073:开场就是回答,正文不再从「先回答你的问题」重来 const ANSWER = [ OPENER, "", `## ${REPORT_HEADING.support}`, "月亮落在白羊座九宫,想法来得快。", "", `## ${REPORT_HEADING.timing}`, "这半年先试,不急着定。", "", `## ${REPORT_HEADING.action}`, "- **这周**:约一位同行聊一次。", "", ].join("\n"); type Part = { text?: string; tool?: { name: string; input: Record } }; type Turn = { parts: Part[]; finish: string; delayMs?: number }; /** A LanguageModelV2 that plays `turns` in order and records every prompt. */ function scriptedModel(turns: Turn[]) { const prompts: unknown[][] = []; let call = 0; const model = { specificationVersion: "v2", provider: "fake", modelId: "fake-single-pass", supportedUrls: {}, async doGenerate() { throw new Error("not used"); }, async doStream(options: { prompt: unknown[]; abortSignal?: AbortSignal }) { prompts.push(options.prompt); const turn = turns[call] ?? { parts: [], finish: "stop" }; call += 1; const signal = options.abortSignal; const stream = new ReadableStream({ async start(controller) { controller.enqueue({ type: "stream-start", warnings: [] }); let textOpen = false; for (const [index, part] of turn.parts.entries()) { if (turn.delayMs) { try { await new Promise((resolve, reject) => { if (signal?.aborted) return reject(signal.reason); const timer = setTimeout(resolve, turn.delayMs); signal?.addEventListener("abort", () => { clearTimeout(timer); reject(signal.reason); }, { once: true }); }); } catch (error) { controller.error(error); return; } } if (part.text !== undefined) { if (!textOpen) { controller.enqueue({ type: "text-start", id: `t${call}` }); textOpen = true; } controller.enqueue({ type: "text-delta", id: `t${call}`, delta: part.text }); } if (part.tool) { if (textOpen) { controller.enqueue({ type: "text-end", id: `t${call}` }); textOpen = false; } controller.enqueue({ type: "tool-call", toolCallId: `call-${call}-${index}`, toolName: part.tool.name, input: JSON.stringify(part.tool.input), }); } } if (textOpen) controller.enqueue({ type: "text-end", id: `t${call}` }); controller.enqueue({ type: "finish", finishReason: turn.finish, usage: { inputTokens: 10, outputTokens: 10, totalTokens: 20 }, }); controller.close(); }, }); return { stream }; }, }; return { model, prompts, calls: () => call }; } const calcCall = (text?: string): Turn => ({ parts: [ ...(text ? [{ text }] : []), { tool: { name: "run-jyotish-consultation", input: { question: "我的事业能换方向吗", domains: ["career"] } } }, ], finish: "tool-calls", }); function pieces(text: string, size = 12) { return (text.match(new RegExp(`[\\s\\S]{1,${size}}`, "g")) ?? []).map((value) => ({ text: value })); } let seq = 0; /** * The natal route's wiring, minus HTTP, auth and billing: the same Agent, * prepareStep, run clock, stream options, step-scoped answer, answer-phase * hand-over, continuation builder and answer retry as `runAgenticConsultation`. */ async function runNatal(turns: Turn[], options: { toolPhaseMs?: number; answerMs?: number } = {}) { seq += 1; const { model, prompts, calls } = scriptedModel(turns); const state = createConsultationRuntimeState({ plannedSteps: AGENT_MAX_STEPS }); const clock = createConsultationRunClock({ toolPhaseMs: options.toolPhaseMs ?? AGENT_TIMEOUT_MS, answerMs: options.answerMs ?? CONSULTATION_ANSWER_TIMEOUT_MS, answerReady: () => state.consultationToolCompleted, }); let workflowRuns = 0; const agentContext = createConsultationAgentContext({ userId: "u", sessionId: "s", requestId: `single-pass-${seq}`, consultationMode: "verified_chart", theme: "career", serverChart: serverChart as never, abortSignal: clock.toolSignal, state, runWorkflow: async () => { workflowRuns += 1; return structuredClone(careerWorkflow) as never; }, }); const agent = getJyotishAgent({ id: `fake-${seq}`, model } as never, agentContext); const baseMessages = [{ role: "user" as const, content: `${natalAnswerShapeInstruction()}\n问题:我的事业能换方向吗` }]; const streamOptions = { runId: `run-${seq}`, maxSteps: AGENT_MAX_STEPS, abortSignal: clock.loopSignal }; const natalStreamOptions = { ...streamOptions, prepareStep: consultationNatalPrepareStep }; const continuations: unknown[] = []; const retryHints: Array = []; let completed: string | null = null; let charges = 0; let errored: unknown = null; const result = await agent.stream(baseMessages as never, natalStreamOptions as never); const response = streamAgentResponse({ runId: `run-${seq}`, requestId: `req-${seq}`, state, stream: result.fullStream as ReadableStream, requireTool: true, stepScopedAnswer: true, onAnswerPhase: () => { clock.answerSignal(); }, pass4Mode: "verified_chart", retryForAnswer: async (retryHint) => { retryHints.push(retryHint); const retried = await agent.stream([ ...baseMessages, { role: "user" as const, content: `服务器计算已经完成,但上一轮没有输出任何回答文本。请重新取回本次计算结果,然后直接给出回答。${retryHint ? `\n${retryHint}` : ""}` }, ] as never, { ...natalStreamOptions, abortSignal: clock.answerSignal() } as never); return retried.fullStream as ReadableStream; }, continueAfterLength: async (output, evidence) => { continuations.push(evidence); const continued = await agent.stream( consultationContinueMessages(baseMessages, output, evidence) as never, { ...streamOptions, abortSignal: clock.answerSignal() } as never, ); return continued.fullStream as ReadableStream; }, toolStatus: () => "ready", receipt: () => ({ runId: `run-${seq}`, runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps: publicConsultationRuntimeSteps(state), stepBudget: consultationStepBudgetReceipt(state), workflow: state.workflowReceipt ?? { route: "pending", status: "blocked", preciseTiming: "blocked", missingLayers: [] }, techniqueTruth: "unknown", }) as never, onComplete: (output) => { completed = output; charges += 1; }, onError: (error) => { errored = error; }, }); const events: Array<{ type: string; code?: string; text?: string; receipt?: unknown }> = []; const parser = createNdjsonParser((event) => events.push(event as never)); parser.finish(await response.text()); const answer = events.filter((event) => event.type === "answer.delta").map((event) => event.text ?? "").join(""); const terminal = events.filter((event) => event.type === "run.completed" || event.type === "run.failed"); return { state, events, answer, terminal, prompts, calls: calls(), continuations, retryHints, workflowRuns, completed: completed as string | null, charges, errored, }; } /** Index of the first prompt that already contains the calculation result. */ function firstPromptWithToolResult(prompts: unknown[][]) { return prompts.findIndex((prompt) => prompt.some((message) => { const value = message as { role?: string; content?: unknown }; return value.role === "tool" && JSON.stringify(value.content).includes("run-jyotish-consultation"); })); } test("the model call that writes the answer has the calculation result in its prompt, and no blind call follows", async () => { const run = await runNatal([calcCall(PRE_TOOL), { parts: pieces(ANSWER), finish: "stop" }]); assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]); assert.equal(run.workflowRuns, 1); // Exactly two model calls: the one that called the tool and the one after it. // The pre-fix route opened a third (compose) whose prompt had no tool result. assert.equal(run.calls, 2); const withResult = firstPromptWithToolResult(run.prompts); assert.equal(withResult, 1, "the call after the tool result sees it"); assert.equal(run.prompts.length - withResult, 1, "exactly one model call after the tool result"); assert.ok(JSON.stringify(run.prompts[1]).includes(GOLDEN_MOON_DEGREE), "the golden chart fact reaches the writer"); // What that call wrote is what the user got, with the four headings intact. assert.equal(run.answer, ANSWER); assert.equal(run.completed, ANSWER); for (const heading of Object.values(REPORT_HEADING)) assert.match(run.answer, new RegExp(`^## ${heading}$`, "m")); assert.ok(run.answer.startsWith(OPENER), "the opener has no heading"); // Pre-tool narration is not the answer. assert.equal(run.answer.includes(PRE_TOOL), false); }); test("the writing instructions travel with the user turn the loop sees", async () => { const run = await runNatal([calcCall(), { parts: pieces(ANSWER), finish: "stop" }]); const writerPrompt = JSON.stringify(run.prompts[1]); for (const heading of Object.values(REPORT_HEADING)) assert.ok(writerPrompt.includes(`## ${heading}`)); assert.match(writerPrompt, /开场不要标题/); // The removed compose prompt claimed a finished calculation it never saw. assert.doesNotMatch(writerPrompt, /服务器计算已经完成。不要再调用排盘工具/); }); test("narration in a step that calls a tool never reaches the answer, even after the calculation", async () => { const run = await runNatal([ calcCall(PRE_TOOL), // After the calculation the model narrates and calls the tool again (a // request-cache hit), then writes the answer. calcCall(BETWEEN_TOOLS), { parts: pieces(ANSWER), finish: "stop" }, ]); assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]); assert.equal(run.workflowRuns, 1, "the second call is served from the request cache"); assert.equal(run.answer, ANSWER); assert.equal(run.answer.includes(PRE_TOOL), false); assert.equal(run.answer.includes(BETWEEN_TOOLS), false); assert.equal(run.charges, 1); }); test("a normal stop completes and charges exactly once", async () => { const run = await runNatal([calcCall(), { parts: pieces(ANSWER), finish: "stop" }]); assert.equal(run.charges, 1); assert.equal(run.errored, null); assert.equal(run.state.composeFinishReason, "stop", "the answer-writing step ended on its own"); assert.equal(run.state.composeAborted, false); assert.equal(run.state.steps.some((step) => step.kind === "abort"), false); }); test("length continues with the calculation result in the continuation prompt", async () => { const cut = ANSWER.indexOf(`## ${REPORT_HEADING.timing}`); const head = ANSWER.slice(0, cut); const tail = ANSWER.slice(cut); const run = await runNatal([ calcCall(), { parts: pieces(head), finish: "length" }, { parts: pieces(tail), finish: "stop" }, ]); assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]); assert.equal(run.continuations.length, 1); assert.ok(run.continuations[0], "the continuation receives the calculation result"); assert.equal(run.calls, 3); const continuationPrompt = JSON.stringify(run.prompts[2]); assert.ok(continuationPrompt.includes(GOLDEN_MOON_DEGREE), "the continuation is not blind"); assert.ok(continuationPrompt.includes(REPORT_HEADING.support), "and it sees the answer so far"); assert.equal(run.completed, ANSWER); assert.equal(run.charges, 1); }); for (const reason of ["content-filter", "tool-calls", "other", "unknown"] as const) { test(`an answer step that ends on ${reason} with visible text is truncated and not charged`, async () => { const run = await runNatal([calcCall(), { parts: pieces(ANSWER.slice(0, 120)), finish: reason }]); assert.deepEqual(run.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]); assert.equal(run.charges, 0); assert.equal(run.continuations.length, 0, "only length is continued"); assert.ok(run.state.steps.some((step) => step.name === "answer-truncated" && step.status === "failed")); }); } test("the answer clock cutting the answer step mid-sentence is truncated, recorded and not charged", async () => { const run = await runNatal( [calcCall(), { parts: pieces(ANSWER, 6), finish: "stop", delayMs: 25 }], { answerMs: 250 }, ); assert.deepEqual(run.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]); assert.equal(run.charges, 0); assert.ok(run.answer.length > 0, "what was written stays"); assert.ok(ANSWER.startsWith(run.answer), "and nothing but the answer was written"); const failure = run.terminal[0]?.receipt as { steps: Array<{ kind: string; name: string; status: string }> }; assert.ok(failure.steps.some((step) => step.kind === "abort" && step.name === "compose-abort")); assert.equal(run.state.modelFinishReason, "tripwire"); assert.equal(run.state.composeAborted, true); }); test("the tool phase's deadline does not starve the answer step", async () => { // The answer step takes about 0.5s; the tool phase's clock is 150ms. The // loop is handed to the answer clock when the calculation result arrives, // so the tool phase expiring mid-answer does not cut it. const run = await runNatal( [calcCall(), { parts: pieces(ANSWER, 6), finish: "stop", delayMs: 12 }], { toolPhaseMs: 150, answerMs: 5_000 }, ); assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]); assert.equal(run.completed, ANSWER); }); test("without the hand-over the tool phase's deadline would cut the answer (control)", async () => { const clock = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 5_000 }); await new Promise((resolve) => setTimeout(resolve, 60)); assert.equal(clock.loopSignal.aborted, true, "a loop never handed over ends with the tool phase"); const handed = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 5_000 }); handed.answerSignal(); await new Promise((resolve) => setTimeout(resolve, 60)); assert.equal(handed.toolSignal.aborted, true, "tools still stop at the tool-phase deadline"); assert.equal(handed.loopSignal.aborted, false, "the answer-writing loop is not cut by it"); // The calculation settled just before the tool-phase timer fired, but the // consumer has not seen the result yet: the loop is handed over, not cut. const raced = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 60, answerReady: () => true }); await new Promise((resolve) => setTimeout(resolve, 45)); assert.equal(raced.loopSignal.aborted, false); await new Promise((resolve) => setTimeout(resolve, 80)); assert.equal(raced.loopSignal.aborted, true, "the answer clock still bounds it"); }); test("an empty answer falls back to the answer retry, which keeps the tools and hits the request cache", async () => { const run = await runNatal([ calcCall(), { parts: [], finish: "stop" }, calcCall(), { parts: pieces(ANSWER), finish: "stop" }, ]); assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]); assert.equal(run.retryHints.length, 1); assert.equal(run.retryHints[0], undefined, "no Pass 4 hint when nothing was rejected"); assert.equal(run.workflowRuns, 1); assert.ok(run.state.steps.some((step) => step.name === "answer-retry")); assert.equal(run.answer, ANSWER); // The retry re-fetch is not announced as a second calculation. assert.equal(run.events.filter((event) => event.type === "tool.started").length, 1); }); test("an answer Pass 4 rejected whole is retried with the rewrite hint", async () => { const run = await runNatal([ calcCall(), { parts: [{ text: "我保证你一定会升职。" }], finish: "stop" }, { parts: pieces(ANSWER), finish: "stop" }, ]); assert.equal(run.answer.includes("一定会升职"), false); assert.deepEqual(run.retryHints, [PASS4_RETRY_HINT]); assert.equal(run.answer, ANSWER); assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]); }); test("the consult route has no separate compose stream and wires the single-pass answer", () => { const route = readFileSync(new URL("../src/app/api/consult/route.ts", import.meta.url), "utf8"); const stream = readFileSync(new URL("../src/lib/stream-agent-response.ts", import.meta.url), "utf8"); assert.doesNotMatch(route, /composeAnswer|interpretFindings|consultationComposePrompt|toolChoice: "none"/); assert.doesNotMatch(stream, /composeAnswer|interpretFindings|drainSpoken|publishFindings/); const natal = route.slice(route.indexOf("if (!prepared.serverChart)")); assert.match(natal, /stepScopedAnswer: true,/); assert.match(natal, /onAnswerPhase: startAnswerPhase,/); assert.match(natal, /consultationContinueMessages\(baseMessages, output, evidence\)/); const window = route.slice(route.indexOf("if (shouldRunDeclaredWindowWorkflow(consultationMode)) {"), route.indexOf("if (!prepared.serverChart)")); assert.match(window, /stepScopedAnswer: true,/); assert.match(window, /windowPacketMessage \? undefined : evidence/); // 原值: assert.match(route, /natalAnswerShapeInstruction\(\)/) // 新值: 用户轮形状经 natalUserTurnShape 按「会话已有回答」切换首轮 / 追问轮,思考栏与工具上下文带 followUpTurn // 原因: TASK-consult-answer-the-question-20260927 D3/D6(BUG-1071) assert.match(route, /natalUserTurnShape\(\{ entrypoint: consultEntrypoint, history \}\)/); assert.match(route, /const followUpTurn = hasPriorAssistantAnswer\(history\);/); assert.match(route, /followUp: followUpTurn,/); assert.match(route, /followUpTurn,\n\s+serverChart: prepared\.serverChart,/); assert.doesNotMatch(route, /natalAnswerShapeInstruction\(\)/); // One run clock: tools on the tool phase, the loop handed to the answer clock. assert.match(route, /const agentAbortSignal = runClock\.toolSignal;/); assert.match(route, /abortSignal: runClock\.loopSignal,/); assert.match(route, /const answerPhaseSignal = runClock\.answerSignal;/); const maxDuration = Number(route.match(/export const maxDuration = (\d+);/)?.[1]); assert.ok(AGENT_TIMEOUT_MS + CONSULTATION_ANSWER_TIMEOUT_MS < maxDuration * 1000); assert.ok(AGENT_TIMEOUT_MS + CONSULTATION_ANSWER_TIMEOUT_MS <= 180_000, "product accepted about three minutes"); });