Files
Jyotisha/frontend/tests/consult-single-pass-answer-20260927.test.ts
T
Jesse_ChenandClaude Opus 5.5 eef0cb7486 fix(consult): write the answer in the step that saw the chart (BUG-1053)
The natal loop's step after run-jyotish-consultation saw the evidence, but
its text was drained and a second, blind compose stream (history + question
only, empty findings) wrote the user-visible answer. Remove compose,
interpret and the drain; keep the loop's own final-step text.

- stepScopedAnswer: per-step holding; text of a step that calls a tool is
  dropped, so narration around tool calls never reaches the answer
- writing shape (opener + four headings) moves into the user turn
- length continuation receives the calculation result; Pass 4 whole-answer
  reject retries through retryForAnswer with the rewrite hint
- createConsultationRunClock: tools keep the 110s tool phase; the loop is
  handed to the 70s answer clock when the calculation result arrives
- settlement judges the step that wrote the answer (BUG-1051 kept)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
2026-09-27 01:01:38 +08:00

441 lines
21 KiB
TypeScript

// BUG-1053: the user-visible consultation answer was written without the chart.
//
// The natal loop's step after `run-jyotish-consultation` saw the evidence, but
// its text was drained and a second `agent.stream` (compose) wrote the answer
// from history + the question only. These tests drive the real personal Agent
// (`getJyotishAgent`, real skill binding, real calculation tool, real
// prepareStep) over a prompt-recording fake model, with the calculation fed by
// the golden engine capture in fixtures/ (AGENTS §7.4). What each model call
// was shown is the assertion: the call that writes the answer must have the
// tool result in its prompt, and there must be no second, blind call.
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
import test from "node:test";
import { getJyotishAgent } from "../src/mastra/index.ts";
import { createNdjsonParser } from "../src/lib/consultation-agent-events.ts";
import {
consultationContinueMessages,
natalAnswerShapeInstruction,
REPORT_HEADING,
} from "../src/lib/consultation-thinking-plan.ts";
import { streamAgentResponse } from "../src/lib/stream-agent-response.ts";
import { PASS4_RETRY_HINT } from "../src/lib/timing-output-guard.ts";
import {
AGENT_MAX_STEPS,
AGENT_TIMEOUT_MS,
CONSULTATION_ANSWER_TIMEOUT_MS,
consultationNatalPrepareStep,
consultationStepBudgetReceipt,
createConsultationAgentContext,
createConsultationRunClock,
createConsultationRuntimeState,
publicConsultationRuntimeSteps,
} from "../src/mastra/consultation-tools.ts";
const golden = JSON.parse(readFileSync(
new URL("./fixtures/consultation-workflow-report-blocked-repairs-golden.json", import.meta.url),
"utf8",
)) as { smoke_birth: Record<string, number | string>; themes: Record<string, Record<string, unknown>> };
const careerWorkflow = golden.themes.career!;
// A fact only the calculation carries: the golden chart's Moon longitude.
const GOLDEN_MOON_DEGREE = "3.2259";
// Fictional capture identity from the fixture itself, not a real person.
const serverChart = {
name: "测试",
toolInput: {
year: 1993, month: 6, day: 15, hour: 10, minute: 30, city: "Handan", lat: 36.42, lon: 114.21, tz: 8,
ayanamsa: "raman" as const, declared_accuracy: "15min" as const, time_source: "family_vague",
},
truth: {
birthDate: "1993-06-15", reportedBirthTime: "10:30", activeBirthTime: null,
selectedTimeKind: "reported" as const, birthTimeSource: "reported", birthTimeStatus: "reported",
placeLabel: "Handan", placeCodes: { countryCode: "CN", provinceCode: null, cityCode: null, districtCode: null },
placeId: null, placeType: "city", placeProvider: "profile", latitude: 36.42, longitude: 114.21,
timezoneId: "Asia/Shanghai", timezoneSource: "profile", timezoneOffset: 8,
},
};
const PRE_TOOL = "我先排一下盘,稍等。";
const BETWEEN_TOOLS = "我再核对一次计算结果。";
const OPENER = "你这盘外面看着稳,底下其实一直在换跑道:表面求安定,底下要自己说了算。";
const ANSWER = [
OPENER,
"",
`## ${REPORT_HEADING.question}`,
"能换,但先换做法,再换岗位。",
"",
`## ${REPORT_HEADING.support}`,
"月亮落在白羊座九宫,想法来得快。",
"",
`## ${REPORT_HEADING.timing}`,
"这半年先试,不急着定。",
"",
`## ${REPORT_HEADING.action}`,
"- **这周**:约一位同行聊一次。",
"",
].join("\n");
type Part = { text?: string; tool?: { name: string; input: Record<string, unknown> } };
type Turn = { parts: Part[]; finish: string; delayMs?: number };
/** A LanguageModelV2 that plays `turns` in order and records every prompt. */
function scriptedModel(turns: Turn[]) {
const prompts: unknown[][] = [];
let call = 0;
const model = {
specificationVersion: "v2",
provider: "fake",
modelId: "fake-single-pass",
supportedUrls: {},
async doGenerate() {
throw new Error("not used");
},
async doStream(options: { prompt: unknown[]; abortSignal?: AbortSignal }) {
prompts.push(options.prompt);
const turn = turns[call] ?? { parts: [], finish: "stop" };
call += 1;
const signal = options.abortSignal;
const stream = new ReadableStream({
async start(controller) {
controller.enqueue({ type: "stream-start", warnings: [] });
let textOpen = false;
for (const [index, part] of turn.parts.entries()) {
if (turn.delayMs) {
try {
await new Promise<void>((resolve, reject) => {
if (signal?.aborted) return reject(signal.reason);
const timer = setTimeout(resolve, turn.delayMs);
signal?.addEventListener("abort", () => {
clearTimeout(timer);
reject(signal.reason);
}, { once: true });
});
} catch (error) {
controller.error(error);
return;
}
}
if (part.text !== undefined) {
if (!textOpen) {
controller.enqueue({ type: "text-start", id: `t${call}` });
textOpen = true;
}
controller.enqueue({ type: "text-delta", id: `t${call}`, delta: part.text });
}
if (part.tool) {
if (textOpen) {
controller.enqueue({ type: "text-end", id: `t${call}` });
textOpen = false;
}
controller.enqueue({
type: "tool-call",
toolCallId: `call-${call}-${index}`,
toolName: part.tool.name,
input: JSON.stringify(part.tool.input),
});
}
}
if (textOpen) controller.enqueue({ type: "text-end", id: `t${call}` });
controller.enqueue({
type: "finish",
finishReason: turn.finish,
usage: { inputTokens: 10, outputTokens: 10, totalTokens: 20 },
});
controller.close();
},
});
return { stream };
},
};
return { model, prompts, calls: () => call };
}
const calcCall = (text?: string): Turn => ({
parts: [
...(text ? [{ text }] : []),
{ tool: { name: "run-jyotish-consultation", input: { question: "我的事业能换方向吗", domains: ["career"] } } },
],
finish: "tool-calls",
});
function pieces(text: string, size = 12) {
return (text.match(new RegExp(`[\\s\\S]{1,${size}}`, "g")) ?? []).map((value) => ({ text: value }));
}
let seq = 0;
/**
* The natal route's wiring, minus HTTP, auth and billing: the same Agent,
* prepareStep, run clock, stream options, step-scoped answer, answer-phase
* hand-over, continuation builder and answer retry as `runAgenticConsultation`.
*/
async function runNatal(turns: Turn[], options: { toolPhaseMs?: number; answerMs?: number } = {}) {
seq += 1;
const { model, prompts, calls } = scriptedModel(turns);
const state = createConsultationRuntimeState({ plannedSteps: AGENT_MAX_STEPS });
const clock = createConsultationRunClock({
toolPhaseMs: options.toolPhaseMs ?? AGENT_TIMEOUT_MS,
answerMs: options.answerMs ?? CONSULTATION_ANSWER_TIMEOUT_MS,
answerReady: () => state.consultationToolCompleted,
});
let workflowRuns = 0;
const agentContext = createConsultationAgentContext({
userId: "u", sessionId: "s", requestId: `single-pass-${seq}`, consultationMode: "verified_chart",
theme: "career", serverChart: serverChart as never, abortSignal: clock.toolSignal, state,
runWorkflow: async () => {
workflowRuns += 1;
return structuredClone(careerWorkflow) as never;
},
});
const agent = getJyotishAgent({ id: `fake-${seq}`, model } as never, agentContext);
const baseMessages = [{ role: "user" as const, content: `${natalAnswerShapeInstruction()}\n问题:我的事业能换方向吗` }];
const streamOptions = { runId: `run-${seq}`, maxSteps: AGENT_MAX_STEPS, abortSignal: clock.loopSignal };
const natalStreamOptions = { ...streamOptions, prepareStep: consultationNatalPrepareStep };
const continuations: unknown[] = [];
const retryHints: Array<string | undefined> = [];
let completed: string | null = null;
let charges = 0;
let errored: unknown = null;
const result = await agent.stream(baseMessages as never, natalStreamOptions as never);
const response = streamAgentResponse({
runId: `run-${seq}`,
requestId: `req-${seq}`,
state,
stream: result.fullStream as ReadableStream<unknown>,
requireTool: true,
stepScopedAnswer: true,
onAnswerPhase: () => { clock.answerSignal(); },
pass4Mode: "verified_chart",
retryForAnswer: async (retryHint) => {
retryHints.push(retryHint);
const retried = await agent.stream([
...baseMessages,
{ role: "user" as const, content: `服务器计算已经完成,但上一轮没有输出任何回答文本。请重新取回本次计算结果,然后直接给出回答。${retryHint ? `\n${retryHint}` : ""}` },
] as never, { ...natalStreamOptions, abortSignal: clock.answerSignal() } as never);
return retried.fullStream as ReadableStream<unknown>;
},
continueAfterLength: async (output, evidence) => {
continuations.push(evidence);
const continued = await agent.stream(
consultationContinueMessages(baseMessages, output, evidence) as never,
{ ...streamOptions, abortSignal: clock.answerSignal() } as never,
);
return continued.fullStream as ReadableStream<unknown>;
},
toolStatus: () => "ready",
receipt: () => ({
runId: `run-${seq}`,
runtime: "mastra-agentic",
skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: publicConsultationRuntimeSteps(state),
stepBudget: consultationStepBudgetReceipt(state),
workflow: state.workflowReceipt ?? { route: "pending", status: "blocked", preciseTiming: "blocked", missingLayers: [] },
techniqueTruth: "unknown",
}) as never,
onComplete: (output) => { completed = output; charges += 1; },
onError: (error) => { errored = error; },
});
const events: Array<{ type: string; code?: string; text?: string; receipt?: unknown }> = [];
const parser = createNdjsonParser((event) => events.push(event as never));
parser.finish(await response.text());
const answer = events.filter((event) => event.type === "answer.delta").map((event) => event.text ?? "").join("");
const terminal = events.filter((event) => event.type === "run.completed" || event.type === "run.failed");
return {
state, events, answer, terminal, prompts, calls: calls(), continuations, retryHints, workflowRuns,
completed: completed as string | null, charges, errored,
};
}
/** Index of the first prompt that already contains the calculation result. */
function firstPromptWithToolResult(prompts: unknown[][]) {
return prompts.findIndex((prompt) => prompt.some((message) => {
const value = message as { role?: string; content?: unknown };
return value.role === "tool" && JSON.stringify(value.content).includes("run-jyotish-consultation");
}));
}
test("the model call that writes the answer has the calculation result in its prompt, and no blind call follows", async () => {
const run = await runNatal([calcCall(PRE_TOOL), { parts: pieces(ANSWER), finish: "stop" }]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.workflowRuns, 1);
// Exactly two model calls: the one that called the tool and the one after it.
// The pre-fix route opened a third (compose) whose prompt had no tool result.
assert.equal(run.calls, 2);
const withResult = firstPromptWithToolResult(run.prompts);
assert.equal(withResult, 1, "the call after the tool result sees it");
assert.equal(run.prompts.length - withResult, 1, "exactly one model call after the tool result");
assert.ok(JSON.stringify(run.prompts[1]).includes(GOLDEN_MOON_DEGREE), "the golden chart fact reaches the writer");
// What that call wrote is what the user got, with the four headings intact.
assert.equal(run.answer, ANSWER);
assert.equal(run.completed, ANSWER);
for (const heading of Object.values(REPORT_HEADING)) assert.match(run.answer, new RegExp(`^## ${heading}$`, "m"));
assert.ok(run.answer.startsWith(OPENER), "the opener has no heading");
// Pre-tool narration is not the answer.
assert.equal(run.answer.includes(PRE_TOOL), false);
});
test("the writing instructions travel with the user turn the loop sees", async () => {
const run = await runNatal([calcCall(), { parts: pieces(ANSWER), finish: "stop" }]);
const writerPrompt = JSON.stringify(run.prompts[1]);
for (const heading of Object.values(REPORT_HEADING)) assert.ok(writerPrompt.includes(`## ${heading}`));
assert.match(writerPrompt, /开场不要标题/);
// The removed compose prompt claimed a finished calculation it never saw.
assert.doesNotMatch(writerPrompt, /服务器计算已经完成。不要再调用排盘工具/);
});
test("narration in a step that calls a tool never reaches the answer, even after the calculation", async () => {
const run = await runNatal([
calcCall(PRE_TOOL),
// After the calculation the model narrates and calls the tool again (a
// request-cache hit), then writes the answer.
calcCall(BETWEEN_TOOLS),
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.workflowRuns, 1, "the second call is served from the request cache");
assert.equal(run.answer, ANSWER);
assert.equal(run.answer.includes(PRE_TOOL), false);
assert.equal(run.answer.includes(BETWEEN_TOOLS), false);
assert.equal(run.charges, 1);
});
test("a normal stop completes and charges exactly once", async () => {
const run = await runNatal([calcCall(), { parts: pieces(ANSWER), finish: "stop" }]);
assert.equal(run.charges, 1);
assert.equal(run.errored, null);
assert.equal(run.state.composeFinishReason, "stop", "the answer-writing step ended on its own");
assert.equal(run.state.composeAborted, false);
assert.equal(run.state.steps.some((step) => step.kind === "abort"), false);
});
test("length continues with the calculation result in the continuation prompt", async () => {
const cut = ANSWER.indexOf(`## ${REPORT_HEADING.timing}`);
const head = ANSWER.slice(0, cut);
const tail = ANSWER.slice(cut);
const run = await runNatal([
calcCall(),
{ parts: pieces(head), finish: "length" },
{ parts: pieces(tail), finish: "stop" },
]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.continuations.length, 1);
assert.ok(run.continuations[0], "the continuation receives the calculation result");
assert.equal(run.calls, 3);
const continuationPrompt = JSON.stringify(run.prompts[2]);
assert.ok(continuationPrompt.includes(GOLDEN_MOON_DEGREE), "the continuation is not blind");
assert.ok(continuationPrompt.includes(REPORT_HEADING.support), "and it sees the answer so far");
assert.equal(run.completed, ANSWER);
assert.equal(run.charges, 1);
});
for (const reason of ["content-filter", "tool-calls", "other", "unknown"] as const) {
test(`an answer step that ends on ${reason} with visible text is truncated and not charged`, async () => {
const run = await runNatal([calcCall(), { parts: pieces(ANSWER.slice(0, 120)), finish: reason }]);
assert.deepEqual(run.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
assert.equal(run.charges, 0);
assert.equal(run.continuations.length, 0, "only length is continued");
assert.ok(run.state.steps.some((step) => step.name === "answer-truncated" && step.status === "failed"));
});
}
test("the answer clock cutting the answer step mid-sentence is truncated, recorded and not charged", async () => {
const run = await runNatal(
[calcCall(), { parts: pieces(ANSWER, 6), finish: "stop", delayMs: 25 }],
{ answerMs: 250 },
);
assert.deepEqual(run.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
assert.equal(run.charges, 0);
assert.ok(run.answer.length > 0, "what was written stays");
assert.ok(ANSWER.startsWith(run.answer), "and nothing but the answer was written");
const failure = run.terminal[0]?.receipt as { steps: Array<{ kind: string; name: string; status: string }> };
assert.ok(failure.steps.some((step) => step.kind === "abort" && step.name === "compose-abort"));
assert.equal(run.state.modelFinishReason, "tripwire");
assert.equal(run.state.composeAborted, true);
});
test("the tool phase's deadline does not starve the answer step", async () => {
// The answer step takes about 0.5s; the tool phase's clock is 150ms. The
// loop is handed to the answer clock when the calculation result arrives,
// so the tool phase expiring mid-answer does not cut it.
const run = await runNatal(
[calcCall(), { parts: pieces(ANSWER, 6), finish: "stop", delayMs: 12 }],
{ toolPhaseMs: 150, answerMs: 5_000 },
);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.completed, ANSWER);
});
test("without the hand-over the tool phase's deadline would cut the answer (control)", async () => {
const clock = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 5_000 });
await new Promise((resolve) => setTimeout(resolve, 60));
assert.equal(clock.loopSignal.aborted, true, "a loop never handed over ends with the tool phase");
const handed = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 5_000 });
handed.answerSignal();
await new Promise((resolve) => setTimeout(resolve, 60));
assert.equal(handed.toolSignal.aborted, true, "tools still stop at the tool-phase deadline");
assert.equal(handed.loopSignal.aborted, false, "the answer-writing loop is not cut by it");
// The calculation settled just before the tool-phase timer fired, but the
// consumer has not seen the result yet: the loop is handed over, not cut.
const raced = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 60, answerReady: () => true });
await new Promise((resolve) => setTimeout(resolve, 45));
assert.equal(raced.loopSignal.aborted, false);
await new Promise((resolve) => setTimeout(resolve, 80));
assert.equal(raced.loopSignal.aborted, true, "the answer clock still bounds it");
});
test("an empty answer falls back to the answer retry, which keeps the tools and hits the request cache", async () => {
const run = await runNatal([
calcCall(),
{ parts: [], finish: "stop" },
calcCall(),
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.retryHints.length, 1);
assert.equal(run.retryHints[0], undefined, "no Pass 4 hint when nothing was rejected");
assert.equal(run.workflowRuns, 1);
assert.ok(run.state.steps.some((step) => step.name === "answer-retry"));
assert.equal(run.answer, ANSWER);
// The retry re-fetch is not announced as a second calculation.
assert.equal(run.events.filter((event) => event.type === "tool.started").length, 1);
});
test("an answer Pass 4 rejected whole is retried with the rewrite hint", async () => {
const run = await runNatal([
calcCall(),
{ parts: [{ text: "我保证你一定会升职。" }], finish: "stop" },
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.equal(run.answer.includes("一定会升职"), false);
assert.deepEqual(run.retryHints, [PASS4_RETRY_HINT]);
assert.equal(run.answer, ANSWER);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
});
test("the consult route has no separate compose stream and wires the single-pass answer", () => {
const route = readFileSync(new URL("../src/app/api/consult/route.ts", import.meta.url), "utf8");
const stream = readFileSync(new URL("../src/lib/stream-agent-response.ts", import.meta.url), "utf8");
assert.doesNotMatch(route, /composeAnswer|interpretFindings|consultationComposePrompt|toolChoice: "none"/);
assert.doesNotMatch(stream, /composeAnswer|interpretFindings|drainSpoken|publishFindings/);
const natal = route.slice(route.indexOf("if (!prepared.serverChart)"));
assert.match(natal, /stepScopedAnswer: true,/);
assert.match(natal, /onAnswerPhase: startAnswerPhase,/);
assert.match(natal, /consultationContinueMessages\(baseMessages, output, evidence\)/);
const window = route.slice(route.indexOf("if (shouldRunDeclaredWindowWorkflow(consultationMode)) {"), route.indexOf("if (!prepared.serverChart)"));
assert.match(window, /stepScopedAnswer: true,/);
assert.match(window, /windowPacketMessage \? undefined : evidence/);
assert.match(route, /natalAnswerShapeInstruction\(\)/);
// One run clock: tools on the tool phase, the loop handed to the answer clock.
assert.match(route, /const agentAbortSignal = runClock\.toolSignal;/);
assert.match(route, /abortSignal: runClock\.loopSignal,/);
assert.match(route, /const answerPhaseSignal = runClock\.answerSignal;/);
const maxDuration = Number(route.match(/export const maxDuration = (\d+);/)?.[1]);
assert.ok(AGENT_TIMEOUT_MS + CONSULTATION_ANSWER_TIMEOUT_MS < maxDuration * 1000);
assert.ok(AGENT_TIMEOUT_MS + CONSULTATION_ANSWER_TIMEOUT_MS <= 180_000, "product accepted about three minutes");
});