Files
Jyotisha/frontend/tests/consult-single-pass-answer-20260927.test.ts
T
Jesse_ChenandClaude Fable 5.1 9d01757f27
Independent Staging Quality Gate / validate (push) Successful in 14m58s
Independent Staging Quality Gate / publish (push) Successful in 3m40s
fix(consult): a stored smalltalk reply is not a prior answer; summary counts as one (acceptance fix for BUG-1071)
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0199rbQDTsUbCVw84wc8BTFe
2026-09-27 23:43:57 +08:00

451 lines
22 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// BUG-1053: the user-visible consultation answer was written without the chart.
//
// The natal loop's step after `run-jyotish-consultation` saw the evidence, but
// its text was drained and a second `agent.stream` (compose) wrote the answer
// from history + the question only. These tests drive the real personal Agent
// (`getJyotishAgent`, real skill binding, real calculation tool, real
// prepareStep) over a prompt-recording fake model, with the calculation fed by
// the golden engine capture in fixtures/ (AGENTS §7.4). What each model call
// was shown is the assertion: the call that writes the answer must have the
// tool result in its prompt, and there must be no second, blind call.
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
import test from "node:test";
import { getJyotishAgent } from "../src/mastra/index.ts";
import { createNdjsonParser } from "../src/lib/consultation-agent-events.ts";
import {
consultationContinueMessages,
natalAnswerShapeInstruction,
REPORT_HEADING,
} from "../src/lib/consultation-thinking-plan.ts";
import { streamAgentResponse } from "../src/lib/stream-agent-response.ts";
import { PASS4_RETRY_HINT } from "../src/lib/timing-output-guard.ts";
import {
AGENT_MAX_STEPS,
AGENT_TIMEOUT_MS,
CONSULTATION_ANSWER_TIMEOUT_MS,
consultationNatalPrepareStep,
consultationStepBudgetReceipt,
createConsultationAgentContext,
createConsultationRunClock,
createConsultationRuntimeState,
publicConsultationRuntimeSteps,
} from "../src/mastra/consultation-tools.ts";
const golden = JSON.parse(readFileSync(
new URL("./fixtures/consultation-workflow-report-blocked-repairs-golden.json", import.meta.url),
"utf8",
)) as { smoke_birth: Record<string, number | string>; themes: Record<string, Record<string, unknown>> };
const careerWorkflow = golden.themes.career!;
// A fact only the calculation carries: the golden chart's Moon longitude.
const GOLDEN_MOON_DEGREE = "3.2259";
// Fictional capture identity from the fixture itself, not a real person.
const serverChart = {
name: "测试",
toolInput: {
year: 1993, month: 6, day: 15, hour: 10, minute: 30, city: "Handan", lat: 36.42, lon: 114.21, tz: 8,
ayanamsa: "raman" as const, declared_accuracy: "15min" as const, time_source: "family_vague",
},
truth: {
birthDate: "1993-06-15", reportedBirthTime: "10:30", activeBirthTime: null,
selectedTimeKind: "reported" as const, birthTimeSource: "reported", birthTimeStatus: "reported",
placeLabel: "Handan", placeCodes: { countryCode: "CN", provinceCode: null, cityCode: null, districtCode: null },
placeId: null, placeType: "city", placeProvider: "profile", latitude: 36.42, longitude: 114.21,
timezoneId: "Asia/Shanghai", timezoneSource: "profile", timezoneOffset: 8,
},
};
const PRE_TOOL = "我先排一下盘,稍等。";
const BETWEEN_TOOLS = "我再核对一次计算结果。";
const OPENER = "你这盘外面看着稳,底下其实一直在换跑道:表面求安定,底下要自己说了算。";
// 原值: 夹具里有 `## ${REPORT_HEADING.question}` 一节
// 新值: 该节删除,夹具从口语开场直接到 `## ${REPORT_HEADING.support}`
// 原因: TASK-consult-answer-the-question-20260927 D8 / BUG-1073:开场就是回答,正文不再从「先回答你的问题」重来
const ANSWER = [
OPENER,
"",
`## ${REPORT_HEADING.support}`,
"月亮落在白羊座九宫,想法来得快。",
"",
`## ${REPORT_HEADING.timing}`,
"这半年先试,不急着定。",
"",
`## ${REPORT_HEADING.action}`,
"- **这周**:约一位同行聊一次。",
"",
].join("\n");
type Part = { text?: string; tool?: { name: string; input: Record<string, unknown> } };
type Turn = { parts: Part[]; finish: string; delayMs?: number };
/** A LanguageModelV2 that plays `turns` in order and records every prompt. */
function scriptedModel(turns: Turn[]) {
const prompts: unknown[][] = [];
let call = 0;
const model = {
specificationVersion: "v2",
provider: "fake",
modelId: "fake-single-pass",
supportedUrls: {},
async doGenerate() {
throw new Error("not used");
},
async doStream(options: { prompt: unknown[]; abortSignal?: AbortSignal }) {
prompts.push(options.prompt);
const turn = turns[call] ?? { parts: [], finish: "stop" };
call += 1;
const signal = options.abortSignal;
const stream = new ReadableStream({
async start(controller) {
controller.enqueue({ type: "stream-start", warnings: [] });
let textOpen = false;
for (const [index, part] of turn.parts.entries()) {
if (turn.delayMs) {
try {
await new Promise<void>((resolve, reject) => {
if (signal?.aborted) return reject(signal.reason);
const timer = setTimeout(resolve, turn.delayMs);
signal?.addEventListener("abort", () => {
clearTimeout(timer);
reject(signal.reason);
}, { once: true });
});
} catch (error) {
controller.error(error);
return;
}
}
if (part.text !== undefined) {
if (!textOpen) {
controller.enqueue({ type: "text-start", id: `t${call}` });
textOpen = true;
}
controller.enqueue({ type: "text-delta", id: `t${call}`, delta: part.text });
}
if (part.tool) {
if (textOpen) {
controller.enqueue({ type: "text-end", id: `t${call}` });
textOpen = false;
}
controller.enqueue({
type: "tool-call",
toolCallId: `call-${call}-${index}`,
toolName: part.tool.name,
input: JSON.stringify(part.tool.input),
});
}
}
if (textOpen) controller.enqueue({ type: "text-end", id: `t${call}` });
controller.enqueue({
type: "finish",
finishReason: turn.finish,
usage: { inputTokens: 10, outputTokens: 10, totalTokens: 20 },
});
controller.close();
},
});
return { stream };
},
};
return { model, prompts, calls: () => call };
}
const calcCall = (text?: string): Turn => ({
parts: [
...(text ? [{ text }] : []),
{ tool: { name: "run-jyotish-consultation", input: { question: "我的事业能换方向吗", domains: ["career"] } } },
],
finish: "tool-calls",
});
function pieces(text: string, size = 12) {
return (text.match(new RegExp(`[\\s\\S]{1,${size}}`, "g")) ?? []).map((value) => ({ text: value }));
}
let seq = 0;
/**
* The natal route's wiring, minus HTTP, auth and billing: the same Agent,
* prepareStep, run clock, stream options, step-scoped answer, answer-phase
* hand-over, continuation builder and answer retry as `runAgenticConsultation`.
*/
async function runNatal(turns: Turn[], options: { toolPhaseMs?: number; answerMs?: number } = {}) {
seq += 1;
const { model, prompts, calls } = scriptedModel(turns);
const state = createConsultationRuntimeState({ plannedSteps: AGENT_MAX_STEPS });
const clock = createConsultationRunClock({
toolPhaseMs: options.toolPhaseMs ?? AGENT_TIMEOUT_MS,
answerMs: options.answerMs ?? CONSULTATION_ANSWER_TIMEOUT_MS,
answerReady: () => state.consultationToolCompleted,
});
let workflowRuns = 0;
const agentContext = createConsultationAgentContext({
userId: "u", sessionId: "s", requestId: `single-pass-${seq}`, consultationMode: "verified_chart",
theme: "career", serverChart: serverChart as never, abortSignal: clock.toolSignal, state,
runWorkflow: async () => {
workflowRuns += 1;
return structuredClone(careerWorkflow) as never;
},
});
const agent = getJyotishAgent({ id: `fake-${seq}`, model } as never, agentContext);
const baseMessages = [{ role: "user" as const, content: `${natalAnswerShapeInstruction()}\n问题:我的事业能换方向吗` }];
const streamOptions = { runId: `run-${seq}`, maxSteps: AGENT_MAX_STEPS, abortSignal: clock.loopSignal };
const natalStreamOptions = { ...streamOptions, prepareStep: consultationNatalPrepareStep };
const continuations: unknown[] = [];
const retryHints: Array<string | undefined> = [];
let completed: string | null = null;
let charges = 0;
let errored: unknown = null;
const result = await agent.stream(baseMessages as never, natalStreamOptions as never);
const response = streamAgentResponse({
runId: `run-${seq}`,
requestId: `req-${seq}`,
state,
stream: result.fullStream as ReadableStream<unknown>,
requireTool: true,
stepScopedAnswer: true,
onAnswerPhase: () => { clock.answerSignal(); },
pass4Mode: "verified_chart",
retryForAnswer: async (retryHint) => {
retryHints.push(retryHint);
const retried = await agent.stream([
...baseMessages,
{ role: "user" as const, content: `服务器计算已经完成,但上一轮没有输出任何回答文本。请重新取回本次计算结果,然后直接给出回答。${retryHint ? `\n${retryHint}` : ""}` },
] as never, { ...natalStreamOptions, abortSignal: clock.answerSignal() } as never);
return retried.fullStream as ReadableStream<unknown>;
},
continueAfterLength: async (output, evidence) => {
continuations.push(evidence);
const continued = await agent.stream(
consultationContinueMessages(baseMessages, output, evidence) as never,
{ ...streamOptions, abortSignal: clock.answerSignal() } as never,
);
return continued.fullStream as ReadableStream<unknown>;
},
toolStatus: () => "ready",
receipt: () => ({
runId: `run-${seq}`,
runtime: "mastra-agentic",
skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: publicConsultationRuntimeSteps(state),
stepBudget: consultationStepBudgetReceipt(state),
workflow: state.workflowReceipt ?? { route: "pending", status: "blocked", preciseTiming: "blocked", missingLayers: [] },
techniqueTruth: "unknown",
}) as never,
onComplete: (output) => { completed = output; charges += 1; },
onError: (error) => { errored = error; },
});
const events: Array<{ type: string; code?: string; text?: string; receipt?: unknown }> = [];
const parser = createNdjsonParser((event) => events.push(event as never));
parser.finish(await response.text());
const answer = events.filter((event) => event.type === "answer.delta").map((event) => event.text ?? "").join("");
const terminal = events.filter((event) => event.type === "run.completed" || event.type === "run.failed");
return {
state, events, answer, terminal, prompts, calls: calls(), continuations, retryHints, workflowRuns,
completed: completed as string | null, charges, errored,
};
}
/** Index of the first prompt that already contains the calculation result. */
function firstPromptWithToolResult(prompts: unknown[][]) {
return prompts.findIndex((prompt) => prompt.some((message) => {
const value = message as { role?: string; content?: unknown };
return value.role === "tool" && JSON.stringify(value.content).includes("run-jyotish-consultation");
}));
}
test("the model call that writes the answer has the calculation result in its prompt, and no blind call follows", async () => {
const run = await runNatal([calcCall(PRE_TOOL), { parts: pieces(ANSWER), finish: "stop" }]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.workflowRuns, 1);
// Exactly two model calls: the one that called the tool and the one after it.
// The pre-fix route opened a third (compose) whose prompt had no tool result.
assert.equal(run.calls, 2);
const withResult = firstPromptWithToolResult(run.prompts);
assert.equal(withResult, 1, "the call after the tool result sees it");
assert.equal(run.prompts.length - withResult, 1, "exactly one model call after the tool result");
assert.ok(JSON.stringify(run.prompts[1]).includes(GOLDEN_MOON_DEGREE), "the golden chart fact reaches the writer");
// What that call wrote is what the user got, with the four headings intact.
assert.equal(run.answer, ANSWER);
assert.equal(run.completed, ANSWER);
for (const heading of Object.values(REPORT_HEADING)) assert.match(run.answer, new RegExp(`^## ${heading}$`, "m"));
assert.ok(run.answer.startsWith(OPENER), "the opener has no heading");
// Pre-tool narration is not the answer.
assert.equal(run.answer.includes(PRE_TOOL), false);
});
test("the writing instructions travel with the user turn the loop sees", async () => {
const run = await runNatal([calcCall(), { parts: pieces(ANSWER), finish: "stop" }]);
const writerPrompt = JSON.stringify(run.prompts[1]);
for (const heading of Object.values(REPORT_HEADING)) assert.ok(writerPrompt.includes(`## ${heading}`));
assert.match(writerPrompt, /开场不要标题/);
// The removed compose prompt claimed a finished calculation it never saw.
assert.doesNotMatch(writerPrompt, /服务器计算已经完成。不要再调用排盘工具/);
});
test("narration in a step that calls a tool never reaches the answer, even after the calculation", async () => {
const run = await runNatal([
calcCall(PRE_TOOL),
// After the calculation the model narrates and calls the tool again (a
// request-cache hit), then writes the answer.
calcCall(BETWEEN_TOOLS),
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.workflowRuns, 1, "the second call is served from the request cache");
assert.equal(run.answer, ANSWER);
assert.equal(run.answer.includes(PRE_TOOL), false);
assert.equal(run.answer.includes(BETWEEN_TOOLS), false);
assert.equal(run.charges, 1);
});
test("a normal stop completes and charges exactly once", async () => {
const run = await runNatal([calcCall(), { parts: pieces(ANSWER), finish: "stop" }]);
assert.equal(run.charges, 1);
assert.equal(run.errored, null);
assert.equal(run.state.composeFinishReason, "stop", "the answer-writing step ended on its own");
assert.equal(run.state.composeAborted, false);
assert.equal(run.state.steps.some((step) => step.kind === "abort"), false);
});
test("length continues with the calculation result in the continuation prompt", async () => {
const cut = ANSWER.indexOf(`## ${REPORT_HEADING.timing}`);
const head = ANSWER.slice(0, cut);
const tail = ANSWER.slice(cut);
const run = await runNatal([
calcCall(),
{ parts: pieces(head), finish: "length" },
{ parts: pieces(tail), finish: "stop" },
]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.continuations.length, 1);
assert.ok(run.continuations[0], "the continuation receives the calculation result");
assert.equal(run.calls, 3);
const continuationPrompt = JSON.stringify(run.prompts[2]);
assert.ok(continuationPrompt.includes(GOLDEN_MOON_DEGREE), "the continuation is not blind");
assert.ok(continuationPrompt.includes(REPORT_HEADING.support), "and it sees the answer so far");
assert.equal(run.completed, ANSWER);
assert.equal(run.charges, 1);
});
for (const reason of ["content-filter", "tool-calls", "other", "unknown"] as const) {
test(`an answer step that ends on ${reason} with visible text is truncated and not charged`, async () => {
const run = await runNatal([calcCall(), { parts: pieces(ANSWER.slice(0, 120)), finish: reason }]);
assert.deepEqual(run.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
assert.equal(run.charges, 0);
assert.equal(run.continuations.length, 0, "only length is continued");
assert.ok(run.state.steps.some((step) => step.name === "answer-truncated" && step.status === "failed"));
});
}
test("the answer clock cutting the answer step mid-sentence is truncated, recorded and not charged", async () => {
const run = await runNatal(
[calcCall(), { parts: pieces(ANSWER, 6), finish: "stop", delayMs: 25 }],
{ answerMs: 250 },
);
assert.deepEqual(run.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
assert.equal(run.charges, 0);
assert.ok(run.answer.length > 0, "what was written stays");
assert.ok(ANSWER.startsWith(run.answer), "and nothing but the answer was written");
const failure = run.terminal[0]?.receipt as { steps: Array<{ kind: string; name: string; status: string }> };
assert.ok(failure.steps.some((step) => step.kind === "abort" && step.name === "compose-abort"));
assert.equal(run.state.modelFinishReason, "tripwire");
assert.equal(run.state.composeAborted, true);
});
test("the tool phase's deadline does not starve the answer step", async () => {
// The answer step takes about 0.5s; the tool phase's clock is 150ms. The
// loop is handed to the answer clock when the calculation result arrives,
// so the tool phase expiring mid-answer does not cut it.
const run = await runNatal(
[calcCall(), { parts: pieces(ANSWER, 6), finish: "stop", delayMs: 12 }],
{ toolPhaseMs: 150, answerMs: 5_000 },
);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.completed, ANSWER);
});
test("without the hand-over the tool phase's deadline would cut the answer (control)", async () => {
const clock = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 5_000 });
await new Promise((resolve) => setTimeout(resolve, 60));
assert.equal(clock.loopSignal.aborted, true, "a loop never handed over ends with the tool phase");
const handed = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 5_000 });
handed.answerSignal();
await new Promise((resolve) => setTimeout(resolve, 60));
assert.equal(handed.toolSignal.aborted, true, "tools still stop at the tool-phase deadline");
assert.equal(handed.loopSignal.aborted, false, "the answer-writing loop is not cut by it");
// The calculation settled just before the tool-phase timer fired, but the
// consumer has not seen the result yet: the loop is handed over, not cut.
const raced = createConsultationRunClock({ toolPhaseMs: 30, answerMs: 60, answerReady: () => true });
await new Promise((resolve) => setTimeout(resolve, 45));
assert.equal(raced.loopSignal.aborted, false);
await new Promise((resolve) => setTimeout(resolve, 80));
assert.equal(raced.loopSignal.aborted, true, "the answer clock still bounds it");
});
test("an empty answer falls back to the answer retry, which keeps the tools and hits the request cache", async () => {
const run = await runNatal([
calcCall(),
{ parts: [], finish: "stop" },
calcCall(),
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(run.retryHints.length, 1);
assert.equal(run.retryHints[0], undefined, "no Pass 4 hint when nothing was rejected");
assert.equal(run.workflowRuns, 1);
assert.ok(run.state.steps.some((step) => step.name === "answer-retry"));
assert.equal(run.answer, ANSWER);
// The retry re-fetch is not announced as a second calculation.
assert.equal(run.events.filter((event) => event.type === "tool.started").length, 1);
});
test("an answer Pass 4 rejected whole is retried with the rewrite hint", async () => {
const run = await runNatal([
calcCall(),
{ parts: [{ text: "我保证你一定会升职。" }], finish: "stop" },
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.equal(run.answer.includes("一定会升职"), false);
assert.deepEqual(run.retryHints, [PASS4_RETRY_HINT]);
assert.equal(run.answer, ANSWER);
assert.deepEqual(run.terminal.map((event) => event.type), ["run.completed"]);
});
test("the consult route has no separate compose stream and wires the single-pass answer", () => {
const route = readFileSync(new URL("../src/app/api/consult/route.ts", import.meta.url), "utf8");
const stream = readFileSync(new URL("../src/lib/stream-agent-response.ts", import.meta.url), "utf8");
assert.doesNotMatch(route, /composeAnswer|interpretFindings|consultationComposePrompt|toolChoice: "none"/);
assert.doesNotMatch(stream, /composeAnswer|interpretFindings|drainSpoken|publishFindings/);
const natal = route.slice(route.indexOf("if (!prepared.serverChart)"));
assert.match(natal, /stepScopedAnswer: true,/);
assert.match(natal, /onAnswerPhase: startAnswerPhase,/);
assert.match(natal, /consultationContinueMessages\(baseMessages, output, evidence\)/);
const window = route.slice(route.indexOf("if (shouldRunDeclaredWindowWorkflow(consultationMode)) {"), route.indexOf("if (!prepared.serverChart)"));
assert.match(window, /stepScopedAnswer: true,/);
assert.match(window, /windowPacketMessage \? undefined : evidence/);
// 原值: assert.match(route, /natalAnswerShapeInstruction\(\)/)
// 新值: 用户轮形状经 natalUserTurnShape 按「会话已有回答」切换首轮 / 追问轮,思考栏与工具上下文带 followUpTurn
// 原因: TASK-consult-answer-the-question-20260927 D3/D6(BUG-1071)
// 原值: natalUserTurnShape({ entrypoint, history }) / hasPriorAssistantAnswer(history)
// 新值: 两处都带 summaryText(Claude 验收补修:寒暄回复不算上一轮解读,摘要算)
// 原因: 「你好」的回复也存成 assistant 消息,只看 role 会把第一个正式问题当追问
assert.match(route, /natalUserTurnShape\(\{ entrypoint: consultEntrypoint, history, summaryText: historyWindow\.summaryText \}\)/);
assert.match(route, /const followUpTurn = hasPriorAssistantAnswer\(history, \{ summaryText: historyWindow\.summaryText \}\);/);
assert.match(route, /followUp: followUpTurn,/);
assert.match(route, /followUpTurn,\n\s+serverChart: prepared\.serverChart,/);
assert.doesNotMatch(route, /natalAnswerShapeInstruction\(\)/);
// One run clock: tools on the tool phase, the loop handed to the answer clock.
assert.match(route, /const agentAbortSignal = runClock\.toolSignal;/);
assert.match(route, /abortSignal: runClock\.loopSignal,/);
assert.match(route, /const answerPhaseSignal = runClock\.answerSignal;/);
const maxDuration = Number(route.match(/export const maxDuration = (\d+);/)?.[1]);
assert.ok(AGENT_TIMEOUT_MS + CONSULTATION_ANSWER_TIMEOUT_MS < maxDuration * 1000);
assert.ok(AGENT_TIMEOUT_MS + CONSULTATION_ANSWER_TIMEOUT_MS <= 180_000, "product accepted about three minutes");
});