Files
Jyotisha/frontend/tests/consult-evidence-lookup-20260927.test.ts
T
Jesse_ChenandClaude Fable 5.1 9a42333a55 feat(consult): answer the sentence asked — no motive guessing, follow-up turns answer directly, first turn says each thing once (BUG-1070~1073)
- Voice: contrast limited to chart structures; drop 「这句我按『……』理解了」; forbid motive/need sentences; yes/no answered first (BUG-1070)
- Follow-up turn (session already holds an answer): ≤200-char direct answer, no skeleton; decided from stored history, no intent regex; tool still runs each turn (BUG-1071)
- Thinking bar first section 「先回答你问的这件事」 (BUG-1072)
- First turn: opener without actions, body headings 盘里支持 / 时间怎么看 / 这周可以做的一件事 only, actions once, ≤900 chars (BUG-1073, D8)
- Checklist is a list to check, not paragraphs to write (T3)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0199rbQDTsUbCVw84wc8BTFe
2026-09-27 23:31:33 +08:00

319 lines
15 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// TASK-consult-evidence-card-20260927 red line 4 and T5.
//
// Red line 4 (BUG-1053 stays fixed with the card): the model call that writes
// the answer has the evidence card in its prompt; narration before a tool call
// is never answer text; a non-`stop` answer step is `answer_truncated` and not
// charged; a length continuation carries the card.
//
// T5 (D8): one read-only lookup of a section outside the card, from this
// request's calculation only, at most once per turn. A lookup during the
// answer phase must not cut or repeat released text and must not reset the
// 110 s tool / 70 s answer clocks.
//
// Real `getJyotishAgent` + prompt-recording fake model; the calculation is the
// golden engine capture of a public AA chart (AGENTS §7.4).
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
import test from "node:test";
import { REPORT_HEADING } from "../src/lib/consultation-thinking-plan.ts";
import { LOOKUP_RESTART_MATCH_CHARS } from "../src/lib/stream-agent-response.ts";
import { evidenceLookupActivityLabel } from "../src/lib/consultation-activity-labels.ts";
import {
CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID,
consultationNatalPrepareStep,
createConsultationRuntimeState,
createConsultationTools,
MAX_EVIDENCE_LOOKUPS_PER_TURN,
} from "../src/mastra/consultation-tools.ts";
import {
consultationWorkflowResponseSchema,
toAgentConsultationContext,
toModelOutput,
} from "../src/mastra/consultation-workflow.ts";
import {
firstPromptWithToolResult,
pieces,
publicServerChart,
runNatalAgent,
type Turn,
} from "./consult-natal-agent-test-support.ts";
type Json = Record<string, unknown>;
const golden = JSON.parse(readFileSync(
new URL("./fixtures/consult-evidence-card-golden.json", import.meta.url),
"utf8",
)) as { charts: Array<{ id: string; workflow: Json }> };
const workflow = golden.charts.find((chart) => chart.id === "steve_jobs")!.workflow;
const modules = (workflow.chart as Json).modules as Json;
// Facts only the calculation carries, read from the engine payload.
const PD_START = String(((modules.dasha_sub_periods as Json).pratyantar_dasha_timeline as Json & { current: Json }).current.start);
const NARAYANA_SIGN = String(((modules.narayana_dasha as Json).current_dasha as Json & { md: Json }).md.sign);
const D60_LAGNA = String((((modules.varga_spectrum as Json).formal as Json).D60 as Json).lagna);
const PRE_TOOL = "我先排一下盘,稍等。";
const LOOKUP_NARRATION = "我再看一眼 D60。";
const OPENER = "你和父母这条线,表面是各过各的,底下一直在较劲:你要自己说了算,他们要你稳。这不是谁对谁错,是同一件事的两种护法,所以叫「隔空守护」。";
// 原值: 夹具里有 `## ${REPORT_HEADING.question}` 一节
// 新值: 该节删除,夹具从口语开场直接到 `## ${REPORT_HEADING.support}`
// 原因: TASK-consult-answer-the-question-20260927 D8 / BUG-1073:开场就是回答,正文不再从「先回答你的问题」重来
const ANSWER = [
OPENER,
"",
`## ${REPORT_HEADING.support}`,
"四宫的主星落在十宫,家里的事常被你当成要办成的事来处理。",
"",
`## ${REPORT_HEADING.timing}`,
"这几个月先试着每周打一次电话,不急着谈结论。",
"",
`## ${REPORT_HEADING.action}`,
"- **这周**:约他们吃一顿饭,只聊近况。",
"",
].join("\n");
// The part of the answer written before a mid-answer lookup: past the first
// heading, so it has already been released to the client.
const HEAD_CHARS = 120;
const calcCall = (text?: string): Turn => ({
parts: [
...(text ? [{ text }] : []),
{ tool: { name: "run-jyotish-consultation", input: { question: "我和父母关系如何", domains: ["parents"] } } },
],
finish: "tool-calls",
});
const lookupCall = (section: string, text?: string): Turn => ({
parts: [
...(text ? pieces(text) : []),
{ tool: { name: CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID, input: { section } } },
],
finish: "tool-calls",
});
const run = (turns: Turn[], options: { toolPhaseMs?: number; answerMs?: number } = {}) => runNatalAgent(turns, { workflow, ...options });
function toolResultText(prompt: unknown[] | undefined, toolName: string) {
return JSON.stringify((prompt ?? []).filter((message) => {
const value = message as { role?: string; content?: unknown };
return value.role === "tool" && JSON.stringify(value.content).includes(toolName);
}));
}
test("the answer-writing call sees the evidence card, not the full result (red line 4)", async () => {
const result = await run([calcCall(PRE_TOOL), { parts: pieces(ANSWER), finish: "stop" }]);
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(result.calls, 2, "no second, blind writer");
const writer = firstPromptWithToolResult(result.prompts, "run-jyotish-consultation");
assert.equal(writer, 1);
const card = toolResultText(result.prompts[writer], "run-jyotish-consultation");
// 原值: "evidence-card-v1" / 新值: "evidence-card-v2" / 原因: 卡 v2(TASK-consult-evidence-card-v2-20260927 T3)
assert.ok(card.includes("evidence-card-v2"), "the card reaches the writer");
assert.ok(card.includes(PD_START), "with the engine's pratyantar start");
assert.ok(card.includes(NARAYANA_SIGN), "and the running Narayana sign");
assert.ok(card.includes("can_answer_direction"), "and the answer contract");
for (const absent of ["technique_audit_table", "western_spectrum", "research_dn"]) {
assert.doesNotMatch(card, new RegExp(`${absent}\\\\*"\\s*:`), `${absent} stays server-side`);
}
assert.equal(result.answer, ANSWER);
assert.equal(result.answer.includes(PRE_TOOL), false, "pre-tool narration is not the answer");
assert.equal(result.charges, 1);
});
test("a non-stop answer step with the card is still truncated and not charged (red line 4)", async () => {
const result = await run([calcCall(), { parts: pieces(ANSWER.slice(0, 200)), finish: "content-filter" }]);
assert.deepEqual(result.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
assert.equal(result.charges, 0);
});
test("a length continuation carries the card (red line 4)", async () => {
const cut = ANSWER.indexOf(`## ${REPORT_HEADING.timing}`);
const result = await run([
calcCall(),
{ parts: pieces(ANSWER.slice(0, cut)), finish: "length" },
{ parts: pieces(ANSWER.slice(cut)), finish: "stop" },
]);
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(result.continuations.length, 1);
const continuation = JSON.stringify(result.prompts[2]);
// 原值: "evidence-card-v1" / 新值: "evidence-card-v2" / 原因: 同上
assert.ok(continuation.includes("evidence-card-v2"));
assert.ok(continuation.includes(PD_START));
assert.equal(result.completed, ANSWER);
});
test("step 0 still exposes only the calculation tool; later steps may use the lookup", () => {
assert.deepEqual(consultationNatalPrepareStep({ stepNumber: 0 }).activeTools, ["run-jyotish-consultation"]);
assert.equal("activeTools" in consultationNatalPrepareStep({ stepNumber: 1 }), false);
assert.equal(MAX_EVIDENCE_LOOKUPS_PER_TURN, 1);
});
test("a lookup before the answer returns the section from the request cache and the answer is written once", async () => {
const result = await run([
calcCall(),
lookupCall("varga:D60", LOOKUP_NARRATION),
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(result.workflowRuns, 1, "the lookup never recalculates");
assert.equal(result.answer, ANSWER);
assert.equal(result.answer.includes(LOOKUP_NARRATION), false, "narration around the lookup is not answer text");
const writer = firstPromptWithToolResult(result.prompts, CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
assert.equal(writer, 2);
const lookup = toolResultText(result.prompts[writer], CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
assert.ok(lookup.includes(D60_LAGNA), "the D60 lagna the engine computed");
assert.match(lookup, /status\\*"\s*:\s*\\*"ok/);
// Receipt step and the live activity row in plain words.
const step = result.state.steps.find((item) => item.name === CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
assert.equal(step?.kind, "tool");
assert.equal(step?.status, "completed");
assert.ok(result.events.some((event) => event.type === "activity" && event.label === evidenceLookupActivityLabel("varga:D60")));
assert.equal(evidenceLookupActivityLabel("varga:D60"), "正在多看一眼:D60 分盘…");
assert.equal(result.state.evidenceLookupCallCount, 1);
assert.equal(result.charges, 1);
});
test("a second lookup in the same turn is refused", async () => {
const result = await run([
calcCall(),
lookupCall("varga:D60"),
lookupCall("western:solar_return"),
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
const second = toolResultText(result.prompts[3], CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
assert.ok(second.includes("lookup_limit_reached"));
assert.equal(result.state.evidenceLookupCallCount, 2);
assert.deepEqual(
result.state.steps.filter((item) => item.name === CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID).map((item) => item.status),
["completed", "failed"],
);
assert.equal(result.answer, ANSWER);
});
test("a lookup with no calculation in the request cache returns unavailable and calculates nothing", async () => {
let workflowRuns = 0;
const state = createConsultationRuntimeState();
const tools = createConsultationTools({
userId: "u", sessionId: "s", requestId: "lookup-miss", consultationMode: "verified_chart",
serverChart: publicServerChart as never, state,
runWorkflow: async () => {
workflowRuns += 1;
return structuredClone(workflow) as never;
},
});
const miss = await tools[CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID].execute!({ section: "varga:D60" } as never, {} as never) as Json;
assert.equal(miss.status, "unavailable");
assert.equal(miss.reason, "calculation_not_in_request_cache");
assert.equal(workflowRuns, 0);
assert.equal(state.steps.at(-1)?.name, CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
assert.equal(state.steps.at(-1)?.status, "failed");
});
test("the lookup returns exactly the projected section it names", async () => {
const state = createConsultationRuntimeState();
const tools = createConsultationTools({
userId: "u", sessionId: "s", requestId: "lookup-hit", consultationMode: "verified_chart",
serverChart: publicServerChart as never, state,
runWorkflow: async () => structuredClone(workflow) as never,
});
await tools["run-jyotish-consultation"].execute!({ question: "q", domains: ["parents"] } as never, { writer: { custom: async () => {} } } as never);
const hit = await tools[CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID].execute!({ section: "varga:D60" } as never, { writer: { custom: async () => {} } } as never) as Json;
const packet = toModelOutput(toAgentConsultationContext(consultationWorkflowResponseSchema.parse(structuredClone(workflow))));
const natal = packet.claim_cards.find((card) => card.category === "natal_foundation")!.evidence as Json;
assert.equal(hit.status, "ok");
assert.deepEqual(hit.data, ((natal.varga_spectrum as Json).formal as Json).D60);
});
test("a lookup after answer text went out: a verbatim restart is dropped, the answer is whole and settles on stop", async () => {
const head = ANSWER.slice(0, HEAD_CHARS);
const result = await run([
calcCall(),
lookupCall("yogas", head),
// The model starts the whole answer over after the lookup.
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(result.answer, ANSWER, "released text is neither cut nor repeated");
assert.equal(result.completed, ANSWER);
assert.equal(result.charges, 1);
assert.ok(result.state.steps.some((step) => step.name === "answer-restart-dropped"));
assert.equal(result.state.composeFinishReason, "stop");
});
test("a lookup after answer text went out: a continuation is kept as written", async () => {
const head = ANSWER.slice(0, HEAD_CHARS);
const result = await run([
calcCall(),
lookupCall("yogas", head),
{ parts: pieces(ANSWER.slice(HEAD_CHARS)), finish: "stop" },
]);
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(result.answer, ANSWER);
assert.equal(result.state.steps.some((step) => step.name === "answer-restart-dropped"), false);
});
test("a step that only opens like the released text is not mistaken for a restart", async () => {
const head = ANSWER.slice(0, HEAD_CHARS);
const shared = OPENER.slice(0, LOOKUP_RESTART_MATCH_CHARS - 10);
const tail = `${shared}——补一句:格局明细里没有新的东西。`;
const result = await run([
calcCall(),
lookupCall("yogas", head),
{ parts: pieces(tail), finish: "stop" },
]);
assert.equal(result.answer, `${head}${tail}`);
assert.equal(result.state.steps.some((step) => step.name === "answer-restart-dropped"), false);
});
test("a lookup does not reset the answer clock: the same 70 s signal bounds the whole answer phase", async () => {
const result = await run([
calcCall(),
{ ...lookupCall("varga:D60"), delayMs: 40 },
{ parts: pieces(ANSWER, 6), finish: "stop", delayMs: 25 },
], { answerMs: 300 });
assert.deepEqual(result.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
assert.equal(result.charges, 0);
assert.equal(result.answerSignals.length, 1, "the answer clock started once");
assert.ok(result.state.steps.some((step) => step.kind === "abort" && step.name === "compose-abort"));
});
test("a lookup in the answer phase is not cut by the tool phase's deadline", async () => {
const result = await run([
calcCall(),
{ ...lookupCall("varga:D60"), delayMs: 30 },
{ parts: pieces(ANSWER, 6), finish: "stop", delayMs: 12 },
], { toolPhaseMs: 150, answerMs: 5_000 });
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(result.clock.toolSignal.aborted, true, "the tool phase did expire");
assert.equal(result.answer, ANSWER);
});
test("BUG-1059: narration without closing punctuation before a tool call never leaks into the answer", async () => {
// The visible-text transformer held the open clause ("…:") across the tool
// call and emitted it glued to the next step's first sentence.
const openCalc = "我先排一下盘:";
const openLookup = "再看一眼 D60 分盘,";
const result = await run([
calcCall(openCalc),
lookupCall("varga:D60", openLookup),
{ parts: pieces(ANSWER), finish: "stop" },
]);
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
assert.equal(result.answer, ANSWER);
assert.equal(result.answer.includes("排一下盘"), false);
assert.equal(result.answer.includes("再看一眼"), false);
});
test("BUG-1059: an open clause of released answer text before a lookup goes out whole, not cut", async () => {
// The head ends mid-sentence; that fragment is answer text, not narration.
const head = ANSWER.slice(0, HEAD_CHARS);
assert.doesNotMatch(head.slice(-1), /[。!?.!?\n]/, "the head really ends mid-clause");
const result = await run([
calcCall(),
lookupCall("yogas", head),
{ parts: pieces(ANSWER.slice(HEAD_CHARS)), finish: "stop" },
]);
assert.equal(result.answer, ANSWER);
});