- Voice: contrast limited to chart structures; drop 「这句我按『……』理解了」; forbid motive/need sentences; yes/no answered first (BUG-1070) - Follow-up turn (session already holds an answer): ≤200-char direct answer, no skeleton; decided from stored history, no intent regex; tool still runs each turn (BUG-1071) - Thinking bar first section 「先回答你问的这件事」 (BUG-1072) - First turn: opener without actions, body headings 盘里支持 / 时间怎么看 / 这周可以做的一件事 only, actions once, ≤900 chars (BUG-1073, D8) - Checklist is a list to check, not paragraphs to write (T3) Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0199rbQDTsUbCVw84wc8BTFe
319 lines
15 KiB
TypeScript
319 lines
15 KiB
TypeScript
// TASK-consult-evidence-card-20260927 red line 4 and T5.
|
||
//
|
||
// Red line 4 (BUG-1053 stays fixed with the card): the model call that writes
|
||
// the answer has the evidence card in its prompt; narration before a tool call
|
||
// is never answer text; a non-`stop` answer step is `answer_truncated` and not
|
||
// charged; a length continuation carries the card.
|
||
//
|
||
// T5 (D8): one read-only lookup of a section outside the card, from this
|
||
// request's calculation only, at most once per turn. A lookup during the
|
||
// answer phase must not cut or repeat released text and must not reset the
|
||
// 110 s tool / 70 s answer clocks.
|
||
//
|
||
// Real `getJyotishAgent` + prompt-recording fake model; the calculation is the
|
||
// golden engine capture of a public AA chart (AGENTS §7.4).
|
||
import assert from "node:assert/strict";
|
||
import { readFileSync } from "node:fs";
|
||
import test from "node:test";
|
||
|
||
import { REPORT_HEADING } from "../src/lib/consultation-thinking-plan.ts";
|
||
import { LOOKUP_RESTART_MATCH_CHARS } from "../src/lib/stream-agent-response.ts";
|
||
import { evidenceLookupActivityLabel } from "../src/lib/consultation-activity-labels.ts";
|
||
import {
|
||
CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID,
|
||
consultationNatalPrepareStep,
|
||
createConsultationRuntimeState,
|
||
createConsultationTools,
|
||
MAX_EVIDENCE_LOOKUPS_PER_TURN,
|
||
} from "../src/mastra/consultation-tools.ts";
|
||
import {
|
||
consultationWorkflowResponseSchema,
|
||
toAgentConsultationContext,
|
||
toModelOutput,
|
||
} from "../src/mastra/consultation-workflow.ts";
|
||
import {
|
||
firstPromptWithToolResult,
|
||
pieces,
|
||
publicServerChart,
|
||
runNatalAgent,
|
||
type Turn,
|
||
} from "./consult-natal-agent-test-support.ts";
|
||
|
||
type Json = Record<string, unknown>;
|
||
const golden = JSON.parse(readFileSync(
|
||
new URL("./fixtures/consult-evidence-card-golden.json", import.meta.url),
|
||
"utf8",
|
||
)) as { charts: Array<{ id: string; workflow: Json }> };
|
||
const workflow = golden.charts.find((chart) => chart.id === "steve_jobs")!.workflow;
|
||
const modules = (workflow.chart as Json).modules as Json;
|
||
// Facts only the calculation carries, read from the engine payload.
|
||
const PD_START = String(((modules.dasha_sub_periods as Json).pratyantar_dasha_timeline as Json & { current: Json }).current.start);
|
||
const NARAYANA_SIGN = String(((modules.narayana_dasha as Json).current_dasha as Json & { md: Json }).md.sign);
|
||
const D60_LAGNA = String((((modules.varga_spectrum as Json).formal as Json).D60 as Json).lagna);
|
||
|
||
const PRE_TOOL = "我先排一下盘,稍等。";
|
||
const LOOKUP_NARRATION = "我再看一眼 D60。";
|
||
const OPENER = "你和父母这条线,表面是各过各的,底下一直在较劲:你要自己说了算,他们要你稳。这不是谁对谁错,是同一件事的两种护法,所以叫「隔空守护」。";
|
||
// 原值: 夹具里有 `## ${REPORT_HEADING.question}` 一节
|
||
// 新值: 该节删除,夹具从口语开场直接到 `## ${REPORT_HEADING.support}`
|
||
// 原因: TASK-consult-answer-the-question-20260927 D8 / BUG-1073:开场就是回答,正文不再从「先回答你的问题」重来
|
||
const ANSWER = [
|
||
OPENER,
|
||
"",
|
||
`## ${REPORT_HEADING.support}`,
|
||
"四宫的主星落在十宫,家里的事常被你当成要办成的事来处理。",
|
||
"",
|
||
`## ${REPORT_HEADING.timing}`,
|
||
"这几个月先试着每周打一次电话,不急着谈结论。",
|
||
"",
|
||
`## ${REPORT_HEADING.action}`,
|
||
"- **这周**:约他们吃一顿饭,只聊近况。",
|
||
"",
|
||
].join("\n");
|
||
|
||
// The part of the answer written before a mid-answer lookup: past the first
|
||
// heading, so it has already been released to the client.
|
||
const HEAD_CHARS = 120;
|
||
|
||
const calcCall = (text?: string): Turn => ({
|
||
parts: [
|
||
...(text ? [{ text }] : []),
|
||
{ tool: { name: "run-jyotish-consultation", input: { question: "我和父母关系如何", domains: ["parents"] } } },
|
||
],
|
||
finish: "tool-calls",
|
||
});
|
||
|
||
const lookupCall = (section: string, text?: string): Turn => ({
|
||
parts: [
|
||
...(text ? pieces(text) : []),
|
||
{ tool: { name: CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID, input: { section } } },
|
||
],
|
||
finish: "tool-calls",
|
||
});
|
||
|
||
const run = (turns: Turn[], options: { toolPhaseMs?: number; answerMs?: number } = {}) => runNatalAgent(turns, { workflow, ...options });
|
||
|
||
function toolResultText(prompt: unknown[] | undefined, toolName: string) {
|
||
return JSON.stringify((prompt ?? []).filter((message) => {
|
||
const value = message as { role?: string; content?: unknown };
|
||
return value.role === "tool" && JSON.stringify(value.content).includes(toolName);
|
||
}));
|
||
}
|
||
|
||
test("the answer-writing call sees the evidence card, not the full result (red line 4)", async () => {
|
||
const result = await run([calcCall(PRE_TOOL), { parts: pieces(ANSWER), finish: "stop" }]);
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
assert.equal(result.calls, 2, "no second, blind writer");
|
||
const writer = firstPromptWithToolResult(result.prompts, "run-jyotish-consultation");
|
||
assert.equal(writer, 1);
|
||
const card = toolResultText(result.prompts[writer], "run-jyotish-consultation");
|
||
// 原值: "evidence-card-v1" / 新值: "evidence-card-v2" / 原因: 卡 v2(TASK-consult-evidence-card-v2-20260927 T3)
|
||
assert.ok(card.includes("evidence-card-v2"), "the card reaches the writer");
|
||
assert.ok(card.includes(PD_START), "with the engine's pratyantar start");
|
||
assert.ok(card.includes(NARAYANA_SIGN), "and the running Narayana sign");
|
||
assert.ok(card.includes("can_answer_direction"), "and the answer contract");
|
||
for (const absent of ["technique_audit_table", "western_spectrum", "research_dn"]) {
|
||
assert.doesNotMatch(card, new RegExp(`${absent}\\\\*"\\s*:`), `${absent} stays server-side`);
|
||
}
|
||
assert.equal(result.answer, ANSWER);
|
||
assert.equal(result.answer.includes(PRE_TOOL), false, "pre-tool narration is not the answer");
|
||
assert.equal(result.charges, 1);
|
||
});
|
||
|
||
test("a non-stop answer step with the card is still truncated and not charged (red line 4)", async () => {
|
||
const result = await run([calcCall(), { parts: pieces(ANSWER.slice(0, 200)), finish: "content-filter" }]);
|
||
assert.deepEqual(result.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
|
||
assert.equal(result.charges, 0);
|
||
});
|
||
|
||
test("a length continuation carries the card (red line 4)", async () => {
|
||
const cut = ANSWER.indexOf(`## ${REPORT_HEADING.timing}`);
|
||
const result = await run([
|
||
calcCall(),
|
||
{ parts: pieces(ANSWER.slice(0, cut)), finish: "length" },
|
||
{ parts: pieces(ANSWER.slice(cut)), finish: "stop" },
|
||
]);
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
assert.equal(result.continuations.length, 1);
|
||
const continuation = JSON.stringify(result.prompts[2]);
|
||
// 原值: "evidence-card-v1" / 新值: "evidence-card-v2" / 原因: 同上
|
||
assert.ok(continuation.includes("evidence-card-v2"));
|
||
assert.ok(continuation.includes(PD_START));
|
||
assert.equal(result.completed, ANSWER);
|
||
});
|
||
|
||
test("step 0 still exposes only the calculation tool; later steps may use the lookup", () => {
|
||
assert.deepEqual(consultationNatalPrepareStep({ stepNumber: 0 }).activeTools, ["run-jyotish-consultation"]);
|
||
assert.equal("activeTools" in consultationNatalPrepareStep({ stepNumber: 1 }), false);
|
||
assert.equal(MAX_EVIDENCE_LOOKUPS_PER_TURN, 1);
|
||
});
|
||
|
||
test("a lookup before the answer returns the section from the request cache and the answer is written once", async () => {
|
||
const result = await run([
|
||
calcCall(),
|
||
lookupCall("varga:D60", LOOKUP_NARRATION),
|
||
{ parts: pieces(ANSWER), finish: "stop" },
|
||
]);
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
assert.equal(result.workflowRuns, 1, "the lookup never recalculates");
|
||
assert.equal(result.answer, ANSWER);
|
||
assert.equal(result.answer.includes(LOOKUP_NARRATION), false, "narration around the lookup is not answer text");
|
||
const writer = firstPromptWithToolResult(result.prompts, CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
|
||
assert.equal(writer, 2);
|
||
const lookup = toolResultText(result.prompts[writer], CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
|
||
assert.ok(lookup.includes(D60_LAGNA), "the D60 lagna the engine computed");
|
||
assert.match(lookup, /status\\*"\s*:\s*\\*"ok/);
|
||
// Receipt step and the live activity row in plain words.
|
||
const step = result.state.steps.find((item) => item.name === CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
|
||
assert.equal(step?.kind, "tool");
|
||
assert.equal(step?.status, "completed");
|
||
assert.ok(result.events.some((event) => event.type === "activity" && event.label === evidenceLookupActivityLabel("varga:D60")));
|
||
assert.equal(evidenceLookupActivityLabel("varga:D60"), "正在多看一眼:D60 分盘…");
|
||
assert.equal(result.state.evidenceLookupCallCount, 1);
|
||
assert.equal(result.charges, 1);
|
||
});
|
||
|
||
test("a second lookup in the same turn is refused", async () => {
|
||
const result = await run([
|
||
calcCall(),
|
||
lookupCall("varga:D60"),
|
||
lookupCall("western:solar_return"),
|
||
{ parts: pieces(ANSWER), finish: "stop" },
|
||
]);
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
const second = toolResultText(result.prompts[3], CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
|
||
assert.ok(second.includes("lookup_limit_reached"));
|
||
assert.equal(result.state.evidenceLookupCallCount, 2);
|
||
assert.deepEqual(
|
||
result.state.steps.filter((item) => item.name === CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID).map((item) => item.status),
|
||
["completed", "failed"],
|
||
);
|
||
assert.equal(result.answer, ANSWER);
|
||
});
|
||
|
||
test("a lookup with no calculation in the request cache returns unavailable and calculates nothing", async () => {
|
||
let workflowRuns = 0;
|
||
const state = createConsultationRuntimeState();
|
||
const tools = createConsultationTools({
|
||
userId: "u", sessionId: "s", requestId: "lookup-miss", consultationMode: "verified_chart",
|
||
serverChart: publicServerChart as never, state,
|
||
runWorkflow: async () => {
|
||
workflowRuns += 1;
|
||
return structuredClone(workflow) as never;
|
||
},
|
||
});
|
||
const miss = await tools[CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID].execute!({ section: "varga:D60" } as never, {} as never) as Json;
|
||
assert.equal(miss.status, "unavailable");
|
||
assert.equal(miss.reason, "calculation_not_in_request_cache");
|
||
assert.equal(workflowRuns, 0);
|
||
assert.equal(state.steps.at(-1)?.name, CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID);
|
||
assert.equal(state.steps.at(-1)?.status, "failed");
|
||
});
|
||
|
||
test("the lookup returns exactly the projected section it names", async () => {
|
||
const state = createConsultationRuntimeState();
|
||
const tools = createConsultationTools({
|
||
userId: "u", sessionId: "s", requestId: "lookup-hit", consultationMode: "verified_chart",
|
||
serverChart: publicServerChart as never, state,
|
||
runWorkflow: async () => structuredClone(workflow) as never,
|
||
});
|
||
await tools["run-jyotish-consultation"].execute!({ question: "q", domains: ["parents"] } as never, { writer: { custom: async () => {} } } as never);
|
||
const hit = await tools[CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID].execute!({ section: "varga:D60" } as never, { writer: { custom: async () => {} } } as never) as Json;
|
||
const packet = toModelOutput(toAgentConsultationContext(consultationWorkflowResponseSchema.parse(structuredClone(workflow))));
|
||
const natal = packet.claim_cards.find((card) => card.category === "natal_foundation")!.evidence as Json;
|
||
assert.equal(hit.status, "ok");
|
||
assert.deepEqual(hit.data, ((natal.varga_spectrum as Json).formal as Json).D60);
|
||
});
|
||
|
||
test("a lookup after answer text went out: a verbatim restart is dropped, the answer is whole and settles on stop", async () => {
|
||
const head = ANSWER.slice(0, HEAD_CHARS);
|
||
const result = await run([
|
||
calcCall(),
|
||
lookupCall("yogas", head),
|
||
// The model starts the whole answer over after the lookup.
|
||
{ parts: pieces(ANSWER), finish: "stop" },
|
||
]);
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
assert.equal(result.answer, ANSWER, "released text is neither cut nor repeated");
|
||
assert.equal(result.completed, ANSWER);
|
||
assert.equal(result.charges, 1);
|
||
assert.ok(result.state.steps.some((step) => step.name === "answer-restart-dropped"));
|
||
assert.equal(result.state.composeFinishReason, "stop");
|
||
});
|
||
|
||
test("a lookup after answer text went out: a continuation is kept as written", async () => {
|
||
const head = ANSWER.slice(0, HEAD_CHARS);
|
||
const result = await run([
|
||
calcCall(),
|
||
lookupCall("yogas", head),
|
||
{ parts: pieces(ANSWER.slice(HEAD_CHARS)), finish: "stop" },
|
||
]);
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
assert.equal(result.answer, ANSWER);
|
||
assert.equal(result.state.steps.some((step) => step.name === "answer-restart-dropped"), false);
|
||
});
|
||
|
||
test("a step that only opens like the released text is not mistaken for a restart", async () => {
|
||
const head = ANSWER.slice(0, HEAD_CHARS);
|
||
const shared = OPENER.slice(0, LOOKUP_RESTART_MATCH_CHARS - 10);
|
||
const tail = `${shared}——补一句:格局明细里没有新的东西。`;
|
||
const result = await run([
|
||
calcCall(),
|
||
lookupCall("yogas", head),
|
||
{ parts: pieces(tail), finish: "stop" },
|
||
]);
|
||
assert.equal(result.answer, `${head}${tail}`);
|
||
assert.equal(result.state.steps.some((step) => step.name === "answer-restart-dropped"), false);
|
||
});
|
||
|
||
test("a lookup does not reset the answer clock: the same 70 s signal bounds the whole answer phase", async () => {
|
||
const result = await run([
|
||
calcCall(),
|
||
{ ...lookupCall("varga:D60"), delayMs: 40 },
|
||
{ parts: pieces(ANSWER, 6), finish: "stop", delayMs: 25 },
|
||
], { answerMs: 300 });
|
||
assert.deepEqual(result.terminal.map((event) => `${event.type}:${event.code ?? ""}`), ["run.failed:answer_truncated"]);
|
||
assert.equal(result.charges, 0);
|
||
assert.equal(result.answerSignals.length, 1, "the answer clock started once");
|
||
assert.ok(result.state.steps.some((step) => step.kind === "abort" && step.name === "compose-abort"));
|
||
});
|
||
|
||
test("a lookup in the answer phase is not cut by the tool phase's deadline", async () => {
|
||
const result = await run([
|
||
calcCall(),
|
||
{ ...lookupCall("varga:D60"), delayMs: 30 },
|
||
{ parts: pieces(ANSWER, 6), finish: "stop", delayMs: 12 },
|
||
], { toolPhaseMs: 150, answerMs: 5_000 });
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
assert.equal(result.clock.toolSignal.aborted, true, "the tool phase did expire");
|
||
assert.equal(result.answer, ANSWER);
|
||
});
|
||
|
||
test("BUG-1059: narration without closing punctuation before a tool call never leaks into the answer", async () => {
|
||
// The visible-text transformer held the open clause ("…:") across the tool
|
||
// call and emitted it glued to the next step's first sentence.
|
||
const openCalc = "我先排一下盘:";
|
||
const openLookup = "再看一眼 D60 分盘,";
|
||
const result = await run([
|
||
calcCall(openCalc),
|
||
lookupCall("varga:D60", openLookup),
|
||
{ parts: pieces(ANSWER), finish: "stop" },
|
||
]);
|
||
assert.deepEqual(result.terminal.map((event) => event.type), ["run.completed"]);
|
||
assert.equal(result.answer, ANSWER);
|
||
assert.equal(result.answer.includes("排一下盘"), false);
|
||
assert.equal(result.answer.includes("再看一眼"), false);
|
||
});
|
||
|
||
test("BUG-1059: an open clause of released answer text before a lookup goes out whole, not cut", async () => {
|
||
// The head ends mid-sentence; that fragment is answer text, not narration.
|
||
const head = ANSWER.slice(0, HEAD_CHARS);
|
||
assert.doesNotMatch(head.slice(-1), /[。!?.!?\n]/, "the head really ends mid-clause");
|
||
const result = await run([
|
||
calcCall(),
|
||
lookupCall("yogas", head),
|
||
{ parts: pieces(ANSWER.slice(HEAD_CHARS)), finish: "stop" },
|
||
]);
|
||
assert.equal(result.answer, ANSWER);
|
||
});
|