/** * TASK-rectification-grounding-20260927 regression tests. * * T1 / BUG-1055 (recurrence of BUG-588): server fact sentences never take part * in the evidence-turn trim, the streamed final state equals the persisted * text, and a model sentence with a clock / range / percentage that is not * this turn's server fact is dropped whole (P3). */ import assert from "node:assert/strict"; import test from "node:test"; import { CASE_ID, SESSION_ID, TURN_ID, USER_ID, activeFocusFixture, candidateSnapshotFixture, CANDIDATE_ID, SECOND_CANDIDATE_ID, conversationSummaryFixture, dossierFixture, fakeAccounting, receiptHandlers, } from "./rectification-v9-test-support.ts"; import { clientStates } from "./rectification-grounding-support.ts"; import { runV9AgentTurn } from "../src/lib/rectification-agentic/v9/agent-run.ts"; import { buildInferenceState } from "../src/lib/rectification-agentic/core/build-state.ts"; import { dropUngroundedFactSentences, sentenceNumbersGrounded, spokenFactWhitelist, } from "../src/lib/rectification-agentic/v9/spoken-grounding.ts"; import { RECTIFICATION_USER_COPY } from "../src/lib/rectification-agentic/user-copy.ts"; const BEFORE: [string, string] = ["04:50", "05:10"]; const AFTER: [string, string] = ["04:55", "05:02"]; const RANGE_SENTENCE = "范围从 04:50–05:10 变为 04:55–05:02。"; function inferenceWithRange(range: [string, string]) { const state = buildInferenceState({ range_start: "04:50", range_end: "05:10", candidates: [ { id: CANDIDATE_ID, time: "05:02", relative_support: 58 }, { id: SECOND_CANDIDATE_ID, time: "04:55", relative_support: 42 }, ], events: [], probes: [], }); return { ...state, credible_range: range }; } type TurnCaseOptions = { body: string; /** Tool results the fake agent reports, in order, after read-case. */ tools?: Array<{ name: string; result?: unknown; failed?: boolean }>; fitPercent?: number; /** Flip the stored range to AFTER when this tool result arrives. */ flipOn?: string; }; /** * Real `runV9AgentTurn` over a fake Mastra agent stream: read-case, then the * listed tools, then one model body. The stored snapshot's credible range is * BEFORE until the `flipOn` tool result, AFTER from then on (the batch / * compare rescore the diagnosis reproduced). */ async function runEvidenceTurn(options: TurnCaseOptions) { process.env.RECTIFICATION_ALGORITHM_VERSION = "rectification-v5"; process.env.RECTIFICATION_DECISION_POLICY_VERSION = "rectification-candidate-policy-v2"; let flipped = false; const dossierFor = (range: [string, string]) => dossierFixture({ latestResult: candidateSnapshotFixture({ representativeTime: "05:02", decisionReceipt: { inference_state: inferenceWithRange(range), ...(typeof options.fitPercent === "number" ? { event_fit_rate: { matched: 4, total: 5, percent: options.fitPercent, band: "high" } } : {}), }, }), conversationSummary: conversationSummaryFixture({ activeFocus: activeFocusFixture({ askedTurnId: null, intent: "collect_method_evidence", expectedAnswerSchema: { prompt: "你大概是哪一年搬的家?", collect: true }, }), }), }); const accounting = fakeAccounting({ ...receiptHandlers, get_agentic_rectification_case_dossier: () => dossierFor(flipped ? AFTER : BEFORE), append_agentic_rectification_turn: () => ({ turn_id: TURN_ID }), finalize_agentic_rectification_turn: () => ({ turn_id: TURN_ID, status: "completed", idempotent: false }), }, { fallback: () => null }); const flipOn = options.flipOn ?? "rectification-compare-candidates"; const tools = options.tools ?? [{ name: "rectification-compare-candidates" }]; const chunks: Array<{ type: string; payload?: Record }> = [ { type: "start" }, { type: "tool-call", payload: { toolName: "rectification-read-case", args: { caseId: CASE_ID } } }, { type: "tool-result", payload: { toolName: "rectification-read-case" } }, ]; for (const tool of tools) { chunks.push({ type: "tool-call", payload: { toolName: tool.name, args: { caseId: CASE_ID } } }); chunks.push(tool.failed ? { type: "tool-error", payload: { toolName: tool.name, error: new Error("engine_request_failed") } } : { type: "tool-result", payload: { toolName: tool.name, result: tool.result ?? {} } }); } chunks.push({ type: "text-delta", payload: { text: options.body } }); chunks.push({ type: "finish" }); const agent = { stream: async () => ({ fullStream: (async function* () { for (const item of chunks) { if (item.type === "tool-result" && item.payload?.toolName === flipOn) flipped = true; yield item; } })(), totalUsage: Promise.resolve({ inputTokens: 10, outputTokens: 20 }), }), getSkill: async () => ({ name: "jyotish-birth-time-rectification", instructions: "skill" }), }; const emitted: Array> = []; const result = await runV9AgentTurn({ userId: USER_ID, caseId: CASE_ID, sessionId: SESSION_ID, requestId: "aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeee1", action: "evidence", message: "2016年9月离开家去北京工作", modelName: "m", accounting: accounting.client as never, billing: { reserve: async () => ({ success: true, status: 200 }), complete: async () => true, release: async () => true }, emit: (event) => { emitted.push(event as never); }, buildAgent: async () => agent as never, }); const finalize = accounting.calls.filter((call) => call.fn === "finalize_agentic_rectification_turn").at(-1); const persisted = String(finalize?.args.p_assistant_message ?? ""); return { result, emitted, persisted, states: clientStates(emitted) }; } /** A fact sentence, once shown, must stay in every later client state. */ function assertNeverWithdrawn(states: readonly string[], fact: string) { const first = states.findIndex((state) => state.includes(fact)); if (first < 0) return; for (const state of states.slice(first)) { assert.ok(state.includes(fact), `fact withdrawn after being shown: ${JSON.stringify(states)}`); } } test("BUG-1055 repro 1: two-sentence body keeps the server range sentence, stream ends where persist does", async () => { const { result, persisted, states } = await runEvidenceTurn({ body: "记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。", }); assert.equal(result.ok, true); // Before: the trim kept only the first sentence and the range sentence was cut. assert.equal(persisted, `记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。${RANGE_SENTENCE}`); assert.equal(result.answerText, persisted); assert.equal(states.at(-1), persisted); assertNeverWithdrawn(states, RANGE_SENTENCE); }); test("BUG-1055 repro 1b: three-sentence body is trimmed to one, range sentence still whole", async () => { const { persisted, states } = await runEvidenceTurn({ body: "记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。接下来再对一件事。", }); assert.equal(persisted, `记下了:2016 年 9 月去北京工作。${RANGE_SENTENCE}`); assert.equal(states.at(-1), persisted); assertNeverWithdrawn(states, RANGE_SENTENCE); }); test("BUG-1055 repro 2: a first sentence carrying the stale range is dropped whole, not persisted", async () => { const { persisted, states } = await runEvidenceTurn({ body: "记下了:2016 年 9 月去北京工作,目前范围在 04:50–05:10。接下来再对一件事。", }); // Before: 「记下了:2016 年 9 月去北京工作,目前范围在 04:50–05:10。」was persisted. assert.doesNotMatch(persisted, /目前范围在 04:50–05:10/); assert.equal(persisted, `接下来再对一件事。${RANGE_SENTENCE}`); assert.equal(states.at(-1), persisted); assertNeverWithdrawn(states, RANGE_SENTENCE); }); test("BUG-1055: an invented fit rate sentence is dropped; a server fit rate sentence is kept", async () => { const invented = await runEvidenceTurn({ body: "记下了:2016 年 9 月去北京工作。目前吻合率 90%。", fitPercent: 80, }); assert.equal(invented.persisted, `记下了:2016 年 9 月去北京工作。${RANGE_SENTENCE}`); const grounded = await runEvidenceTurn({ body: "记下了:2016 年 9 月去北京工作。目前吻合率 80%。", fitPercent: 80, }); assert.equal(grounded.persisted, `记下了:2016 年 9 月去北京工作。目前吻合率 80%。${RANGE_SENTENCE}`); }); test("BUG-1055: the post-rescore range and representative minute may be repeated; only mismatches go", async () => { const kept = await runEvidenceTurn({ body: "记下了:2016 年 9 月去北京工作。代表分钟是 05:02,范围 04:55–05:02。", }); assert.equal(kept.persisted, `记下了:2016 年 9 月去北京工作。代表分钟是 05:02,范围 04:55–05:02。${RANGE_SENTENCE}`); }); test("BUG-1055: when the whitelist leaves no model body the batch recap stands in", async () => { const { persisted } = await runEvidenceTurn({ body: "目前范围 04:50–05:10,吻合率 90%。", tools: [{ name: "rectification-record-evidence-batch", result: { items: [{ index: 0, outcome: "accepted" }], accepted_recaps: [{ display_date_label: "2016 年 9 月", event_phrase: "去北京工作" }], accepted_count: 1, rescore: { status: "completed" }, }, }], flipOn: "rectification-record-evidence-batch", }); assert.equal(persisted, `记下了:2016 年 9 月 去北京工作。${RANGE_SENTENCE}`); }); test("BUG-1055: 这次没有重新比较 survives a multi-sentence body and replaces the range sentence", async () => { const { persisted, states } = await runEvidenceTurn({ body: "记下了:2016 年 9 月去北京工作。这件事拿去对照了。还有一句。", tools: [{ name: "rectification-record-evidence-batch", result: { items: [{ index: 0, outcome: "accepted" }], accepted_recaps: [{ display_date_label: "2016 年 9 月", event_phrase: "去北京工作" }], accepted_count: 1, rescore: { status: "failed", error_code: "engine_request_failed" }, }, }], flipOn: "never", }); assert.equal(persisted, `记下了:2016 年 9 月去北京工作。${RECTIFICATION_USER_COPY.rescoreSkipped}`); assert.equal(states.at(-1), persisted); assertNeverWithdrawn(states, RECTIFICATION_USER_COPY.rescoreSkipped); }); test("BUG-1055: 候选比较这次没跑成 survives a multi-sentence body", async () => { const { persisted, states } = await runEvidenceTurn({ body: "2016 年 9 月去北京工作这件事拿去对照了。它落在那年的变动里。还有一句。", tools: [{ name: "rectification-compare-candidates", failed: true }], flipOn: "never", }); assert.equal(persisted, `2016 年 9 月去北京工作这件事拿去对照了。\n\n${RECTIFICATION_USER_COPY.compareFailedRetry}`); assert.equal(states.at(-1), persisted); assertNeverWithdrawn(states, RECTIFICATION_USER_COPY.compareFailedRetry); }); test("P3 whitelist is a gate on whole sentences: clocks, ranges and percentages", () => { const whitelist = spokenFactWhitelist({ credibleRange: ["04:55", "05:02"], representativeTime: "04:58", fitPercent: 79.6, }); assert.deepEqual(whitelist.percents, [80]); assert.equal(sentenceNumbersGrounded("范围 04:55–05:02。", whitelist), true); assert.equal(sentenceNumbersGrounded("范围 04:55 到 05:02。", whitelist), true); assert.equal(sentenceNumbersGrounded("代表分钟 04:58。", whitelist), true); assert.equal(sentenceNumbersGrounded("吻合率 80%。", whitelist), true); assert.equal(sentenceNumbersGrounded("吻合率百分之80。", whitelist), true); assert.equal(sentenceNumbersGrounded("2016 年 9 月入学。", whitelist), true); assert.equal(sentenceNumbersGrounded("范围 04:50–05:02。", whitelist), false); assert.equal(sentenceNumbersGrounded("范围 04:55–05:10。", whitelist), false); assert.equal(sentenceNumbersGrounded("代表分钟 05:00。", whitelist), false); assert.equal(sentenceNumbersGrounded("吻合率 90%。", whitelist), false); assert.equal(sentenceNumbersGrounded("吻合率 80.5%。", whitelist), false); // Whole sentence out; the neighbours stay untouched (no in-sentence deletion). assert.equal( dropUngroundedFactSentences("记下了:2016 年入学。范围在 09:00–09:05,吻合率 90%。还有吗", whitelist), "记下了:2016 年入学。还有吗", ); assert.equal(dropUngroundedFactSentences("范围在 09:00–09:05。", spokenFactWhitelist({})), ""); });