Files
Jyotisha/frontend/tests/rectification-grounding-20260927.test.ts
T
Jesse_ChenandClaude Opus 5.5 0e0baaa74c fix(rectification): server fact sentences skip the evidence trim; model numbers must match server facts (BUG-1055)
T1 of TASK-rectification-grounding-20260927 (recurrence of BUG-588).
- The attempt no longer streams range-changed / rescore-skipped /
  compare-failed sentences; the finish whitelists and trims the model body,
  then joins the server facts, and emits one final replace equal to the
  persisted text.
- P3 whitelist (spoken-grounding.ts): a model sentence with a clock, clock
  range or percentage that is not this turn's server fact is dropped whole;
  the batch recap stands in when nothing is left.
- record-evidence-batch returns range_after_rescore (post-rescore
  credible_range, representative minute, fit percent, delivers_range_this_turn);
  the receipt fingerprint stays over the old shape.
- System prompt: range is said by the server; the delivery three sentences
  only when the batch says this turn delivers.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
2026-09-27 03:28:27 +08:00

275 lines
13 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* TASK-rectification-grounding-20260927 regression tests.
*
* T1 / BUG-1055 (recurrence of BUG-588): server fact sentences never take part
* in the evidence-turn trim, the streamed final state equals the persisted
* text, and a model sentence with a clock / range / percentage that is not
* this turn's server fact is dropped whole (P3).
*/
import assert from "node:assert/strict";
import test from "node:test";
import {
CASE_ID,
SESSION_ID,
TURN_ID,
USER_ID,
activeFocusFixture,
candidateSnapshotFixture,
CANDIDATE_ID,
SECOND_CANDIDATE_ID,
conversationSummaryFixture,
dossierFixture,
fakeAccounting,
receiptHandlers,
} from "./rectification-v9-test-support.ts";
import { clientStates } from "./rectification-grounding-support.ts";
import { runV9AgentTurn } from "../src/lib/rectification-agentic/v9/agent-run.ts";
import { buildInferenceState } from "../src/lib/rectification-agentic/core/build-state.ts";
import {
dropUngroundedFactSentences,
sentenceNumbersGrounded,
spokenFactWhitelist,
} from "../src/lib/rectification-agentic/v9/spoken-grounding.ts";
import { RECTIFICATION_USER_COPY } from "../src/lib/rectification-agentic/user-copy.ts";
const BEFORE: [string, string] = ["04:50", "05:10"];
const AFTER: [string, string] = ["04:55", "05:02"];
const RANGE_SENTENCE = "范围从 04:50–05:10 变为 04:55–05:02。";
function inferenceWithRange(range: [string, string]) {
const state = buildInferenceState({
range_start: "04:50",
range_end: "05:10",
candidates: [
{ id: CANDIDATE_ID, time: "05:02", relative_support: 58 },
{ id: SECOND_CANDIDATE_ID, time: "04:55", relative_support: 42 },
],
events: [],
probes: [],
});
return { ...state, credible_range: range };
}
type TurnCaseOptions = {
body: string;
/** Tool results the fake agent reports, in order, after read-case. */
tools?: Array<{ name: string; result?: unknown; failed?: boolean }>;
fitPercent?: number;
/** Flip the stored range to AFTER when this tool result arrives. */
flipOn?: string;
};
/**
* Real `runV9AgentTurn` over a fake Mastra agent stream: read-case, then the
* listed tools, then one model body. The stored snapshot's credible range is
* BEFORE until the `flipOn` tool result, AFTER from then on (the batch /
* compare rescore the diagnosis reproduced).
*/
async function runEvidenceTurn(options: TurnCaseOptions) {
process.env.RECTIFICATION_ALGORITHM_VERSION = "rectification-v5";
process.env.RECTIFICATION_DECISION_POLICY_VERSION = "rectification-candidate-policy-v2";
let flipped = false;
const dossierFor = (range: [string, string]) => dossierFixture({
latestResult: candidateSnapshotFixture({
representativeTime: "05:02",
decisionReceipt: {
inference_state: inferenceWithRange(range),
...(typeof options.fitPercent === "number"
? { event_fit_rate: { matched: 4, total: 5, percent: options.fitPercent, band: "high" } }
: {}),
},
}),
conversationSummary: conversationSummaryFixture({
activeFocus: activeFocusFixture({
askedTurnId: null,
intent: "collect_method_evidence",
expectedAnswerSchema: { prompt: "你大概是哪一年搬的家?", collect: true },
}),
}),
});
const accounting = fakeAccounting({
...receiptHandlers,
get_agentic_rectification_case_dossier: () => dossierFor(flipped ? AFTER : BEFORE),
append_agentic_rectification_turn: () => ({ turn_id: TURN_ID }),
finalize_agentic_rectification_turn: () => ({ turn_id: TURN_ID, status: "completed", idempotent: false }),
}, { fallback: () => null });
const flipOn = options.flipOn ?? "rectification-compare-candidates";
const tools = options.tools ?? [{ name: "rectification-compare-candidates" }];
const chunks: Array<{ type: string; payload?: Record<string, unknown> }> = [
{ type: "start" },
{ type: "tool-call", payload: { toolName: "rectification-read-case", args: { caseId: CASE_ID } } },
{ type: "tool-result", payload: { toolName: "rectification-read-case" } },
];
for (const tool of tools) {
chunks.push({ type: "tool-call", payload: { toolName: tool.name, args: { caseId: CASE_ID } } });
chunks.push(tool.failed
? { type: "tool-error", payload: { toolName: tool.name, error: new Error("engine_request_failed") } }
: { type: "tool-result", payload: { toolName: tool.name, result: tool.result ?? {} } });
}
chunks.push({ type: "text-delta", payload: { text: options.body } });
chunks.push({ type: "finish" });
const agent = {
stream: async () => ({
fullStream: (async function* () {
for (const item of chunks) {
if (item.type === "tool-result" && item.payload?.toolName === flipOn) flipped = true;
yield item;
}
})(),
totalUsage: Promise.resolve({ inputTokens: 10, outputTokens: 20 }),
}),
getSkill: async () => ({ name: "jyotish-birth-time-rectification", instructions: "skill" }),
};
const emitted: Array<Record<string, unknown>> = [];
const result = await runV9AgentTurn({
userId: USER_ID,
caseId: CASE_ID,
sessionId: SESSION_ID,
requestId: "aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeee1",
action: "evidence",
message: "2016年9月离开家去北京工作",
modelName: "m",
accounting: accounting.client as never,
billing: { reserve: async () => ({ success: true, status: 200 }), complete: async () => true, release: async () => true },
emit: (event) => { emitted.push(event as never); },
buildAgent: async () => agent as never,
});
const finalize = accounting.calls.filter((call) => call.fn === "finalize_agentic_rectification_turn").at(-1);
const persisted = String(finalize?.args.p_assistant_message ?? "");
return { result, emitted, persisted, states: clientStates(emitted) };
}
/** A fact sentence, once shown, must stay in every later client state. */
function assertNeverWithdrawn(states: readonly string[], fact: string) {
const first = states.findIndex((state) => state.includes(fact));
if (first < 0) return;
for (const state of states.slice(first)) {
assert.ok(state.includes(fact), `fact withdrawn after being shown: ${JSON.stringify(states)}`);
}
}
test("BUG-1055 repro 1: two-sentence body keeps the server range sentence, stream ends where persist does", async () => {
const { result, persisted, states } = await runEvidenceTurn({
body: "记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。",
});
assert.equal(result.ok, true);
// Before: the trim kept only the first sentence and the range sentence was cut.
assert.equal(persisted, `记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。${RANGE_SENTENCE}`);
assert.equal(result.answerText, persisted);
assert.equal(states.at(-1), persisted);
assertNeverWithdrawn(states, RANGE_SENTENCE);
});
test("BUG-1055 repro 1b: three-sentence body is trimmed to one, range sentence still whole", async () => {
const { persisted, states } = await runEvidenceTurn({
body: "记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。接下来再对一件事。",
});
assert.equal(persisted, `记下了:2016 年 9 月去北京工作。${RANGE_SENTENCE}`);
assert.equal(states.at(-1), persisted);
assertNeverWithdrawn(states, RANGE_SENTENCE);
});
test("BUG-1055 repro 2: a first sentence carrying the stale range is dropped whole, not persisted", async () => {
const { persisted, states } = await runEvidenceTurn({
body: "记下了:2016 年 9 月去北京工作,目前范围在 04:50–05:10。接下来再对一件事。",
});
// Before: 「记下了:2016 年 9 月去北京工作,目前范围在 04:50–05:10。」was persisted.
assert.doesNotMatch(persisted, /目前范围在 04:50–05:10/);
assert.equal(persisted, `接下来再对一件事。${RANGE_SENTENCE}`);
assert.equal(states.at(-1), persisted);
assertNeverWithdrawn(states, RANGE_SENTENCE);
});
test("BUG-1055: an invented fit rate sentence is dropped; a server fit rate sentence is kept", async () => {
const invented = await runEvidenceTurn({
body: "记下了:2016 年 9 月去北京工作。目前吻合率 90%。",
fitPercent: 80,
});
assert.equal(invented.persisted, `记下了:2016 年 9 月去北京工作。${RANGE_SENTENCE}`);
const grounded = await runEvidenceTurn({
body: "记下了:2016 年 9 月去北京工作。目前吻合率 80%。",
fitPercent: 80,
});
assert.equal(grounded.persisted, `记下了:2016 年 9 月去北京工作。目前吻合率 80%。${RANGE_SENTENCE}`);
});
test("BUG-1055: the post-rescore range and representative minute may be repeated; only mismatches go", async () => {
const kept = await runEvidenceTurn({
body: "记下了:2016 年 9 月去北京工作。代表分钟是 05:02,范围 04:55–05:02。",
});
assert.equal(kept.persisted, `记下了:2016 年 9 月去北京工作。代表分钟是 05:02,范围 04:55–05:02。${RANGE_SENTENCE}`);
});
test("BUG-1055: when the whitelist leaves no model body the batch recap stands in", async () => {
const { persisted } = await runEvidenceTurn({
body: "目前范围 04:50–05:10,吻合率 90%。",
tools: [{
name: "rectification-record-evidence-batch",
result: {
items: [{ index: 0, outcome: "accepted" }],
accepted_recaps: [{ display_date_label: "2016 年 9 月", event_phrase: "去北京工作" }],
accepted_count: 1,
rescore: { status: "completed" },
},
}],
flipOn: "rectification-record-evidence-batch",
});
assert.equal(persisted, `记下了:2016 年 9 月 去北京工作。${RANGE_SENTENCE}`);
});
test("BUG-1055: 这次没有重新比较 survives a multi-sentence body and replaces the range sentence", async () => {
const { persisted, states } = await runEvidenceTurn({
body: "记下了:2016 年 9 月去北京工作。这件事拿去对照了。还有一句。",
tools: [{
name: "rectification-record-evidence-batch",
result: {
items: [{ index: 0, outcome: "accepted" }],
accepted_recaps: [{ display_date_label: "2016 年 9 月", event_phrase: "去北京工作" }],
accepted_count: 1,
rescore: { status: "failed", error_code: "engine_request_failed" },
},
}],
flipOn: "never",
});
assert.equal(persisted, `记下了:2016 年 9 月去北京工作。${RECTIFICATION_USER_COPY.rescoreSkipped}`);
assert.equal(states.at(-1), persisted);
assertNeverWithdrawn(states, RECTIFICATION_USER_COPY.rescoreSkipped);
});
test("BUG-1055: 候选比较这次没跑成 survives a multi-sentence body", async () => {
const { persisted, states } = await runEvidenceTurn({
body: "2016 年 9 月去北京工作这件事拿去对照了。它落在那年的变动里。还有一句。",
tools: [{ name: "rectification-compare-candidates", failed: true }],
flipOn: "never",
});
assert.equal(persisted, `2016 年 9 月去北京工作这件事拿去对照了。\n\n${RECTIFICATION_USER_COPY.compareFailedRetry}`);
assert.equal(states.at(-1), persisted);
assertNeverWithdrawn(states, RECTIFICATION_USER_COPY.compareFailedRetry);
});
test("P3 whitelist is a gate on whole sentences: clocks, ranges and percentages", () => {
const whitelist = spokenFactWhitelist({
credibleRange: ["04:55", "05:02"],
representativeTime: "04:58",
fitPercent: 79.6,
});
assert.deepEqual(whitelist.percents, [80]);
assert.equal(sentenceNumbersGrounded("范围 04:55–05:02。", whitelist), true);
assert.equal(sentenceNumbersGrounded("范围 04:55 到 05:02。", whitelist), true);
assert.equal(sentenceNumbersGrounded("代表分钟 04:58。", whitelist), true);
assert.equal(sentenceNumbersGrounded("吻合率 80%。", whitelist), true);
assert.equal(sentenceNumbersGrounded("吻合率百分之80。", whitelist), true);
assert.equal(sentenceNumbersGrounded("2016 年 9 月入学。", whitelist), true);
assert.equal(sentenceNumbersGrounded("范围 04:50–05:02。", whitelist), false);
assert.equal(sentenceNumbersGrounded("范围 04:55–05:10。", whitelist), false);
assert.equal(sentenceNumbersGrounded("代表分钟 05:00。", whitelist), false);
assert.equal(sentenceNumbersGrounded("吻合率 90%。", whitelist), false);
assert.equal(sentenceNumbersGrounded("吻合率 80.5%。", whitelist), false);
// Whole sentence out; the neighbours stay untouched (no in-sentence deletion).
assert.equal(
dropUngroundedFactSentences("记下了:2016 年入学。范围在 09:00–09:05,吻合率 90%。还有吗", whitelist),
"记下了:2016 年入学。还有吗",
);
assert.equal(dropUngroundedFactSentences("范围在 09:00–09:05。", spokenFactWhitelist({})), "");
});