T1 of TASK-rectification-grounding-20260927 (recurrence of BUG-588). - The attempt no longer streams range-changed / rescore-skipped / compare-failed sentences; the finish whitelists and trims the model body, then joins the server facts, and emits one final replace equal to the persisted text. - P3 whitelist (spoken-grounding.ts): a model sentence with a clock, clock range or percentage that is not this turn's server fact is dropped whole; the batch recap stands in when nothing is left. - record-evidence-batch returns range_after_rescore (post-rescore credible_range, representative minute, fit percent, delivers_range_this_turn); the receipt fingerprint stays over the old shape. - System prompt: range is said by the server; the delivery three sentences only when the batch says this turn delivers. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
275 lines
13 KiB
TypeScript
275 lines
13 KiB
TypeScript
/**
|
||
* TASK-rectification-grounding-20260927 regression tests.
|
||
*
|
||
* T1 / BUG-1055 (recurrence of BUG-588): server fact sentences never take part
|
||
* in the evidence-turn trim, the streamed final state equals the persisted
|
||
* text, and a model sentence with a clock / range / percentage that is not
|
||
* this turn's server fact is dropped whole (P3).
|
||
*/
|
||
import assert from "node:assert/strict";
|
||
import test from "node:test";
|
||
import {
|
||
CASE_ID,
|
||
SESSION_ID,
|
||
TURN_ID,
|
||
USER_ID,
|
||
activeFocusFixture,
|
||
candidateSnapshotFixture,
|
||
CANDIDATE_ID,
|
||
SECOND_CANDIDATE_ID,
|
||
conversationSummaryFixture,
|
||
dossierFixture,
|
||
fakeAccounting,
|
||
receiptHandlers,
|
||
} from "./rectification-v9-test-support.ts";
|
||
import { clientStates } from "./rectification-grounding-support.ts";
|
||
import { runV9AgentTurn } from "../src/lib/rectification-agentic/v9/agent-run.ts";
|
||
import { buildInferenceState } from "../src/lib/rectification-agentic/core/build-state.ts";
|
||
import {
|
||
dropUngroundedFactSentences,
|
||
sentenceNumbersGrounded,
|
||
spokenFactWhitelist,
|
||
} from "../src/lib/rectification-agentic/v9/spoken-grounding.ts";
|
||
import { RECTIFICATION_USER_COPY } from "../src/lib/rectification-agentic/user-copy.ts";
|
||
|
||
const BEFORE: [string, string] = ["04:50", "05:10"];
|
||
const AFTER: [string, string] = ["04:55", "05:02"];
|
||
const RANGE_SENTENCE = "范围从 04:50–05:10 变为 04:55–05:02。";
|
||
|
||
function inferenceWithRange(range: [string, string]) {
|
||
const state = buildInferenceState({
|
||
range_start: "04:50",
|
||
range_end: "05:10",
|
||
candidates: [
|
||
{ id: CANDIDATE_ID, time: "05:02", relative_support: 58 },
|
||
{ id: SECOND_CANDIDATE_ID, time: "04:55", relative_support: 42 },
|
||
],
|
||
events: [],
|
||
probes: [],
|
||
});
|
||
return { ...state, credible_range: range };
|
||
}
|
||
|
||
type TurnCaseOptions = {
|
||
body: string;
|
||
/** Tool results the fake agent reports, in order, after read-case. */
|
||
tools?: Array<{ name: string; result?: unknown; failed?: boolean }>;
|
||
fitPercent?: number;
|
||
/** Flip the stored range to AFTER when this tool result arrives. */
|
||
flipOn?: string;
|
||
};
|
||
|
||
/**
|
||
* Real `runV9AgentTurn` over a fake Mastra agent stream: read-case, then the
|
||
* listed tools, then one model body. The stored snapshot's credible range is
|
||
* BEFORE until the `flipOn` tool result, AFTER from then on (the batch /
|
||
* compare rescore the diagnosis reproduced).
|
||
*/
|
||
async function runEvidenceTurn(options: TurnCaseOptions) {
|
||
process.env.RECTIFICATION_ALGORITHM_VERSION = "rectification-v5";
|
||
process.env.RECTIFICATION_DECISION_POLICY_VERSION = "rectification-candidate-policy-v2";
|
||
let flipped = false;
|
||
const dossierFor = (range: [string, string]) => dossierFixture({
|
||
latestResult: candidateSnapshotFixture({
|
||
representativeTime: "05:02",
|
||
decisionReceipt: {
|
||
inference_state: inferenceWithRange(range),
|
||
...(typeof options.fitPercent === "number"
|
||
? { event_fit_rate: { matched: 4, total: 5, percent: options.fitPercent, band: "high" } }
|
||
: {}),
|
||
},
|
||
}),
|
||
conversationSummary: conversationSummaryFixture({
|
||
activeFocus: activeFocusFixture({
|
||
askedTurnId: null,
|
||
intent: "collect_method_evidence",
|
||
expectedAnswerSchema: { prompt: "你大概是哪一年搬的家?", collect: true },
|
||
}),
|
||
}),
|
||
});
|
||
const accounting = fakeAccounting({
|
||
...receiptHandlers,
|
||
get_agentic_rectification_case_dossier: () => dossierFor(flipped ? AFTER : BEFORE),
|
||
append_agentic_rectification_turn: () => ({ turn_id: TURN_ID }),
|
||
finalize_agentic_rectification_turn: () => ({ turn_id: TURN_ID, status: "completed", idempotent: false }),
|
||
}, { fallback: () => null });
|
||
const flipOn = options.flipOn ?? "rectification-compare-candidates";
|
||
const tools = options.tools ?? [{ name: "rectification-compare-candidates" }];
|
||
const chunks: Array<{ type: string; payload?: Record<string, unknown> }> = [
|
||
{ type: "start" },
|
||
{ type: "tool-call", payload: { toolName: "rectification-read-case", args: { caseId: CASE_ID } } },
|
||
{ type: "tool-result", payload: { toolName: "rectification-read-case" } },
|
||
];
|
||
for (const tool of tools) {
|
||
chunks.push({ type: "tool-call", payload: { toolName: tool.name, args: { caseId: CASE_ID } } });
|
||
chunks.push(tool.failed
|
||
? { type: "tool-error", payload: { toolName: tool.name, error: new Error("engine_request_failed") } }
|
||
: { type: "tool-result", payload: { toolName: tool.name, result: tool.result ?? {} } });
|
||
}
|
||
chunks.push({ type: "text-delta", payload: { text: options.body } });
|
||
chunks.push({ type: "finish" });
|
||
const agent = {
|
||
stream: async () => ({
|
||
fullStream: (async function* () {
|
||
for (const item of chunks) {
|
||
if (item.type === "tool-result" && item.payload?.toolName === flipOn) flipped = true;
|
||
yield item;
|
||
}
|
||
})(),
|
||
totalUsage: Promise.resolve({ inputTokens: 10, outputTokens: 20 }),
|
||
}),
|
||
getSkill: async () => ({ name: "jyotish-birth-time-rectification", instructions: "skill" }),
|
||
};
|
||
const emitted: Array<Record<string, unknown>> = [];
|
||
const result = await runV9AgentTurn({
|
||
userId: USER_ID,
|
||
caseId: CASE_ID,
|
||
sessionId: SESSION_ID,
|
||
requestId: "aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeee1",
|
||
action: "evidence",
|
||
message: "2016年9月离开家去北京工作",
|
||
modelName: "m",
|
||
accounting: accounting.client as never,
|
||
billing: { reserve: async () => ({ success: true, status: 200 }), complete: async () => true, release: async () => true },
|
||
emit: (event) => { emitted.push(event as never); },
|
||
buildAgent: async () => agent as never,
|
||
});
|
||
const finalize = accounting.calls.filter((call) => call.fn === "finalize_agentic_rectification_turn").at(-1);
|
||
const persisted = String(finalize?.args.p_assistant_message ?? "");
|
||
return { result, emitted, persisted, states: clientStates(emitted) };
|
||
}
|
||
|
||
/** A fact sentence, once shown, must stay in every later client state. */
|
||
function assertNeverWithdrawn(states: readonly string[], fact: string) {
|
||
const first = states.findIndex((state) => state.includes(fact));
|
||
if (first < 0) return;
|
||
for (const state of states.slice(first)) {
|
||
assert.ok(state.includes(fact), `fact withdrawn after being shown: ${JSON.stringify(states)}`);
|
||
}
|
||
}
|
||
|
||
test("BUG-1055 repro 1: two-sentence body keeps the server range sentence, stream ends where persist does", async () => {
|
||
const { result, persisted, states } = await runEvidenceTurn({
|
||
body: "记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。",
|
||
});
|
||
assert.equal(result.ok, true);
|
||
// Before: the trim kept only the first sentence and the range sentence was cut.
|
||
assert.equal(persisted, `记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。${RANGE_SENTENCE}`);
|
||
assert.equal(result.answerText, persisted);
|
||
assert.equal(states.at(-1), persisted);
|
||
assertNeverWithdrawn(states, RANGE_SENTENCE);
|
||
});
|
||
|
||
test("BUG-1055 repro 1b: three-sentence body is trimmed to one, range sentence still whole", async () => {
|
||
const { persisted, states } = await runEvidenceTurn({
|
||
body: "记下了:2016 年 9 月去北京工作。这件事我拿去和星盘对照了。接下来再对一件事。",
|
||
});
|
||
assert.equal(persisted, `记下了:2016 年 9 月去北京工作。${RANGE_SENTENCE}`);
|
||
assert.equal(states.at(-1), persisted);
|
||
assertNeverWithdrawn(states, RANGE_SENTENCE);
|
||
});
|
||
|
||
test("BUG-1055 repro 2: a first sentence carrying the stale range is dropped whole, not persisted", async () => {
|
||
const { persisted, states } = await runEvidenceTurn({
|
||
body: "记下了:2016 年 9 月去北京工作,目前范围在 04:50–05:10。接下来再对一件事。",
|
||
});
|
||
// Before: 「记下了:2016 年 9 月去北京工作,目前范围在 04:50–05:10。」was persisted.
|
||
assert.doesNotMatch(persisted, /目前范围在 04:50–05:10/);
|
||
assert.equal(persisted, `接下来再对一件事。${RANGE_SENTENCE}`);
|
||
assert.equal(states.at(-1), persisted);
|
||
assertNeverWithdrawn(states, RANGE_SENTENCE);
|
||
});
|
||
|
||
test("BUG-1055: an invented fit rate sentence is dropped; a server fit rate sentence is kept", async () => {
|
||
const invented = await runEvidenceTurn({
|
||
body: "记下了:2016 年 9 月去北京工作。目前吻合率 90%。",
|
||
fitPercent: 80,
|
||
});
|
||
assert.equal(invented.persisted, `记下了:2016 年 9 月去北京工作。${RANGE_SENTENCE}`);
|
||
const grounded = await runEvidenceTurn({
|
||
body: "记下了:2016 年 9 月去北京工作。目前吻合率 80%。",
|
||
fitPercent: 80,
|
||
});
|
||
assert.equal(grounded.persisted, `记下了:2016 年 9 月去北京工作。目前吻合率 80%。${RANGE_SENTENCE}`);
|
||
});
|
||
|
||
test("BUG-1055: the post-rescore range and representative minute may be repeated; only mismatches go", async () => {
|
||
const kept = await runEvidenceTurn({
|
||
body: "记下了:2016 年 9 月去北京工作。代表分钟是 05:02,范围 04:55–05:02。",
|
||
});
|
||
assert.equal(kept.persisted, `记下了:2016 年 9 月去北京工作。代表分钟是 05:02,范围 04:55–05:02。${RANGE_SENTENCE}`);
|
||
});
|
||
|
||
test("BUG-1055: when the whitelist leaves no model body the batch recap stands in", async () => {
|
||
const { persisted } = await runEvidenceTurn({
|
||
body: "目前范围 04:50–05:10,吻合率 90%。",
|
||
tools: [{
|
||
name: "rectification-record-evidence-batch",
|
||
result: {
|
||
items: [{ index: 0, outcome: "accepted" }],
|
||
accepted_recaps: [{ display_date_label: "2016 年 9 月", event_phrase: "去北京工作" }],
|
||
accepted_count: 1,
|
||
rescore: { status: "completed" },
|
||
},
|
||
}],
|
||
flipOn: "rectification-record-evidence-batch",
|
||
});
|
||
assert.equal(persisted, `记下了:2016 年 9 月 去北京工作。${RANGE_SENTENCE}`);
|
||
});
|
||
|
||
test("BUG-1055: 这次没有重新比较 survives a multi-sentence body and replaces the range sentence", async () => {
|
||
const { persisted, states } = await runEvidenceTurn({
|
||
body: "记下了:2016 年 9 月去北京工作。这件事拿去对照了。还有一句。",
|
||
tools: [{
|
||
name: "rectification-record-evidence-batch",
|
||
result: {
|
||
items: [{ index: 0, outcome: "accepted" }],
|
||
accepted_recaps: [{ display_date_label: "2016 年 9 月", event_phrase: "去北京工作" }],
|
||
accepted_count: 1,
|
||
rescore: { status: "failed", error_code: "engine_request_failed" },
|
||
},
|
||
}],
|
||
flipOn: "never",
|
||
});
|
||
assert.equal(persisted, `记下了:2016 年 9 月去北京工作。${RECTIFICATION_USER_COPY.rescoreSkipped}`);
|
||
assert.equal(states.at(-1), persisted);
|
||
assertNeverWithdrawn(states, RECTIFICATION_USER_COPY.rescoreSkipped);
|
||
});
|
||
|
||
test("BUG-1055: 候选比较这次没跑成 survives a multi-sentence body", async () => {
|
||
const { persisted, states } = await runEvidenceTurn({
|
||
body: "2016 年 9 月去北京工作这件事拿去对照了。它落在那年的变动里。还有一句。",
|
||
tools: [{ name: "rectification-compare-candidates", failed: true }],
|
||
flipOn: "never",
|
||
});
|
||
assert.equal(persisted, `2016 年 9 月去北京工作这件事拿去对照了。\n\n${RECTIFICATION_USER_COPY.compareFailedRetry}`);
|
||
assert.equal(states.at(-1), persisted);
|
||
assertNeverWithdrawn(states, RECTIFICATION_USER_COPY.compareFailedRetry);
|
||
});
|
||
|
||
test("P3 whitelist is a gate on whole sentences: clocks, ranges and percentages", () => {
|
||
const whitelist = spokenFactWhitelist({
|
||
credibleRange: ["04:55", "05:02"],
|
||
representativeTime: "04:58",
|
||
fitPercent: 79.6,
|
||
});
|
||
assert.deepEqual(whitelist.percents, [80]);
|
||
assert.equal(sentenceNumbersGrounded("范围 04:55–05:02。", whitelist), true);
|
||
assert.equal(sentenceNumbersGrounded("范围 04:55 到 05:02。", whitelist), true);
|
||
assert.equal(sentenceNumbersGrounded("代表分钟 04:58。", whitelist), true);
|
||
assert.equal(sentenceNumbersGrounded("吻合率 80%。", whitelist), true);
|
||
assert.equal(sentenceNumbersGrounded("吻合率百分之80。", whitelist), true);
|
||
assert.equal(sentenceNumbersGrounded("2016 年 9 月入学。", whitelist), true);
|
||
assert.equal(sentenceNumbersGrounded("范围 04:50–05:02。", whitelist), false);
|
||
assert.equal(sentenceNumbersGrounded("范围 04:55–05:10。", whitelist), false);
|
||
assert.equal(sentenceNumbersGrounded("代表分钟 05:00。", whitelist), false);
|
||
assert.equal(sentenceNumbersGrounded("吻合率 90%。", whitelist), false);
|
||
assert.equal(sentenceNumbersGrounded("吻合率 80.5%。", whitelist), false);
|
||
// Whole sentence out; the neighbours stay untouched (no in-sentence deletion).
|
||
assert.equal(
|
||
dropUngroundedFactSentences("记下了:2016 年入学。范围在 09:00–09:05,吻合率 90%。还有吗", whitelist),
|
||
"记下了:2016 年入学。还有吗",
|
||
);
|
||
assert.equal(dropUngroundedFactSentences("范围在 09:00–09:05。", spokenFactWhitelist({})), "");
|
||
});
|