From 13d93020be7615730e465ff286b860675f50c725 Mon Sep 17 00:00:00 2001 From: Jesse_Chen Date: Sun, 27 Sep 2026 02:59:31 +0800 Subject: [PATCH] fix(rectification): read-case keeps conversation memory with many candidates (BUG-1057) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit T3 of TASK-rectification-grounding-20260927 (red line 3). The turn-decision read-case is the model's only conversation memory; with about 7+ candidates (9 in the public AA case) it exceeded 6 KB and cleared recent_turns and relevant_evidence_summary first. Candidates in the model-visible inference now carry time / score / status / cluster_range only (the last round names candidates by time), candidate_summary.candidates (a duplicate) is gone, and over budget the order is: drop cluster ranges → keep the best six candidates → shorten turns/evidence → clear them. Stored inference and receipts are untouched. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8 --- .../rectification-agentic/v9/turn-decision.ts | 116 +++++++++++-- ...ation-grounding-read-case-20260927.test.ts | 152 ++++++++++++++++++ 2 files changed, 256 insertions(+), 12 deletions(-) create mode 100644 frontend/tests/rectification-grounding-read-case-20260927.test.ts diff --git a/frontend/src/lib/rectification-agentic/v9/turn-decision.ts b/frontend/src/lib/rectification-agentic/v9/turn-decision.ts index 4f5e703b..bb6d8044 100644 --- a/frontend/src/lib/rectification-agentic/v9/turn-decision.ts +++ b/frontend/src/lib/rectification-agentic/v9/turn-decision.ts @@ -21,6 +21,15 @@ export const TURN_DECISION_MAX_BYTES = 6 * 1024; export const TURN_DECISION_RECENT_TURNS = 6; export const TURN_DECISION_TURN_TEXT_LIMIT = 400; export const TURN_DECISION_EVIDENCE_LIMIT = 6; +/** + * The only fields a candidate carries in the model-visible inference + * (BUG-1057). `score` is the inference probability; the model must still call + * it relative support, never a probability (Skill §9). Candidate ids, window + * positions and raw posteriors stay in the stored receipt. + */ +export const TURN_DECISION_CANDIDATE_FIELDS = ["time", "score", "status", "cluster_range"] as const; +/** When the payload is over budget, at most this many candidates survive (best first). */ +export const TURN_DECISION_BUDGET_CANDIDATES = 6; export type ReadCaseProjection = "turn_decision" | "full_diagnostics"; @@ -59,6 +68,52 @@ function clipText(value: string | null | undefined, max: number): string | null return text.length <= max ? text : `${text.slice(0, max)}…`; } +function roundScore(value: unknown): number | null { + return typeof value === "number" && Number.isFinite(value) ? Math.round(value * 1000) / 1000 : null; +} + +/** + * Model-visible slice of `compactInferenceProjection` (BUG-1057): candidates + * keep time / score / status / cluster_range only, and the last round names + * candidates by time instead of by id. Everything else is unchanged. The + * stored inference state and receipts are not touched. + */ +export function modelVisibleInference( + compact: Record | null | undefined, +): Record | null { + if (!compact) return null; + const rows = Array.isArray(compact.candidates) ? compact.candidates as Array> : []; + const timeById = new Map(); + for (const row of rows) { + if (typeof row.id === "string" && typeof row.time === "string") timeById.set(row.id, row.time); + } + const candidates = rows.map((row) => ({ + time: row.time, + score: roundScore(row.probability), + status: row.status, + cluster_range: row.cluster_range, + })); + const round = compact.last_inference_round && typeof compact.last_inference_round === "object" + ? compact.last_inference_round as Record + : null; + const lastRound = round + ? { + kind: round.kind, + entropy_before: round.entropy_before, + entropy_after: round.entropy_after, + eliminated_times: (Array.isArray(round.eliminated_ids) ? round.eliminated_ids : []) + .flatMap((id) => (typeof id === "string" && timeById.has(id) ? [timeById.get(id)!] : [])), + score_deltas: Object.fromEntries( + Object.entries(round.score_deltas && typeof round.score_deltas === "object" + ? round.score_deltas as Record + : {}) + .flatMap(([id, delta]) => (timeById.has(id) ? [[timeById.get(id)!, roundScore(delta)]] : [])), + ), + } + : null; + return { ...compact, candidates, last_inference_round: lastRound }; +} + function looksLikeChoiceSchema(schema: Readonly> | null | undefined): boolean { if (!schema) return false; const choice = schema.choice; @@ -192,11 +247,6 @@ export function projectTurnDecision( const decision = decideFromDossier(dossier, { currentEvidenceFingerprint: evidenceLedgerFingerprint(dossier.evidence), }); - const candidates = (dossier.latestResult?.candidates ?? []).slice(0, 6).map((item) => ({ - time: item.time, - relative_support: item.relativeSupport, - rank: item.rank, - })); const recentTurns = dossier.turns.slice(-TURN_DECISION_RECENT_TURNS).flatMap((turn) => { const text = clipText(turn.text, TURN_DECISION_TURN_TEXT_LIMIT); if (!text || (turn.status !== "completed" && !(turn.role === "user" && turn.status === "pending"))) { @@ -217,7 +267,7 @@ export function projectTurnDecision( const renderableQuestion = isRenderableChoiceQuestion(currentQuestion) ? currentQuestion : null; - const inferenceProjection = compactInferenceProjection(inference); + const inferenceProjection = modelVisibleInference(compactInferenceProjection(inference)); const payload: Record = { projection: "turn_decision", case_id: dossier.case.caseId, @@ -230,7 +280,7 @@ export function projectTurnDecision( selection_allowed: decision.selectionAllowed, completion_status: decision.completionStatus, validated: decision.validated, - candidates, + // BUG-1057: the per-candidate list lives once, in inference.candidates. entropy: inference?.entropy ?? null, }, inference: renderableQuestion || !inferenceProjection @@ -265,15 +315,57 @@ export function projectTurnDecision( return enforceTurnDecisionBudget(payload); } +function withInferenceCandidates( + payload: Record, + shape: (candidates: Array>) => { candidates: Array>; omitted?: number }, +): Record { + const inference = payload.inference && typeof payload.inference === "object" && !Array.isArray(payload.inference) + ? payload.inference as Record + : null; + if (!inference || !Array.isArray(inference.candidates)) return payload; + const shaped = shape(inference.candidates as Array>); + return { + ...payload, + inference: { + ...inference, + candidates: shaped.candidates, + ...(shaped.omitted ? { candidates_omitted: shaped.omitted } : {}), + }, + }; +} + +/** + * Over budget, candidate detail goes first and the conversation memory last + * (BUG-1057). The model gets no message history; `recent_turns` and + * `relevant_evidence_summary` are its only memory of the conversation. + * Order: drop each candidate's cluster_range → keep the best + * TURN_DECISION_BUDGET_CANDIDATES candidates (active before eliminated) → + * shorten turns/evidence → clear them (last resort, `truncated: true`). + */ export function enforceTurnDecisionBudget(payload: Record): Record { if (utf8Bytes(payload) <= TURN_DECISION_MAX_BYTES) return payload; + const noRanges = withInferenceCandidates(payload, (candidates) => ({ + candidates: candidates.map(({ cluster_range: _range, ...rest }) => rest), + })); + if (utf8Bytes(noRanges) <= TURN_DECISION_MAX_BYTES) return noRanges; + const topCandidates = withInferenceCandidates(noRanges, (candidates) => { + const ranked = [...candidates].sort((left, right) => { + const leftActive = left.status === "eliminated" ? 1 : 0; + const rightActive = right.status === "eliminated" ? 1 : 0; + if (leftActive !== rightActive) return leftActive - rightActive; + return (Number(right.score) || 0) - (Number(left.score) || 0); + }); + const kept = ranked.slice(0, TURN_DECISION_BUDGET_CANDIDATES); + return { candidates: kept, omitted: candidates.length - kept.length }; + }); + if (utf8Bytes(topCandidates) <= TURN_DECISION_MAX_BYTES) return topCandidates; const shrunk = { - ...payload, - recent_turns: Array.isArray(payload.recent_turns) - ? (payload.recent_turns as unknown[]).slice(-4) + ...topCandidates, + recent_turns: Array.isArray(topCandidates.recent_turns) + ? (topCandidates.recent_turns as unknown[]).slice(-4) : [], - relevant_evidence_summary: Array.isArray(payload.relevant_evidence_summary) - ? (payload.relevant_evidence_summary as unknown[]).slice(-3) + relevant_evidence_summary: Array.isArray(topCandidates.relevant_evidence_summary) + ? (topCandidates.relevant_evidence_summary as unknown[]).slice(-3) : [], }; if (utf8Bytes(shrunk) <= TURN_DECISION_MAX_BYTES) return shrunk; diff --git a/frontend/tests/rectification-grounding-read-case-20260927.test.ts b/frontend/tests/rectification-grounding-read-case-20260927.test.ts new file mode 100644 index 00000000..0eee733d --- /dev/null +++ b/frontend/tests/rectification-grounding-read-case-20260927.test.ts @@ -0,0 +1,152 @@ +/** + * BUG-1057 (TASK-rectification-grounding-20260927 T3, red line 3): the + * turn-decision read-case is the model's only conversation memory (the model + * messages carry no history). With 9 candidates it used to exceed 6 KB and + * clear `recent_turns` and `relevant_evidence_summary` first. Now candidates + * carry time / score / status / cluster_range only, the duplicate + * `candidate_summary.candidates` is gone, and over budget the candidate detail + * goes before the conversation. + * + * Fixture: a real local engine response for a public AA chart (9 candidates). + */ +import assert from "node:assert/strict"; +import test from "node:test"; +import { + CASE_ID, + TURN_ID, + USER_ID, + dossierFixture, + fakeAccounting, + receiptHandlers, +} from "./rectification-v9-test-support.ts"; +import { + AA_EVIDENCE_ROWS, + AA_RANGE, + buildGoldenLatest, + setFocusHandler, + stubGoldenFetch, +} from "./rectification-grounding-support.ts"; +import { createRectificationV9Tools } from "../src/mastra/rectification-v9-tools.ts"; +import { + TURN_DECISION_CANDIDATE_FIELDS, + TURN_DECISION_MAX_BYTES, + enforceTurnDecisionBudget, + turnDecisionByteLength, +} from "../src/lib/rectification-agentic/v9/turn-decision.ts"; + +const SIX_TURNS = [ + ["user", "2000 年拿了一个很重要的表演奖,那一年整个人的事业一下子起来了,很多人开始找我合作。"], + ["assistant", "记下了:2000 年获奖。"], + ["user", "2013 年做了一次预防性的大手术,前后休养了大半年,那年身体状态很差。"], + ["assistant", "记下了:2013 年做手术。"], + ["user", "2014 年 8 月结婚,是在法国办的婚礼,家里人都去了,算是那几年最大的一件事。"], + ["assistant", "记下了:2014 年 8 月结婚。"], +].map(([role, text], index) => ({ + id: `77777777-7777-4777-8777-77777777777${index}`, + role, + text, + status: "completed", + created_at: `2026-08-12T10:0${index}:00.000Z`, + completed_at: `2026-08-12T10:0${index}:05.000Z`, +})); + +async function readCaseFor(turns: unknown[]) { + const { latest, compute } = await buildGoldenLatest(); + const restore = stubGoldenFetch(); + try { + const accounting = fakeAccounting({ + ...receiptHandlers, + get_agentic_rectification_case_dossier: () => dossierFixture({ + evidence: AA_EVIDENCE_ROWS, + latestResult: latest, + turns: turns as never, + candidateRange: AA_RANGE as never, + }), + get_agentic_rectification_case_compute: () => compute, + set_agentic_rectification_conversation_focus: setFocusHandler, + }, { fallback: () => null }); + const tools = createRectificationV9Tools({ + userId: USER_ID, + caseId: CASE_ID, + turnId: TURN_ID, + userMessage: "2014 年结婚", + accounting: accounting.client as never, + }) as Record> }>; + const readCase = await tools["rectification-read-case"].execute({ caseId: CASE_ID }); + return { readCase, candidateCount: latest.candidates.length }; + } finally { + restore(); + } +} + +test("red line 3: 9 candidates and six turns keep recent_turns and relevant_evidence_summary", async () => { + const { readCase, candidateCount } = await readCaseFor(SIX_TURNS); + assert.equal(candidateCount, 9); + assert.ok(turnDecisionByteLength(readCase) <= TURN_DECISION_MAX_BYTES); + assert.equal(readCase.truncated, undefined); + const turns = readCase.recent_turns as Array<{ role: string; text: string }>; + assert.equal(turns.length, 6); + assert.match(turns[4].text, /2014 年 8 月结婚/); + const evidence = readCase.relevant_evidence_summary as unknown[]; + assert.equal(evidence.length, 3); + const inference = readCase.inference as { candidates: Array> }; + assert.equal(inference.candidates.length, 9); + for (const candidate of inference.candidates) { + assert.deepEqual(Object.keys(candidate), [...TURN_DECISION_CANDIDATE_FIELDS]); + assert.match(String(candidate.time), /^\d{2}:\d{2}$/); + } + const summary = readCase.candidate_summary as Record; + assert.equal("candidates" in summary, false); + const text = JSON.stringify(readCase); + assert.doesNotMatch(text, /cluster_intervals|window_offset_minutes|posterior_score|candidate_date/); + console.log(JSON.stringify({ scope: "BUG-1057 read-case size", bytes: turnDecisionByteLength(readCase) })); +}); + +test("red line 3: over budget, candidate detail goes before recent turns and evidence", () => { + const clock = (index: number) => `${String(4 + Math.floor(index / 60)).padStart(2, "0")}:${String(index % 60).padStart(2, "0")}`; + const candidates = Array.from({ length: 200 }, (_, index) => ({ + time: clock(index), + score: index === 7 ? 0.5 : 0.01, + status: index % 3 === 0 ? "eliminated" : "active", + cluster_range: [clock(index), clock(index)], + })); + const turns = Array.from({ length: 6 }, (_, index) => ({ role: index % 2 ? "assistant" : "user", text: `第 ${index} 句话`.repeat(10) })); + const evidence = Array.from({ length: 6 }, (_, index) => ({ domain: "career", summary: `事件 ${index}`.repeat(8) })); + const payload = { + projection: "turn_decision", + inference: { credible_range: ["05:00", "05:39"], candidates }, + recent_turns: turns, + relevant_evidence_summary: evidence, + }; + assert.ok(turnDecisionByteLength(payload) > TURN_DECISION_MAX_BYTES); + const trimmed = enforceTurnDecisionBudget(payload); + assert.ok(turnDecisionByteLength(trimmed) <= TURN_DECISION_MAX_BYTES); + assert.deepEqual(trimmed.recent_turns, turns); + assert.deepEqual(trimmed.relevant_evidence_summary, evidence); + assert.equal(trimmed.truncated, undefined); + const kept = (trimmed.inference as { candidates: Array>; candidates_omitted?: number }); + assert.ok(kept.candidates.every((candidate) => !("cluster_range" in candidate))); + assert.ok(kept.candidates.length < 200); + assert.equal(kept.candidates[0]?.time, "04:07"); + assert.ok(kept.candidates.every((candidate) => candidate.status === "active")); + assert.equal(kept.candidates_omitted, 200 - kept.candidates.length); +}); + +test("red line 3: first budget step only drops cluster ranges; every candidate and turn stays", () => { + const clock = (index: number) => `${String(4 + Math.floor(index / 60)).padStart(2, "0")}:${String(index % 60).padStart(2, "0")}`; + const candidates = Array.from({ length: 90 }, (_, index) => ({ + time: clock(index), + score: 0.01, + status: "active", + cluster_range: [clock(index), clock(index)], + })); + const turns = [{ role: "user", text: "2014 年 8 月结婚。" }]; + const payload = { inference: { candidates }, recent_turns: turns, relevant_evidence_summary: [] }; + assert.ok(turnDecisionByteLength(payload) > TURN_DECISION_MAX_BYTES); + const trimmed = enforceTurnDecisionBudget(payload); + const kept = trimmed.inference as { candidates: Array>; candidates_omitted?: number }; + assert.equal(kept.candidates.length, 90); + assert.equal(kept.candidates_omitted, undefined); + assert.ok(kept.candidates.every((candidate) => !("cluster_range" in candidate))); + assert.deepEqual(trimmed.recent_turns, turns); +});