T5 of TASK-rectification-grounding-20260927. - rectification-offer-candidates returned the full projection (~100 KB on the public AA case: *_pre_inference, 31 KB contrast packet, probe lists). It now returns the compare stripping (offerModelProjection); 100,012 → 28,799 bytes. - agentVisibleLatestProjection also drops engine_indistinguishable_width_minutes (audit-only since BUG-593), keeps the verification Markdown once (skill_verification_report; range_delivery.verification_markdown was a byte-identical copy) and shows the slim candidate list. - Receipt fingerprints stay over the full payloads; the case API projection (latestResultToolProjection) is unchanged. Contract tests on a real local engine response (public AA chart). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
143 lines
7.0 KiB
TypeScript
143 lines
7.0 KiB
TypeScript
/**
|
|
* TASK-rectification-grounding-20260927 T5 (R1, BUG-593 side path): what
|
|
* compare and offer hand the model. Fixture: a real local engine response for
|
|
* a public AA chart (9 candidates), AGENTS §7.4.
|
|
*
|
|
* - offer used to return the full projection (~100 KB): `*_pre_inference`,
|
|
* the contrast packet (31 KB) and the probe lists. It now gets the compare
|
|
* stripping.
|
|
* - the verification Markdown is kept once (skill_verification_report);
|
|
* range_delivery carried a byte-identical copy.
|
|
* - `engine_indistinguishable_width_minutes` is audit-only (BUG-593) and no
|
|
* longer reaches the model.
|
|
* - batch returns the post-rescore range and whether this turn delivers.
|
|
* The full projection used by the case API and the stored receipts is
|
|
* unchanged.
|
|
*/
|
|
import assert from "node:assert/strict";
|
|
import { readFileSync } from "node:fs";
|
|
import test from "node:test";
|
|
import {
|
|
CASE_ID,
|
|
TURN_ID,
|
|
USER_ID,
|
|
dossierFixture,
|
|
fakeAccounting,
|
|
receiptHandlers,
|
|
} from "./rectification-v9-test-support.ts";
|
|
import {
|
|
AA_EVIDENCE_ROWS,
|
|
AA_RANGE,
|
|
AA_TURNS,
|
|
buildGoldenLatest,
|
|
setFocusHandler,
|
|
stubGoldenFetch,
|
|
} from "./rectification-grounding-support.ts";
|
|
import {
|
|
batchRangeAfterRescore,
|
|
createRectificationV9Tools,
|
|
latestResultToolProjection,
|
|
offerModelProjection,
|
|
} from "../src/mastra/rectification-v9-tools.ts";
|
|
import { parseV9CaseDossier } from "../src/lib/rectification-agentic/v9/tool-service.ts";
|
|
import { decideFromDossier } from "../src/lib/rectification-agentic/v9/decision-from-dossier.ts";
|
|
|
|
const bytes = (value: unknown) => Buffer.byteLength(JSON.stringify(value), "utf8");
|
|
|
|
const MODEL_HIDDEN_KEYS = [
|
|
"candidate_contrast_packet",
|
|
"discriminating_event_probes",
|
|
"event_clarification_probes",
|
|
"evidence_collection_probes",
|
|
"candidate_contrast_opportunities",
|
|
"dasha_agreement_pre_inference",
|
|
"representative_time_pre_inference",
|
|
"house_table_pre_inference",
|
|
"natal_recast_pre_inference",
|
|
"engine_indistinguishable_width_minutes",
|
|
] as const;
|
|
|
|
function assertModelView(view: Record<string, unknown>) {
|
|
for (const key of MODEL_HIDDEN_KEYS) assert.equal(key in view, false, key);
|
|
const report = view.skill_verification_report as { markdown?: string } | undefined;
|
|
assert.ok(report?.markdown && report.markdown.length > 200, "report Markdown kept once");
|
|
const delivery = view.range_delivery as Record<string, unknown> | undefined;
|
|
if (delivery) assert.equal("verification_markdown" in delivery, false);
|
|
const needle = JSON.stringify(report.markdown.slice(0, 80)).slice(1, -1);
|
|
assert.equal(JSON.stringify(view).split(needle).length - 1, 1, "Markdown appears exactly once");
|
|
const inference = view.inference_state as { candidates?: Array<Record<string, unknown>> } | null;
|
|
for (const candidate of inference?.candidates ?? []) {
|
|
assert.deepEqual(Object.keys(candidate), ["time", "score", "status", "cluster_range"]);
|
|
}
|
|
}
|
|
|
|
test("R1: offer's model view drops pre-inference, packet, probes, duplicate Markdown and engine width", async () => {
|
|
const { latest } = await buildGoldenLatest();
|
|
const parsed = parseV9CaseDossier(dossierFixture({ evidence: AA_EVIDENCE_ROWS, latestResult: latest, turns: AA_TURNS as never, candidateRange: AA_RANGE as never }))!;
|
|
const decision = decideFromDossier(parsed as never);
|
|
const full = latestResultToolProjection(parsed.latestResult as never, decision as never) as Record<string, unknown>;
|
|
// The full projection (case API, receipts) keeps its audit fields.
|
|
assert.ok("candidate_contrast_packet" in full);
|
|
assert.ok("engine_indistinguishable_width_minutes" in full);
|
|
const view = offerModelProjection({ ...full, candidate_snapshot: { is_current: true } });
|
|
assertModelView(view);
|
|
// Conclusion fields the delivery turn needs are still there.
|
|
for (const key of ["candidates", "representative_time", "credible_range", "session_outcome", "can_adopt", "event_fit_rate", "range_delivery"]) {
|
|
assert.ok(key in view, key);
|
|
}
|
|
console.log(JSON.stringify({ scope: "R1 offer model view", before_bytes: bytes(full), after_bytes: bytes(view) }));
|
|
assert.ok(bytes(view) < bytes(full) / 3);
|
|
});
|
|
|
|
test("R1: compare's model view (real tool, golden engine response) keeps the Markdown once and no engine width", async () => {
|
|
const { latest, compute } = await buildGoldenLatest();
|
|
const restore = stubGoldenFetch();
|
|
try {
|
|
const accounting = fakeAccounting({
|
|
...receiptHandlers,
|
|
get_agentic_rectification_case_dossier: () => dossierFixture({
|
|
evidence: AA_EVIDENCE_ROWS,
|
|
latestResult: latest,
|
|
turns: AA_TURNS as never,
|
|
candidateRange: AA_RANGE as never,
|
|
}),
|
|
get_agentic_rectification_case_compute: () => compute,
|
|
set_agentic_rectification_conversation_focus: setFocusHandler,
|
|
refresh_agentic_rectification_vedastro_validation: () => ({ result_id: latest.result_id, decision_receipt: latest.decision_receipt }),
|
|
}, { fallback: () => null });
|
|
const tools = createRectificationV9Tools({
|
|
userId: USER_ID,
|
|
caseId: CASE_ID,
|
|
turnId: TURN_ID,
|
|
userMessage: "2000 年拿了奖",
|
|
accounting: accounting.client as never,
|
|
}) as unknown as Record<string, { execute(input: unknown): Promise<Record<string, unknown>> }>;
|
|
await tools["rectification-read-case"].execute({ caseId: CASE_ID });
|
|
const compare = await tools["rectification-compare-candidates"].execute({ caseId: CASE_ID });
|
|
assertModelView(compare);
|
|
console.log(JSON.stringify({ scope: "R1 compare model view", after_bytes: bytes(compare) }));
|
|
} finally {
|
|
restore();
|
|
}
|
|
});
|
|
|
|
test("T1/R3: batch range_after_rescore names the post-rescore range and whether this turn delivers", async () => {
|
|
const { latest } = await buildGoldenLatest();
|
|
const parsed = parseV9CaseDossier(dossierFixture({ evidence: AA_EVIDENCE_ROWS, latestResult: latest, candidateRange: AA_RANGE as never }))!;
|
|
const decision = decideFromDossier(parsed as never);
|
|
const withQuestion = batchRangeAfterRescore(decision as never, latest.decision_receipt, { question_id: "q" });
|
|
assert.deepEqual(withQuestion?.credible_range, decision.credibleRange);
|
|
assert.equal(withQuestion?.representative_time, decision.representativeTime ?? null);
|
|
assert.equal(withQuestion?.delivers_range_this_turn, false);
|
|
const deliverable = { ...decision, canAdopt: true, sessionOutcome: "adopt_representative" };
|
|
assert.equal(batchRangeAfterRescore(deliverable as never, latest.decision_receipt, null)?.delivers_range_this_turn, true);
|
|
assert.equal(batchRangeAfterRescore(null, latest.decision_receipt, null), null);
|
|
const tools = readFileSync(new URL("../src/mastra/rectification-v9-tools.ts", import.meta.url), "utf8");
|
|
// The receipt keeps hashing the pre-existing shape; only the returned value grows.
|
|
const batchReceipt = tools.indexOf("executedMethods: [...rescore.executedMethods],");
|
|
const batchReturn = tools.indexOf("range_after_rescore: rescore.rangeAfter");
|
|
assert.ok(batchReceipt > 0 && batchReturn > batchReceipt);
|
|
assert.match(tools.slice(batchReceipt - 120, batchReceipt), /resultFingerprint: hashResult\(projection\)/);
|
|
assert.match(tools, /return offerModelProjection\(payload\);/);
|
|
});
|