fix(rectification): close round A/2 tail gaps in tests, CI, and stale-score reuse
Independent Staging Quality Gate / validate (pull_request) Failing after 6m13s
Independent Staging Quality Gate / publish (pull_request) Has been skipped

Window_scan assertions now match the public from_sign/to_sign contract, the
staging quick gate runs the rectification Python suite, and compare-candidates
rescores when stored policy lags the live engine identity.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Jesse_Chen
2026-09-01 18:17:33 +08:00
co-authored by Cursor
parent 26ea3f06cb
commit c48a965640
9 changed files with 250 additions and 7 deletions
@@ -11,6 +11,10 @@ import {
} from "../src/lib/rectification-agentic/core/snapshot-source.ts";
import type { ConflictProbe } from "../src/lib/rectification-agentic/core/types.ts";
import { contrastPacketFromLatestResult } from "../src/lib/rectification-agentic/v9/decision-from-dossier.ts";
import {
cachedEngineScoreIsReusable,
liveEngineScoringIdentityFromEnv,
} from "../src/lib/rectification-agentic/v9/engine-client.ts";
import {
parseDiscriminatingEventProbes,
parseProspectiveProbes,
@@ -84,6 +88,39 @@ test("scoring policy version changes stale stored snapshots", () => {
assert.equal(storedSnapshotIsCurrent(current, current), true);
});
test("matching fingerprints still rescore when stored policy lags the live engine", () => {
const fingerprints = {
evidenceLedgerFingerprint: "b".repeat(64),
candidateRangeFingerprint: "c".repeat(64),
};
const stored = {
...fingerprints,
algorithmVersion: "rectification-v5-matrix-scoring-6",
policyVersion: "rectification-candidate-policy-v2",
};
assert.equal(
cachedEngineScoreIsReusable(stored, fingerprints, {
algorithmVersion: "rectification-v5-matrix-scoring-6",
policyVersion: "rectification-candidate-policy-v2",
}),
true,
);
assert.equal(
cachedEngineScoreIsReusable(stored, fingerprints, {
algorithmVersion: "rectification-v5-matrix-scoring-7",
policyVersion: "rectification-candidate-policy-v3",
}),
false,
);
const fromEnv = liveEngineScoringIdentityFromEnv({
RECTIFICATION_ENGINE_VERSION: "rectification-v5",
RECTIFICATION_DECISION_POLICY_VERSION: "rectification-candidate-policy-v3",
});
assert.equal(fromEnv.algorithmVersion, null);
assert.equal(fromEnv.policyVersion, "rectification-candidate-policy-v3");
assert.equal(cachedEngineScoreIsReusable(stored, fingerprints, fromEnv), false);
});
test("anchored known_event_quality distinguish probes stay in the public packet", () => {
const parsed = parseDiscriminatingEventProbes([qualityProbe()]);
assert.equal(parsed.length, 1);
@@ -692,7 +692,17 @@ test("compare-candidates reuses a matching fingerprint without calling the engin
const evidenceFp = evidenceLedgerFingerprint(parsed.evidence);
const rangeFp = candidateRangeFingerprint(parsed.case.candidateRange, compute.baselineProfileFingerprint);
const previous = globalThis.fetch;
globalThis.fetch = (async () => {
globalThis.fetch = (async (input: RequestInfo | URL) => {
const url = String(input);
if (url.includes("/api/rectification/v5/versions")) {
return {
ok: true,
json: async () => ({
algorithm_version: "rectification-v5",
decision_policy_version: "rectification-candidate-policy-v2",
}),
} as Response;
}
throw new Error("engine must not run on a matching fingerprint");
}) as unknown as typeof fetch;
try {
@@ -729,6 +739,65 @@ test("compare-candidates reuses a matching fingerprint without calling the engin
}
});
test("compare-candidates rescores when stored policy lags the live engine", async () => {
const rawDossier = dossierFixture();
const parsed = parseV9CaseDossier(rawDossier);
const compute = parseV9ComputeProjection(computeFixture());
assert.ok(parsed);
assert.ok(compute);
assert.ok(parsed.case.candidateRange);
const evidenceFp = evidenceLedgerFingerprint(parsed.evidence);
const rangeFp = candidateRangeFingerprint(parsed.case.candidateRange, compute.baselineProfileFingerprint);
const previous = globalThis.fetch;
let scoreRequested = false;
globalThis.fetch = (async (input: RequestInfo | URL) => {
const url = String(input);
if (url.includes("/api/rectification/v5/versions")) {
return {
ok: true,
json: async () => ({
algorithm_version: "rectification-v5-matrix-scoring-7",
decision_policy_version: "rectification-candidate-policy-v3",
}),
} as Response;
}
if (url.includes("/api/rectification/v5/score")) {
scoreRequested = true;
}
throw new Error("stale policy must rescore");
}) as unknown as typeof fetch;
try {
const accounting = fakeAccounting({
...receiptHandlers,
get_agentic_rectification_case_dossier: () => dossierFixture({
latestResult: {
...candidateSnapshotFixture({ evidenceLedgerFingerprint: evidenceFp }),
candidate_range_fingerprint: rangeFp,
},
}),
get_agentic_rectification_case_compute: () => computeFixture(),
persist_agentic_rectification_candidate_v2: () => {
throw new Error("persist must not run until the live engine scores");
},
});
const tools = createRectificationV9Tools({
userId: USER_ID,
caseId: CASE_ID,
turnId: TURN_ID,
accounting: accounting.client as never,
});
await assert.rejects(
(tools["rectification-compare-candidates"] as unknown as {
execute(input: unknown): Promise<unknown>;
}).execute({ caseId: CASE_ID }),
(error: unknown) => error instanceof Error && error.message.includes("stale policy must rescore"),
);
assert.equal(scoreRequested, true);
} finally {
globalThis.fetch = previous;
}
});
test("compare-candidates refuses to run without scorable evidence", async () => {
const accounting = fakeAccounting({
...receiptHandlers,