fix(rectification): enforce persisted convergence budgets

This commit is contained in:
Jesse_Chen
2026-08-31 05:50:17 +08:00
parent 79304c31fc
commit 86ba17ee18
5 changed files with 190 additions and 5 deletions
@@ -11,6 +11,11 @@ export type DynamicStopInput = {
readonly forcedReason: "user_finished" | "generation_unavailable" | null;
};
// Keep the existing dynamic-stop safety cap as a shared server-owned value.
// Rectification's authoritative decision consumes the same cap; it does not
// define a second threshold.
export const EFFECTIVE_ANSWER_SAFETY_CAP = 10;
export type DynamicStopDecision =
| {
readonly kind: "finish";
@@ -42,7 +47,9 @@ export function decideDynamicStop(input: DynamicStopInput): DynamicStopDecision
if (input.result?.confidence === "high" && input.result.canApply) {
return { kind: "finish", reason: "high_confidence", plateauCount };
}
if (input.effectiveAnswerCount >= 10) return { kind: "finish", reason: "safety_cap", plateauCount };
if (input.effectiveAnswerCount >= EFFECTIVE_ANSWER_SAFETY_CAP) {
return { kind: "finish", reason: "safety_cap", plateauCount };
}
if (plateauCount >= 2) return { kind: "finish", reason: "plateau", plateauCount };
if (input.usefulOpportunityCount === 0) return { kind: "finish", reason: "no_information_gain", plateauCount };
if (input.repeatedOnly) return { kind: "finish", reason: "repeated_partition", plateauCount };
@@ -27,6 +27,9 @@ export type DecideNextActionInput = Readonly<{
holdoutValidation?: HoldoutValidationStatus;
accepted?: boolean;
inferenceCredibleRange?: readonly [string, string] | null;
inferenceRounds?: number;
effectiveAnswerCount?: number;
plateauRounds?: number;
}>;
export type RectificationNextAction = Readonly<{
@@ -55,6 +58,9 @@ export function decideNextAction(input: DecideNextActionInput): RectificationNex
inferenceCredibleRange: input.inferenceCredibleRange,
engineAcceptAllowed: input.selectionAllowed,
engineProposeAllowed: input.proposeAllowed,
inferenceRounds: input.inferenceRounds,
effectiveAnswerCount: input.effectiveAnswerCount,
plateauRounds: input.plateauRounds,
});
return { type: decision.nextAction, separation: decision.separation, probe: decision.probe };
}
@@ -14,7 +14,13 @@ import {
} from "./candidate-separation.ts";
import { rangeFromTimes } from "./credible-range.ts";
import type { DroppedProbe } from "../v9/probe-question-contract.ts";
import type { RectificationPhase, ResultStatus } from "./types.ts";
import { EFFECTIVE_ANSWER_SAFETY_CAP } from "../../birth-time-dynamic-stop-policy.ts";
import { RECTIFICATION_POLICY } from "../../rectification-policy.ts";
import {
DEFAULT_MAX_DISCRIMINATION_ROUNDS,
type RectificationPhase,
type ResultStatus,
} from "./types.ts";
export type RectificationNextActionType =
| "ask_fact_collection"
@@ -81,6 +87,10 @@ export type DecideRectificationInput = Readonly<{
engineAcceptAllowed?: boolean;
engineProposeAllowed?: boolean;
datedMethodCollectOpen?: boolean;
/** Server-persisted inference counters; never infer these from chat text. */
inferenceRounds?: number;
effectiveAnswerCount?: number;
plateauRounds?: number;
}>;
export function decideRectification(input: DecideRectificationInput): RectificationDecision {
@@ -136,7 +146,13 @@ export function decideRectification(input: DecideRectificationInput): Rectificat
}
if (input.snapshotCurrent === false) {
if (probe && !userStopped && input.trainingGateOpen !== false) {
return discriminate(separation, holdout, range, probe);
return discriminateOrExhaust(
input,
separation,
holdout,
range,
probe,
);
}
if (!(userStopped && input.candidateScores.length > 0)) {
return collect(separation, holdout, range, probe);
@@ -144,7 +160,7 @@ export function decideRectification(input: DecideRectificationInput): Rectificat
}
if (!separation.sufficient) {
if (probe && !userStopped) {
return discriminate(separation, holdout, range, probe);
return discriminateOrExhaust(input, separation, holdout, range, probe);
}
return completeWithRange(separation, holdout, range, userStopped ? "user_stopped" : "offer");
}
@@ -153,7 +169,7 @@ export function decideRectification(input: DecideRectificationInput): Rectificat
}
if (holdout === "failed") {
if (probe && !userStopped) {
return discriminate(separation, holdout, range, probe);
return discriminateOrExhaust(input, separation, holdout, range, probe);
}
return completeWithRange(separation, holdout, range, "exhausted");
}
@@ -183,6 +199,25 @@ export function decideRectification(input: DecideRectificationInput): Rectificat
});
}
function budgetExhausted(input: DecideRectificationInput): boolean {
return (input.inferenceRounds ?? 0) >= DEFAULT_MAX_DISCRIMINATION_ROUNDS
|| (input.effectiveAnswerCount ?? 0) >= EFFECTIVE_ANSWER_SAFETY_CAP
|| (input.plateauRounds ?? 0) >= RECTIFICATION_POLICY.maxPlateauRounds;
}
function discriminateOrExhaust(
input: DecideRectificationInput,
separation: CandidateSeparation,
holdout: HoldoutValidationStatus,
range: readonly [string, string] | null,
probe: CandidateDiscriminatorProbe,
): RectificationDecision {
if (budgetExhausted(input)) {
return completeWithRange(separation, holdout, range, "exhausted");
}
return discriminate(separation, holdout, range, probe);
}
function collect(
separation: CandidateSeparation,
holdout: HoldoutValidationStatus,
@@ -355,6 +355,22 @@ export type DecideFromDossierOptions = Readonly<{
birthDate?: string | null;
}>;
function decisionBudgetFromInference(inference: InferenceState | null | undefined) {
const rounds = inference?.rounds ?? [];
let plateauRounds = 0;
for (let index = rounds.length - 1; index >= 0; index -= 1) {
if (rounds[index]?.kind !== "low_information") break;
plateauRounds += 1;
}
return {
// The existing convergence evaluator budgets informative rounds. Keep the
// authoritative decision aligned with that persisted interpretation.
inferenceRounds: rounds.filter((item) => item.kind === "informative").length,
effectiveAnswerCount: inference?.answered_probes.length ?? 0,
plateauRounds,
};
}
function completedProbeForSemanticKey(
semanticKey: string,
packet: CandidateContrastPacket | null | undefined,
@@ -440,6 +456,7 @@ export function decideFromDossier(
sessionOutcome: "collect_evidence",
});
const latest = dossier.latestResult;
const decisionBudget = decisionBudgetFromInference(inference);
const snapshotCurrent = scoreableSnapshotCurrentFromDossier(dossier, options, inference);
const confirmationGate = buildConfirmationGate({
engineConfirmationAllowed: latest?.confirmationAllowed === true,
@@ -480,6 +497,7 @@ export function decideFromDossier(
|| latest?.decisionReceipt?.accept_allowed === true,
engineProposeAllowed: latest?.decisionReceipt?.propose_allowed === true,
datedMethodCollectOpen: datedMethodCollectOpen(collecting.methods),
...decisionBudget,
}),
droppedProbes: gated.dropped,
};
@@ -503,6 +521,7 @@ export function decideAfterInferenceChange(input: {
trainingGateOpen: trainingScoreableGate(input.dossier.evidence).open,
candidateScores: [],
userStopped: input.userStopped,
...decisionBudgetFromInference(null),
});
}
const training = input.state.events.filter((item) => item.usage === "training");
@@ -543,6 +562,7 @@ export function decideAfterInferenceChange(input: {
|| input.dossier.latestResult?.decisionReceipt?.accept_allowed === true,
engineProposeAllowed: input.dossier.latestResult?.decisionReceipt?.propose_allowed === true,
datedMethodCollectOpen: datedMethodCollectOpen(collecting.methods),
...decisionBudgetFromInference(input.state),
}),
droppedProbes: gated.dropped,
};
@@ -0,0 +1,117 @@
import assert from "node:assert/strict";
import test from "node:test";
import { EFFECTIVE_ANSWER_SAFETY_CAP } from "../src/lib/birth-time-dynamic-stop-policy.ts";
import { DEFAULT_MAX_DISCRIMINATION_ROUNDS } from "../src/lib/rectification-agentic/core/types.ts";
import { decideRectification } from "../src/lib/rectification-agentic/core/rectification-decision.ts";
import { RECTIFICATION_POLICY } from "../src/lib/rectification-policy.ts";
import type { CandidateDiscriminatorProbe } from "../src/lib/rectification-agentic/core/candidate-contrast-packet.ts";
const PROBE: CandidateDiscriminatorProbe = {
probeId: "probe-1",
candidateSetVersion: "set-1",
question: "这件事更接近哪一种情况?",
expectedOutcomes: [
{
outcomeId: "yes",
supportsCandidateIds: ["05:00"],
conflictsCandidateIds: ["05:06", "05:07"],
},
{
outcomeId: "no",
supportsCandidateIds: ["05:06", "05:07"],
conflictsCandidateIds: ["05:00"],
},
],
candidateSplitHash: "05:00|05:06|05:07",
informationGain: 0.5,
sourceFeatures: [{ technique: "test", calculationResultId: null }],
domain: "career",
year: 2020,
semanticKey: "career.2020.test",
};
const BASE_INPUT = {
methodCoverageAll: true,
trainingGateOpen: true,
candidateScores: [
{ time: "05:00", score: 34 },
{ time: "05:06", score: 33 },
{ time: "05:07", score: 33 },
],
discriminatorProbe: PROBE,
holdoutValidation: "unavailable" as const,
};
function decideWithBudget(budget: {
inferenceRounds?: number;
effectiveAnswerCount?: number;
plateauRounds?: number;
}) {
return decideRectification({ ...BASE_INPUT, ...budget });
}
test("every persisted discrimination budget terminates before asking another probe", () => {
for (const budget of [
{ inferenceRounds: DEFAULT_MAX_DISCRIMINATION_ROUNDS },
{ effectiveAnswerCount: EFFECTIVE_ANSWER_SAFETY_CAP },
{ plateauRounds: RECTIFICATION_POLICY.maxPlateauRounds },
]) {
const decision = decideWithBudget(budget);
assert.equal(decision.nextAction, "complete_with_range");
assert.equal(decision.sessionOutcome, "completed_with_range");
assert.notEqual(decision.nextAction, "ask_candidate_discriminator");
}
});
test("repeated declined or unsure answers reach the existing plateau terminal", () => {
const decision = decideWithBudget({
inferenceRounds: 0,
effectiveAnswerCount: RECTIFICATION_POLICY.maxPlateauRounds,
plateauRounds: RECTIFICATION_POLICY.maxPlateauRounds,
});
assert.equal(decision.nextAction, "complete_with_range");
assert.equal(decision.sessionOutcome, "completed_with_range");
});
test("exhausted discrimination still delivers the credible candidate range", () => {
const decision = decideWithBudget({
inferenceRounds: DEFAULT_MAX_DISCRIMINATION_ROUNDS,
effectiveAnswerCount: 0,
plateauRounds: 0,
});
assert.equal(decision.resultStatus, "completed_with_range");
assert.equal(decision.sessionOutcome, "completed_with_range");
assert.equal(decision.canOfferRange, true);
assert.equal(decision.canAdopt, true);
assert.deepEqual(decision.credibleRange, ["05:00", "05:07"]);
});
test("additional score evidence never widens the credible range", () => {
const before = decideRectification({
...BASE_INPUT,
candidateScores: [
{ time: "05:00", score: 34 },
{ time: "05:06", score: 33 },
{ time: "05:07", score: 33 },
],
inferenceRounds: DEFAULT_MAX_DISCRIMINATION_ROUNDS,
});
const after = decideRectification({
...BASE_INPUT,
candidateScores: [
{ time: "05:00", score: 42 },
{ time: "05:06", score: 33 },
{ time: "05:07", score: 33 },
],
inferenceRounds: DEFAULT_MAX_DISCRIMINATION_ROUNDS,
});
const width = (range: readonly [string, string] | null) => {
assert.ok(range);
const toMinutes = (time: string) => Number(time.slice(0, 2)) * 60 + Number(time.slice(3, 5));
return toMinutes(range[1]) - toMinutes(range[0]);
};
assert.ok(width(after.credibleRange) <= width(before.credibleRange));
});