fix(rectification): server fact sentences skip the evidence trim; model numbers must match server facts (BUG-1055)

T1 of TASK-rectification-grounding-20260927 (recurrence of BUG-588).
- The attempt no longer streams range-changed / rescore-skipped /
  compare-failed sentences; the finish whitelists and trims the model body,
  then joins the server facts, and emits one final replace equal to the
  persisted text.
- P3 whitelist (spoken-grounding.ts): a model sentence with a clock, clock
  range or percentage that is not this turn's server fact is dropped whole;
  the batch recap stands in when nothing is left.
- record-evidence-batch returns range_after_rescore (post-rescore
  credible_range, representative minute, fit percent, delivers_range_this_turn);
  the receipt fingerprint stays over the old shape.
- System prompt: range is said by the server; the delivery three sentences
  only when the batch says this turn delivers.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
Jesse_Chen
2026-09-27 03:28:27 +08:00
co-authored by Claude Opus 5.5
parent 1937608998
commit 0e0baaa74c
11 changed files with 868 additions and 46 deletions
@@ -21,9 +21,7 @@ import {
isAcceptableOpeningBody,
openingRangeFromCandidateRange,
openingSpokenBody,
withCompareFailedRetryNotice,
withRangeChangedAfterEvidence,
withRescoreSkippedNotice,
stripRangeProgressClaims,
} from "../user-copy";
import { attemptTimeoutForRemainingBudget, canStartRetryAttempt } from "../../rectification-run-budget.ts";
import { stripQuestionSentences, stripVerbalWindowChange } from "./collect-prompt";
@@ -56,8 +54,12 @@ import {
turnExpectsEvidenceWrite,
} from "./host-fallback";
import { applyStepAnswerChunk, createStepAnswerState, flushStepAnswerOnStreamFinish } from "./step-answer";
import { refinementFromDecisionReceipt } from "./refinement-packet";
import { spokenFactWhitelist, type SpokenFactWhitelist } from "./spoken-grounding";
import {
type AttemptOutcome,
type TurnServerFacts,
composeSpokenWithServerFacts,
MAX_ATTEMPTS,
streamFinishReason,
publicToolCallKey,
@@ -150,9 +152,16 @@ export async function streamV9Attempt(
const emittedActivities = new Set<string>();
const repeatedCalls = new Map<string, number>();
let phaseSequence = 0;
const rangeBeforeCompare = previousInferenceFromReceipt(
dossier.latestResult?.decisionReceipt ?? null,
)?.credible_range ?? null;
// BUG-1058: a typed answer to a choice card that also carries a dated event
// persists the choice before this run; the route hands over the range from
// before that answer so one server sentence covers both changes.
const rangeBeforeCompare = options.rangeBeforeTurn !== undefined
? options.rangeBeforeTurn
: previousInferenceFromReceipt(
dossier.latestResult?.decisionReceipt ?? null,
)?.credible_range ?? null;
let serverFacts: TurnServerFacts | null = null;
let spokenFacts: SpokenFactWhitelist | null = null;
const recordPhase = async (phase: string, tool: string | null = null) => {
if (
@@ -411,9 +420,11 @@ export async function streamV9Attempt(
if (flushed.kind === "publish") await publishSpokenStep(flushed.pieces);
}
if (toolTerminalStatus.get("rectification-compare-candidates") === "failed" && !batchRescoreFailed(batchToolResult)) {
await emitVisibleSpoken(withCompareFailedRetryNotice(answerText));
}
// BUG-1055: server fact sentences are not streamed here. They join the
// body after the finish has trimmed and whitelisted it, so a streamed
// fact is never withdrawn by a later replace.
const compareFailed = toolTerminalStatus.get("rectification-compare-candidates") === "failed"
&& !batchRescoreFailed(batchToolResult);
const completeAttempt = async (settleBilling = true): Promise<AttemptOutcome> => {
let inputTokens = 0;
@@ -435,6 +446,14 @@ export async function streamV9Attempt(
errorCode: null,
usage: { inputTokens, outputTokens, ...(cache ? { cache } : {}) },
answerText,
streamedText: visibleEmitted,
serverFacts: serverFacts ?? (compareFailed
? { compareFailed: true, rescoreSkipped: false, rangeBefore: null, rangeAfter: null }
: null),
spokenFacts,
hostRecap: toolTerminalStatus.get("rectification-record-evidence-batch") === "completed"
? composeHostFallbackNarration(batchToolResult ?? {})
: null,
answerDeltas,
phases,
toolsUsed: [...toolsUsed],
@@ -586,21 +605,35 @@ export async function streamV9Attempt(
answerText = openingSpokenBody(openingRange);
}
}
const rangeAfterEvidence = previousInferenceFromReceipt(
const latestInference = previousInferenceFromReceipt(
latestDossier.latestResult?.decisionReceipt ?? null,
)?.credible_range ?? null;
if (batchRescoreFailed(batchToolResult)) {
answerText = withRescoreSkippedNotice(answerText);
} else {
answerText = withRangeChangedAfterEvidence(
answerText,
rangeBeforeCompare,
rangeAfterEvidence,
);
);
const rangeAfterEvidence = latestInference?.credible_range ?? null;
const rescoreSkipped = batchRescoreFailed(batchToolResult);
if (rescoreSkipped) answerText = stripRangeProgressClaims(answerText);
serverFacts = {
compareFailed,
rescoreSkipped,
rangeBefore: rangeBeforeCompare,
rangeAfter: rangeAfterEvidence,
};
// P3: the only clocks and percentages a model sentence may carry are this
// turn's server facts (range, representative minute, event fit rate).
const fitRate = refinementFromDecisionReceipt(latestDossier.latestResult?.decisionReceipt ?? null).event_fit_rate;
const whitelistSources = [
{ credibleRange: rangeAfterEvidence, credibleIntervals: latestInference?.credible_intervals ?? null,
representativeTime: latestInference?.representative_time ?? latestDossier.latestResult?.representativeTime ?? null },
{ credibleRange: decision.credibleRange ?? null, representativeTime: decision.representativeTime ?? null },
].map((source) => spokenFactWhitelist({ ...source, fitPercent: fitRate?.percent ?? null }));
spokenFacts = {
ranges: [...new Set(whitelistSources.flatMap((item) => item.ranges))],
clocks: [...new Set(whitelistSources.flatMap((item) => item.clocks))],
percents: [...new Set(whitelistSources.flatMap((item) => item.percents))],
};
answerText = stripVerbalWindowChange(answerText);
if (!answerText && !composeSpokenWithServerFacts("", serverFacts)) {
answerText = RECTIFICATION_USER_COPY.declaredWindowLockedReply;
}
answerText = stripVerbalWindowChange(answerText)
|| RECTIFICATION_USER_COPY.declaredWindowLockedReply;
if (answerText !== visibleEmitted) await emitVisibleSpoken(answerText);
return completeAttempt();
} finally {
clearTimeout(timeout);
@@ -16,9 +16,12 @@ import { userFacingRunFailure } from "./run-diagnostic";
import { reportTurnProgress } from "./turn-instrumentation.ts";
import {
type AttemptOutcome,
composeSpokenWithServerFacts,
safeErrorCode,
v9TurnReceipts,
} from "./agent-run-support";
import { dropUngroundedFactSentences } from "./spoken-grounding";
import { RECTIFICATION_USER_COPY } from "../user-copy";
import type { V9AgentRunResult } from "./agent-run";
import type { V9TimedTurn } from "./agent-run-prepare";
@@ -126,7 +129,13 @@ export async function finishV9AgentTurn(
null,
outcome.usage,
);
const answerText = outcome.answerText;
// P3 (BUG-1055): on an evidence turn a model sentence whose clock, clock
// range or percentage is not this turn's server fact is dropped whole. When
// nothing of the model body is left, the batch recap (server data) stands
// in for it.
const answerText = action === "evidence" && outcome.spokenFacts
? dropUngroundedFactSentences(outcome.answerText, outcome.spokenFacts) || outcome.hostRecap || ""
: outcome.answerText;
let spokenAnswer = answerText;
let interviewIdle: Awaited<ReturnType<typeof persistNextInterviewIfIdle>> | null = null;
let interviewSettled = false;
@@ -167,13 +176,18 @@ export async function finishV9AgentTurn(
}
}
if (action === "evidence") {
// BUG-606 / BUG-615: the trim applies to the model body only.
spokenAnswer = trimSpokenTurnForInterview(answerText, interviewIdle?.terminalNote === true);
if (interviewIdle?.terminalNote && interviewIdle.hostNarration) {
spokenAnswer = composeIdleGapIntoSpoken(spokenAnswer, interviewIdle.hostNarration);
}
if (spokenAnswer !== answerText) {
await emit({ type: "answer.delta", text: spokenAnswer, replace: true });
}
}
// BUG-1055: server fact sentences join after the trim, so they are never
// cut and never streamed before this final text.
spokenAnswer = composeSpokenWithServerFacts(spokenAnswer, outcome.serverFacts)
|| RECTIFICATION_USER_COPY.collectHandoff;
if (action === "evidence" && interviewIdle?.terminalNote && interviewIdle.hostNarration) {
spokenAnswer = composeIdleGapIntoSpoken(spokenAnswer, interviewIdle.hostNarration);
}
if (spokenAnswer !== (outcome.streamedText ?? outcome.answerText)) {
await emit({ type: "answer.delta", text: spokenAnswer, replace: true });
}
await finalizeTurn("completed", spokenAnswer, outcome.attemptId, outcome.attemptId, true);
@@ -8,6 +8,12 @@ import { insertV9RunPhase, RectificationToolServiceError, type RectificationRpcC
import { promptCacheUsage } from "../../agent-generation-settings.ts";
import { toAgentModelFinishReason } from "../../agent-observability.ts";
import type { PublicStreamEvent } from "./stream-mapping";
import type { SpokenFactWhitelist } from "./spoken-grounding";
import {
withCompareFailedRetryNotice,
withRangeChangedAfterEvidence,
withRescoreSkippedNotice,
} from "../user-copy";
export type AttemptStatus = "completed" | "failed" | "retryable";
export type Usage = Readonly<{ inputTokens: number; outputTokens: number; cache?: ReturnType<typeof promptCacheUsage> }>;
@@ -25,8 +31,40 @@ export type AttemptOutcome = Readonly<{
caseLoaded: boolean;
attemptId: string;
settleBilling?: boolean;
/** What the client shows when the attempt ends; server fact sentences are never streamed by the attempt. */
streamedText?: string;
/** Server fact sentences this turn owes the user, joined after the body is trimmed (BUG-1055). */
serverFacts?: TurnServerFacts | null;
/** This turn's server numbers a model sentence may repeat (P3 whitelist). */
spokenFacts?: SpokenFactWhitelist | null;
/** Server recap of the batch write, used when the whitelist leaves no model body. */
hostRecap?: string | null;
}>;
/**
* Server-owned sentences of a turn: compare failed, rescore skipped, or the
* range change. They are never trimmed with the model body and never streamed
* before the body is final (BUG-1055, recurrence of BUG-588).
*/
export type TurnServerFacts = Readonly<{
compareFailed: boolean;
rescoreSkipped: boolean;
rangeBefore: readonly [string, string] | null;
rangeAfter: readonly [string, string] | null;
}>;
/** Join the server fact sentences after an already trimmed and whitelisted body. */
export function composeSpokenWithServerFacts(
body: string,
facts: TurnServerFacts | null | undefined,
): string {
let spoken = body.trim();
if (!facts) return spoken;
if (facts.compareFailed) spoken = withCompareFailedRetryNotice(spoken);
if (facts.rescoreSkipped) return withRescoreSkippedNotice(spoken);
return withRangeChangedAfterEvidence(spoken, facts.rangeBefore, facts.rangeAfter);
}
export const MAX_ATTEMPTS = 2;
export const RETRYABLE_ERROR_CODES = new Set([
"stream_aborted",
@@ -79,6 +79,12 @@ export type V9AgentRunOptions = Readonly<{
collectIntent?: "classified" | "unclassified" | null;
/** Classifier timing from the route preflight, for the run diagnostic only. */
classifierDiagnostic?: TurnIntentClassifierDiagnostic | null;
/**
* Credible range before this user message, when the route already persisted
* part of the message (a typed choice answer that also carries a dated
* event) before the run. Undefined: read it from the prepared dossier.
*/
rangeBeforeTurn?: readonly [string, string] | null;
}>;
export {
@@ -0,0 +1,135 @@
/**
* Number whitelist for the model-written body of an evidence / delivery turn
* (TASK-rectification-grounding-20260927 P3, red line 2).
*
* Range, representative minute and fit rate are server facts. A model
* sentence that carries a clock (HH:MM), a clock range or a percentage is
* kept only when every such number equals this turn's server facts;
* otherwise the whole sentence is dropped. The regexes here only decide
* whether a sentence passes. They never delete a word inside a sentence
* (AGENTS three-channel rule: a regex is a gate, not a knife).
*
* Server fact sentences (range changed, rescore skipped, compare failed) are
* appended after this gate and never pass through it.
*/
import { splitAfterSentencePunctuation } from "../../sentence-split.ts";
export type SpokenFactWhitelist = Readonly<{
/** Clock ranges the server states this turn, as `HH:MM–HH:MM` keys. */
ranges: readonly string[];
/** Single clocks the server states this turn (range ends, representative minute). */
clocks: readonly string[];
/** Whole-number percentages the server states this turn (event fit rate). */
percents: readonly number[];
}>;
export type SpokenFactInput = Readonly<{
credibleRange?: readonly [string, string] | null;
credibleIntervals?: readonly Readonly<{ start_at?: unknown; end_at?: unknown }>[] | null;
representativeTime?: string | null;
fitPercent?: number | null;
}>;
type ClockHit = Readonly<{ start: number; end: number; clock: string }>;
const CLOCK_RE = /(\d{1,2})[::](\d{2})/g;
const RANGE_JOINER_RE = /^\s*(?:[–—\-~~至]|到)\s*$/;
const PERCENT_RE = /(\d{1,3}(?:\.\d+)?)\s*[%%]/g;
const CHINESE_PERCENT_RE = /百分之\s*(\d{1,3}(?:\.\d+)?)/g;
function isDigitOrColon(char: string | undefined): boolean {
return char !== undefined && /[\d::.]/.test(char);
}
/** Clock tokens (00:00–23:59) not glued to other digits; no lookbehind (Safari floor). */
function clockHits(sentence: string): ClockHit[] {
const hits: ClockHit[] = [];
for (const match of sentence.matchAll(CLOCK_RE)) {
const start = match.index ?? 0;
const end = start + match[0].length;
if (isDigitOrColon(sentence[start - 1]) || isDigitOrColon(sentence[end])) continue;
const hours = Number(match[1]);
const minutes = Number(match[2]);
if (hours > 23 || minutes > 59) continue;
hits.push({ start, end, clock: normalizeClock(match[1], match[2]) });
}
return hits;
}
function percentValues(sentence: string): number[] {
const values: number[] = [];
for (const match of sentence.matchAll(PERCENT_RE)) {
const start = match.index ?? 0;
if (/[\d.]/.test(sentence[start - 1] ?? "")) continue;
values.push(Number(match[1]));
}
for (const match of sentence.matchAll(CHINESE_PERCENT_RE)) values.push(Number(match[1]));
return values;
}
function normalizeClock(hours: string, minutes: string): string {
return `${hours.padStart(2, "0")}:${minutes}`;
}
function clockOf(value: unknown): string | null {
if (typeof value !== "string") return null;
const match = /(?:T|^)([01]\d|2[0-3]):([0-5]\d)/.exec(value.trim());
return match ? `${match[1]}:${match[2]}` : null;
}
function rangeKey(start: string, end: string): string {
return `${start}–${end}`;
}
export function spokenFactWhitelist(input: SpokenFactInput): SpokenFactWhitelist {
const ranges = new Set<string>();
const clocks = new Set<string>();
const addRange = (start: string | null, end: string | null) => {
if (!start || !end) return;
ranges.add(rangeKey(start, end));
clocks.add(start);
clocks.add(end);
};
const range = input.credibleRange;
if (range) addRange(clockOf(range[0]), clockOf(range[1]));
for (const interval of input.credibleIntervals ?? []) {
addRange(clockOf(interval?.start_at), clockOf(interval?.end_at));
}
const representative = clockOf(input.representativeTime ?? null);
if (representative) clocks.add(representative);
const percents = new Set<number>();
if (typeof input.fitPercent === "number" && Number.isFinite(input.fitPercent)) {
percents.add(Math.round(input.fitPercent));
}
return { ranges: [...ranges], clocks: [...clocks], percents: [...percents] };
}
/** True when every clock, clock range and percentage in the sentence is a server fact. */
export function sentenceNumbersGrounded(sentence: string, whitelist: SpokenFactWhitelist): boolean {
const clocks = clockHits(sentence);
for (let index = 0; index < clocks.length; index += 1) {
const hit = clocks[index];
const next = clocks[index + 1];
if (next && RANGE_JOINER_RE.test(sentence.slice(hit.end, next.start))) {
if (!whitelist.ranges.includes(rangeKey(hit.clock, next.clock))) return false;
index += 1;
continue;
}
if (!whitelist.clocks.includes(hit.clock)) return false;
}
for (const value of percentValues(sentence)) {
if (!Number.isFinite(value) || Math.round(value) !== value) return false;
if (!whitelist.percents.includes(value)) return false;
}
return true;
}
/** Drop, whole, every body sentence whose clocks or percentages are not this turn's server facts. */
export function dropUngroundedFactSentences(body: string, whitelist: SpokenFactWhitelist): string {
const spoken = body.trim();
if (!spoken) return "";
const kept = splitAfterSentencePunctuation(spoken, "。!??\n")
.filter((part) => !part.trim() || sentenceNumbersGrounded(part, whitelist));
return kept.join("").trim();
}