fix(rectification): server fact sentences skip the evidence trim; model numbers must match server facts (BUG-1055)
T1 of TASK-rectification-grounding-20260927 (recurrence of BUG-588). - The attempt no longer streams range-changed / rescore-skipped / compare-failed sentences; the finish whitelists and trims the model body, then joins the server facts, and emits one final replace equal to the persisted text. - P3 whitelist (spoken-grounding.ts): a model sentence with a clock, clock range or percentage that is not this turn's server fact is dropped whole; the batch recap stands in when nothing is left. - record-evidence-batch returns range_after_rescore (post-rescore credible_range, representative minute, fit percent, delivers_range_this_turn); the receipt fingerprint stays over the old shape. - System prompt: range is said by the server; the delivery three sentences only when the batch says this turn delivers. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
1937608998
commit
0e0baaa74c
@@ -21,9 +21,7 @@ import {
|
||||
isAcceptableOpeningBody,
|
||||
openingRangeFromCandidateRange,
|
||||
openingSpokenBody,
|
||||
withCompareFailedRetryNotice,
|
||||
withRangeChangedAfterEvidence,
|
||||
withRescoreSkippedNotice,
|
||||
stripRangeProgressClaims,
|
||||
} from "../user-copy";
|
||||
import { attemptTimeoutForRemainingBudget, canStartRetryAttempt } from "../../rectification-run-budget.ts";
|
||||
import { stripQuestionSentences, stripVerbalWindowChange } from "./collect-prompt";
|
||||
@@ -56,8 +54,12 @@ import {
|
||||
turnExpectsEvidenceWrite,
|
||||
} from "./host-fallback";
|
||||
import { applyStepAnswerChunk, createStepAnswerState, flushStepAnswerOnStreamFinish } from "./step-answer";
|
||||
import { refinementFromDecisionReceipt } from "./refinement-packet";
|
||||
import { spokenFactWhitelist, type SpokenFactWhitelist } from "./spoken-grounding";
|
||||
import {
|
||||
type AttemptOutcome,
|
||||
type TurnServerFacts,
|
||||
composeSpokenWithServerFacts,
|
||||
MAX_ATTEMPTS,
|
||||
streamFinishReason,
|
||||
publicToolCallKey,
|
||||
@@ -150,9 +152,16 @@ export async function streamV9Attempt(
|
||||
const emittedActivities = new Set<string>();
|
||||
const repeatedCalls = new Map<string, number>();
|
||||
let phaseSequence = 0;
|
||||
const rangeBeforeCompare = previousInferenceFromReceipt(
|
||||
dossier.latestResult?.decisionReceipt ?? null,
|
||||
)?.credible_range ?? null;
|
||||
// BUG-1058: a typed answer to a choice card that also carries a dated event
|
||||
// persists the choice before this run; the route hands over the range from
|
||||
// before that answer so one server sentence covers both changes.
|
||||
const rangeBeforeCompare = options.rangeBeforeTurn !== undefined
|
||||
? options.rangeBeforeTurn
|
||||
: previousInferenceFromReceipt(
|
||||
dossier.latestResult?.decisionReceipt ?? null,
|
||||
)?.credible_range ?? null;
|
||||
let serverFacts: TurnServerFacts | null = null;
|
||||
let spokenFacts: SpokenFactWhitelist | null = null;
|
||||
|
||||
const recordPhase = async (phase: string, tool: string | null = null) => {
|
||||
if (
|
||||
@@ -411,9 +420,11 @@ export async function streamV9Attempt(
|
||||
if (flushed.kind === "publish") await publishSpokenStep(flushed.pieces);
|
||||
}
|
||||
|
||||
if (toolTerminalStatus.get("rectification-compare-candidates") === "failed" && !batchRescoreFailed(batchToolResult)) {
|
||||
await emitVisibleSpoken(withCompareFailedRetryNotice(answerText));
|
||||
}
|
||||
// BUG-1055: server fact sentences are not streamed here. They join the
|
||||
// body after the finish has trimmed and whitelisted it, so a streamed
|
||||
// fact is never withdrawn by a later replace.
|
||||
const compareFailed = toolTerminalStatus.get("rectification-compare-candidates") === "failed"
|
||||
&& !batchRescoreFailed(batchToolResult);
|
||||
|
||||
const completeAttempt = async (settleBilling = true): Promise<AttemptOutcome> => {
|
||||
let inputTokens = 0;
|
||||
@@ -435,6 +446,14 @@ export async function streamV9Attempt(
|
||||
errorCode: null,
|
||||
usage: { inputTokens, outputTokens, ...(cache ? { cache } : {}) },
|
||||
answerText,
|
||||
streamedText: visibleEmitted,
|
||||
serverFacts: serverFacts ?? (compareFailed
|
||||
? { compareFailed: true, rescoreSkipped: false, rangeBefore: null, rangeAfter: null }
|
||||
: null),
|
||||
spokenFacts,
|
||||
hostRecap: toolTerminalStatus.get("rectification-record-evidence-batch") === "completed"
|
||||
? composeHostFallbackNarration(batchToolResult ?? {})
|
||||
: null,
|
||||
answerDeltas,
|
||||
phases,
|
||||
toolsUsed: [...toolsUsed],
|
||||
@@ -586,21 +605,35 @@ export async function streamV9Attempt(
|
||||
answerText = openingSpokenBody(openingRange);
|
||||
}
|
||||
}
|
||||
const rangeAfterEvidence = previousInferenceFromReceipt(
|
||||
const latestInference = previousInferenceFromReceipt(
|
||||
latestDossier.latestResult?.decisionReceipt ?? null,
|
||||
)?.credible_range ?? null;
|
||||
if (batchRescoreFailed(batchToolResult)) {
|
||||
answerText = withRescoreSkippedNotice(answerText);
|
||||
} else {
|
||||
answerText = withRangeChangedAfterEvidence(
|
||||
answerText,
|
||||
rangeBeforeCompare,
|
||||
rangeAfterEvidence,
|
||||
);
|
||||
);
|
||||
const rangeAfterEvidence = latestInference?.credible_range ?? null;
|
||||
const rescoreSkipped = batchRescoreFailed(batchToolResult);
|
||||
if (rescoreSkipped) answerText = stripRangeProgressClaims(answerText);
|
||||
serverFacts = {
|
||||
compareFailed,
|
||||
rescoreSkipped,
|
||||
rangeBefore: rangeBeforeCompare,
|
||||
rangeAfter: rangeAfterEvidence,
|
||||
};
|
||||
// P3: the only clocks and percentages a model sentence may carry are this
|
||||
// turn's server facts (range, representative minute, event fit rate).
|
||||
const fitRate = refinementFromDecisionReceipt(latestDossier.latestResult?.decisionReceipt ?? null).event_fit_rate;
|
||||
const whitelistSources = [
|
||||
{ credibleRange: rangeAfterEvidence, credibleIntervals: latestInference?.credible_intervals ?? null,
|
||||
representativeTime: latestInference?.representative_time ?? latestDossier.latestResult?.representativeTime ?? null },
|
||||
{ credibleRange: decision.credibleRange ?? null, representativeTime: decision.representativeTime ?? null },
|
||||
].map((source) => spokenFactWhitelist({ ...source, fitPercent: fitRate?.percent ?? null }));
|
||||
spokenFacts = {
|
||||
ranges: [...new Set(whitelistSources.flatMap((item) => item.ranges))],
|
||||
clocks: [...new Set(whitelistSources.flatMap((item) => item.clocks))],
|
||||
percents: [...new Set(whitelistSources.flatMap((item) => item.percents))],
|
||||
};
|
||||
answerText = stripVerbalWindowChange(answerText);
|
||||
if (!answerText && !composeSpokenWithServerFacts("", serverFacts)) {
|
||||
answerText = RECTIFICATION_USER_COPY.declaredWindowLockedReply;
|
||||
}
|
||||
answerText = stripVerbalWindowChange(answerText)
|
||||
|| RECTIFICATION_USER_COPY.declaredWindowLockedReply;
|
||||
if (answerText !== visibleEmitted) await emitVisibleSpoken(answerText);
|
||||
return completeAttempt();
|
||||
} finally {
|
||||
clearTimeout(timeout);
|
||||
|
||||
@@ -16,9 +16,12 @@ import { userFacingRunFailure } from "./run-diagnostic";
|
||||
import { reportTurnProgress } from "./turn-instrumentation.ts";
|
||||
import {
|
||||
type AttemptOutcome,
|
||||
composeSpokenWithServerFacts,
|
||||
safeErrorCode,
|
||||
v9TurnReceipts,
|
||||
} from "./agent-run-support";
|
||||
import { dropUngroundedFactSentences } from "./spoken-grounding";
|
||||
import { RECTIFICATION_USER_COPY } from "../user-copy";
|
||||
import type { V9AgentRunResult } from "./agent-run";
|
||||
import type { V9TimedTurn } from "./agent-run-prepare";
|
||||
|
||||
@@ -126,7 +129,13 @@ export async function finishV9AgentTurn(
|
||||
null,
|
||||
outcome.usage,
|
||||
);
|
||||
const answerText = outcome.answerText;
|
||||
// P3 (BUG-1055): on an evidence turn a model sentence whose clock, clock
|
||||
// range or percentage is not this turn's server fact is dropped whole. When
|
||||
// nothing of the model body is left, the batch recap (server data) stands
|
||||
// in for it.
|
||||
const answerText = action === "evidence" && outcome.spokenFacts
|
||||
? dropUngroundedFactSentences(outcome.answerText, outcome.spokenFacts) || outcome.hostRecap || ""
|
||||
: outcome.answerText;
|
||||
let spokenAnswer = answerText;
|
||||
let interviewIdle: Awaited<ReturnType<typeof persistNextInterviewIfIdle>> | null = null;
|
||||
let interviewSettled = false;
|
||||
@@ -167,13 +176,18 @@ export async function finishV9AgentTurn(
|
||||
}
|
||||
}
|
||||
if (action === "evidence") {
|
||||
// BUG-606 / BUG-615: the trim applies to the model body only.
|
||||
spokenAnswer = trimSpokenTurnForInterview(answerText, interviewIdle?.terminalNote === true);
|
||||
if (interviewIdle?.terminalNote && interviewIdle.hostNarration) {
|
||||
spokenAnswer = composeIdleGapIntoSpoken(spokenAnswer, interviewIdle.hostNarration);
|
||||
}
|
||||
if (spokenAnswer !== answerText) {
|
||||
await emit({ type: "answer.delta", text: spokenAnswer, replace: true });
|
||||
}
|
||||
}
|
||||
// BUG-1055: server fact sentences join after the trim, so they are never
|
||||
// cut and never streamed before this final text.
|
||||
spokenAnswer = composeSpokenWithServerFacts(spokenAnswer, outcome.serverFacts)
|
||||
|| RECTIFICATION_USER_COPY.collectHandoff;
|
||||
if (action === "evidence" && interviewIdle?.terminalNote && interviewIdle.hostNarration) {
|
||||
spokenAnswer = composeIdleGapIntoSpoken(spokenAnswer, interviewIdle.hostNarration);
|
||||
}
|
||||
if (spokenAnswer !== (outcome.streamedText ?? outcome.answerText)) {
|
||||
await emit({ type: "answer.delta", text: spokenAnswer, replace: true });
|
||||
}
|
||||
await finalizeTurn("completed", spokenAnswer, outcome.attemptId, outcome.attemptId, true);
|
||||
|
||||
|
||||
@@ -8,6 +8,12 @@ import { insertV9RunPhase, RectificationToolServiceError, type RectificationRpcC
|
||||
import { promptCacheUsage } from "../../agent-generation-settings.ts";
|
||||
import { toAgentModelFinishReason } from "../../agent-observability.ts";
|
||||
import type { PublicStreamEvent } from "./stream-mapping";
|
||||
import type { SpokenFactWhitelist } from "./spoken-grounding";
|
||||
import {
|
||||
withCompareFailedRetryNotice,
|
||||
withRangeChangedAfterEvidence,
|
||||
withRescoreSkippedNotice,
|
||||
} from "../user-copy";
|
||||
|
||||
export type AttemptStatus = "completed" | "failed" | "retryable";
|
||||
export type Usage = Readonly<{ inputTokens: number; outputTokens: number; cache?: ReturnType<typeof promptCacheUsage> }>;
|
||||
@@ -25,8 +31,40 @@ export type AttemptOutcome = Readonly<{
|
||||
caseLoaded: boolean;
|
||||
attemptId: string;
|
||||
settleBilling?: boolean;
|
||||
/** What the client shows when the attempt ends; server fact sentences are never streamed by the attempt. */
|
||||
streamedText?: string;
|
||||
/** Server fact sentences this turn owes the user, joined after the body is trimmed (BUG-1055). */
|
||||
serverFacts?: TurnServerFacts | null;
|
||||
/** This turn's server numbers a model sentence may repeat (P3 whitelist). */
|
||||
spokenFacts?: SpokenFactWhitelist | null;
|
||||
/** Server recap of the batch write, used when the whitelist leaves no model body. */
|
||||
hostRecap?: string | null;
|
||||
}>;
|
||||
|
||||
/**
|
||||
* Server-owned sentences of a turn: compare failed, rescore skipped, or the
|
||||
* range change. They are never trimmed with the model body and never streamed
|
||||
* before the body is final (BUG-1055, recurrence of BUG-588).
|
||||
*/
|
||||
export type TurnServerFacts = Readonly<{
|
||||
compareFailed: boolean;
|
||||
rescoreSkipped: boolean;
|
||||
rangeBefore: readonly [string, string] | null;
|
||||
rangeAfter: readonly [string, string] | null;
|
||||
}>;
|
||||
|
||||
/** Join the server fact sentences after an already trimmed and whitelisted body. */
|
||||
export function composeSpokenWithServerFacts(
|
||||
body: string,
|
||||
facts: TurnServerFacts | null | undefined,
|
||||
): string {
|
||||
let spoken = body.trim();
|
||||
if (!facts) return spoken;
|
||||
if (facts.compareFailed) spoken = withCompareFailedRetryNotice(spoken);
|
||||
if (facts.rescoreSkipped) return withRescoreSkippedNotice(spoken);
|
||||
return withRangeChangedAfterEvidence(spoken, facts.rangeBefore, facts.rangeAfter);
|
||||
}
|
||||
|
||||
export const MAX_ATTEMPTS = 2;
|
||||
export const RETRYABLE_ERROR_CODES = new Set([
|
||||
"stream_aborted",
|
||||
|
||||
@@ -79,6 +79,12 @@ export type V9AgentRunOptions = Readonly<{
|
||||
collectIntent?: "classified" | "unclassified" | null;
|
||||
/** Classifier timing from the route preflight, for the run diagnostic only. */
|
||||
classifierDiagnostic?: TurnIntentClassifierDiagnostic | null;
|
||||
/**
|
||||
* Credible range before this user message, when the route already persisted
|
||||
* part of the message (a typed choice answer that also carries a dated
|
||||
* event) before the run. Undefined: read it from the prepared dossier.
|
||||
*/
|
||||
rangeBeforeTurn?: readonly [string, string] | null;
|
||||
}>;
|
||||
|
||||
export {
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
/**
|
||||
* Number whitelist for the model-written body of an evidence / delivery turn
|
||||
* (TASK-rectification-grounding-20260927 P3, red line 2).
|
||||
*
|
||||
* Range, representative minute and fit rate are server facts. A model
|
||||
* sentence that carries a clock (HH:MM), a clock range or a percentage is
|
||||
* kept only when every such number equals this turn's server facts;
|
||||
* otherwise the whole sentence is dropped. The regexes here only decide
|
||||
* whether a sentence passes. They never delete a word inside a sentence
|
||||
* (AGENTS three-channel rule: a regex is a gate, not a knife).
|
||||
*
|
||||
* Server fact sentences (range changed, rescore skipped, compare failed) are
|
||||
* appended after this gate and never pass through it.
|
||||
*/
|
||||
|
||||
import { splitAfterSentencePunctuation } from "../../sentence-split.ts";
|
||||
|
||||
export type SpokenFactWhitelist = Readonly<{
|
||||
/** Clock ranges the server states this turn, as `HH:MM–HH:MM` keys. */
|
||||
ranges: readonly string[];
|
||||
/** Single clocks the server states this turn (range ends, representative minute). */
|
||||
clocks: readonly string[];
|
||||
/** Whole-number percentages the server states this turn (event fit rate). */
|
||||
percents: readonly number[];
|
||||
}>;
|
||||
|
||||
export type SpokenFactInput = Readonly<{
|
||||
credibleRange?: readonly [string, string] | null;
|
||||
credibleIntervals?: readonly Readonly<{ start_at?: unknown; end_at?: unknown }>[] | null;
|
||||
representativeTime?: string | null;
|
||||
fitPercent?: number | null;
|
||||
}>;
|
||||
|
||||
type ClockHit = Readonly<{ start: number; end: number; clock: string }>;
|
||||
|
||||
const CLOCK_RE = /(\d{1,2})[::](\d{2})/g;
|
||||
const RANGE_JOINER_RE = /^\s*(?:[–—\-~~至]|到)\s*$/;
|
||||
const PERCENT_RE = /(\d{1,3}(?:\.\d+)?)\s*[%%]/g;
|
||||
const CHINESE_PERCENT_RE = /百分之\s*(\d{1,3}(?:\.\d+)?)/g;
|
||||
|
||||
function isDigitOrColon(char: string | undefined): boolean {
|
||||
return char !== undefined && /[\d::.]/.test(char);
|
||||
}
|
||||
|
||||
/** Clock tokens (00:00–23:59) not glued to other digits; no lookbehind (Safari floor). */
|
||||
function clockHits(sentence: string): ClockHit[] {
|
||||
const hits: ClockHit[] = [];
|
||||
for (const match of sentence.matchAll(CLOCK_RE)) {
|
||||
const start = match.index ?? 0;
|
||||
const end = start + match[0].length;
|
||||
if (isDigitOrColon(sentence[start - 1]) || isDigitOrColon(sentence[end])) continue;
|
||||
const hours = Number(match[1]);
|
||||
const minutes = Number(match[2]);
|
||||
if (hours > 23 || minutes > 59) continue;
|
||||
hits.push({ start, end, clock: normalizeClock(match[1], match[2]) });
|
||||
}
|
||||
return hits;
|
||||
}
|
||||
|
||||
function percentValues(sentence: string): number[] {
|
||||
const values: number[] = [];
|
||||
for (const match of sentence.matchAll(PERCENT_RE)) {
|
||||
const start = match.index ?? 0;
|
||||
if (/[\d.]/.test(sentence[start - 1] ?? "")) continue;
|
||||
values.push(Number(match[1]));
|
||||
}
|
||||
for (const match of sentence.matchAll(CHINESE_PERCENT_RE)) values.push(Number(match[1]));
|
||||
return values;
|
||||
}
|
||||
|
||||
function normalizeClock(hours: string, minutes: string): string {
|
||||
return `${hours.padStart(2, "0")}:${minutes}`;
|
||||
}
|
||||
|
||||
function clockOf(value: unknown): string | null {
|
||||
if (typeof value !== "string") return null;
|
||||
const match = /(?:T|^)([01]\d|2[0-3]):([0-5]\d)/.exec(value.trim());
|
||||
return match ? `${match[1]}:${match[2]}` : null;
|
||||
}
|
||||
|
||||
function rangeKey(start: string, end: string): string {
|
||||
return `${start}–${end}`;
|
||||
}
|
||||
|
||||
export function spokenFactWhitelist(input: SpokenFactInput): SpokenFactWhitelist {
|
||||
const ranges = new Set<string>();
|
||||
const clocks = new Set<string>();
|
||||
const addRange = (start: string | null, end: string | null) => {
|
||||
if (!start || !end) return;
|
||||
ranges.add(rangeKey(start, end));
|
||||
clocks.add(start);
|
||||
clocks.add(end);
|
||||
};
|
||||
const range = input.credibleRange;
|
||||
if (range) addRange(clockOf(range[0]), clockOf(range[1]));
|
||||
for (const interval of input.credibleIntervals ?? []) {
|
||||
addRange(clockOf(interval?.start_at), clockOf(interval?.end_at));
|
||||
}
|
||||
const representative = clockOf(input.representativeTime ?? null);
|
||||
if (representative) clocks.add(representative);
|
||||
const percents = new Set<number>();
|
||||
if (typeof input.fitPercent === "number" && Number.isFinite(input.fitPercent)) {
|
||||
percents.add(Math.round(input.fitPercent));
|
||||
}
|
||||
return { ranges: [...ranges], clocks: [...clocks], percents: [...percents] };
|
||||
}
|
||||
|
||||
/** True when every clock, clock range and percentage in the sentence is a server fact. */
|
||||
export function sentenceNumbersGrounded(sentence: string, whitelist: SpokenFactWhitelist): boolean {
|
||||
const clocks = clockHits(sentence);
|
||||
for (let index = 0; index < clocks.length; index += 1) {
|
||||
const hit = clocks[index];
|
||||
const next = clocks[index + 1];
|
||||
if (next && RANGE_JOINER_RE.test(sentence.slice(hit.end, next.start))) {
|
||||
if (!whitelist.ranges.includes(rangeKey(hit.clock, next.clock))) return false;
|
||||
index += 1;
|
||||
continue;
|
||||
}
|
||||
if (!whitelist.clocks.includes(hit.clock)) return false;
|
||||
}
|
||||
for (const value of percentValues(sentence)) {
|
||||
if (!Number.isFinite(value) || Math.round(value) !== value) return false;
|
||||
if (!whitelist.percents.includes(value)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Drop, whole, every body sentence whose clocks or percentages are not this turn's server facts. */
|
||||
export function dropUngroundedFactSentences(body: string, whitelist: SpokenFactWhitelist): string {
|
||||
const spoken = body.trim();
|
||||
if (!spoken) return "";
|
||||
const kept = splitAfterSentencePunctuation(spoken, "。!??\n")
|
||||
.filter((part) => !part.trim() || sentenceNumbersGrounded(part, whitelist));
|
||||
return kept.join("").trim();
|
||||
}
|
||||
Reference in New Issue
Block a user