T1 of TASK-rectification-grounding-20260927 (recurrence of BUG-588). - The attempt no longer streams range-changed / rescore-skipped / compare-failed sentences; the finish whitelists and trims the model body, then joins the server facts, and emits one final replace equal to the persisted text. - P3 whitelist (spoken-grounding.ts): a model sentence with a clock, clock range or percentage that is not this turn's server fact is dropped whole; the batch recap stands in when nothing is left. - record-evidence-batch returns range_after_rescore (post-rescore credible_range, representative minute, fit percent, delivers_range_this_turn); the receipt fingerprint stays over the old shape. - System prompt: range is said by the server; the delivery three sentences only when the batch says this turn delivers. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
273 lines
10 KiB
TypeScript
273 lines
10 KiB
TypeScript
/**
|
||
* Shared fixtures for TASK-rectification-grounding-20260927.
|
||
*
|
||
* `rectification-grounding-aa-30min.golden.json` is a real local engine
|
||
* response (`runV9CandidateScore` request + response) for a public AA chart
|
||
* (Angelina Jolie, 1975-06-04 Los Angeles, 08:54–09:24 search window, three
|
||
* public dated events). No private data.
|
||
*/
|
||
import { readFileSync } from "node:fs";
|
||
import {
|
||
CASE_ID,
|
||
FOCUS_ID,
|
||
RESULT_ID,
|
||
TURN_ID,
|
||
computeFixture,
|
||
dossierFixture,
|
||
} from "./rectification-v9-test-support.ts";
|
||
import { runV9CandidateScore } from "../src/lib/rectification-agentic/v9/engine-client.ts";
|
||
import { buildCaseInferenceState } from "../src/lib/rectification-agentic/v9/inference-adapter.ts";
|
||
import { requestCandidateIntervals, candidatePositionFields } from "../src/lib/rectification-agentic/core/candidate-window.ts";
|
||
import { candidateRangeFingerprint, evidenceLedgerFingerprint, parseV9CaseDossier } from "../src/lib/rectification-agentic/v9/tool-service.ts";
|
||
|
||
export const AA_GOLDEN = JSON.parse(readFileSync(
|
||
new URL("./fixtures/rectification-grounding-aa-30min.golden.json", import.meta.url),
|
||
"utf8",
|
||
)) as { request: Record<string, unknown> & { events: Array<Record<string, string>> }; response: Record<string, unknown> };
|
||
|
||
const req = AA_GOLDEN.request as Record<string, unknown> & { events: Array<Record<string, string>> };
|
||
|
||
export const AA_BASELINE = {
|
||
birth_date: req.birth_date,
|
||
latitude: req.lat,
|
||
longitude: req.lon,
|
||
timezone_offset: req.tz,
|
||
timezone_id: "America/Los_Angeles",
|
||
birth_time_source: "family_vague",
|
||
reported_birth_time: "09:09",
|
||
active_birth_time: null,
|
||
uncertainty_before_minutes: 15,
|
||
uncertainty_after_minutes: 15,
|
||
birth_place_label: "Los Angeles",
|
||
};
|
||
|
||
const baseRange = { start_time: String(req.start_time), end_time: String(req.end_time) };
|
||
export const AA_RANGE = {
|
||
...baseRange,
|
||
candidate_intervals: requestCandidateIntervals(AA_BASELINE as never, baseRange as never),
|
||
};
|
||
|
||
const VEDASTRO_PASS = {
|
||
status: "passed",
|
||
can_confirm_exact_minute: false,
|
||
event_validation: { search_events_primary_supports_local_winner: true },
|
||
minute_sensitive_validation: { status: "passed" },
|
||
};
|
||
|
||
/** Serve the golden engine response (and a passing VedAstro check) to every fetch. */
|
||
export function stubGoldenFetch(): () => void {
|
||
const previous = globalThis.fetch;
|
||
globalThis.fetch = (async (url: unknown) => {
|
||
const body = String(url).includes("vedastro") ? VEDASTRO_PASS : AA_GOLDEN.response;
|
||
return new Response(JSON.stringify(body), { status: 200 });
|
||
}) as typeof fetch;
|
||
return () => {
|
||
globalThis.fetch = previous;
|
||
};
|
||
}
|
||
|
||
export const AA_EVIDENCE_ROWS = req.events.map((event, index) => ({
|
||
id: `44444444-4444-4444-8444-44444444444${index}`,
|
||
source_turn_id: TURN_ID,
|
||
subject: "self",
|
||
event_kind: event.event_kind,
|
||
domain: event.domain,
|
||
occurred_from: event.date_start,
|
||
occurred_to: null,
|
||
date_precision: event.precision,
|
||
summary: event.summary,
|
||
status: "confirmed",
|
||
supersedes_evidence_id: null,
|
||
created_at: "2026-08-12T10:00:06.000Z",
|
||
}));
|
||
|
||
export const AA_TURNS = [
|
||
{
|
||
id: TURN_ID,
|
||
role: "user",
|
||
text: "2000 年拿了一个大奖;2014 年 8 月结婚。",
|
||
status: "completed",
|
||
created_at: "2026-08-12T10:00:00.000Z",
|
||
completed_at: "2026-08-12T10:00:05.000Z",
|
||
},
|
||
{
|
||
id: "77777777-7777-4777-8777-777777777771",
|
||
role: "assistant",
|
||
text: "记下了:2000 年获奖、2014 年 8 月结婚。",
|
||
status: "completed",
|
||
created_at: "2026-08-12T10:00:06.000Z",
|
||
completed_at: "2026-08-12T10:00:07.000Z",
|
||
},
|
||
];
|
||
|
||
/** The stored snapshot a real run of the golden case persists (9 candidates). */
|
||
export async function buildGoldenLatest() {
|
||
const restore = stubGoldenFetch();
|
||
try {
|
||
process.env.RECTIFICATION_ALGORITHM_VERSION = String(AA_GOLDEN.response.algorithm_version);
|
||
process.env.RECTIFICATION_DECISION_POLICY_VERSION = String(AA_GOLDEN.response.decision_policy_version);
|
||
const scored = await runV9CandidateScore({
|
||
baselineBirthSnapshot: AA_BASELINE,
|
||
candidateRange: AA_RANGE as never,
|
||
events: req.events,
|
||
} as never);
|
||
const parsedEvidence = parseV9CaseDossier(dossierFixture({ evidence: AA_EVIDENCE_ROWS }))!.evidence;
|
||
const inference = buildCaseInferenceState({
|
||
range: baseRange,
|
||
candidates: scored.candidates as never,
|
||
evidence: parsedEvidence as never,
|
||
probes: [],
|
||
transitions: scored.windowScan?.transitions as never,
|
||
});
|
||
const compute = { ...computeFixture({ baselineBirthSnapshot: AA_BASELINE as never }), candidate_range: AA_RANGE };
|
||
const latest = {
|
||
result_id: RESULT_ID,
|
||
candidates: scored.candidates.map((candidate) => ({
|
||
...candidatePositionFields(candidate as never),
|
||
candidate_id: candidate.candidateId,
|
||
time: candidate.time,
|
||
rank: candidate.rank,
|
||
relative_support: candidate.relativeSupport,
|
||
tied_minute_count: candidate.tiedMinuteCount,
|
||
...(candidate.clusterTimes ? { cluster_times: candidate.clusterTimes } : {}),
|
||
...(candidate.clusterStart ? { cluster_start: candidate.clusterStart } : {}),
|
||
...(candidate.clusterEnd ? { cluster_end: candidate.clusterEnd } : {}),
|
||
})),
|
||
overall_confidence: scored.overallConfidence,
|
||
selection_allowed: scored.selectionAllowed,
|
||
confirmation_allowed: false,
|
||
decision_receipt: { ...scored.decisionReceipt, inference_state: inference },
|
||
execution_ledger: scored.executionLedger,
|
||
representative_time: scored.representativeTime,
|
||
selected_time: null,
|
||
selection_kind: null,
|
||
evidence_ledger_fingerprint: evidenceLedgerFingerprint(parsedEvidence),
|
||
candidate_range_fingerprint: candidateRangeFingerprint(AA_RANGE as never, compute.baseline_profile_fingerprint),
|
||
skill_version: "9.0.0",
|
||
algorithm_version: scored.algorithmVersion,
|
||
event_contract_version: scored.eventContractVersion,
|
||
decision_policy_version: scored.policyVersion,
|
||
created_at: "2026-08-12T10:05:00.000Z",
|
||
invalidated_at: null,
|
||
};
|
||
return { latest, compute, scored };
|
||
} finally {
|
||
restore();
|
||
}
|
||
}
|
||
|
||
export function setFocusHandler(_fn: string, args: Record<string, unknown>) {
|
||
return {
|
||
focus: {
|
||
id: FOCUS_ID,
|
||
case_id: CASE_ID,
|
||
question_id: args.p_question_id,
|
||
intent: args.p_intent,
|
||
target_evidence_id: args.p_target_evidence_id,
|
||
target_domain: args.p_target_domain,
|
||
target_kind: args.p_target_kind,
|
||
expected_answer_schema: args.p_expected_answer_schema,
|
||
status: "active",
|
||
asked_at: "2026-08-27T00:00:00.000Z",
|
||
resolved_at: null,
|
||
asked_turn_id: args.p_asked_turn_id ?? null,
|
||
},
|
||
idempotent: false,
|
||
};
|
||
}
|
||
|
||
export type ScriptPart = { text?: string; tool?: { name: string; input: Record<string, unknown> } };
|
||
export type ScriptTurn = { parts: ScriptPart[]; finish: string };
|
||
export type RecordedCall = { prompt: Array<{ role: string; content: unknown }>; tools: unknown };
|
||
|
||
/** An AI SDK v2 language model that replays a script and records every prompt it is sent. */
|
||
export function scriptedModel(script: ScriptTurn[], onCall?: (callNumber: number) => void) {
|
||
const calls: RecordedCall[] = [];
|
||
let call = 0;
|
||
const next = (options: { prompt: unknown[]; tools?: unknown }) => {
|
||
calls.push({ prompt: options.prompt as RecordedCall["prompt"], tools: options.tools });
|
||
const turn = script[call] ?? { parts: [{ text: "" }], finish: "stop" };
|
||
call += 1;
|
||
onCall?.(call);
|
||
return turn;
|
||
};
|
||
const model = {
|
||
specificationVersion: "v2",
|
||
provider: "fake",
|
||
modelId: "fake-rectification",
|
||
supportedUrls: {},
|
||
async doGenerate(options: { prompt: unknown[]; tools?: unknown }) {
|
||
const turn = next(options);
|
||
const content = turn.parts.map((part, index) => part.tool
|
||
? { type: "tool-call", toolCallId: `g${call}-${index}`, toolName: part.tool.name, input: JSON.stringify(part.tool.input) }
|
||
: { type: "text", text: part.text ?? "" });
|
||
return { content, finishReason: turn.finish, usage: { inputTokens: 10, outputTokens: 10, totalTokens: 20 }, warnings: [] };
|
||
},
|
||
async doStream(options: { prompt: unknown[]; tools?: unknown }) {
|
||
const turn = next(options);
|
||
const id = `t${call}`;
|
||
const stream = new ReadableStream({
|
||
start(controller) {
|
||
controller.enqueue({ type: "stream-start", warnings: [] });
|
||
let open = false;
|
||
for (const [index, part] of turn.parts.entries()) {
|
||
if (part.text !== undefined) {
|
||
if (!open) {
|
||
controller.enqueue({ type: "text-start", id });
|
||
open = true;
|
||
}
|
||
controller.enqueue({ type: "text-delta", id, delta: part.text });
|
||
}
|
||
if (part.tool) {
|
||
if (open) {
|
||
controller.enqueue({ type: "text-end", id });
|
||
open = false;
|
||
}
|
||
controller.enqueue({
|
||
type: "tool-call",
|
||
toolCallId: `c${call}-${index}`,
|
||
toolName: part.tool.name,
|
||
input: JSON.stringify(part.tool.input),
|
||
});
|
||
}
|
||
}
|
||
if (open) controller.enqueue({ type: "text-end", id });
|
||
controller.enqueue({ type: "finish", finishReason: turn.finish, usage: { inputTokens: 10, outputTokens: 10, totalTokens: 20 } });
|
||
controller.close();
|
||
},
|
||
});
|
||
return { stream };
|
||
},
|
||
};
|
||
return { model, calls };
|
||
}
|
||
|
||
/** The text a client ends up showing after applying `answer.delta` events in order. */
|
||
export function clientStates(events: ReadonlyArray<Record<string, unknown>>): string[] {
|
||
let text = "";
|
||
const states: string[] = [];
|
||
for (const event of events) {
|
||
if (event.type !== "answer.delta") continue;
|
||
const delta = String(event.text ?? "");
|
||
text = event.replace === true ? delta : `${text}${delta}`;
|
||
states.push(text);
|
||
}
|
||
return states;
|
||
}
|
||
|
||
const CJK = /[⺀-鿿豈- -〿-]/g;
|
||
|
||
/**
|
||
* Token estimate used for the before/after size tables: one token per CJK
|
||
* character (including full-width punctuation) plus one token per four other
|
||
* characters. It is an estimate, not a provider tokenizer.
|
||
*/
|
||
export function estimateTokens(text: string): number {
|
||
const cjk = (text.match(CJK) ?? []).length;
|
||
return Math.round(cjk + (text.length - cjk) / 4);
|
||
}
|
||
|
||
export function contentText(content: unknown): string {
|
||
return typeof content === "string" ? content : JSON.stringify(content);
|
||
}
|