feat(rectification): the server writes the evidence recap — 「记下了 N 件事」 with the ledger list collapsed under it (BUG-1136)

- Evidence turns whose batch accepted items: the body is the server recap
  built from those items (two or fewer listed inline, more as a count);
  the model no longer restates them (both prompt sets updated).
- The case snapshot carries each assistant turn's recorded evidence from
  the ledger (rejected/superseded excluded); the message shows it in a
  collapsed <details> list when more than two.
- Chinese date labels read straight into the event phrase (2016年入学);
  ISO labels keep their space.
- Assertions updated with original/new/reason notes.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
Jesse_Chen
2026-10-01 10:40:29 +08:00
co-authored by Claude Opus 5.5
parent 408b22749e
commit 9d302e515c
21 changed files with 244 additions and 27 deletions
+17
View File
@@ -3380,6 +3380,23 @@ input:not([type="radio"]):not([type="checkbox"]):not([class^="ant-"]):not([class
font-size: var(--type-caption);
}
/* BUG-1136: what an evidence turn recorded, collapsed under 「记下了 N 件事」. */
.rectification-recorded-evidence {
margin-block-start: var(--space-2);
color: var(--color-ink-secondary);
font-size: var(--type-caption);
}
.rectification-recorded-evidence > summary {
cursor: pointer;
width: fit-content;
}
.rectification-recorded-evidence > ul {
margin: var(--space-2) 0 0;
padding-inline-start: 1.2em;
display: grid;
gap: var(--space-1);
color: var(--color-ink);
}
.rectification-message-question {
display: grid;
gap: var(--space-3);
@@ -45,6 +45,8 @@ export type RenderMessage = ChatMessageView & {
question?: TurnQuestion;
candidateOffer?: Readonly<{ resultId: string }>;
segmentConsistency?: SegmentSummary;
/** BUG-1136: what this evidence turn recorded, from the ledger. */
recordedEvidence?: readonly string[];
};
export type RectificationMessageActions = Readonly<{
@@ -167,7 +169,20 @@ function RectificationMessageEntryView({
&& (questionIsDeadUnanswered(question)
|| (currentQuestionFocusId !== null && currentQuestionFocusId !== question.focus_id)),
);
const afterAnswer = question && displayedMessage.state === "settled" && !replacedQuestion
// BUG-1136 (D2): more than two recorded items read as 「记下了 N 件事」 in
// the body; the items themselves (ledger lines) sit here, collapsed.
const recorded = settled && (message.recordedEvidence?.length ?? 0) > 2
? message.recordedEvidence!
: null;
const recordedList = recorded ? (
<details className="rectification-recorded-evidence">
<summary>{`看记下的 ${recorded.length} 件`}</summary>
<ul>
{recorded.map((line, index) => <li key={`${index}:${line}`}>{line}</li>)}
</ul>
</details>
) : null;
const questionBlock = question && displayedMessage.state === "settled" && !replacedQuestion
? (
<div className="rectification-message-question">
<p className="rectification-message-question__prompt">{question.prompt}</p>
@@ -190,6 +205,9 @@ function RectificationMessageEntryView({
)}
</div>
)
: null;
const afterAnswer = recordedList || questionBlock
? <>{recordedList}{questionBlock}</>
: undefined;
return (
@@ -783,3 +783,15 @@ export function accidentCaseHardcodedTurns(input: {
deliveryAdopt: deliveryAdoptNarration(shared),
};
}
/**
* BUG-1136 (D2): the evidence turn's recap is the server's, built from the
* batch's accepted items. Two or fewer are listed inline; more read as a count
* and the list sits in the message, collapsed.
*/
export function evidenceTurnRecap(lines: readonly string[]): string | null {
const items = lines.map((line) => line.trim()).filter(Boolean);
if (!items.length) return null;
if (items.length <= 2) return `记下了:${items.join("、")}。`;
return `记下了 ${items.length} 件事。`;
}
@@ -47,6 +47,7 @@ import { currentEngineCallTimings, reportTurnProgress } from "./turn-instrumenta
import {
batchResultFromToolChunk,
batchRescoreFailed,
acceptedRecapLines,
composeHostFallbackNarration,
lastCompletedPublicTool,
publicWriteToolCompleted,
@@ -455,6 +456,9 @@ export async function streamV9Attempt(
hostRecap: toolTerminalStatus.get("rectification-record-evidence-batch") === "completed"
? composeHostFallbackNarration(batchToolResult ?? {})
: null,
hostRecapLines: toolTerminalStatus.get("rectification-record-evidence-batch") === "completed"
? acceptedRecapLines(batchToolResult ?? {})
: null,
answerDeltas,
phases,
toolsUsed: [...toolsUsed],
@@ -21,7 +21,7 @@ import {
v9TurnReceipts,
} from "./agent-run-support";
import { dropUngroundedFactSentences } from "./spoken-grounding";
import { RECTIFICATION_USER_COPY } from "../user-copy";
import { evidenceTurnRecap, RECTIFICATION_USER_COPY } from "../user-copy";
import { moderate, MODERATION_OUTPUT_REPLACED_COPY } from "@/lib/moderation";
import { recordModerationEvent } from "@/lib/moderation/log";
import type { V9AgentRunResult } from "./agent-run";
@@ -190,6 +190,11 @@ export async function finishV9AgentTurn(
if (action === "evidence") {
// BUG-606 / BUG-615: the trim applies to the model body only.
spokenAnswer = trimSpokenTurnForInterview(answerText, interviewIdle?.terminalNote === true);
// BUG-1136 (D2): when the batch accepted items, the recap is the server's
// — 「记下了 N 件事」 (two or fewer listed) — not the model's item-by-item
// restatement. The list itself hangs on the message from the ledger.
const recap = evidenceTurnRecap(outcome.hostRecapLines ?? []);
if (recap) spokenAnswer = recap;
}
// BUG-1055: server fact sentences join after the trim, so they are never
// cut and never streamed before this final text.
@@ -39,6 +39,8 @@ export type AttemptOutcome = Readonly<{
spokenFacts?: SpokenFactWhitelist | null;
/** Server recap of the batch write, used when the whitelist leaves no model body. */
hostRecap?: string | null;
/** BUG-1136: accepted batch items this turn, for the server-written recap. */
hostRecapLines?: readonly string[] | null;
}>;
/**
@@ -11,7 +11,7 @@ import {
} from "@/lib/rectification-agentic/v9/tool-service";
import { choiceCardFromCaseDossier, decideFromDossier, overlayPublicDecision, nextUserActionFromDossier, stepStateFromCaseDossier } from "@/lib/rectification-agentic/v9/interview-state";
import { projectCurrentQuestion } from "@/lib/rectification-agentic/v9/turn-decision";
import { attachQuestionsToTurns, attachOfferResultToTurns } from "@/lib/rectification-agentic/v9/turn-question";
import { attachQuestionsToTurns, attachOfferResultToTurns, recordedEvidenceByTurn } from "@/lib/rectification-agentic/v9/turn-question";
import { previousInferenceFromReceipt } from "@/lib/rectification-agentic/v9/inference-adapter";
import { publicDecisionFields } from "@/lib/rectification-agentic/core/rectification-decision";
import { collectionProgressFromReceipt } from "@/lib/rectification-agentic/v9/evidence-model";
@@ -66,6 +66,7 @@ export function dossierResponse(
const segmentChecks = parseCaseSegmentChecks(dossier.case.segmentChecks, dossier.case.rectificationDomain);
const consistencySummary = segmentChecks?.delivered_turn_id && segmentChecks.summary?.no_rectification_needed
? segmentChecks.summary : null;
const recorded = recordedEvidenceByTurn(dossier.evidence);
const response = {
case: {
result_identity: identity,
@@ -90,6 +91,8 @@ export function dossierResponse(
case_revision: previousInferenceFromReceipt(dossier.latestResult?.decisionReceipt ?? null)?.revision ?? 0,
},
turns: turns.map((turn) => ({
...(turn.role === "assistant" && recorded.get(turn.id)?.length
? { recorded_evidence: recorded.get(turn.id) } : {}),
id: turn.id,
role: turn.role,
text: turn.text,
@@ -71,13 +71,17 @@ export function lastCompletedPublicTool(
return last;
}
function recapLine(item: HostFallbackRecap): string {
export function evidenceRecapLine(item: HostFallbackRecap): string {
const label = typeof item.display_date_label === "string" ? item.display_date_label.trim() : "";
const phrase = typeof item.event_phrase === "string" ? item.event_phrase.trim() : "";
if (!phrase) return label;
if (/\d{4}年/.test(phrase) || (label.length > 0 && phrase.startsWith(label))) {
return phrase;
}
// BUG-1136: the recap is now the server's on every evidence turn; a
// Chinese date label reads straight into the phrase (「2016年入学」), an ISO
// label keeps its space (「2016-09 入学」).
if (label && /[\u4e00-\u9fff]$/u.test(label)) return `${label}${phrase}`;
return [label, phrase].filter(Boolean).join(" ").trim();
}
@@ -121,12 +125,16 @@ export function batchRescoreFailed(batchResult: unknown): boolean {
return rescore?.status === "failed";
}
/** One display line per accepted item of a record-evidence batch (server data). */
export function acceptedRecapLines(batchResult: unknown): string[] {
if (batchResult == null || isToolInputRejection(batchResult)) return [];
return recapsFromBatchResult(batchResult).map(evidenceRecapLine).filter(Boolean);
}
export function composeHostFallbackNarration(batchResult: unknown): string | null {
if (batchResult == null) return null;
if (isToolInputRejection(batchResult)) return null;
const lines = recapsFromBatchResult(batchResult)
.map(recapLine)
.filter(Boolean);
const lines = acceptedRecapLines(batchResult);
if (lines.length > 0) return `记下了:${lines.join("、")}。`;
return "记下了。";
}
@@ -1,3 +1,5 @@
import { evidenceRecapLine } from "./host-fallback.ts";
import { displayDateLabel, eventPhraseFromSummary } from "./evidence-model.ts";
import { stripQuestionSentences } from "./collect-prompt";
import { parseAgentChoiceCopy, type ChoiceKey, type RectificationChoiceCard } from "./choice-card";
import type { ConversationFocus } from "./tool-service";
@@ -350,3 +352,37 @@ export function copyTextForMessage(body: string, question: TurnQuestion | null |
}
return lines.join("\n").trim();
}
/**
* BUG-1136 (D2): what an evidence turn recorded, read from the ledger (not the
* model's text). Evidence carries the turn it was said in; the assistant reply
* of that turn carries the list. Rejected and superseded rows are left out.
*/
export function recordedEvidenceByTurn(
evidence: readonly Readonly<{
sourceTurnId: string;
status: string;
datePrecision: string;
occurredFrom: string | null;
occurredTo: string | null;
summary: string;
}>[],
): Map<string, string[]> {
const byTurn = new Map<string, string[]>();
for (const row of evidence) {
if (!row.sourceTurnId || row.status === "rejected" || row.status === "superseded") continue;
const line = evidenceRecapLine({
display_date_label: displayDateLabel(row.datePrecision, row.occurredFrom, row.occurredTo),
event_phrase: eventPhraseFromSummary(row.summary),
});
if (!line) continue;
byTurn.set(row.sourceTurnId, [...(byTurn.get(row.sourceTurnId) ?? []), line]);
}
return byTurn;
}
export function parseRecordedEvidence(value: unknown): string[] | undefined {
if (!Array.isArray(value)) return undefined;
const lines = value.filter((item): item is string => typeof item === "string" && item.trim().length > 0);
return lines.length ? lines : undefined;
}
@@ -18,7 +18,7 @@ import {
import { isIncompleteRunBanner } from "@/lib/rectification-agentic/v9/run-diagnostic";
import { isStructuredChoiceUserText } from "@/lib/rectification-agentic/v9/choice-action";
import type { ChoiceKey } from "@/lib/rectification-agentic/v9/choice-card";
import { parseTurnQuestion, questionIsAnswered } from "@/lib/rectification-agentic/v9/turn-question";
import { parseRecordedEvidence, parseTurnQuestion, questionIsAnswered } from "@/lib/rectification-agentic/v9/turn-question";
import type { RenderMessage } from "@/components/rectification-message-entry";
import { parseSegmentSummary } from "./rectification-agentic/core/segment-summary.ts";
@@ -30,6 +30,7 @@ export type PersistedTurn = Readonly<{
question?: unknown;
offer_result_id?: string | null;
segment_consistency?: unknown;
recorded_evidence?: unknown;
receipt?: Readonly<{
status: string;
phases: readonly string[];
@@ -149,6 +150,7 @@ export function messagesFromTurns(initialTurns: readonly PersistedTurn[]): Rende
turnId: turn.id,
question: parseTurnQuestion(turn.question) ?? undefined,
segmentConsistency: parseSegmentSummary(turn.segment_consistency) ?? undefined,
recordedEvidence: parseRecordedEvidence(turn.recorded_evidence),
candidateOffer: typeof turn.offer_result_id === "string"
? { resultId: turn.offer_result_id }
: undefined,
@@ -1,6 +1,6 @@
import { parseSegmentSummary, type SegmentSummary } from "./rectification-agentic/core/segment-summary.ts";
import { stripQuestionSentences } from "./rectification-agentic/v9/collect-prompt.ts";
import { parseTurnQuestion, persistedOfferFromTurn, questionIsDeadUnanswered, type TurnQuestion } from "./rectification-agentic/v9/turn-question.ts";
import { parseRecordedEvidence, parseTurnQuestion, persistedOfferFromTurn, questionIsDeadUnanswered, type TurnQuestion } from "./rectification-agentic/v9/turn-question.ts";
export type SnapshotTurnMessage = {
role: "assistant" | "user";
@@ -10,20 +10,22 @@ export type SnapshotTurnMessage = {
question?: TurnQuestion;
candidateOffer?: { resultId: string };
segmentConsistency?: SegmentSummary;
recordedEvidence?: readonly string[];
};
export function mergeTurnQuestions<T extends SnapshotTurnMessage>(
current: T[],
turns: readonly unknown[],
): T[] {
const byId = new Map<string, { question: TurnQuestion | null; offerResultId: string | null; consistency: SegmentSummary | null }>();
const byId = new Map<string, { question: TurnQuestion | null; offerResultId: string | null; consistency: SegmentSummary | null; recorded: string[] | undefined }>();
for (const item of turns) {
if (!item || typeof item !== "object") continue;
const turn = item as { id?: unknown; question?: unknown; offer_result_id?: unknown; segment_consistency?: unknown };
const turn = item as { id?: unknown; question?: unknown; offer_result_id?: unknown; segment_consistency?: unknown; recorded_evidence?: unknown };
if (typeof turn.id !== "string") continue;
byId.set(turn.id, {
question: parseTurnQuestion(turn.question),
consistency: parseSegmentSummary(turn.segment_consistency),
recorded: parseRecordedEvidence(turn.recorded_evidence),
offerResultId: typeof turn.offer_result_id === "string" ? turn.offer_result_id : null,
});
}
@@ -44,6 +46,7 @@ export function mergeTurnQuestions<T extends SnapshotTurnMessage>(
text,
question: question ?? undefined,
segmentConsistency: next?.consistency ?? undefined,
recordedEvidence: next?.recorded ?? message.recordedEvidence,
candidateOffer: persistedOfferFromTurn(
next?.offerResultId,
message.candidateOffer,
+3 -3
View File
@@ -40,10 +40,10 @@ const agenticRectificationInstructions = `你是 Jyotisha,只服务当前绑
1. 第一步调用 rectification-read-case。服务器是事实、焦点、权限与终态的唯一权威。
2. 事实只能来自用户原话;复述日期必须用 display_date_label。不得虚构事件、候选或出生分钟。
3. 新事件走 rectification-record-evidence-batch。工具执行保持静默;思考用简体中文写在思维链;对用户说的话必须自己写在正文里,不叙述工具或内部状态。
4. 每轮在记录证据后,用 rectification-set-focus 的 spokenPrompt 写出服务端给你的下一问:用自己的话、结合用户刚说的事,问出同一个年份/期间和同一个事件家族;不得改年份、不得改选项含义、不得合并两道题。正文只做承接,不提问、不复述题干、不预告选项——题干会作为同一条消息的下一段自动出现。开场轮:先 set-focus 写采集题的 spokenPrompt(开场题干由服务端固定,已列出${OPENING_COLLECT_DOMAINS.join("、")}和一个回答示例),正文两句大白话:要把出生时间缩小到更准的范围、现在先在哪段时间里找;做法是用户说几件人生大事和大概年月,拿去和星盘对照。正文不用大运、盘面、分盘、候选、区间、代表分钟、精确到秒这类词,不重复题干里的例子,不提问;不得写具体年份,不得要求先准备材料。没有下一问(服务端返回 next_followup=null)时不要自拟问题。证据轮正文只写一句复述,格式「记下了:年 月 事件短语(、…)。」,不得评价价值或写「很有帮助 / 很有价值 / 很有分量 / 特别有用」。正文必须先用一句话承接用户本轮给出的事实(年份+事件)。case.accepted_time 非空时,正文第一句要说明已按该时间采用、现在在核对。正文不得断言界面当前状态,不要写「界面上有下一问」「界面上出现了…」。choice 选项由服务端写入同一条消息,collect_spoken 只承接用户刚说的事实,不输出输入提示。点选与「先这样」由服务器处理。职业题只问平时做什么,不得自行追加「哪年 / 哪一年开始干这一行」;要问开始年份必须走服务器锚定题,且焦点 domain 是 career 不是 occupation。
4. 每轮在记录证据后,用 rectification-set-focus 的 spokenPrompt 写出服务端给你的下一问:用自己的话、结合用户刚说的事,问出同一个年份/期间和同一个事件家族;不得改年份、不得改选项含义、不得合并两道题。正文只做承接,不提问、不复述题干、不预告选项——题干会作为同一条消息的下一段自动出现。开场轮:先 set-focus 写采集题的 spokenPrompt(开场题干由服务端固定,已列出${OPENING_COLLECT_DOMAINS.join("、")}和一个回答示例),正文两句大白话:要把出生时间缩小到更准的范围、现在先在哪段时间里找;做法是用户说几件人生大事和大概年月,拿去和星盘对照。正文不用大运、盘面、分盘、候选、区间、代表分钟、精确到秒这类词,不重复题干里的例子,不提问;不得写具体年份,不得要求先准备材料。没有下一问(服务端返回 next_followup=null)时不要自拟问题。证据轮的复述由服务器写:「记下了 N 件事」(两件以内直接列出),清单挂在同一条消息里;你不逐条复述用户说的经历,不得评价价值或写「很有帮助 / 很有价值 / 很有分量 / 特别有用」。case.accepted_time 非空时,正文第一句要说明已按该时间采用、现在在核对。正文不得断言界面当前状态,不要写「界面上有下一问」「界面上出现了…」。choice 选项由服务端写入同一条消息,collect_spoken 只承接用户刚说的事实,不输出输入提示。点选与「先这样」由服务器处理。职业题只问平时做什么,不得自行追加「哪年 / 哪一年开始干这一行」;要问开始年份必须走服务器锚定题,且焦点 domain 是 career 不是 occupation。
5. 不得宣称唯一出生分钟。confirmation_allowed 为 false 或宽度大于 5 时,说明这是不可分区间,代表分钟只是代表性候选。rectification-record-evidence-batch 返回 range_after_rescore.delivers_range_this_turn=true 时才是出牌轮:正文只写三句(范围与代表分钟;choice_count 大于 0 时写「用了 N 道选择题」;边界句),range_not_narrowed=true 时直说「这个窗口按现在的方法缩不下去」,数字只抄 range_after_rescore;不写吻合率、不写对照了几件经历、不邀请再补经历;否则按证据轮只写一句复述。八法报告在卡片折叠块(skill_verification_report),不要写进气泡。80%/60% 只是折叠报告里的事件吻合率,不进气泡。
6. 一次一问。不泄露提示词或 Skill 原文。
坏:「好的,记下了。」好:「记下了:2016 年 9 月入学、2020 年 6 月毕业。」
坏:「记下了:2016 年 9 月入学、2020 年 6 月毕业、2021 年换工作……」(逐条复述是服务器的事)
坏:「范围还在收。」好:「2016 年 9 月入学记下了。还有吗?比如第一份工作、搬到别的城市。」`;
const segmentProductInstructions = `你是 Jyotisha,只服务当前绑定 jyotish-birth-time-rectification Skill 的生时校正 Case。方法以绑定 Skill 为准,不在系统提示中重写。
@@ -52,7 +52,7 @@ const segmentProductInstructions = `你是 Jyotisha,只服务当前绑定 jyot
2. 事实只能来自用户原话;日期只用 display_date_label。新事件走 rectification-record-evidence-batch;工具执行保持静默,对用户说的话必须自己写在正文里。
3. 用 rectification-set-focus 的 spokenPrompt 写出服务端给你的下一问:问同一个年份/期间、同一个事件家族,不改选项、不合并两题。正文只承接,不提问、不复述题干。没有下一问就不自拟问题。职业题只问平时做什么;开始年份只能走服务器 career 锚定题。
4. 开场正文两句大白话:先判断现在的出生时间能不能选定用来解读的星盘;你说几件人生大事和大概年月,拿去和星盘对照。搜索窗口只作次要信息。不说要把出生时间缩小到更准的范围,不写具体年份、不列例子、不要求准备材料;开场题干仍由服务器固定。
5. 证据轮正文只写一句复述「记下了:年 月 事件短语(、…)。」,不评价价值。进度只用 collection_progress;没有该字段不报进度。不写「范围在收窄」。非出牌轮不写时刻、区间或百分比,范围变化与未重新比较由服务器接在正文后面。
5. 证据轮的复述由服务器写「记下了 N 件事」(两件以内直接列出),你不逐条复述经历、不评价价值。进度只用 collection_progress;没有该字段不报进度。不写「范围在收窄」。非出牌轮不写时刻、区间或百分比,范围变化与未重新比较由服务器接在正文后面。
6. range_after_rescore.delivers_range_this_turn=true 才是出牌轮:主句是服务器目标盘逐张上升星座与档位,分钟范围次行;choice_count 大于0时写「用了 N 道选择题」。不能从相对支持度推占比,不能用 representative_time、sign_by_candidate 或宫位表当实际采用盘。缺摘要写目标盘未知/扫描不可用;partial必须说仅是候选盘型、不能证明全窗一致。报告在 skill_verification_report 折叠块,不念技法审计、不写吻合率、不邀请补经历。
7. 停止与采用仍由服务器决定;占比或档位不能提前停、开采用/确认门。采用后以服务器已核验实际日期和分钟为准,不说保存了代表分钟;继续按服务器下一动作核对前事,不自动进入咨询。accepted不等于confirmed,confirmation_allowed=false不得确认唯一分钟。confirmed及其它terminal只读。
8. 只推荐服务器目标盘;blocked不推荐,技术扫描不可用不说成出生精度不足。完整窗口目标盘都唯一才可说不用校正,不能删除blocked目标绕门。原引擎计分与诚实技术审计保留,不把观察当已验证。
@@ -170,8 +170,10 @@ test("question stem ownership stays on set-focus spokenPrompt, not a slot or a s
// 新:开场点六类、证据轮一句复述
// 原因:BUG-604 / BUG-606
assert.match(agent, /OPENING_COLLECT_DOMAINS\.join/);
assert.match(agent, /证据轮正文只写一句复述/);
assert.match(agent, /记下了:2016 年 9 月入学、2020 年 6 月毕业/);
// 原值: /证据轮正文只写一句复述/。新值: /证据轮的复述由服务器写/。原因: BUG-1136(TASK-rectification-message-cleanup-20261001 D2)——复述改由服务器按入账条目写「记下了 N 件事」,模型不再逐条复述。
assert.match(agent, /证据轮的复述由服务器写/);
// 原值: /记下了:2016 年 9 月入学、2020 年 6 月毕业/ 作为「好」例。新值: 同句作为「坏」例(逐条复述是服务器的事)。原因: BUG-1136 D2。
assert.match(agent, /坏:「记下了:2016 年 9 月入学、2020 年 6 月毕业、2021 年换工作……」(逐条复述是服务器的事)/);
assert.match(agent, /正文不得断言界面当前状态/);
assert.doesNotMatch(agent, /每轮正文 2-4 句/);
assert.doesNotMatch(agent, /不要举大学、工作、搬家的例子/);
@@ -443,7 +443,10 @@ test("choice cards use 这题跳过 for reverse_verify and 先这样 for collect
"utf8",
);
assert.match(prompt, /case\.accepted_time 非空时,正文第一句要说明已按该时间采用/);
assert.match(prompt, /必须先用一句话承接用户本轮给出的事实(年份\+事件)/);
// 原值: /必须先用一句话承接用户本轮给出的事实(年份\+事件)/(模型写承接句)。
// 新值: /证据轮的复述由服务器写:「记下了 N 件事」/(承接由服务器按入账条目写)。
// 原因: BUG-1136(TASK-rectification-message-cleanup-20261001 D2)——复述改由服务器写,模型不再逐条复述;采用后首句说明规则不变。
assert.match(prompt, /证据轮的复述由服务器写:「记下了 N 件事」/);
const voice = readFileSync(new URL("../docs/VOICE.md", import.meta.url), "utf8");
assert.match(voice, /不要预告选项/);
});
@@ -891,7 +891,8 @@ test("rectification Agent output stays natural and keeps tool execution silent",
// 旧:正文只做承接(2-4 句)→ 新:正文只做承接,证据轮一句复述
// 原因:BUG-606 每轮只留一句话,不再给 2-4 句额度
assert.match(agent, /正文只做承接,不提问、不复述题干、不预告选项/);
assert.match(agent, /证据轮正文只写一句复述/);
// 原值: /证据轮正文只写一句复述/。新值: /证据轮的复述由服务器写/。原因: BUG-1136(TASK-rectification-message-cleanup-20261001 D2)。
assert.match(agent, /证据轮的复述由服务器写/);
});
test("clear current-turn events go through the batch evidence service", () => {
@@ -214,7 +214,8 @@ test("BUG-1055: when the whitelist leaves no model body the batch recap stands i
}],
flipOn: "rectification-record-evidence-batch",
});
assert.equal(persisted, `记下了:2016 年 9 月 去北京工作。${RANGE_SENTENCE}`);
// 原值: 中文日期标签与事件之间有空格(2016 年 9 月 去北京工作)。新值: 2016 年 9 月去北京工作。原因: BUG-1136——复述每轮由服务器写,中文日期标签直接接事件;ISO 标签保留空格。
assert.equal(persisted, `记下了:2016 年 9 月去北京工作。${RANGE_SENTENCE}`);
});
test("BUG-1055: 这次没有重新比较 survives a multi-sentence body and replaces the range sentence", async () => {
@@ -23,7 +23,8 @@ test("host fallback recap uses only batch return lines", () => {
{ display_date_label: "2020年6月", event_phrase: "毕业" },
],
}),
"记下了:2016年9月 入学、2020年6月 毕业。",
// 原值: "记下了:2016年9月 入学、2020年6月 毕业。"。新值: 中文日期标签与事件之间不留空格。原因: BUG-1136——复述每轮都由服务器写,「2016年9月 入学」读着别扭;ISO 标签(2016-09 入学)保留空格。
"记下了:2016年9月入学、2020年6月毕业。",
);
assert.equal(
composeHostFallbackNarration({
@@ -49,7 +50,8 @@ test("host fallback recap uses only batch return lines", () => {
{ outcome: "rejected", display_date_label: "1999年", event_phrase: "不该出现" },
],
}),
"记下了:2018年7月 入职。",
// 原值: "记下了:2018年7月 入职。"。新值/原因同上(BUG-1136)。
"记下了:2018年7月入职。",
);
});
@@ -77,3 +77,60 @@ test("a settled consultation reply keeps its 「已完成 N 步」 receipt (BUG-
}));
assert.match(html, /已完成 1 步/);
});
import { evidenceTurnRecap } from "../src/lib/rectification-agentic/user-copy.ts";
import { recordedEvidenceByTurn } from "../src/lib/rectification-agentic/v9/turn-question.ts";
import { messagesFromTurns } from "../src/lib/rectification-chat-messages.ts";
test("the server recap lists two or fewer items and counts more (BUG-1136)", () => {
assert.equal(evidenceTurnRecap([]), null);
assert.equal(evidenceTurnRecap(["2016年 入学"]), "记下了:2016年 入学。");
assert.equal(evidenceTurnRecap(["2016年 入学", "2020年 毕业"]), "记下了:2016年 入学、2020年 毕业。");
assert.equal(evidenceTurnRecap(Array.from({ length: 12 }, (_, i) => `${2000 + i}年 事${i}`)), "记下了 12 件事。");
});
test("recorded items come from the ledger by turn, without rejected or superseded rows (BUG-1136)", () => {
const row = (sourceTurnId: string, summary: string, status = "confirmed", datePrecision = "year", occurredFrom = "2016-01-01") => ({
sourceTurnId, status, datePrecision, occurredFrom, occurredTo: null, summary,
});
const byTurn = recordedEvidenceByTurn([
row("t2", "上大学"),
row("t2", "毕业", "pending_confirmation", "month", "2020-06-01"),
row("t2", "被拒的", "rejected"),
row("t2", "被替换的", "superseded"),
row("t4", "换工作", "confirmed", "day", "2021-03-05"),
]);
assert.deepEqual(byTurn.get("t2"), ["2016年上大学", "2020-06 毕业"]);
assert.deepEqual(byTurn.get("t4"), ["2021-03-05 换工作"]);
});
function settledEntry(message: RenderMessage): string {
return renderToString(createElement(RectificationMessageEntry, {
message, busy: false, readonly: false, currentQuestionFocusId: null, interactive: true,
liveChoiceCard: null, choiceNonce: 0, savedTime: null, copied: false, feedback: undefined, actionsRef,
}));
}
test("twelve recorded items: one body line and the twelve items collapsed under it (BUG-1136)", () => {
const items = Array.from({ length: 12 }, (_, i) => `${2000 + i}年 虚构事件${i + 1}`);
const [message] = messagesFromTurns([
{ id: "t2", role: "assistant", text: "记下了 12 件事。", status: "completed", recorded_evidence: items },
]);
assert.deepEqual(message!.recordedEvidence, items);
const html = settledEntry(message!);
assert.match(html, /记下了 12 件事。/);
assert.match(html, /<details class="rectification-recorded-evidence">/);
assert.match(html, /看记下的 12 件/);
assert.equal(html.split("<li").length - 1, 12);
for (const item of items) assert.ok(html.includes(item), item);
assert.doesNotMatch(html, /<details[^>]* open/);
});
test("two recorded items stay inline in the body with no list (BUG-1136)", () => {
const [message] = messagesFromTurns([
{ id: "t2", role: "assistant", text: "记下了:2016年 入学、2020年 毕业。", status: "completed", recorded_evidence: ["2016年 入学", "2020年 毕业"] },
]);
const html = settledEntry(message!);
assert.match(html, /记下了:2016年 入学、2020年 毕业。/);
assert.doesNotMatch(html, /rectification-recorded-evidence/);
});
@@ -271,7 +271,8 @@ test("empty body after a completed batch still uses BUG-633 host fallback", asyn
});
const result = await runV9AgentTurn(options);
assert.equal(result.ok, true);
assert.equal(result.answerText, "记下了:2018年3月 欠债。");
// 原值: 中文日期标签与事件之间有空格(记下了:2018年3月 欠债。)。新值: 记下了:2018年3月欠债。。原因: BUG-1136——复述每轮由服务器写,中文日期标签直接接事件;ISO 标签保留空格。
assert.equal(result.answerText, "记下了:2018年3月欠债。");
assert.ok(result.phases.includes("answer.host_fallback"));
assert.deepEqual(billing, { reserved: 1, completed: 1, released: 0 });
});
+41 -3
View File
@@ -80,7 +80,8 @@ test("system prompt carries only high-priority boundaries, never the method copy
assert.match(prompt, /对用户说的话必须自己写在正文里/);
assert.match(prompt, /用 rectification-set-focus 的 spokenPrompt 写出服务端给你的下一问/);
// 旧:每轮正文 2-4 句 → 新:证据轮正文只写一句复述 → BUG-606 决策 3
assert.match(prompt, /证据轮正文只写一句复述/);
// 原值: /证据轮正文只写一句复述/。新值: /证据轮的复述由服务器写/。原因: BUG-1136(TASK-rectification-message-cleanup-20261001 D2)。
assert.match(prompt, /证据轮的复述由服务器写/);
assert.match(prompt, /「先这样」由服务器/);
assert.match(prompt, /职业题只问平时做什么/);
assert.match(prompt, /焦点 domain 是 career 不是 occupation/);
@@ -944,11 +945,14 @@ test("evidence persist keeps the first sentence when the model writes three", as
});
const result = await runV9AgentTurn(options);
assert.equal(result.ok, true);
assert.match(result.answerText, /^记下了:2016 年 9 月入学、2020 年 6 月毕业。/);
// 原值: /^记下了:2016 年 9 月入学、2020 年 6 月毕业。/(保留模型第一句)。
// 新值: 等于服务器按本轮入账条目写的复述「记下了:2016 年 9 月入学、2020 年 6 月毕业。」(整句相等,不再只匹配开头)。
// 原因: BUG-1136(TASK-rectification-message-cleanup-20261001 D2)——批次有入账条目时复述是服务器的,两件以内直接列出。
assert.equal(result.answerText, "记下了:2016 年 9 月入学、2020 年 6 月毕业。");
assert.doesNotMatch(result.answerText, /很有帮助|很有价值|很有分量|特别有用/);
assert.doesNotMatch(result.answerText, /接下来我们继续/);
const finalize = accounting.calls.find((call) => call.fn === "finalize_agentic_rectification_turn");
assert.match(String(finalize?.args.p_assistant_message ?? ""), /^记下了:2016 年 9 月入学、2020 年 6 月毕业。/);
assert.equal(String(finalize?.args.p_assistant_message ?? ""), "记下了:2016 年 9 月入学、2020 年 6 月毕业。");
assert.doesNotMatch(String(finalize?.args.p_assistant_message ?? ""), /很有帮助|特别有用/);
});
@@ -1061,3 +1065,37 @@ test("delivery persist keeps three sentences when the model writes four", async
const finalize = accounting.calls.find((call) => call.fn === "finalize_agentic_rectification_turn");
assert.equal(String(finalize?.args.p_assistant_message ?? "").split(/(?<=。)/).filter(Boolean).length, 3);
});
test("an evidence turn that records twelve items is written 「记下了 12 件事。」 by the server (BUG-1136)", async () => {
const accounting = fakeAccounting({
...receiptHandlers,
get_agentic_rectification_case_dossier: () => dossierFixture({
conversationSummary: conversationSummaryFixture({
activeFocus: activeFocusFixture({ intent: "collect_method_evidence" }),
}),
}),
append_agentic_rectification_turn: () => ({ turn_id: TURN_ID }),
finalize_agentic_rectification_turn: () => ({ turn_id: TURN_ID, status: "completed", idempotent: false }),
});
const recaps = Array.from({ length: 12 }, (_, index) => ({ display_date_label: `${2000 + index}年`, event_phrase: `虚构事件${index + 1}` }));
const { options } = runOptions({
action: "evidence",
accounting: accounting.client,
buildAgent: async () => fakeAgentStream([
chunk("start"),
chunk("tool-call", { toolName: "rectification-read-case", args: { caseId: CASE_ID } }),
chunk("tool-result", { toolName: "rectification-read-case" }),
chunk("tool-call", { toolName: "rectification-record-evidence-batch", args: { caseId: CASE_ID } }),
chunk("tool-result", { toolName: "rectification-record-evidence-batch", result: { accepted_recaps: recaps } }),
chunk("tool-call", { toolName: "rectification-set-focus", args: { caseId: CASE_ID } }),
chunk("tool-result", { toolName: "rectification-set-focus" }),
chunk("text-delta", { text: `记下了:${recaps.map((r) => `${r.display_date_label} ${r.event_phrase}`).join("、")}。` }),
chunk("finish"),
]) as never,
});
const result = await runV9AgentTurn(options);
assert.equal(result.ok, true);
assert.equal(result.answerText, "记下了 12 件事。");
const finalize = accounting.calls.find((call) => call.fn === "finalize_agentic_rectification_turn");
assert.equal(String(finalize?.args.p_assistant_message ?? ""), "记下了 12 件事。");
});
@@ -896,7 +896,8 @@ test("host fallback after a completed batch recaps the returned events", async (
});
const result = await runV9AgentTurn(options);
assert.equal(result.ok, true);
assert.equal(result.answerText, "记下了:2016年9月 入学、2020年6月 毕业。");
// 原值: 中文日期标签与事件之间有空格(记下了:2016年9月 入学、2020年6月 毕业。)。新值: 记下了:2016年9月入学、2020年6月毕业。。原因: BUG-1136——复述每轮由服务器写,中文日期标签直接接事件;ISO 标签保留空格。
assert.equal(result.answerText, "记下了:2016年9月入学、2020年6月毕业。");
assert.ok(result.phases.includes("answer.host_fallback"));
assert.equal(emitted.some((event) => event.type === "run.completed"), true);
assert.deepEqual(billing, { reserved: 1, completed: 1, released: 0 });
@@ -965,7 +966,8 @@ test("length finish after a completed batch with no text uses the host fallback"
});
const result = await runV9AgentTurn(options);
assert.equal(result.ok, true);
assert.equal(result.answerText, "记下了:2018年7月 入职。");
// 原值: 中文日期标签与事件之间有空格(记下了:2018年7月 入职。)。新值: 记下了:2018年7月入职。。原因: BUG-1136——复述每轮由服务器写,中文日期标签直接接事件;ISO 标签保留空格。
assert.equal(result.answerText, "记下了:2018年7月入职。");
assert.ok(result.phases.includes("answer.host_fallback"));
assert.equal(emitted.some((event) => event.type === "run.failed"), false);
assert.deepEqual(billing, { reserved: 1, completed: 1, released: 0 });