feat(consult): sections presuppose nothing, examples carry no reading rule or job shape, corrections are acknowledged

- natalAnswerShapeBody: NATAL_SECTION_RULE replaces 「这个人是什么样、你们怎么相处、
  事情会怎么走」; the shared voice quotes the same constant (BUG-1164)
- product-voice / VOICE.md: no 逆行 in examples, parents example leads with the
  affliction and asks to confirm, married-presupposing Good example removed (BUG-1165)
- follow-up shape: a user correction is acknowledged, never defended (BUG-1166)
- career examples carry no job shape (升职 / 专业能力 / 做深现有工作) (BUG-1168)
- biography backtest golden regenerated with the real engine so the card carries
  brief 1's aspects, combustion, yoga domains and saturn_from_moon_house;
  backtest script gains --correction follow-up turns

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
This commit is contained in:
Jesse_Chen
2026-10-02 00:38:01 +08:00
co-authored by Claude Opus 5.5
parent 585c7370f8
commit 955bab691a
7 changed files with 215 additions and 42 deletions
@@ -28,9 +28,14 @@
* DS_KEY=... ./node_modules/.bin/tsx scripts/research/consult-biography-backtest.mts \
* [--out <dir>] [--repeats 2] [--figures a,b] [--domains parents,marriage,health,career] \
* [--model deepseek-flash] [--base-url https://api.deepseek.com] [--concurrency 4] [--dump-cards]
* [--correction barack_obama,marilyn_monroe] [--correction-text 我爸其实没怎么管过我]
*
* --dump-cards writes the model-visible tool result per figure x domain and
* calls no model. --recheck <dir> re-runs the deterministic checks of an
* calls no model. --correction adds, for each listed figure's parents answer,
* one follow-up turn in the same session (history = that question and answer)
* with --correction-text (T3 / D4 of the task brief); those answers are written
* as parents-correction-run<N>.md and listed in the summary for reading.
* --recheck <dir> re-runs the deterministic checks of an
* earlier run against the current rubric and rewrites its summary. The key is read from DS_KEY (or BACKTEST_MODEL_KEY) only and
* is never written anywhere. Output defaults to <repo>/scratch/ (gitignored);
* only the summary may be quoted into docs.
@@ -74,7 +79,7 @@ export const GAPS = [
"引擎调用换成 golden(同 capture 路径、raman 岁差、mean 交点、参考日 2026-09-27);产品用用户档案的岁差设置。",
"用户回合不带「用户称呼」:真实路由会带档案名,这里故意不给名人名字,免得模型凭记忆答生平。",
"natalToolInstruction 一句与 currentTimeContext 是 route.ts 内部函数,未导出,脚本里按原文复刻(见 routeToolInstruction / routeCurrentTime)。",
"没有内容审核(moderateOutput)、计费、标题旁路事件、历史摘要;单轮、无历史。",
"没有内容审核(moderateOutput)、计费、标题旁路事件、历史摘要;主回测单轮、无历史;--correction 纠正追问轮的历史只有上一问一答。",
"性别一律按档案未填(性别未知)处理。",
"模型经 Mastra 的 OpenAI 兼容通道直连 DeepSeek(model id 由参数给),不经产品数据库的模型目录。",
] as const;
@@ -123,6 +128,8 @@ function parseArgs(argv: readonly string[]) {
concurrency: Math.max(1, Number(value("concurrency") ?? 4)),
dumpCards: argv.includes("--dump-cards"),
recheck: value("recheck"),
correction: value("correction")?.split(",").filter(Boolean) ?? [],
correctionText: value("correction-text") ?? "我爸其实没怎么管过我",
};
}
@@ -229,7 +236,9 @@ function model(args: ReturnType<typeof parseArgs>, apiKey: string): ResolvedLang
};
}
type Job = { figure: GoldenFigure; domain: string; question: string; run: number };
/** A follow-up turn in the same session: the earlier question and answer are the history. */
type FollowUp = { priorQuestion: string; priorAnswer: string };
type Job = { figure: GoldenFigure; domain: string; question: string; run: number; followUp?: FollowUp };
type Usage = { inputTokens?: number; outputTokens?: number; reasoningTokens?: number; cachedInputTokens?: number };
@@ -249,17 +258,27 @@ function agentContext(job: Job, state: ConsultationRuntimeState, executed: strin
consultationMode: "verified_chart",
plan: createConsultationPlan({ userIntent: job.question, theme: job.domain as ConsultationDomain }),
theme: job.domain as ConsultationDomain,
followUpTurn: false,
followUpTurn: Boolean(job.followUp),
serverChart: serverChart(job.figure),
state,
runWorkflow: stubRunWorkflow(job.figure, executed) as never,
});
}
function userTurn(question: string) {
function historyOf(followUp: FollowUp | undefined) {
return followUp
? [
{ role: "user" as const, text: followUp.priorQuestion },
{ role: "assistant" as const, text: followUp.priorAnswer },
]
: [];
}
function userTurn(question: string, followUp?: FollowUp) {
return consultationUserTurnContent({
currentTime: routeCurrentTime(REQUEST_TIME),
instruction: `${routeToolInstruction}${natalUserTurnShape({ history: [] })}`,
// natalUserTurnShape reads the stored history, as route.ts does: a prior answer -> follow-up shape.
instruction: `${routeToolInstruction}${natalUserTurnShape({ history: historyOf(followUp) })}`,
question,
});
}
@@ -270,7 +289,11 @@ async function runJob(job: Job, resolved: ResolvedLanguageModel) {
const ctx = agentContext(job, state, executed);
const agent = getJyotishAgent(resolved, ctx);
const runId = ctx.requestId;
const baseMessages = [{ role: "user" as const, content: userTurn(job.question) }];
// route.ts consultationBaseMessages: stored history as plain messages, then this turn.
const baseMessages = [
...historyOf(job.followUp).map((message) => ({ role: message.role, content: message.text })),
{ role: "user" as const, content: userTurn(job.question, job.followUp) },
];
const streamOptions = {
runId,
maxSteps: AGENT_MAX_STEPS,
@@ -518,8 +541,41 @@ async function main() {
console.log(`${job.figure.id} ${job.domain} run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`);
return { figure: job.figure.id, domain: job.domain, run: job.run, chars: result.output.length, ...result, output: undefined, checks };
});
const corrections = await pool(
jobs.filter((job) => job.domain === "parents" && args.correction.includes(job.figure.id)),
args.concurrency,
async (prior) => {
const priorFile = join(args.out, prior.figure.id, `parents-run${prior.run}.md`);
const priorText = readFileSync(priorFile, "utf8");
const priorAnswer = priorText.slice(priorText.indexOf("\n---\n\n") + 6).trim();
const job: Job = { ...prior, question: args.correctionText, followUp: { priorQuestion: prior.question, priorAnswer } };
let result: Awaited<ReturnType<typeof runJob>>;
try {
result = await runJob(job, resolved);
} catch (error) {
result = { output: "", failure: error instanceof Error ? error.message : String(error), executed: [], seconds: 0, usage: {}, cardChars: null, visibleChars: null, methodologySections: 0, lookups: 0, finish: null };
}
const file = join(args.out, job.figure.id, `parents-correction-run${job.run}.md`);
writeFileSync(file, [
`# ${job.figure.label} · parents 纠正追问 · run ${job.run}`,
"",
`- 上一轮:${priorFile}`,
`- 追问:${job.question}`,
`- 模型:${args.model};用时 ${result.seconds.toFixed(1)} s;结束:${result.finish ?? "?"};查卡 ${result.lookups} 次`,
`- 用量:${JSON.stringify(result.usage)}`,
...(result.failure ? [`- 失败:${result.failure}`] : []),
"",
"---",
"",
result.output || "(无回答)",
"",
].join("\n"));
console.log(`${job.figure.id} parents correction run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`);
return { figure: job.figure.id, run: job.run, chars: result.output.length, seconds: result.seconds, failure: result.failure, usage: result.usage };
},
);
const total: Usage = {};
for (const row of rows) addUsage(total, row.usage);
for (const row of [...rows, ...corrections]) addUsage(total, row.usage);
const summary = {
model: args.model,
started_at: new Date(started).toISOString(),
@@ -529,6 +585,7 @@ async function main() {
usage: total,
gaps: GAPS,
rows,
corrections,
};
writeSummary(args.out, summary as unknown as Json & { rows: Row[] }, args.model);
console.log(`summary: ${join(args.out, "summary.md")}`);