Files
Jyotisha/frontend/scripts/research/consult-biography-backtest.mts
T
Jesse_ChenandClaude Opus 5.5 7806c72f2c feat(consult): cards carry functional profile v2; close round-4 backtest residue (BUG-1158, BUG-1176, BUG-1182)
- Evidence-card and biography-backtest goldens regenerated with the capture
  scripts (PYTHONHASHSEED=0, cache TTL 0; byte-identical twice); only the
  functional-role fields change.
- CAREER_FIELD_ASK_RULE: A10 / AL / Sun / 10th lord in 4/6/8/12 say
  resistance or cost only, never "not on stage / go deep", and houses are not
  translated into fields.
- AFFLICTION_RANGE_RULE: no portrait of an afflicted person from the lord's
  house or a benefic occupant ("managing / arranging / cares about your
  future"), actions do not presuppose reaching them, one ask per answer.
- Examples that paired "4th lord in 10th" with "her attention is on your
  future / close" now state the line is unafflicted (Monroe's chart had that
  exact placement and copied it). Retrograde line also bans "drags on".
- Backtest script: several follow-up domains in one run.
Changed assertions carry 原值 / 新值 / 原因 in the test comments.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
2026-10-02 12:32:02 +08:00

611 lines
30 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Biography backtest for the ordinary (natal) consultation answer
* (TASK-consult-no-presupposition-and-backtest-20261001 T4, D5).
*
* Runs the product's real consultation agent against public Rodden AA charts
* and writes each answer for a human / Claude to score against
* docs/research/consult_biography_backtest_rubric.json. Nothing here copies
* prompt text: the system prompt, the skill method block, the tools, the
* evidence card, the methodology checklist, the user-turn shape and the answer
* stream handling all come from the product modules, so a run always tests the
* prompts on the current checkout.
*
* What is real Where it comes from
* system prompt + bound skill method getJyotishAgent (src/mastra/index.ts)
* tool loop, prepareStep, settings Mastra Agent.stream + consultationNatalPrepareStep
* + consultationGenerationSettings (route.ts natalStreamOptions)
* tool result (contract, card, createConsultationTools -> toModelEvidenceView
* methodology checklist) (engine call replaced by the golden workflow)
* evidence lookup tool the real read-consultation-evidence
* user turn consultationUserTurnContent + natalUserTurnShape
* answer extraction, pass 4, retries, streamAgentResponse (stepScopedAnswer, pass4Mode,
* length continuation retry / retryForAnswer / continueAfterLength as route.ts)
*
* Gaps against app/api/consult/route.ts are listed in GAPS below and in
* docs/testing/consult-affliction-backtest-20261001.md.
*
* Usage (from frontend/, Node 22):
* DS_KEY=... ./node_modules/.bin/tsx scripts/research/consult-biography-backtest.mts \
* [--out <dir>] [--repeats 2] [--figures a,b] [--domains parents,marriage,health,career] \
* [--model deepseek-flash] [--base-url https://api.deepseek.com] [--concurrency 4] [--dump-cards]
* [--correction barack_obama,marilyn_monroe] [--correction-text 我爸其实没怎么管过我]
* [--correction-domain career] [--correction-texts "zinedine_zidane=我是足球运动员|barack_obama=我是政治家"]
* Several follow-up domains in one run: --correction-domain parents,career with
* domain-qualified entries, e.g. --correction "parents:barack_obama,career:zinedine_zidane"
* --correction-texts "career:zinedine_zidane=我是足球运动员" (an unqualified entry or text
* applies to every listed domain).
*
* --dump-cards writes the model-visible tool result per figure x domain and
* calls no model. --correction adds, for each listed figure's parents answer,
* one follow-up turn in the same session (history = that question and answer)
* with --correction-text (T3 / D4 of the task brief); those answers are written
* as parents-correction-run<N>.md and listed in the summary for reading.
* --recheck <dir> re-runs the deterministic checks of an
* earlier run against the current rubric and rewrites its summary. The key is read from DS_KEY (or BACKTEST_MODEL_KEY) only and
* is never written anywhere. Output defaults to <repo>/scratch/ (gitignored);
* only the summary may be quoted into docs.
*/
import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
import { dirname, join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { getJyotishAgent } from "../../src/mastra/index.ts";
import {
AGENT_MAX_STEPS,
consultationContinueGenerationSettings,
consultationGenerationSettings,
consultationNatalPrepareStep,
consultationStepBudgetReceipt,
createConsultationAgentContext,
createConsultationRuntimeState,
publicConsultationRuntimeSteps,
type ConsultationRuntimeState,
} from "../../src/mastra/consultation-tools.ts";
import {
consultationWorkflowResponseSchema,
minuteSensitiveThemesFromBirthTimeSensitivity,
} from "../../src/mastra/consultation-workflow.ts";
import type { ResolvedLanguageModel } from "../../src/mastra/model.ts";
import { createConsultationPlan } from "../../src/lib/consultation-plan.ts";
import { consultationUserTurnContent } from "../../src/lib/consultation-session-history.ts";
import { consultationContinueMessages, natalUserTurnShape } from "../../src/lib/consultation-thinking-plan.ts";
import { streamAgentResponse } from "../../src/lib/stream-agent-response.ts";
import type { ConsultationDomain } from "../../src/lib/consultation-domain-registry.ts";
import type { ServerChartConsultation } from "../../src/lib/consultation-route-service.ts";
const HERE = dirname(fileURLToPath(import.meta.url));
const FRONTEND = resolve(HERE, "../..");
const ROOT = resolve(FRONTEND, "..");
const GOLDEN = join(FRONTEND, "tests/fixtures/consult-biography-backtest-golden.json");
const RUBRIC = join(ROOT, "docs/research/consult_biography_backtest_rubric.json");
/** Where this harness differs from the live route; repeated in the summary. */
export const GAPS = [
"引擎调用换成 golden(同 capture 路径、raman 岁差、mean 交点、参考日 2026-09-27);产品用用户档案的岁差设置。",
"用户回合不带「用户称呼」:真实路由会带档案名,这里故意不给名人名字,免得模型凭记忆答生平。",
"natalToolInstruction 一句与 currentTimeContext 是 route.ts 内部函数,未导出,脚本里按原文复刻(见 routeToolInstruction / routeCurrentTime)。",
"没有内容审核(moderateOutput)、计费、标题旁路事件、历史摘要;主回测单轮、无历史;--correction 纠正追问轮的历史只有上一问一答。",
"性别一律按档案未填(性别未知)处理。",
"模型经 Mastra 的 OpenAI 兼容通道直连 DeepSeek(model id 由参数给),不经产品数据库的模型目录。",
] as const;
// route.ts (natalToolInstruction, entrypoint undefined = ordinary chat). Not exported there.
const routeToolInstruction = "如需新的个人星盘结论,必须调用服务器绑定的排盘工具。";
// route.ts currentTimeContext. Not exported there.
function routeCurrentTime(now: Date) {
const chinaTime = new Date(now.getTime() + 8 * 60 * 60 * 1000).toISOString().replace("T", " ").slice(0, 19);
return `服务端当前时间(权威):${now.toISOString()};中国标准时间(UTC+8):${chinaTime}。涉及“现在、今天、今年、未来几个月”等相对时间时,以此为准。`;
}
// The golden is computed for this reference date; the request clock matches it.
const REQUEST_TIME = new Date("2026-09-27T04:00:00.000Z");
type Json = Record<string, unknown>;
type GoldenRoute = { domain: string; engine_route: string; question: string };
type GoldenFigure = {
id: string;
label: string;
rodden_rating: string;
shared_chart: Json;
routes: Record<string, Json & { chart?: Json; chart_modules?: Json }>;
};
type Golden = { routes: GoldenRoute[]; figures: GoldenFigure[] };
type RubricFigure = {
id: string;
domains: Record<string, { must_not?: { claim: string; literals?: string[] }[] }>;
};
function parseArgs(argv: readonly string[]) {
const value = (name: string) => {
const index = argv.indexOf(`--${name}`);
return index >= 0 ? argv[index + 1] : undefined;
};
const stamp = new Date().toISOString().replace(/[:.]/g, "-");
return {
out: resolve(value("out") ?? join(ROOT, "scratch/consult-biography-backtest", stamp)),
repeats: Math.max(1, Number(value("repeats") ?? 2)),
figures: value("figures")?.split(",").filter(Boolean) ?? null,
domains: value("domains")?.split(",").filter(Boolean) ?? ["parents", "marriage", "health", "career"],
model: value("model") ?? "deepseek-flash",
baseUrl: value("base-url") ?? "https://api.deepseek.com",
concurrency: Math.max(1, Number(value("concurrency") ?? 4)),
dumpCards: argv.includes("--dump-cards"),
recheck: value("recheck"),
correction: value("correction")?.split(",").filter(Boolean) ?? [],
correctionText: value("correction-text") ?? "我爸其实没怎么管过我",
// Follow-up domain (default parents) and optional per-figure texts
// "figure=text|figure=text" (career follow-up: the user states the real job).
correctionDomains: (value("correction-domain") ?? "parents").split(",").filter(Boolean),
correctionTexts: Object.fromEntries((value("correction-texts") ?? "").split("|").filter(Boolean).map((pair) => {
const at = pair.indexOf("=");
return [pair.slice(0, at), pair.slice(at + 1)];
})) as Record<string, string>,
};
}
/** One route's engine response, with the figure's shared chart put back exactly as captured. */
function routeWorkflow(figure: GoldenFigure, domain: string): Json {
const route = figure.routes[domain];
if (!route) throw new Error(`golden has no ${domain} route for ${figure.id}`);
const { chart: chartOverrides, chart_modules: moduleOverrides, ...rest } = structuredClone(route);
const shared = structuredClone(figure.shared_chart);
const chart = {
...shared,
...(chartOverrides ?? {}),
modules: { ...(shared.modules as Json), ...(moduleOverrides ?? {}) },
};
return { ...rest, chart };
}
// The engine route each product domain runs as (consultation-domain-registry):
// parents / children / family all run the engine's family route.
const GOLDEN_ROUTE_FOR_DOMAIN: Record<string, string> = {
parents: "parents",
family: "parents",
children: "parents",
marriage: "marriage",
health: "health",
career: "career",
};
function stubRunWorkflow(figure: GoldenFigure, executed: string[]) {
return async (input: { theme: string }) => {
const key = GOLDEN_ROUTE_FOR_DOMAIN[input.theme];
if (!key) throw new Error(`backtest golden has no engine route for domain ${input.theme}`);
executed.push(input.theme);
// runConsultationWorkflow: schema parse, then minute-sensitive themes on the policy.
const data = consultationWorkflowResponseSchema.parse(routeWorkflow(figure, key));
const consumer = data.consumer_context as Json;
return {
...data,
consumer_context: {
...consumer,
answer_policy: {
...(consumer.answer_policy as Json),
minute_sensitive_themes: minuteSensitiveThemesFromBirthTimeSensitivity(data.birth_time_sensitivity),
},
},
} as unknown as typeof data;
};
}
function serverChart(figure: GoldenFigure): ServerChartConsultation {
const birth = (figure.shared_chart.birth ?? {}) as Json;
const route = routeWorkflow(figure, "parents");
const body = ((route.chart as Json).birth ?? birth) as Json;
const num = (key: string, fallback = 0) => (typeof body[key] === "number" ? body[key] as number : fallback);
// Birth fields only shape the tool input schema; the stub ignores them.
const toolInput = {
year: num("year", 1950),
month: num("month", 1),
day: num("day", 1),
hour: num("hour"),
minute: num("minute"),
city: "public chart",
lat: num("lat", num("latitude")),
lon: num("lon", num("longitude")),
tz: num("tz", num("timezone")),
ayanamsa: "raman",
declared_accuracy: "minute",
time_source: "birth_record",
} as unknown as ServerChartConsultation["toolInput"];
return {
name: "backtest",
toolInput,
truth: {
birthDate: "1950-01-01",
reportedBirthTime: null,
activeBirthTime: null,
selectedTimeKind: "reported",
birthTimeSource: "birth_record",
birthTimeStatus: "reported",
placeLabel: "public chart",
placeCodes: { countryCode: null, provinceCode: null, cityCode: null, districtCode: null },
placeId: null,
placeType: null,
placeProvider: null,
timezoneId: null,
timezoneSource: null,
latitude: toolInput.lat,
longitude: toolInput.lon,
timezoneOffset: toolInput.tz,
},
gender: null,
} as ServerChartConsultation;
}
function model(args: ReturnType<typeof parseArgs>, apiKey: string): ResolvedLanguageModel {
return {
id: `backtest-${args.model}`,
label: args.model,
description: "biography backtest",
creditCost: 0,
isDefault: false,
mode: "compatible",
model: { providerId: "deepseek", modelId: args.model, url: args.baseUrl, apiKey } as ResolvedLanguageModel["model"],
};
}
/** A follow-up turn in the same session: the earlier question and answer are the history. */
type FollowUp = { priorQuestion: string; priorAnswer: string };
type Job = { figure: GoldenFigure; domain: string; question: string; run: number; followUp?: FollowUp };
type Usage = { inputTokens?: number; outputTokens?: number; reasoningTokens?: number; cachedInputTokens?: number };
function addUsage(total: Usage, usage: unknown) {
const row = (usage ?? {}) as Record<string, unknown>;
for (const key of ["inputTokens", "outputTokens", "reasoningTokens", "cachedInputTokens"] as const) {
const value = row[key];
if (typeof value === "number") total[key] = (total[key] ?? 0) + value;
}
}
function agentContext(job: Job, state: ConsultationRuntimeState, executed: string[]) {
return createConsultationAgentContext({
userId: "biography-backtest",
sessionId: `backtest-${job.figure.id}`,
requestId: `backtest-${job.figure.id}-${job.domain}-${job.run}`,
consultationMode: "verified_chart",
plan: createConsultationPlan({ userIntent: job.question, theme: job.domain as ConsultationDomain }),
theme: job.domain as ConsultationDomain,
followUpTurn: Boolean(job.followUp),
serverChart: serverChart(job.figure),
state,
runWorkflow: stubRunWorkflow(job.figure, executed) as never,
});
}
function historyOf(followUp: FollowUp | undefined) {
return followUp
? [
{ role: "user" as const, text: followUp.priorQuestion },
{ role: "assistant" as const, text: followUp.priorAnswer },
]
: [];
}
function userTurn(question: string, followUp?: FollowUp) {
return consultationUserTurnContent({
currentTime: routeCurrentTime(REQUEST_TIME),
// natalUserTurnShape reads the stored history, as route.ts does: a prior answer -> follow-up shape.
instruction: `${routeToolInstruction}${natalUserTurnShape({ history: historyOf(followUp) })}`,
question,
});
}
async function runJob(job: Job, resolved: ResolvedLanguageModel) {
const state = createConsultationRuntimeState();
const executed: string[] = [];
const ctx = agentContext(job, state, executed);
const agent = getJyotishAgent(resolved, ctx);
const runId = ctx.requestId;
// route.ts consultationBaseMessages: stored history as plain messages, then this turn.
const baseMessages = [
...historyOf(job.followUp).map((message) => (message.role === "user"
? { role: "user" as const, content: message.text }
: { role: "assistant" as const, content: message.text })),
{ role: "user" as const, content: userTurn(job.question, job.followUp) },
];
const streamOptions = {
runId,
maxSteps: AGENT_MAX_STEPS,
...consultationGenerationSettings(resolved.model),
};
const natalStreamOptions = { ...streamOptions, prepareStep: consultationNatalPrepareStep };
const usages: Promise<unknown>[] = [];
const startedAt = Date.now();
let output: string | null = null;
let failure: string | null = null;
const first = await agent.stream(baseMessages, natalStreamOptions);
usages.push(first.totalUsage);
const response = streamAgentResponse({
runId,
requestId: runId,
state,
stream: first.fullStream as never,
requireTool: true,
retry: async () => {
const retried = await agent.stream([
...baseMessages,
{ role: "user" as const, content: "运行合同不完整:本次尚未取得服务器计算结果。请调用 run-jyotish-consultation 完成计算,再据此回答;不要在工具参数中添加出生资料。" },
], natalStreamOptions);
usages.push(retried.totalUsage);
return retried.fullStream as never;
},
retryForAnswer: async (retryHint?: string) => {
const retried = await agent.stream([
...baseMessages,
{ role: "user" as const, content: `服务器计算已经完成,但上一轮没有输出任何回答文本。请重新取回本次计算结果,然后直接给出回答;不要只描述过程或工具调用。${retryHint ? `\n${retryHint}` : ""}` },
], natalStreamOptions);
usages.push(retried.totalUsage);
return retried.fullStream as never;
},
continueAfterLength: async (text: string, evidence?: unknown) => {
const continued = await agent.stream(consultationContinueMessages(baseMessages, text, evidence), {
...streamOptions,
...consultationContinueGenerationSettings(resolved.model),
});
usages.push(continued.totalUsage);
return continued.fullStream as never;
},
stepScopedAnswer: true,
pass4Mode: "verified_chart",
toolStatus: () => {
const status = state.workflowReceipt?.status;
return status === "ready" || status === "degraded" ? status : "blocked";
},
receipt: () => ({
runId,
runtime: "mastra-agentic",
skill: {
name: "jyotish-vedic-astrology",
loaded: state.jyotishSkillBound,
referenceReads: state.skillReferenceReadCount,
methodologySections: state.methodologySectionCount,
},
steps: publicConsultationRuntimeSteps(state),
stepBudget: consultationStepBudgetReceipt(state),
workflow: state.workflowReceipt ?? { route: job.domain, status: "blocked", preciseTiming: "blocked", missingLayers: [], domains: [job.domain as ConsultationDomain] },
techniqueTruth: state.techniqueTruth ?? "unknown",
}) as never,
onComplete: (text: string) => {
output = text;
},
onError: (error: unknown) => {
failure = error instanceof Error ? error.message : String(error);
},
});
// Drain the public event stream; that is what drives the run.
await response.text();
const usage: Usage = {};
for (const item of await Promise.allSettled(usages)) {
if (item.status === "fulfilled") addUsage(usage, item.value);
}
// Assigned inside the stream callbacks, which TypeScript cannot see.
const settledOutput = output as string | null;
const settledFailure = failure as string | null;
return {
output: settledOutput ?? "",
failure: settledFailure,
executed,
seconds: (Date.now() - startedAt) / 1000,
usage,
cardChars: state.evidenceCardChars ?? null,
visibleChars: state.modelVisibleChars ?? null,
methodologySections: state.methodologySectionCount,
lookups: state.evidenceLookupCallCount,
finish: state.composeFinishReason ?? state.modelFinishReason ?? null,
};
}
// ---- deterministic checks (judgment scoring is done by reading the answers) ----
const BANNED = [
{ id: "retrograde_repeat", pattern: /逆行[^。!?\n]{0,30}(反复|打回来|来回|拉回|折返|回头)/gu },
{ id: "repeat_retrograde", pattern: /(反复|打回来|来回)[^。!?\n]{0,12}逆行/gu },
];
// 他 / 她 when the card says 性别未知: only for the partner (marriage) or children.
const PRONOUN = /(?<![其吉])[他她](?![们人])/gu;
export function deterministicChecks(answer: string, domain: string, mustNot: { claim: string; literals?: string[] }[]) {
const context = (index: number) => answer.slice(Math.max(0, index - 14), index + 16).replace(/\s+/g, " ");
const banned = BANNED.flatMap((rule) => [...answer.matchAll(rule.pattern)].map((hit) => ({ rule: rule.id, text: hit[0] })));
const pronouns = domain === "marriage" || domain === "children"
? [...answer.matchAll(PRONOUN)].map((hit) => context(hit.index ?? 0))
: [];
const literalHits = mustNot.flatMap((item) => (item.literals ?? [])
.filter((literal) => answer.includes(literal))
.map((literal) => ({ claim: item.claim, literal })));
return { banned, pronouns, literalHits };
}
async function pool<T, R>(items: readonly T[], size: number, fn: (item: T) => Promise<R>) {
const results: R[] = new Array(items.length);
let next = 0;
await Promise.all(Array.from({ length: Math.min(size, items.length) }, async () => {
while (next < items.length) {
const index = next++;
results[index] = await fn(items[index]!);
}
}));
return results;
}
async function dumpCards(jobs: readonly Job[], out: string) {
for (const job of jobs.filter((item) => item.run === 1)) {
const state = createConsultationRuntimeState();
const ctx = agentContext(job, state, []);
const fake = model({ ...parseArgs([]), model: "card-dump" }, "unused");
const agent = getJyotishAgent(fake, ctx);
const tools = await agent.listTools();
const tool = tools["run-jyotish-consultation"] as unknown as { execute: (input: unknown, context: unknown) => Promise<unknown> };
const view = await tool.execute({ question: job.question, domains: [job.domain] }, {});
const path = join(out, "cards", `${job.figure.id}-${job.domain}.json`);
mkdirSync(dirname(path), { recursive: true });
writeFileSync(path, `${JSON.stringify(view, null, 1)}\n`);
console.log(`card ${path} (${JSON.stringify(view).length} chars)`);
}
}
// The product logs every reasoning delta (consultation-budget.ts); a batch run
// would drown its own progress lines in them.
const productInfo = console.info.bind(console);
console.info = (...items: unknown[]) => {
if (typeof items[0] === "string" && items[0].startsWith("[consult-reasoning]")) return;
productInfo(...items);
};
type Row = { figure: string; domain: string; run: number; chars: number; seconds: number; checks: ReturnType<typeof deterministicChecks> };
function writeSummary(out: string, summary: Json & { rows: Row[] }, modelId: string) {
const rows = summary.rows;
summary.banned_hits = rows.reduce((sum, row) => sum + row.checks.banned.length, 0);
summary.pronoun_hits = rows.reduce((sum, row) => sum + row.checks.pronouns.length, 0);
summary.must_not_literal_hits = rows.reduce((sum, row) => sum + row.checks.literalHits.length, 0);
writeFileSync(join(out, "summary.json"), `${JSON.stringify(summary, null, 1)}\n`);
writeFileSync(join(out, "summary.md"), [
`# 名人生平回测 · ${modelId}`,
"",
`开始 ${summary.started_at},${rows.length} 份,墙钟 ${Number(summary.wall_seconds).toFixed(0)} s,失败 ${summary.failures} 份。`,
`用量合计:${JSON.stringify(summary.usage)}`,
`确定性检查:禁句 ${summary.banned_hits},性别代词(婚恋/子女)${summary.pronoun_hits},must_not 原文 ${summary.must_not_literal_hits}。判断类评分需逐份阅读。`,
"",
"| 名人 | 领域 | run | 字数 | 秒 | 禁句 | 代词 | must_not |",
"| --- | --- | --- | --- | --- | --- | --- | --- |",
...rows.map((row) => `| ${row.figure} | ${row.domain} | ${row.run} | ${row.chars} | ${row.seconds.toFixed(0)} | ${row.checks.banned.length} | ${row.checks.pronouns.length} | ${row.checks.literalHits.length} |`),
"",
"## must_not 原文命中",
"",
...rows.flatMap((row) => row.checks.literalHits.map((hit) => `- ${row.figure} · ${row.domain} · run${row.run}:「${hit.literal}」(${hit.claim})`)),
"",
"## 与真实路由的差异",
"",
...GAPS.map((gap) => `- ${gap}`),
"",
].join("\n"));
}
function recheck(out: string, rubric: { figures: RubricFigure[] }) {
const summary = JSON.parse(readFileSync(join(out, "summary.json"), "utf8")) as Json & { rows: Row[]; model: string };
for (const row of summary.rows) {
const text = readFileSync(join(out, row.figure, `${row.domain}-run${row.run}.md`), "utf8");
const answer = text.slice(text.indexOf("\n---\n\n") + 6).trim();
const mustNot = rubric.figures.find((item) => item.id === row.figure)?.domains[row.domain]?.must_not ?? [];
row.checks = deterministicChecks(answer === "(无回答)" ? "" : answer, row.domain, mustNot);
}
writeSummary(out, summary, summary.model);
console.log(`rechecked: ${join(out, "summary.md")}`);
}
async function main() {
const args = parseArgs(process.argv.slice(2));
const golden = JSON.parse(readFileSync(GOLDEN, "utf8")) as Golden;
const rubric = JSON.parse(readFileSync(RUBRIC, "utf8")) as { figures: RubricFigure[] };
if (args.recheck) {
recheck(resolve(args.recheck), rubric);
return;
}
const questions = new Map(golden.routes.map((route) => [route.domain, route.question]));
const figures = golden.figures.filter((figure) => !args.figures || args.figures.includes(figure.id));
const jobs: Job[] = figures.flatMap((figure) => args.domains.flatMap((domain) => {
const question = questions.get(domain);
if (!question) throw new Error(`no golden question for domain ${domain}`);
return Array.from({ length: args.repeats }, (_, index) => ({ figure, domain, question, run: index + 1 }));
}));
mkdirSync(args.out, { recursive: true });
if (args.dumpCards) {
await dumpCards(jobs, args.out);
return;
}
const apiKey = process.env.DS_KEY ?? process.env.BACKTEST_MODEL_KEY;
if (!apiKey) throw new Error("Set DS_KEY (or BACKTEST_MODEL_KEY); without a key the backtest is an environment gap.");
const resolved = model(args, apiKey);
const started = Date.now();
const rows = await pool(jobs, args.concurrency, async (job) => {
let result: Awaited<ReturnType<typeof runJob>>;
try {
result = await runJob(job, resolved);
} catch (error) {
result = { output: "", failure: error instanceof Error ? error.message : String(error), executed: [], seconds: 0, usage: {}, cardChars: null, visibleChars: null, methodologySections: 0, lookups: 0, finish: null };
}
const mustNot = rubric.figures.find((item) => item.id === job.figure.id)?.domains[job.domain]?.must_not ?? [];
const checks = deterministicChecks(result.output, job.domain, mustNot);
const file = join(args.out, job.figure.id, `${job.domain}-run${job.run}.md`);
mkdirSync(dirname(file), { recursive: true });
writeFileSync(file, [
`# ${job.figure.label} · ${job.domain} · run ${job.run}`,
"",
`- 问题:${job.question}`,
`- 模型:${args.model};用时 ${result.seconds.toFixed(1)} s;结束:${result.finish ?? "?"};查卡 ${result.lookups} 次;清单段 ${result.methodologySections}`,
`- 执行领域:${result.executed.join(", ") || "无"};卡 ${result.cardChars ?? "?"} 字符,工具结果 ${result.visibleChars ?? "?"} 字符`,
`- 用量:${JSON.stringify(result.usage)}`,
`- 确定性检查:禁句 ${checks.banned.length},性别代词 ${checks.pronouns.length},must_not 原文 ${checks.literalHits.length}`,
...(result.failure ? [`- 失败:${result.failure}`] : []),
...checks.banned.map((hit) => ` - 禁句 ${hit.rule}:「${hit.text}」`),
...checks.pronouns.map((hit) => ` - 代词:「${hit}」`),
...checks.literalHits.map((hit) => ` - must_not:「${hit.literal}」(${hit.claim})`),
"",
"---",
"",
result.output || "(无回答)",
"",
].join("\n"));
console.log(`${job.figure.id} ${job.domain} run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`);
return { figure: job.figure.id, domain: job.domain, run: job.run, chars: result.output.length, ...result, output: undefined, checks };
});
const corrections = await pool(
jobs.filter((job) => args.correctionDomains.includes(job.domain)
&& (args.correction.includes(job.figure.id) || args.correction.includes(`${job.domain}:${job.figure.id}`))),
args.concurrency,
async (prior) => {
const priorFile = join(args.out, prior.figure.id, `${prior.domain}-run${prior.run}.md`);
const priorText = readFileSync(priorFile, "utf8");
const priorAnswer = priorText.slice(priorText.indexOf("\n---\n\n") + 6).trim();
const question = args.correctionTexts[`${prior.domain}:${prior.figure.id}`] ?? args.correctionTexts[prior.figure.id] ?? args.correctionText;
const job: Job = { ...prior, question, followUp: { priorQuestion: prior.question, priorAnswer } };
let result: Awaited<ReturnType<typeof runJob>>;
try {
result = await runJob(job, resolved);
} catch (error) {
result = { output: "", failure: error instanceof Error ? error.message : String(error), executed: [], seconds: 0, usage: {}, cardChars: null, visibleChars: null, methodologySections: 0, lookups: 0, finish: null };
}
const file = join(args.out, job.figure.id, `${job.domain}-correction-run${job.run}.md`);
writeFileSync(file, [
`# ${job.figure.label} · ${job.domain} 纠正追问 · run ${job.run}`,
"",
`- 上一轮:${priorFile}`,
`- 追问:${job.question}`,
`- 模型:${args.model};用时 ${result.seconds.toFixed(1)} s;结束:${result.finish ?? "?"};查卡 ${result.lookups} 次`,
`- 用量:${JSON.stringify(result.usage)}`,
...(result.failure ? [`- 失败:${result.failure}`] : []),
"",
"---",
"",
result.output || "(无回答)",
"",
].join("\n"));
console.log(`${job.figure.id} ${job.domain} correction run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`);
return { figure: job.figure.id, domain: job.domain, run: job.run, chars: result.output.length, seconds: result.seconds, failure: result.failure, usage: result.usage };
},
);
const total: Usage = {};
for (const row of [...rows, ...corrections]) addUsage(total, row.usage);
const summary = {
model: args.model,
started_at: new Date(started).toISOString(),
wall_seconds: (Date.now() - started) / 1000,
jobs: rows.length,
failures: rows.filter((row) => row.failure || row.chars === 0).length,
usage: total,
gaps: GAPS,
rows,
corrections,
};
writeSummary(args.out, summary as unknown as Json & { rows: Row[] }, args.model);
console.log(`summary: ${join(args.out, "summary.md")}`);
}
await main();