- Evidence-card and biography-backtest goldens regenerated with the capture
scripts (PYTHONHASHSEED=0, cache TTL 0; byte-identical twice); only the
functional-role fields change.
- CAREER_FIELD_ASK_RULE: A10 / AL / Sun / 10th lord in 4/6/8/12 say
resistance or cost only, never "not on stage / go deep", and houses are not
translated into fields.
- AFFLICTION_RANGE_RULE: no portrait of an afflicted person from the lord's
house or a benefic occupant ("managing / arranging / cares about your
future"), actions do not presuppose reaching them, one ask per answer.
- Examples that paired "4th lord in 10th" with "her attention is on your
future / close" now state the line is unafflicted (Monroe's chart had that
exact placement and copied it). Retrograde line also bans "drags on".
- Backtest script: several follow-up domains in one run.
Changed assertions carry 原值 / 新值 / 原因 in the test comments.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
611 lines
30 KiB
TypeScript
611 lines
30 KiB
TypeScript
/**
|
||
* Biography backtest for the ordinary (natal) consultation answer
|
||
* (TASK-consult-no-presupposition-and-backtest-20261001 T4, D5).
|
||
*
|
||
* Runs the product's real consultation agent against public Rodden AA charts
|
||
* and writes each answer for a human / Claude to score against
|
||
* docs/research/consult_biography_backtest_rubric.json. Nothing here copies
|
||
* prompt text: the system prompt, the skill method block, the tools, the
|
||
* evidence card, the methodology checklist, the user-turn shape and the answer
|
||
* stream handling all come from the product modules, so a run always tests the
|
||
* prompts on the current checkout.
|
||
*
|
||
* What is real Where it comes from
|
||
* system prompt + bound skill method getJyotishAgent (src/mastra/index.ts)
|
||
* tool loop, prepareStep, settings Mastra Agent.stream + consultationNatalPrepareStep
|
||
* + consultationGenerationSettings (route.ts natalStreamOptions)
|
||
* tool result (contract, card, createConsultationTools -> toModelEvidenceView
|
||
* methodology checklist) (engine call replaced by the golden workflow)
|
||
* evidence lookup tool the real read-consultation-evidence
|
||
* user turn consultationUserTurnContent + natalUserTurnShape
|
||
* answer extraction, pass 4, retries, streamAgentResponse (stepScopedAnswer, pass4Mode,
|
||
* length continuation retry / retryForAnswer / continueAfterLength as route.ts)
|
||
*
|
||
* Gaps against app/api/consult/route.ts are listed in GAPS below and in
|
||
* docs/testing/consult-affliction-backtest-20261001.md.
|
||
*
|
||
* Usage (from frontend/, Node 22):
|
||
* DS_KEY=... ./node_modules/.bin/tsx scripts/research/consult-biography-backtest.mts \
|
||
* [--out <dir>] [--repeats 2] [--figures a,b] [--domains parents,marriage,health,career] \
|
||
* [--model deepseek-flash] [--base-url https://api.deepseek.com] [--concurrency 4] [--dump-cards]
|
||
* [--correction barack_obama,marilyn_monroe] [--correction-text 我爸其实没怎么管过我]
|
||
* [--correction-domain career] [--correction-texts "zinedine_zidane=我是足球运动员|barack_obama=我是政治家"]
|
||
* Several follow-up domains in one run: --correction-domain parents,career with
|
||
* domain-qualified entries, e.g. --correction "parents:barack_obama,career:zinedine_zidane"
|
||
* --correction-texts "career:zinedine_zidane=我是足球运动员" (an unqualified entry or text
|
||
* applies to every listed domain).
|
||
*
|
||
* --dump-cards writes the model-visible tool result per figure x domain and
|
||
* calls no model. --correction adds, for each listed figure's parents answer,
|
||
* one follow-up turn in the same session (history = that question and answer)
|
||
* with --correction-text (T3 / D4 of the task brief); those answers are written
|
||
* as parents-correction-run<N>.md and listed in the summary for reading.
|
||
* --recheck <dir> re-runs the deterministic checks of an
|
||
* earlier run against the current rubric and rewrites its summary. The key is read from DS_KEY (or BACKTEST_MODEL_KEY) only and
|
||
* is never written anywhere. Output defaults to <repo>/scratch/ (gitignored);
|
||
* only the summary may be quoted into docs.
|
||
*/
|
||
import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
||
import { dirname, join, resolve } from "node:path";
|
||
import { fileURLToPath } from "node:url";
|
||
|
||
import { getJyotishAgent } from "../../src/mastra/index.ts";
|
||
import {
|
||
AGENT_MAX_STEPS,
|
||
consultationContinueGenerationSettings,
|
||
consultationGenerationSettings,
|
||
consultationNatalPrepareStep,
|
||
consultationStepBudgetReceipt,
|
||
createConsultationAgentContext,
|
||
createConsultationRuntimeState,
|
||
publicConsultationRuntimeSteps,
|
||
type ConsultationRuntimeState,
|
||
} from "../../src/mastra/consultation-tools.ts";
|
||
import {
|
||
consultationWorkflowResponseSchema,
|
||
minuteSensitiveThemesFromBirthTimeSensitivity,
|
||
} from "../../src/mastra/consultation-workflow.ts";
|
||
import type { ResolvedLanguageModel } from "../../src/mastra/model.ts";
|
||
import { createConsultationPlan } from "../../src/lib/consultation-plan.ts";
|
||
import { consultationUserTurnContent } from "../../src/lib/consultation-session-history.ts";
|
||
import { consultationContinueMessages, natalUserTurnShape } from "../../src/lib/consultation-thinking-plan.ts";
|
||
import { streamAgentResponse } from "../../src/lib/stream-agent-response.ts";
|
||
import type { ConsultationDomain } from "../../src/lib/consultation-domain-registry.ts";
|
||
import type { ServerChartConsultation } from "../../src/lib/consultation-route-service.ts";
|
||
|
||
const HERE = dirname(fileURLToPath(import.meta.url));
|
||
const FRONTEND = resolve(HERE, "../..");
|
||
const ROOT = resolve(FRONTEND, "..");
|
||
const GOLDEN = join(FRONTEND, "tests/fixtures/consult-biography-backtest-golden.json");
|
||
const RUBRIC = join(ROOT, "docs/research/consult_biography_backtest_rubric.json");
|
||
|
||
/** Where this harness differs from the live route; repeated in the summary. */
|
||
export const GAPS = [
|
||
"引擎调用换成 golden(同 capture 路径、raman 岁差、mean 交点、参考日 2026-09-27);产品用用户档案的岁差设置。",
|
||
"用户回合不带「用户称呼」:真实路由会带档案名,这里故意不给名人名字,免得模型凭记忆答生平。",
|
||
"natalToolInstruction 一句与 currentTimeContext 是 route.ts 内部函数,未导出,脚本里按原文复刻(见 routeToolInstruction / routeCurrentTime)。",
|
||
"没有内容审核(moderateOutput)、计费、标题旁路事件、历史摘要;主回测单轮、无历史;--correction 纠正追问轮的历史只有上一问一答。",
|
||
"性别一律按档案未填(性别未知)处理。",
|
||
"模型经 Mastra 的 OpenAI 兼容通道直连 DeepSeek(model id 由参数给),不经产品数据库的模型目录。",
|
||
] as const;
|
||
|
||
// route.ts (natalToolInstruction, entrypoint undefined = ordinary chat). Not exported there.
|
||
const routeToolInstruction = "如需新的个人星盘结论,必须调用服务器绑定的排盘工具。";
|
||
|
||
// route.ts currentTimeContext. Not exported there.
|
||
function routeCurrentTime(now: Date) {
|
||
const chinaTime = new Date(now.getTime() + 8 * 60 * 60 * 1000).toISOString().replace("T", " ").slice(0, 19);
|
||
return `服务端当前时间(权威):${now.toISOString()};中国标准时间(UTC+8):${chinaTime}。涉及“现在、今天、今年、未来几个月”等相对时间时,以此为准。`;
|
||
}
|
||
|
||
// The golden is computed for this reference date; the request clock matches it.
|
||
const REQUEST_TIME = new Date("2026-09-27T04:00:00.000Z");
|
||
|
||
type Json = Record<string, unknown>;
|
||
type GoldenRoute = { domain: string; engine_route: string; question: string };
|
||
type GoldenFigure = {
|
||
id: string;
|
||
label: string;
|
||
rodden_rating: string;
|
||
shared_chart: Json;
|
||
routes: Record<string, Json & { chart?: Json; chart_modules?: Json }>;
|
||
};
|
||
type Golden = { routes: GoldenRoute[]; figures: GoldenFigure[] };
|
||
|
||
type RubricFigure = {
|
||
id: string;
|
||
domains: Record<string, { must_not?: { claim: string; literals?: string[] }[] }>;
|
||
};
|
||
|
||
function parseArgs(argv: readonly string[]) {
|
||
const value = (name: string) => {
|
||
const index = argv.indexOf(`--${name}`);
|
||
return index >= 0 ? argv[index + 1] : undefined;
|
||
};
|
||
const stamp = new Date().toISOString().replace(/[:.]/g, "-");
|
||
return {
|
||
out: resolve(value("out") ?? join(ROOT, "scratch/consult-biography-backtest", stamp)),
|
||
repeats: Math.max(1, Number(value("repeats") ?? 2)),
|
||
figures: value("figures")?.split(",").filter(Boolean) ?? null,
|
||
domains: value("domains")?.split(",").filter(Boolean) ?? ["parents", "marriage", "health", "career"],
|
||
model: value("model") ?? "deepseek-flash",
|
||
baseUrl: value("base-url") ?? "https://api.deepseek.com",
|
||
concurrency: Math.max(1, Number(value("concurrency") ?? 4)),
|
||
dumpCards: argv.includes("--dump-cards"),
|
||
recheck: value("recheck"),
|
||
correction: value("correction")?.split(",").filter(Boolean) ?? [],
|
||
correctionText: value("correction-text") ?? "我爸其实没怎么管过我",
|
||
// Follow-up domain (default parents) and optional per-figure texts
|
||
// "figure=text|figure=text" (career follow-up: the user states the real job).
|
||
correctionDomains: (value("correction-domain") ?? "parents").split(",").filter(Boolean),
|
||
correctionTexts: Object.fromEntries((value("correction-texts") ?? "").split("|").filter(Boolean).map((pair) => {
|
||
const at = pair.indexOf("=");
|
||
return [pair.slice(0, at), pair.slice(at + 1)];
|
||
})) as Record<string, string>,
|
||
};
|
||
}
|
||
|
||
/** One route's engine response, with the figure's shared chart put back exactly as captured. */
|
||
function routeWorkflow(figure: GoldenFigure, domain: string): Json {
|
||
const route = figure.routes[domain];
|
||
if (!route) throw new Error(`golden has no ${domain} route for ${figure.id}`);
|
||
const { chart: chartOverrides, chart_modules: moduleOverrides, ...rest } = structuredClone(route);
|
||
const shared = structuredClone(figure.shared_chart);
|
||
const chart = {
|
||
...shared,
|
||
...(chartOverrides ?? {}),
|
||
modules: { ...(shared.modules as Json), ...(moduleOverrides ?? {}) },
|
||
};
|
||
return { ...rest, chart };
|
||
}
|
||
|
||
// The engine route each product domain runs as (consultation-domain-registry):
|
||
// parents / children / family all run the engine's family route.
|
||
const GOLDEN_ROUTE_FOR_DOMAIN: Record<string, string> = {
|
||
parents: "parents",
|
||
family: "parents",
|
||
children: "parents",
|
||
marriage: "marriage",
|
||
health: "health",
|
||
career: "career",
|
||
};
|
||
|
||
function stubRunWorkflow(figure: GoldenFigure, executed: string[]) {
|
||
return async (input: { theme: string }) => {
|
||
const key = GOLDEN_ROUTE_FOR_DOMAIN[input.theme];
|
||
if (!key) throw new Error(`backtest golden has no engine route for domain ${input.theme}`);
|
||
executed.push(input.theme);
|
||
// runConsultationWorkflow: schema parse, then minute-sensitive themes on the policy.
|
||
const data = consultationWorkflowResponseSchema.parse(routeWorkflow(figure, key));
|
||
const consumer = data.consumer_context as Json;
|
||
return {
|
||
...data,
|
||
consumer_context: {
|
||
...consumer,
|
||
answer_policy: {
|
||
...(consumer.answer_policy as Json),
|
||
minute_sensitive_themes: minuteSensitiveThemesFromBirthTimeSensitivity(data.birth_time_sensitivity),
|
||
},
|
||
},
|
||
} as unknown as typeof data;
|
||
};
|
||
}
|
||
|
||
function serverChart(figure: GoldenFigure): ServerChartConsultation {
|
||
const birth = (figure.shared_chart.birth ?? {}) as Json;
|
||
const route = routeWorkflow(figure, "parents");
|
||
const body = ((route.chart as Json).birth ?? birth) as Json;
|
||
const num = (key: string, fallback = 0) => (typeof body[key] === "number" ? body[key] as number : fallback);
|
||
// Birth fields only shape the tool input schema; the stub ignores them.
|
||
const toolInput = {
|
||
year: num("year", 1950),
|
||
month: num("month", 1),
|
||
day: num("day", 1),
|
||
hour: num("hour"),
|
||
minute: num("minute"),
|
||
city: "public chart",
|
||
lat: num("lat", num("latitude")),
|
||
lon: num("lon", num("longitude")),
|
||
tz: num("tz", num("timezone")),
|
||
ayanamsa: "raman",
|
||
declared_accuracy: "minute",
|
||
time_source: "birth_record",
|
||
} as unknown as ServerChartConsultation["toolInput"];
|
||
return {
|
||
name: "backtest",
|
||
toolInput,
|
||
truth: {
|
||
birthDate: "1950-01-01",
|
||
reportedBirthTime: null,
|
||
activeBirthTime: null,
|
||
selectedTimeKind: "reported",
|
||
birthTimeSource: "birth_record",
|
||
birthTimeStatus: "reported",
|
||
placeLabel: "public chart",
|
||
placeCodes: { countryCode: null, provinceCode: null, cityCode: null, districtCode: null },
|
||
placeId: null,
|
||
placeType: null,
|
||
placeProvider: null,
|
||
timezoneId: null,
|
||
timezoneSource: null,
|
||
latitude: toolInput.lat,
|
||
longitude: toolInput.lon,
|
||
timezoneOffset: toolInput.tz,
|
||
},
|
||
gender: null,
|
||
} as ServerChartConsultation;
|
||
}
|
||
|
||
function model(args: ReturnType<typeof parseArgs>, apiKey: string): ResolvedLanguageModel {
|
||
return {
|
||
id: `backtest-${args.model}`,
|
||
label: args.model,
|
||
description: "biography backtest",
|
||
creditCost: 0,
|
||
isDefault: false,
|
||
mode: "compatible",
|
||
model: { providerId: "deepseek", modelId: args.model, url: args.baseUrl, apiKey } as ResolvedLanguageModel["model"],
|
||
};
|
||
}
|
||
|
||
/** A follow-up turn in the same session: the earlier question and answer are the history. */
|
||
type FollowUp = { priorQuestion: string; priorAnswer: string };
|
||
type Job = { figure: GoldenFigure; domain: string; question: string; run: number; followUp?: FollowUp };
|
||
|
||
type Usage = { inputTokens?: number; outputTokens?: number; reasoningTokens?: number; cachedInputTokens?: number };
|
||
|
||
function addUsage(total: Usage, usage: unknown) {
|
||
const row = (usage ?? {}) as Record<string, unknown>;
|
||
for (const key of ["inputTokens", "outputTokens", "reasoningTokens", "cachedInputTokens"] as const) {
|
||
const value = row[key];
|
||
if (typeof value === "number") total[key] = (total[key] ?? 0) + value;
|
||
}
|
||
}
|
||
|
||
function agentContext(job: Job, state: ConsultationRuntimeState, executed: string[]) {
|
||
return createConsultationAgentContext({
|
||
userId: "biography-backtest",
|
||
sessionId: `backtest-${job.figure.id}`,
|
||
requestId: `backtest-${job.figure.id}-${job.domain}-${job.run}`,
|
||
consultationMode: "verified_chart",
|
||
plan: createConsultationPlan({ userIntent: job.question, theme: job.domain as ConsultationDomain }),
|
||
theme: job.domain as ConsultationDomain,
|
||
followUpTurn: Boolean(job.followUp),
|
||
serverChart: serverChart(job.figure),
|
||
state,
|
||
runWorkflow: stubRunWorkflow(job.figure, executed) as never,
|
||
});
|
||
}
|
||
|
||
function historyOf(followUp: FollowUp | undefined) {
|
||
return followUp
|
||
? [
|
||
{ role: "user" as const, text: followUp.priorQuestion },
|
||
{ role: "assistant" as const, text: followUp.priorAnswer },
|
||
]
|
||
: [];
|
||
}
|
||
|
||
function userTurn(question: string, followUp?: FollowUp) {
|
||
return consultationUserTurnContent({
|
||
currentTime: routeCurrentTime(REQUEST_TIME),
|
||
// natalUserTurnShape reads the stored history, as route.ts does: a prior answer -> follow-up shape.
|
||
instruction: `${routeToolInstruction}${natalUserTurnShape({ history: historyOf(followUp) })}`,
|
||
question,
|
||
});
|
||
}
|
||
|
||
async function runJob(job: Job, resolved: ResolvedLanguageModel) {
|
||
const state = createConsultationRuntimeState();
|
||
const executed: string[] = [];
|
||
const ctx = agentContext(job, state, executed);
|
||
const agent = getJyotishAgent(resolved, ctx);
|
||
const runId = ctx.requestId;
|
||
// route.ts consultationBaseMessages: stored history as plain messages, then this turn.
|
||
const baseMessages = [
|
||
...historyOf(job.followUp).map((message) => (message.role === "user"
|
||
? { role: "user" as const, content: message.text }
|
||
: { role: "assistant" as const, content: message.text })),
|
||
{ role: "user" as const, content: userTurn(job.question, job.followUp) },
|
||
];
|
||
const streamOptions = {
|
||
runId,
|
||
maxSteps: AGENT_MAX_STEPS,
|
||
...consultationGenerationSettings(resolved.model),
|
||
};
|
||
const natalStreamOptions = { ...streamOptions, prepareStep: consultationNatalPrepareStep };
|
||
const usages: Promise<unknown>[] = [];
|
||
const startedAt = Date.now();
|
||
let output: string | null = null;
|
||
let failure: string | null = null;
|
||
const first = await agent.stream(baseMessages, natalStreamOptions);
|
||
usages.push(first.totalUsage);
|
||
const response = streamAgentResponse({
|
||
runId,
|
||
requestId: runId,
|
||
state,
|
||
stream: first.fullStream as never,
|
||
requireTool: true,
|
||
retry: async () => {
|
||
const retried = await agent.stream([
|
||
...baseMessages,
|
||
{ role: "user" as const, content: "运行合同不完整:本次尚未取得服务器计算结果。请调用 run-jyotish-consultation 完成计算,再据此回答;不要在工具参数中添加出生资料。" },
|
||
], natalStreamOptions);
|
||
usages.push(retried.totalUsage);
|
||
return retried.fullStream as never;
|
||
},
|
||
retryForAnswer: async (retryHint?: string) => {
|
||
const retried = await agent.stream([
|
||
...baseMessages,
|
||
{ role: "user" as const, content: `服务器计算已经完成,但上一轮没有输出任何回答文本。请重新取回本次计算结果,然后直接给出回答;不要只描述过程或工具调用。${retryHint ? `\n${retryHint}` : ""}` },
|
||
], natalStreamOptions);
|
||
usages.push(retried.totalUsage);
|
||
return retried.fullStream as never;
|
||
},
|
||
continueAfterLength: async (text: string, evidence?: unknown) => {
|
||
const continued = await agent.stream(consultationContinueMessages(baseMessages, text, evidence), {
|
||
...streamOptions,
|
||
...consultationContinueGenerationSettings(resolved.model),
|
||
});
|
||
usages.push(continued.totalUsage);
|
||
return continued.fullStream as never;
|
||
},
|
||
stepScopedAnswer: true,
|
||
pass4Mode: "verified_chart",
|
||
toolStatus: () => {
|
||
const status = state.workflowReceipt?.status;
|
||
return status === "ready" || status === "degraded" ? status : "blocked";
|
||
},
|
||
receipt: () => ({
|
||
runId,
|
||
runtime: "mastra-agentic",
|
||
skill: {
|
||
name: "jyotish-vedic-astrology",
|
||
loaded: state.jyotishSkillBound,
|
||
referenceReads: state.skillReferenceReadCount,
|
||
methodologySections: state.methodologySectionCount,
|
||
},
|
||
steps: publicConsultationRuntimeSteps(state),
|
||
stepBudget: consultationStepBudgetReceipt(state),
|
||
workflow: state.workflowReceipt ?? { route: job.domain, status: "blocked", preciseTiming: "blocked", missingLayers: [], domains: [job.domain as ConsultationDomain] },
|
||
techniqueTruth: state.techniqueTruth ?? "unknown",
|
||
}) as never,
|
||
onComplete: (text: string) => {
|
||
output = text;
|
||
},
|
||
onError: (error: unknown) => {
|
||
failure = error instanceof Error ? error.message : String(error);
|
||
},
|
||
});
|
||
// Drain the public event stream; that is what drives the run.
|
||
await response.text();
|
||
const usage: Usage = {};
|
||
for (const item of await Promise.allSettled(usages)) {
|
||
if (item.status === "fulfilled") addUsage(usage, item.value);
|
||
}
|
||
// Assigned inside the stream callbacks, which TypeScript cannot see.
|
||
const settledOutput = output as string | null;
|
||
const settledFailure = failure as string | null;
|
||
return {
|
||
output: settledOutput ?? "",
|
||
failure: settledFailure,
|
||
executed,
|
||
seconds: (Date.now() - startedAt) / 1000,
|
||
usage,
|
||
cardChars: state.evidenceCardChars ?? null,
|
||
visibleChars: state.modelVisibleChars ?? null,
|
||
methodologySections: state.methodologySectionCount,
|
||
lookups: state.evidenceLookupCallCount,
|
||
finish: state.composeFinishReason ?? state.modelFinishReason ?? null,
|
||
};
|
||
}
|
||
|
||
// ---- deterministic checks (judgment scoring is done by reading the answers) ----
|
||
|
||
const BANNED = [
|
||
{ id: "retrograde_repeat", pattern: /逆行[^。!?\n]{0,30}(反复|打回来|来回|拉回|折返|回头)/gu },
|
||
{ id: "repeat_retrograde", pattern: /(反复|打回来|来回)[^。!?\n]{0,12}逆行/gu },
|
||
];
|
||
// 他 / 她 when the card says 性别未知: only for the partner (marriage) or children.
|
||
const PRONOUN = /(?<![其吉])[他她](?![们人])/gu;
|
||
|
||
export function deterministicChecks(answer: string, domain: string, mustNot: { claim: string; literals?: string[] }[]) {
|
||
const context = (index: number) => answer.slice(Math.max(0, index - 14), index + 16).replace(/\s+/g, " ");
|
||
const banned = BANNED.flatMap((rule) => [...answer.matchAll(rule.pattern)].map((hit) => ({ rule: rule.id, text: hit[0] })));
|
||
const pronouns = domain === "marriage" || domain === "children"
|
||
? [...answer.matchAll(PRONOUN)].map((hit) => context(hit.index ?? 0))
|
||
: [];
|
||
const literalHits = mustNot.flatMap((item) => (item.literals ?? [])
|
||
.filter((literal) => answer.includes(literal))
|
||
.map((literal) => ({ claim: item.claim, literal })));
|
||
return { banned, pronouns, literalHits };
|
||
}
|
||
|
||
async function pool<T, R>(items: readonly T[], size: number, fn: (item: T) => Promise<R>) {
|
||
const results: R[] = new Array(items.length);
|
||
let next = 0;
|
||
await Promise.all(Array.from({ length: Math.min(size, items.length) }, async () => {
|
||
while (next < items.length) {
|
||
const index = next++;
|
||
results[index] = await fn(items[index]!);
|
||
}
|
||
}));
|
||
return results;
|
||
}
|
||
|
||
async function dumpCards(jobs: readonly Job[], out: string) {
|
||
for (const job of jobs.filter((item) => item.run === 1)) {
|
||
const state = createConsultationRuntimeState();
|
||
const ctx = agentContext(job, state, []);
|
||
const fake = model({ ...parseArgs([]), model: "card-dump" }, "unused");
|
||
const agent = getJyotishAgent(fake, ctx);
|
||
const tools = await agent.listTools();
|
||
const tool = tools["run-jyotish-consultation"] as unknown as { execute: (input: unknown, context: unknown) => Promise<unknown> };
|
||
const view = await tool.execute({ question: job.question, domains: [job.domain] }, {});
|
||
const path = join(out, "cards", `${job.figure.id}-${job.domain}.json`);
|
||
mkdirSync(dirname(path), { recursive: true });
|
||
writeFileSync(path, `${JSON.stringify(view, null, 1)}\n`);
|
||
console.log(`card ${path} (${JSON.stringify(view).length} chars)`);
|
||
}
|
||
}
|
||
|
||
// The product logs every reasoning delta (consultation-budget.ts); a batch run
|
||
// would drown its own progress lines in them.
|
||
const productInfo = console.info.bind(console);
|
||
console.info = (...items: unknown[]) => {
|
||
if (typeof items[0] === "string" && items[0].startsWith("[consult-reasoning]")) return;
|
||
productInfo(...items);
|
||
};
|
||
|
||
type Row = { figure: string; domain: string; run: number; chars: number; seconds: number; checks: ReturnType<typeof deterministicChecks> };
|
||
|
||
function writeSummary(out: string, summary: Json & { rows: Row[] }, modelId: string) {
|
||
const rows = summary.rows;
|
||
summary.banned_hits = rows.reduce((sum, row) => sum + row.checks.banned.length, 0);
|
||
summary.pronoun_hits = rows.reduce((sum, row) => sum + row.checks.pronouns.length, 0);
|
||
summary.must_not_literal_hits = rows.reduce((sum, row) => sum + row.checks.literalHits.length, 0);
|
||
writeFileSync(join(out, "summary.json"), `${JSON.stringify(summary, null, 1)}\n`);
|
||
writeFileSync(join(out, "summary.md"), [
|
||
`# 名人生平回测 · ${modelId}`,
|
||
"",
|
||
`开始 ${summary.started_at},${rows.length} 份,墙钟 ${Number(summary.wall_seconds).toFixed(0)} s,失败 ${summary.failures} 份。`,
|
||
`用量合计:${JSON.stringify(summary.usage)}`,
|
||
`确定性检查:禁句 ${summary.banned_hits},性别代词(婚恋/子女)${summary.pronoun_hits},must_not 原文 ${summary.must_not_literal_hits}。判断类评分需逐份阅读。`,
|
||
"",
|
||
"| 名人 | 领域 | run | 字数 | 秒 | 禁句 | 代词 | must_not |",
|
||
"| --- | --- | --- | --- | --- | --- | --- | --- |",
|
||
...rows.map((row) => `| ${row.figure} | ${row.domain} | ${row.run} | ${row.chars} | ${row.seconds.toFixed(0)} | ${row.checks.banned.length} | ${row.checks.pronouns.length} | ${row.checks.literalHits.length} |`),
|
||
"",
|
||
"## must_not 原文命中",
|
||
"",
|
||
...rows.flatMap((row) => row.checks.literalHits.map((hit) => `- ${row.figure} · ${row.domain} · run${row.run}:「${hit.literal}」(${hit.claim})`)),
|
||
"",
|
||
"## 与真实路由的差异",
|
||
"",
|
||
...GAPS.map((gap) => `- ${gap}`),
|
||
"",
|
||
].join("\n"));
|
||
}
|
||
|
||
function recheck(out: string, rubric: { figures: RubricFigure[] }) {
|
||
const summary = JSON.parse(readFileSync(join(out, "summary.json"), "utf8")) as Json & { rows: Row[]; model: string };
|
||
for (const row of summary.rows) {
|
||
const text = readFileSync(join(out, row.figure, `${row.domain}-run${row.run}.md`), "utf8");
|
||
const answer = text.slice(text.indexOf("\n---\n\n") + 6).trim();
|
||
const mustNot = rubric.figures.find((item) => item.id === row.figure)?.domains[row.domain]?.must_not ?? [];
|
||
row.checks = deterministicChecks(answer === "(无回答)" ? "" : answer, row.domain, mustNot);
|
||
}
|
||
writeSummary(out, summary, summary.model);
|
||
console.log(`rechecked: ${join(out, "summary.md")}`);
|
||
}
|
||
|
||
async function main() {
|
||
const args = parseArgs(process.argv.slice(2));
|
||
const golden = JSON.parse(readFileSync(GOLDEN, "utf8")) as Golden;
|
||
const rubric = JSON.parse(readFileSync(RUBRIC, "utf8")) as { figures: RubricFigure[] };
|
||
if (args.recheck) {
|
||
recheck(resolve(args.recheck), rubric);
|
||
return;
|
||
}
|
||
const questions = new Map(golden.routes.map((route) => [route.domain, route.question]));
|
||
const figures = golden.figures.filter((figure) => !args.figures || args.figures.includes(figure.id));
|
||
const jobs: Job[] = figures.flatMap((figure) => args.domains.flatMap((domain) => {
|
||
const question = questions.get(domain);
|
||
if (!question) throw new Error(`no golden question for domain ${domain}`);
|
||
return Array.from({ length: args.repeats }, (_, index) => ({ figure, domain, question, run: index + 1 }));
|
||
}));
|
||
mkdirSync(args.out, { recursive: true });
|
||
if (args.dumpCards) {
|
||
await dumpCards(jobs, args.out);
|
||
return;
|
||
}
|
||
const apiKey = process.env.DS_KEY ?? process.env.BACKTEST_MODEL_KEY;
|
||
if (!apiKey) throw new Error("Set DS_KEY (or BACKTEST_MODEL_KEY); without a key the backtest is an environment gap.");
|
||
const resolved = model(args, apiKey);
|
||
const started = Date.now();
|
||
const rows = await pool(jobs, args.concurrency, async (job) => {
|
||
let result: Awaited<ReturnType<typeof runJob>>;
|
||
try {
|
||
result = await runJob(job, resolved);
|
||
} catch (error) {
|
||
result = { output: "", failure: error instanceof Error ? error.message : String(error), executed: [], seconds: 0, usage: {}, cardChars: null, visibleChars: null, methodologySections: 0, lookups: 0, finish: null };
|
||
}
|
||
const mustNot = rubric.figures.find((item) => item.id === job.figure.id)?.domains[job.domain]?.must_not ?? [];
|
||
const checks = deterministicChecks(result.output, job.domain, mustNot);
|
||
const file = join(args.out, job.figure.id, `${job.domain}-run${job.run}.md`);
|
||
mkdirSync(dirname(file), { recursive: true });
|
||
writeFileSync(file, [
|
||
`# ${job.figure.label} · ${job.domain} · run ${job.run}`,
|
||
"",
|
||
`- 问题:${job.question}`,
|
||
`- 模型:${args.model};用时 ${result.seconds.toFixed(1)} s;结束:${result.finish ?? "?"};查卡 ${result.lookups} 次;清单段 ${result.methodologySections}`,
|
||
`- 执行领域:${result.executed.join(", ") || "无"};卡 ${result.cardChars ?? "?"} 字符,工具结果 ${result.visibleChars ?? "?"} 字符`,
|
||
`- 用量:${JSON.stringify(result.usage)}`,
|
||
`- 确定性检查:禁句 ${checks.banned.length},性别代词 ${checks.pronouns.length},must_not 原文 ${checks.literalHits.length}`,
|
||
...(result.failure ? [`- 失败:${result.failure}`] : []),
|
||
...checks.banned.map((hit) => ` - 禁句 ${hit.rule}:「${hit.text}」`),
|
||
...checks.pronouns.map((hit) => ` - 代词:「${hit}」`),
|
||
...checks.literalHits.map((hit) => ` - must_not:「${hit.literal}」(${hit.claim})`),
|
||
"",
|
||
"---",
|
||
"",
|
||
result.output || "(无回答)",
|
||
"",
|
||
].join("\n"));
|
||
console.log(`${job.figure.id} ${job.domain} run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`);
|
||
return { figure: job.figure.id, domain: job.domain, run: job.run, chars: result.output.length, ...result, output: undefined, checks };
|
||
});
|
||
const corrections = await pool(
|
||
jobs.filter((job) => args.correctionDomains.includes(job.domain)
|
||
&& (args.correction.includes(job.figure.id) || args.correction.includes(`${job.domain}:${job.figure.id}`))),
|
||
args.concurrency,
|
||
async (prior) => {
|
||
const priorFile = join(args.out, prior.figure.id, `${prior.domain}-run${prior.run}.md`);
|
||
const priorText = readFileSync(priorFile, "utf8");
|
||
const priorAnswer = priorText.slice(priorText.indexOf("\n---\n\n") + 6).trim();
|
||
const question = args.correctionTexts[`${prior.domain}:${prior.figure.id}`] ?? args.correctionTexts[prior.figure.id] ?? args.correctionText;
|
||
const job: Job = { ...prior, question, followUp: { priorQuestion: prior.question, priorAnswer } };
|
||
let result: Awaited<ReturnType<typeof runJob>>;
|
||
try {
|
||
result = await runJob(job, resolved);
|
||
} catch (error) {
|
||
result = { output: "", failure: error instanceof Error ? error.message : String(error), executed: [], seconds: 0, usage: {}, cardChars: null, visibleChars: null, methodologySections: 0, lookups: 0, finish: null };
|
||
}
|
||
const file = join(args.out, job.figure.id, `${job.domain}-correction-run${job.run}.md`);
|
||
writeFileSync(file, [
|
||
`# ${job.figure.label} · ${job.domain} 纠正追问 · run ${job.run}`,
|
||
"",
|
||
`- 上一轮:${priorFile}`,
|
||
`- 追问:${job.question}`,
|
||
`- 模型:${args.model};用时 ${result.seconds.toFixed(1)} s;结束:${result.finish ?? "?"};查卡 ${result.lookups} 次`,
|
||
`- 用量:${JSON.stringify(result.usage)}`,
|
||
...(result.failure ? [`- 失败:${result.failure}`] : []),
|
||
"",
|
||
"---",
|
||
"",
|
||
result.output || "(无回答)",
|
||
"",
|
||
].join("\n"));
|
||
console.log(`${job.figure.id} ${job.domain} correction run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`);
|
||
return { figure: job.figure.id, domain: job.domain, run: job.run, chars: result.output.length, seconds: result.seconds, failure: result.failure, usage: result.usage };
|
||
},
|
||
);
|
||
const total: Usage = {};
|
||
for (const row of [...rows, ...corrections]) addUsage(total, row.usage);
|
||
const summary = {
|
||
model: args.model,
|
||
started_at: new Date(started).toISOString(),
|
||
wall_seconds: (Date.now() - started) / 1000,
|
||
jobs: rows.length,
|
||
failures: rows.filter((row) => row.failure || row.chars === 0).length,
|
||
usage: total,
|
||
gaps: GAPS,
|
||
rows,
|
||
corrections,
|
||
};
|
||
writeSummary(args.out, summary as unknown as Json & { rows: Row[] }, args.model);
|
||
console.log(`summary: ${join(args.out, "summary.md")}`);
|
||
}
|
||
|
||
await main();
|