perf(rectification): evidence turns get Skill §5/§7 only; drop Mastra <available_skills> injection (R4)

T6 of TASK-rectification-grounding-20260927.
- skill-slice.ts: per-action slice rule as a code constant, keyed on heading
  titles (evidence: 「ConversationFocus 与意图承接」「批量证据与日期真实性」, i.e.
  §5/§7 of the 10.0.x layout); other actions and any bound Skill without those
  headings (9.0.0) get the whole body, so historical Cases still run with their
  exact bound Skill (BUG-621).
- The rectification Agent declares providesSkillDiscovery "on-demand"
  (rectificationSkillBoundProcessor): no <available_skills> block with a temp
  path and no "call the skill tool" system message; getSkill still loads the
  bound package.
- Measured on a real Agent + recording model (public AA case, estimate = CJK
  chars + other chars / 4): fixed overhead per call 12,002 → 6,471 tokens
  (step 0: 8,440 → 2,909). Skill text unchanged; no version bump.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
Jesse_Chen
2026-09-27 03:28:27 +08:00
co-authored by Claude Opus 5.5
parent 1ed3e55579
commit c4bb394bbf
4 changed files with 301 additions and 1 deletions
@@ -10,6 +10,7 @@ import type { V9CaseDossier } from "./tool-service";
import { cachedSystemMessage } from "../../agent-generation-settings.ts";
import { OPENING_COLLECT_DOMAINS } from "../user-copy";
import { retryConstraintForAttempt } from "./host-fallback";
import { sliceRectificationSkill } from "./skill-slice";
import type { V9AgentRunOptions } from "./agent-run";
function clockWindow(range: { start_time?: string | null; end_time?: string | null } | null | undefined): string | null {
@@ -54,9 +55,12 @@ export function buildAgentMessages(
const timeContext = options.timeContext
?? `服务端当前时间(权威):${new Date().toISOString()}。涉及“现在、今天、今年、未来几个月”等相对时间时,以此为准。`;
const caseContext = `【服务端 Case ID】${options.caseId}。所有 rectification 工具调用的 caseId 必须原样使用此值。`;
// T6 (TASK-rectification-grounding-20260927): an evidence turn gets only
// the Skill sections it uses; other actions keep the whole bound body.
const boundSkill = sliceRectificationSkill(skillInstructions, options.action).text;
const bootstrapContent = [
"【服务器已绑定当前 Case 的精确 Skill】运行器已在本 attempt 内加载并核验下列指令;不要重复调用 skill。第一步必须调用 rectification-read-case。",
skillInstructions,
boundSkill,
...(attempt > 1 ? [retryConstraintForAttempt(previousErrorCode)] : []),
].join("\n\n");
const bootstrap = cachedSystemMessage(bootstrapContent, options.generationModel)
@@ -0,0 +1,71 @@
/**
* Per-turn slice of the bound rectification Skill (TASK-rectification-
* grounding-20260927 T6).
*
* The runner used to send the whole bound SKILL.md body (about 10.7K
* characters for 10.0.31) on every model call. An evidence turn only uses
* §5 (ConversationFocus and intent) and §7 (batch evidence and date
* truthfulness); the rest describes the opening, case status table, fields
* that only exist in `full_diagnostics` (method_followup_plan,
* CaseConversationSummary, guided_collect_windows …), the candidate language
* of offer turns, and upstream sync. Other actions keep the whole body.
*
* The slice is keyed on `## N. <title>` heading titles, not on numbers:
* the 9.0.0 snapshot numbers its sections differently (its §5 is the
* candidate language). If any wanted title is missing, the whole body is
* sent unchanged, so a historical Case still runs with its exact bound Skill
* (BUG-621). All 10.0.x snapshots share the 10.0.31 headings.
*/
import type { RectificationAgentAction } from "./step-budget";
/** Heading titles per action; evidence = §5 and §7 of the 10.0.x layout. */
export const RECTIFICATION_SKILL_SECTIONS_BY_ACTION: Readonly<
Record<RectificationAgentAction, readonly string[] | "all">
> = {
opening: "all",
read_only: "all",
evidence: ["ConversationFocus 与意图承接", "批量证据与日期真实性"],
rescore: "all",
accept: "all",
confirm: "all",
};
export type RectificationSkillSlice = Readonly<{
text: string;
/** Heading lines sent, or "all" when the whole body went out. */
sections: readonly string[] | "all";
}>;
function headingTitle(line: string): string | null {
const match = /^##\s+(?:\d+\.\s*)?(.+?)\s*$/.exec(line);
return match ? match[1] : null;
}
export function sliceRectificationSkill(
instructions: string,
action: RectificationAgentAction,
): RectificationSkillSlice {
const wanted = RECTIFICATION_SKILL_SECTIONS_BY_ACTION[action];
if (wanted === "all") return { text: instructions, sections: "all" };
const preamble: string[] = [];
const sections: Array<{ title: string; lines: string[] }> = [];
for (const line of instructions.split("\n")) {
const title = line.startsWith("## ") ? headingTitle(line) : null;
if (title !== null) {
sections.push({ title, lines: [line] });
continue;
}
if (sections.length > 0) sections.at(-1)!.lines.push(line);
else preamble.push(line);
}
const picked = wanted.map((title) => sections.find((section) => section.title === title));
if (picked.some((section) => !section)) return { text: instructions, sections: "all" };
const title = preamble.find((line) => line.startsWith("# ")) ?? "";
const headings = picked.map((section) => section!.lines[0]!.trim());
const note = `(本轮是证据轮:以下只附本轮适用的 Skill 章节「${picked.map((section) => section!.title).join("」「")}」;其余章节不适用于本轮,按服务器投影与工具说明执行。)`;
const body = picked.map((section) => section!.lines.join("\n").trim()).join("\n\n");
return {
text: [title, note, body].filter(Boolean).join("\n\n"),
sections: headings,
};
}
@@ -1,4 +1,5 @@
import { Agent } from "@mastra/core/agent";
import type { Processor } from "@mastra/core/processors";
import type { ResolvedLanguageModel } from "./model";
import {
resolveActiveSkillPackage,
@@ -56,8 +57,30 @@ export function getRectificationV9Agent(
model: model.model,
instructions: agenticRectificationInstructions,
skills: [resolveSkillPackageRuntimePath(skillPackage)],
inputProcessors: [rectificationSkillBoundProcessor],
tools: createRectificationV9AgentTools(ctx),
});
}
/**
* R4 (TASK-rectification-grounding-20260927): Mastra's default skills
* processor added an `<available_skills>` block (with a temporary package
* path) and "call the `skill` tool" to every step, contradicting the runner's
* "do not call skill" and costing about 1K characters per call.
* `providesSkillDiscovery: "on-demand"` declares that the caller owns skill
* loading, which is true: the runner loads the exact bound Skill through
* `agent.getSkill` and quotes it (sliced per turn) in the bootstrap message.
* With this marker Mastra adds neither the system block nor the `skill` /
* `skill_search` tools (same mechanism as the consultation agents'
* `jyotishSkillBoundProcessor`).
*/
export const rectificationSkillBoundProcessor: Processor & { processInputStep: NonNullable<Processor["processInputStep"]> } = {
id: "rectification-skill-bound",
name: "Rectification Skill Bound",
providesSkillDiscovery: "on-demand",
processInputStep() {
// Nothing to inject: the bound Skill text rides in the runner's bootstrap message.
},
};
export { RECTIFICATION_V9_SKILL_NAME };
@@ -0,0 +1,202 @@
/**
* TASK-rectification-grounding-20260927 T6 (R4 + Skill slicing): what a real
* rectification Agent sends the model on an evidence turn. Real
* `runV9AgentTurn` + real `getRectificationV9Agent` (real bound Skill, real
* tools) + a prompt-recording scripted model; case data is a real local
* engine response for a public AA chart (9 candidates).
*
* Token estimate (for the size table): one token per CJK character plus one
* per four other characters (`estimateTokens`). "Fixed overhead" is what every
* call of the turn re-sends regardless of tool results: the system messages
* plus the tool schemas.
*/
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
import test from "node:test";
import {
CASE_ID,
SESSION_ID,
TURN_ID,
USER_ID,
dossierFixture,
fakeAccounting,
receiptHandlers,
} from "./rectification-v9-test-support.ts";
import {
AA_EVIDENCE_ROWS,
AA_RANGE,
AA_TURNS,
buildGoldenLatest,
contentText,
estimateTokens,
scriptedModel,
setFocusHandler,
stubGoldenFetch,
type RecordedCall,
} from "./rectification-grounding-support.ts";
import { getRectificationV9Agent } from "../src/mastra/agentic-rectification.ts";
import { runV9AgentTurn } from "../src/lib/rectification-agentic/v9/agent-run.ts";
import {
RECTIFICATION_SKILL_SECTIONS_BY_ACTION,
sliceRectificationSkill,
} from "../src/lib/rectification-agentic/v9/skill-slice.ts";
import {
resolveActiveSkillPackage,
resolveExactSkillPackage,
} from "../src/lib/skill-package-registry.ts";
import { resolve } from "node:path";
export type CallSize = Readonly<{
call: number;
systemChars: number;
systemTokens: number;
toolsBytes: number;
toolsTokens: number;
toolNames: string[];
fixedTokens: number;
totalTokens: number;
}>;
export function callSizes(calls: readonly RecordedCall[]): CallSize[] {
return calls.map((call, index) => {
const system = call.prompt.filter((message) => message.role === "system").map((message) => contentText(message.content)).join("\n");
const tools = JSON.stringify(call.tools ?? []);
const all = call.prompt.map((message) => contentText(message.content)).join("\n");
return {
call: index + 1,
systemChars: system.length,
systemTokens: estimateTokens(system),
toolsBytes: Buffer.byteLength(tools, "utf8"),
toolsTokens: estimateTokens(tools),
toolNames: Array.isArray(call.tools) ? (call.tools as Array<{ name: string }>).map((tool) => tool.name) : [],
fixedTokens: estimateTokens(system) + estimateTokens(tools),
totalTokens: estimateTokens(all) + estimateTokens(tools),
};
});
}
/** One evidence turn: read-case, then one short body (no write claimed). */
export async function recordEvidenceTurn() {
const { latest, compute } = await buildGoldenLatest();
const { model, calls } = scriptedModel([
{ parts: [{ tool: { name: "rectification-read-case", input: { caseId: CASE_ID } } }], finish: "tool-calls" },
{ parts: [{ text: "2014 年 8 月结婚这件事拿去和星盘对照了。" }], finish: "stop" },
]);
const active = resolveActiveSkillPackage("jyotish-birth-time-rectification");
const accounting = fakeAccounting({
...receiptHandlers,
// The case is bound to the active Skill (10.0.x layout), as a new case is.
get_agentic_rectification_skill_identity: () => ({
skill_name: active.name,
skill_version: active.version,
skill_sha256: active.sha256,
skill_source_commit: active.sourceCommit,
}),
insert_agentic_rectification_skill_run_receipt: () => ({
receipt_id: "88888888-8888-4888-8888-888888888888",
skill_name: active.name,
skill_version: active.version,
skill_sha256: active.sha256,
source_commit: active.sourceCommit,
}),
get_agentic_rectification_case_dossier: () => {
const dossier = dossierFixture({
evidence: AA_EVIDENCE_ROWS,
latestResult: latest,
turns: AA_TURNS as never,
candidateRange: AA_RANGE as never,
}) as { case: Record<string, unknown> };
return { ...dossier, case: { ...dossier.case, skill_version: active.version } };
},
get_agentic_rectification_case_compute: () => compute,
set_agentic_rectification_conversation_focus: setFocusHandler,
append_agentic_rectification_turn: () => ({ turn_id: TURN_ID, should_execute: true, status: "pending" }),
finalize_agentic_rectification_turn: () => ({ turn_id: TURN_ID, status: "completed", idempotent: false }),
}, { fallback: () => null });
const restore = stubGoldenFetch();
try {
await runV9AgentTurn({
userId: USER_ID,
caseId: CASE_ID,
sessionId: SESSION_ID,
requestId: "req-grounding-prompt",
action: "evidence",
message: "2014 年 8 月结婚。",
modelName: "fake",
accounting: accounting.client as never,
billing: { reserve: async () => ({ success: true, status: 200 }), complete: async () => true, release: async () => true },
emit: () => {},
expectedWrite: "none",
buildAgent: async (turnId, skillPackage, attemptId) => getRectificationV9Agent(
{ id: "fake", label: "fake", description: "", creditCost: 1, isDefault: true, mode: "compatible", model: model as never } as never,
{ userId: USER_ID, caseId: CASE_ID, turnId, attemptId, userMessage: "2014 年 8 月结婚。", accounting: accounting.client as never },
skillPackage,
),
});
} finally {
restore();
}
return calls;
}
function activeSkillBody(): string {
const skill = resolveActiveSkillPackage("jyotish-birth-time-rectification");
const raw = readFileSync(resolve(skill.resolvedPath, "SKILL.md"), "utf8");
return raw.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n/, "").trim();
}
test("R4: an evidence-turn call carries no <available_skills> block and no skill tools", async () => {
const calls = await recordEvidenceTurn();
assert.ok(calls.length >= 2);
for (const call of calls) {
const system = call.prompt.filter((message) => message.role === "system").map((message) => contentText(message.content)).join("\n");
assert.doesNotMatch(system, /<available_skills>/);
assert.doesNotMatch(system, /Skills are NOT tools|call the `skill` tool/);
const names = Array.isArray(call.tools) ? (call.tools as Array<{ name: string }>).map((tool) => tool.name) : [];
assert.equal(names.includes("skill"), false);
assert.equal(names.includes("skill_search"), false);
}
});
test("T6: the evidence-turn bootstrap carries Skill §5 and §7 only", async () => {
const calls = await recordEvidenceTurn();
const system = calls[0]!.prompt.filter((message) => message.role === "system").map((message) => contentText(message.content)).join("\n");
assert.match(system, /## 5\. ConversationFocus 与意图承接/);
assert.match(system, /## 7\. 批量证据与日期真实性/);
for (const heading of ["## 1. ", "## 2. ", "## 3. ", "## 4. ", "## 6. ", "## 8. ", "## 9. ", "## 10. ", "## 11. "]) {
assert.equal(system.includes(heading), false, `${heading} in ${system.slice(system.indexOf(heading) - 200, system.indexOf(heading) + 200)}`);
}
// full_diagnostics-only fields described in §6 are not in the evidence turn.
assert.doesNotMatch(system, /guided_collect_windows|method_followup_plan/);
assert.match(system, /本轮是证据轮/);
assert.match(system, /# Jyotish 生时校正(V10)/);
});
test("T6: fixed overhead per evidence-turn call is under 8K estimated tokens", async () => {
const sizes = callSizes(await recordEvidenceTurn());
console.log(JSON.stringify({ scope: "T6 evidence turn sizes", sizes }));
for (const size of sizes) assert.ok(size.fixedTokens < 8_000, JSON.stringify(size));
});
test("T6: the slice rule is a code constant; other actions and unknown layouts get the whole Skill", () => {
assert.deepEqual(RECTIFICATION_SKILL_SECTIONS_BY_ACTION.evidence, ["ConversationFocus 与意图承接", "批量证据与日期真实性"]);
assert.equal(RECTIFICATION_SKILL_SECTIONS_BY_ACTION.opening, "all");
assert.equal(RECTIFICATION_SKILL_SECTIONS_BY_ACTION.read_only, "all");
const body = activeSkillBody();
assert.equal(sliceRectificationSkill(body, "opening").text, body);
const evidence = sliceRectificationSkill(body, "evidence");
assert.deepEqual(evidence.sections, ["## 5. ConversationFocus 与意图承接", "## 7. 批量证据与日期真实性"]);
assert.ok(evidence.text.length < body.length / 2);
const oldLayout = "# 旧版\n\n## 概述\n\n没有编号的章节。";
assert.equal(sliceRectificationSkill(oldLayout, "evidence").text, oldLayout);
// BUG-621: the oldest deprecated snapshot still resolves and still slices safely.
const oldest = resolveExactSkillPackage(
"jyotish-birth-time-rectification",
"9.0.0",
"5acb3103e80993ea611b93d2c1746e70b74aff8f8636bcffc80a837b954b470d",
);
const oldestBody = readFileSync(resolve(oldest.resolvedPath, "SKILL.md"), "utf8");
// 9.0.0 numbers its sections differently (its §5 is the candidate language):
// no title match, so the whole bound body goes out unchanged.
assert.equal(sliceRectificationSkill(oldestBody, "evidence").text, oldestBody);
});