feat(consult): one read-only evidence lookup per turn; settle open clauses at tool calls (BUG-1059)

read-consultation-evidence returns one closed-enum section (a formal varga,
research/extended vargas, a Western layer, yogas, Ashtakavarga, Shadbala,
transits, Chara Dasha, arudha, karakas, KP, gulika, kakshya, mahadashas,
thematic evidence) from this request's finished calculation, never
recalculates, answers unavailable on a cache miss and refuses a second call.
The receipt records the step and the write row shows 「正在多看一眼:…」.
A lookup after answer text went out keeps the released text whole: a verbatim
restart is dropped as it arrives (40-char confirmation), a continuation is
kept, and settlement still reads the step that wrote the answer; the lookup
runs on the answer clock without resetting it. A length continuation carries
the lookup result with the card. BUG-1059: the visible-text transformer's open
clause is settled at each tool call, so unpunctuated narration no longer
leaks into the answer.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
Jesse_Chen
2026-09-27 02:54:11 +08:00
co-authored by Claude Opus 5.5
parent 147ebc1789
commit cb3ee55837
9 changed files with 873 additions and 7 deletions
@@ -14,3 +14,47 @@ export function consultationWriteLabel(heading: string, live: boolean): string {
const title = heading.trim() || "回答";
return live ? `正在写${title}…` : `写${title}`;
}
const WESTERN_LAYER_LABELS: Readonly<Record<string, string>> = {
natal: "西洋本命",
transits: "西洋行运",
solar_return: "太阳回归",
secondary_progressions: "次限推运",
solar_arc_directions: "太阳弧",
converse_secondary_progressions: "逆推次限",
converse_solar_arc_directions: "逆推太阳弧",
midpoints: "中点",
lunar_return: "月亮回归",
transit_duration_scan: "行运时长扫描",
parans: "共升共落",
};
const LOOKUP_SECTION_LABELS: Readonly<Record<string, string>> = {
"varga:research_dn": "研究用分盘",
"varga:extended": "D81 / D108 / D144 分盘",
yogas: "格局明细",
ashtakavarga: "八分法(Ashtakavarga)",
shadbala: "六力(Shadbala)",
transits: "行运触发",
chara_dasha: "Chara 大运",
arudha_padas: "映点(Arudha)",
chara_karakas: "七个代表星(Chara Karaka)",
kp_cusps: "KP 宫头",
gulika: "Gulika",
kakshya: "Kakshya",
vimshottari_mahadashas: "全部大运起止",
domain_thematic_evidence: "这个主题的证据明细",
};
/** Plain words for one evidence-lookup section (TASK-consult-evidence-card-20260927 T5). */
export function evidenceLookupSectionLabel(section: string): string {
if (section.startsWith("varga:D")) return `${section.slice("varga:".length)} 分盘`;
if (section.startsWith("western:")) return WESTERN_LAYER_LABELS[section.slice("western:".length)] ?? "西洋层";
return LOOKUP_SECTION_LABELS[section] ?? "一段盘面数据";
}
/** The activity row while the answer model looks up one section outside the card. */
export function evidenceLookupActivityLabel(section: string, live = true): string {
const label = evidenceLookupSectionLabel(section);
return live ? `正在多看一眼:${label}…` : `多看了一眼:${label}`;
}
+107 -2
View File
@@ -1,5 +1,6 @@
import {
appendConsultationRuntimeStep,
CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID,
CONSULTATION_NATAL_CALC_TOOL_ID,
CONSULTATION_WINDOW_CALC_TOOL_ID,
type ConsultationRuntimeState,
@@ -446,6 +447,14 @@ function isCalculationTool(toolName: unknown) {
return toolName === CONSULTATION_NATAL_CALC_TOOL_ID || toolName === CONSULTATION_WINDOW_CALC_TOOL_ID;
}
/**
* How many characters a step after an evidence lookup must repeat, from the
* start of the answer already put out in this attempt, before it counts as a
* restart and the repeat is dropped. Two answers that merely open alike
* ("你这盘…") diverge well before this.
*/
export const LOOKUP_RESTART_MATCH_CHARS = 40;
function stepFinishReason(chunk: Chunk) {
const payload = chunk.payload as { stepResult?: { reason?: unknown } } | undefined;
return payload?.stepResult?.reason;
@@ -466,6 +475,9 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
// The calculation result the model saw, kept so a length continuation
// writes with the same evidence (BUG-1053). Never sent to the client.
let calculationEvidence: unknown;
// The one-shot evidence lookup's result, if the model used it; carried into
// a length continuation with the card (TASK-consult-evidence-card-20260927).
let lookupEvidence: unknown;
// Pass 4 buffers only the current open sentence. Closed sentences are
// classified and either sent whole or dropped whole. Whole-answer rewrite
// is allowed only before any answer.delta has gone out; after the first
@@ -602,6 +614,16 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
// the cut answer read as finished (BUG-1051 carried into the loop).
let stepWrote = false;
let answerStepReason: AgentModelFinishReason | undefined;
// Evidence lookup during the answer (T5): if the model had already put
// answer text out in this attempt and then called the lookup, the next
// step may start the answer over. A verbatim restart of the text already
// out is dropped as it arrives, so nothing released is repeated; anything
// that diverges within LOOKUP_RESTART_MATCH_CHARS is kept as written.
let attemptText = "";
let restartCheck = false;
let restartPos = 0;
let restartHeld = "";
let restartConfirmed = false;
const resetStep = () => {
stepText = "";
stepReleased = false;
@@ -619,6 +641,7 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
return;
}
uncontractedText = "";
attemptText += text;
held += text;
if (!held) return;
if (/\S/.test(held)) {
@@ -643,7 +666,58 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
if (/\S/.test(held)) emitted = true;
held = "";
};
const acceptText = async (text: string) => {
const filterRestart = (text: string) => {
let index = 0;
while (index < text.length && restartCheck) {
const char = text[index]!;
if (restartPos === 0 && !restartConfirmed && /\s/.test(char)) {
restartHeld += char;
index += 1;
continue;
}
if (restartPos < attemptText.length && char === attemptText[restartPos]) {
restartPos += 1;
index += 1;
if (!restartConfirmed) {
restartHeld += char;
if (restartPos >= LOOKUP_RESTART_MATCH_CHARS) {
restartConfirmed = true;
restartHeld = "";
appendConsultationRuntimeStep(options.state, {
kind: "validation",
name: "answer-restart-dropped",
status: "completed",
});
}
}
if (restartPos >= attemptText.length) restartCheck = false;
continue;
}
restartCheck = false;
}
// Still repeating: everything so far is held (not yet a confirmed
// restart) or dropped (confirmed).
if (restartCheck) return "";
// The check ended inside this chunk. An unconfirmed match was a
// coincidence and goes out as written; a confirmed repeat stays dropped.
const kept = `${restartConfirmed ? "" : restartHeld}${text.slice(index)}`;
restartHeld = "";
return kept;
};
const acceptText = async (raw: string) => {
const text = restartCheck ? filterRestart(raw) : raw;
await acceptAnswerText(text);
};
const flushRestartHeld = async () => {
// Only a step that has started repeating ends the check here; the
// lookup's own step ends before the step that might restart begins.
if (!restartCheck || (restartPos === 0 && !restartHeld)) return;
restartCheck = false;
const held = restartConfirmed ? "" : restartHeld;
restartHeld = "";
if (held) await acceptAnswerText(held);
};
const acceptAnswerText = async (text: string) => {
if (!options.stepScopedAnswer || !contractReady(options) || stepReleased) {
await outputText(text);
return;
@@ -659,6 +733,7 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
// The step ended (or the stream did): its held text is the answer unless
// the step called a tool.
const settleStep = async (reason?: unknown) => {
await flushRestartHeld();
const pending = stepText;
const toolStep = stepCalledTool || reason === "tool-calls";
resetStep();
@@ -699,6 +774,13 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
) {
calculationEvidence = chunk.payload?.result;
}
if (
chunk.type === "tool-result"
&& chunk.payload?.toolName === CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID
&& !isToolInputRejection(chunk.payload?.result)
) {
lookupEvidence = chunk.payload?.result;
}
markAnswerPhase();
flushThinkingPlan(controller);
if (chunk.type === "step-start") {
@@ -706,9 +788,26 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
stepWrote = false;
}
if (chunk.type === "tool-call") {
// The visible-text transformer holds the step's open clause until a
// sentence boundary. It used to carry that clause into the next
// step's first text, so narration without closing punctuation
// ("我先排一下盘:") leaked into the answer (BUG-1059). Settle it with
// this step: answer text it continues goes out whole, anything else
// is narration and dropped with the step.
const openClause = visible.finish("");
if (openClause && (!contractReady(options) || stepReleased)) await acceptAnswerText(openClause);
// Whatever this step said before calling a tool is narration.
stepCalledTool = true;
stepText = "";
if (
chunk.payload?.toolName === CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID
&& /\S/.test(attemptText)
) {
restartCheck = true;
restartPos = 0;
restartHeld = "";
restartConfirmed = false;
}
}
if (chunk.type === "abort" && !outcome.aborted) {
// Mastra's own abort chunk: the signal fired and the stream is about
@@ -784,7 +883,13 @@ export function streamAgentResponse(options: StreamAgentResponseOptions) {
phase: "answer-composition",
label: heading ? consultationWriteLabel(heading, true) : "正在组织回答",
});
await consumeAttempt(controller, await options.continueAfterLength(pendingAnswer(), calculationEvidence), {
const evidence = lookupEvidence !== undefined
&& calculationEvidence
&& typeof calculationEvidence === "object"
&& !Array.isArray(calculationEvidence)
? { ...calculationEvidence, evidence_lookup: lookupEvidence }
: calculationEvidence;
await consumeAttempt(controller, await options.continueAfterLength(pendingAnswer(), evidence), {
suppressCompositionActivity: true,
answerPhase: true,
});
+153 -3
View File
@@ -16,8 +16,13 @@ import type { TechniqueAuditRow, WorkflowReceipt } from "../lib/consultation-age
import { normalizeTechniqueAuditRows } from "../lib/consultation-technique-audit.ts";
import type { AgentModelFinishReason } from "../lib/agent-observability.ts";
import { agentGenerationSettings, AGENT_SLICE_ANSWER_OUTPUT_TOKENS, AGENT_SLICE_THINKING_OUTPUT_TOKENS } from "../lib/agent-generation-settings.ts";
import { chartCalculationProgressLabel } from "../lib/consultation-activity-labels.ts";
import { buildEvidenceCard, type EvidenceCard } from "../lib/consultation-evidence-card.ts";
import { chartCalculationProgressLabel, evidenceLookupActivityLabel } from "../lib/consultation-activity-labels.ts";
import {
buildEvidenceCard,
EVIDENCE_LOOKUP_SECTIONS,
type EvidenceCard,
type EvidenceLookupSection,
} from "../lib/consultation-evidence-card.ts";
import { pinsConsultationDomains, type ConsultationEntrypoint } from "../lib/consultation-entrypoint.ts";
import {
dailyConsultationThinkingPlan,
@@ -135,6 +140,13 @@ export function createConsultationRunClock(options: {
return Object.freeze({ toolSignal, loopSignal: loop.signal, answerSignal });
}
export const CONSULTATION_NATAL_CALC_TOOL_ID = "run-jyotish-consultation";
/**
* The one-shot, read-only lookup of a section outside the evidence card
* (TASK-consult-evidence-card-20260927 D8). It reads this request's finished
* calculation only; it never calculates.
*/
export const CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID = "read-consultation-evidence";
export const MAX_EVIDENCE_LOOKUPS_PER_TURN = 1;
export const CONSULTATION_WINDOW_CALC_TOOL_ID = "run-jyotish-window-consultation";
/**
@@ -276,6 +288,8 @@ export type ConsultationRuntimeState = {
evidenceCardChars?: number;
/** Characters of the whole tool result the model saw. */
modelVisibleChars?: number;
/** Calls to the one-shot evidence lookup this turn, including refused ones. */
evidenceLookupCallCount: number;
};
export function createConsultationRuntimeState(options: { plannedSteps?: number; reservedValidationSteps?: number } = {}): ConsultationRuntimeState {
@@ -294,6 +308,7 @@ export function createConsultationRuntimeState(options: { plannedSteps?: number;
stepBudget: { planned, reservedValidation, total: planned + reservedValidation },
stepsTruncated: false,
modelStepCount: 0,
evidenceLookupCallCount: 0,
};
// Binding is the run's first step and it costs no model step, so it is
// recorded here rather than observed from the stream. It carries no duration
@@ -845,8 +860,91 @@ export function toModelEvidenceView(full: FullDomainPlanContext, card: EvidenceC
export type ConsultationModelEvidenceView = ReturnType<typeof toModelEvidenceView>;
type EvidenceLookupCache = {
executions: readonly DomainExecution[];
};
function lookupRecord(value: unknown): Record<string, unknown> {
return value && typeof value === "object" && !Array.isArray(value) ? value as Record<string, unknown> : {};
}
function packetEvidence(execution: DomainExecution, category: string) {
const card = execution.modelOutput.claim_cards.find((item) => item.category === category);
return lookupRecord(card?.evidence);
}
function pickDefined(source: Record<string, unknown>, keys: readonly string[]) {
const out: Record<string, unknown> = {};
for (const key of keys) if (source[key] !== undefined) out[key] = source[key];
return Object.keys(out).length ? out : undefined;
}
/**
* One section from this request's calculation, copied from the model
* projection or the engine context exactly as they hold it. `undefined` means
* the calculation did not produce it; nothing is recomputed.
*/
export function readEvidenceLookupSection(
cache: EvidenceLookupCache,
section: EvidenceLookupSection,
): unknown {
const first = cache.executions[0];
if (!first) return undefined;
const natal = packetEvidence(first, "natal_foundation");
const timing = packetEvidence(first, "timing");
const spectrum = lookupRecord(natal.varga_spectrum);
const western = lookupRecord(timing.western_spectrum);
const layers = lookupRecord(first.context.local_layers);
if (section.startsWith("varga:D")) return lookupRecord(spectrum.formal)[section.slice("varga:".length)];
if (section === "varga:research_dn") return spectrum.research_dn;
if (section === "varga:extended") return spectrum.extended;
if (section === "western:natal") return pickDefined(western, ["zodiac", "house_system", "natal", "boundary"]);
if (section.startsWith("western:")) return lookupRecord(western.techniques)[section.slice("western:".length)];
switch (section) {
case "yogas":
case "arudha_padas":
case "kp_cusps":
case "gulika":
case "kakshya":
return natal[section];
case "transits":
case "chara_dasha":
return timing[section];
case "vimshottari_mahadashas":
return timing.dasha;
case "ashtakavarga":
return pickDefined(lookupRecord(layers.ashtakavarga), ["sav", "bav", "house_scores", "strongest_signs", "weakest_signs"]);
case "shadbala":
return pickDefined(
{ summary: first.context.chart.shadbala, boundary: layers.shadbala_boundary },
["summary", "boundary"],
);
case "chara_karakas": {
const table = lookupRecord(lookupRecord(layers.jaimini).chara_karakas);
return pickDefined(table, ["AK", "AmK", "BK", "MK", "PK", "GK", "DK"]);
}
case "domain_thematic_evidence": {
const byDomain: Record<string, unknown> = {};
for (const execution of cache.executions) {
const evidence = packetEvidence(execution, "domain");
if (Object.keys(evidence).length) byDomain[execution.domain] = evidence;
}
return Object.keys(byDomain).length ? byDomain : undefined;
}
default:
return undefined;
}
}
const evidenceLookupInputSchema = z.object({
section: z.enum(EVIDENCE_LOOKUP_SECTIONS),
}).strict();
export function createConsultationTools(ctx: ConsultationAgentContext) {
let calculation: Promise<ConsultationModelEvidenceView> | null = null;
// Request-scoped: filled when this request's calculation finishes, read by
// the lookup tool, dropped with the request.
let lookupCache: EvidenceLookupCache | null = null;
const consultationTool = createTool({
id: "run-jyotish-consultation",
description: `Run one server-validated plan of personal Jyotish consultation domains. Send only question and domains. The single ordered domains array is the only way to select domains: list every domain the question needs, in priority order, up to ${MAX_CONSULTATION_DOMAIN_PLAN_VALUES}. Do not drop a relevant domain to shorten the plan. Use only the ids enumerated in the schema; workflow or checklist names from the skill's methodology are not domain ids. Questions about one's parents (父母) use parents and questions about one's children (子女) use children; family is only for the household as a whole (家里、家庭氛围). Domains execute one after another and each costs about ${Math.round(CONSULTATION_DOMAIN_DURATION_MS / 1000)}s of the run's wall clock; the server executes as many as that clock can pay for (about ${MAX_CONSULTATION_DOMAINS}) and returns the rest in omitted_domains. Birth data is server-bound and must never be supplied. The result always carries one top-level answer contract—status, evidence_contract, claim_cards, rectification—which for several domains is the most restrictive merge of the executed ones, with each domain's own contract in consultations. claim_cards is the evidence card: the base chart (ascendant, house signs, placements, functional benefics/malefics, the running Vimshottari and Narayana periods) and a section per domain; evidence_card lists what else this calculation holds. One calculation is executed per request and reused, so repeating the call with different parameters cannot change the result.`,
@@ -939,6 +1037,7 @@ export function createConsultationTools(ctx: ConsultationAgentContext) {
ctx.state.consultationToolSuccessCount += 1;
appendConsultationRuntimeStep(ctx.state, { kind: "tool", name: "run-jyotish-consultation", status: "completed", durationMs: ctx.state.consultationToolDurationMs });
const modelContext = toModelDomainPlanContext(executions, omittedDomains);
lookupCache = { executions };
const card = buildEvidenceCard(executions.map((execution) => ({
domain: execution.domain,
packet: execution.modelOutput,
@@ -986,7 +1085,58 @@ export function createConsultationTools(ctx: ConsultationAgentContext) {
}
},
});
return { "run-jyotish-consultation": consultationTool };
const lookupTool = createTool({
id: CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID,
description: `Read one section of this request's finished chart calculation that is not on the evidence card, for example a varga the card does not carry (varga:D60), a Western layer (western:solar_return), or the yoga details. It never calculates: it returns what run-jyotish-consultation already computed, or status unavailable. At most ${MAX_EVIDENCE_LOOKUPS_PER_TURN} call per turn; a second call is refused. Call it before writing any answer text, and only when the question needs that section; the answer contract and card come from run-jyotish-consultation.`,
inputSchema: evidenceLookupInputSchema,
execute: async (input, context) => {
const startedAt = Date.now();
ctx.state.evidenceLookupCallCount += 1;
const note = "If you had already started writing the answer, continue from where you stopped; never repeat text already written.";
if (ctx.state.evidenceLookupCallCount > MAX_EVIDENCE_LOOKUPS_PER_TURN) {
if (ctx.state.evidenceLookupCallCount === MAX_EVIDENCE_LOOKUPS_PER_TURN + 1) {
appendConsultationRuntimeStep(ctx.state, {
kind: "tool",
name: CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID,
status: "failed",
durationMs: 0,
failureCode: "lookup_limit_reached",
});
}
return { status: "refused" as const, reason: "lookup_limit_reached", section: input.section, note };
}
if (!lookupCache) {
appendConsultationRuntimeStep(ctx.state, {
kind: "tool",
name: CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID,
status: "failed",
durationMs: Math.max(0, Date.now() - startedAt),
failureCode: "calculation_not_cached",
});
return { status: "unavailable" as const, reason: "calculation_not_in_request_cache", section: input.section, note };
}
await (context as { writer?: { custom?: (value: unknown) => unknown } } | undefined)?.writer?.custom?.({
type: "data-jyotish-activity",
data: { phase: "answer-composition", label: evidenceLookupActivityLabel(input.section) },
});
const data = readEvidenceLookupSection(lookupCache, input.section);
const found = data !== undefined && data !== null;
appendConsultationRuntimeStep(ctx.state, {
kind: "tool",
name: CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID,
status: found ? "completed" : "failed",
durationMs: Math.max(0, Date.now() - startedAt),
...(found ? {} : { failureCode: "section_not_computed" }),
});
return found
? { status: "ok" as const, section: input.section, data, note }
: { status: "unavailable" as const, reason: "section_not_computed", section: input.section, note };
},
});
return {
"run-jyotish-consultation": consultationTool,
[CONSULTATION_EVIDENCE_LOOKUP_TOOL_ID]: lookupTool,
};
}
async function executeWindowConsultation(