From 48fb160fc938a3779dd801e87a823c1ce957954c Mon Sep 17 00:00:00 2001 From: Jesse_Chen Date: Sun, 27 Sep 2026 02:46:22 +0800 Subject: [PATCH] feat(consult): six-field evidence-card record in agent observability (log only) After each natal turn the route logs a separate [agent-observability] event {runId, agentVersion, evidenceCard: {domains, cardVersion, cardChars, cardTokenEstimate, citedFieldIds, feedback}}. Cited field ids follow the research R5 rule (ISO date, degree, planet-in-sign phrase of 8+ chars appearing verbatim in a finished answer); matched text is discarded. Thumbs are client state only today, so feedback is "none"; storing thumbs per turn needs a table and is left to a follow-up. The strict schema has no user, session, question or answer field. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8 --- frontend/src/app/api/consult/route.ts | 35 +++++-- frontend/src/lib/agent-observability.ts | 28 ++++++ frontend/src/mastra/consultation-tools.ts | 25 ++++- ...t-evidence-card-telemetry-20260927.test.ts | 91 +++++++++++++++++++ 4 files changed, 169 insertions(+), 10 deletions(-) create mode 100644 frontend/tests/consult-evidence-card-telemetry-20260927.test.ts diff --git a/frontend/src/app/api/consult/route.ts b/frontend/src/app/api/consult/route.ts index be2c4739..7cf599ff 100644 --- a/frontend/src/app/api/consult/route.ts +++ b/frontend/src/app/api/consult/route.ts @@ -62,6 +62,7 @@ import { createWindowConsultationAgentContext, precomputeWindowConsultation, windowPrecomputedPacketMessage, + consultationEvidenceCardTelemetry, consultationModelStepTelemetry, consultationStepBudgetReceipt, createConsultationRuntimeHooks, @@ -879,6 +880,10 @@ export async function POST(request: Request) { let firstActivityMs = -1; let firstTextMs = -1; let logged = false; + // The finished natal answer, read once by the evidence-card record to find + // which card fields it quoted; never logged. A failed or cut run leaves it + // empty, so only finished answers count citations. + let answerForTelemetry = ""; const markFirstActivity = () => { if (firstActivityMs < 0) firstActivityMs = Date.now() - agentStartedAt; }; const markFirstText = () => { if (firstTextMs < 0) firstTextMs = Date.now() - agentStartedAt; }; const consultEntrypoint = parsed.success ? parsed.data.entrypoint : undefined; @@ -948,6 +953,15 @@ export async function POST(request: Request) { themeCoverage: state.workflowReceipt?.domains ?? [consultationTheme], billingSettlementResult: settlementResult, }); + // D7: a separate six-field record, no session or user id in it. + const evidenceCard = consultationEvidenceCardTelemetry(state, answerForTelemetry); + if (evidenceCard) { + logAgentObservability({ + runId: requestId, + agentVersion: "consultation-evidence-card-v1", + evidenceCard, + }); + } }; const settleRun = async ( action: () => Promise, @@ -1385,15 +1399,18 @@ export async function POST(request: Request) { headers: { "x-jyotish-birth-time-mode": consultationMode }, onFirstActivity: markFirstActivity, onFirstOutput: markFirstText, - onComplete: (output, agentExecutionReceipt, thinkingText, thinkingSections) => settleRun(() => completeResponse( - output, - mergeUsage(usages), - state.techniqueTruth ?? "unknown", - state.workflowReceipt ?? workflowReceipt, - agentExecutionReceipt, - thinkingText, - thinkingSections, - ), undefined), + onComplete: (output, agentExecutionReceipt, thinkingText, thinkingSections) => { + answerForTelemetry = output; + return settleRun(() => completeResponse( + output, + mergeUsage(usages), + state.techniqueTruth ?? "unknown", + state.workflowReceipt ?? workflowReceipt, + agentExecutionReceipt, + thinkingText, + thinkingSections, + ), undefined); + }, onError: (error) => settleRun( cancel, toAgentObservabilityErrorCode(error), diff --git a/frontend/src/lib/agent-observability.ts b/frontend/src/lib/agent-observability.ts index 4fcafa8c..14b058df 100644 --- a/frontend/src/lib/agent-observability.ts +++ b/frontend/src/lib/agent-observability.ts @@ -113,6 +113,33 @@ export const agentObservabilityContractPhaseSchema = z.object({ status: z.enum(agentObservabilityStepStatuses), }).strict().readonly(); +/** + * The evidence-card record (TASK-consult-evidence-card-20260927 D7): exactly + * six fields, numbers / enums / card field names only. No question, answer, + * birth data, name, email, user or session id, and never the matched text. + */ +export const EVIDENCE_CARD_TELEMETRY_FIELDS = [ + "domains", + "cardVersion", + "cardChars", + "cardTokenEstimate", + "citedFieldIds", + "feedback", +] as const; + +export const evidenceCardTelemetrySchema = z.object({ + domains: z.array(consultationDomainSchema).min(1).max(2), + cardVersion: z.literal("evidence-card-v1"), + cardChars: countSchema, + cardTokenEstimate: countSchema, + citedFieldIds: z.array(z.string().min(1).max(120).regex(/^[A-Za-z][A-Za-z0-9._-]*$/, "invalid card field id")).max(64), + // Thumbs are client state only today (not stored), so a turn is logged + // with `none`; storing thumbs per turn needs a table and is a separate task. + feedback: z.enum(["up", "down", "none"]), +}).strict().readonly(); + +export type EvidenceCardTelemetry = z.infer; + export const agentObservabilityEventSchema = z.object({ runId: opaqueIdSchema.optional(), requestId: opaqueIdSchema.optional(), @@ -155,6 +182,7 @@ export const agentObservabilityEventSchema = z.object({ reportJobDurationMs: durationMsSchema.optional(), reportJobPeakMemoryBytes: z.number().int().min(0).max(Number.MAX_SAFE_INTEGER).optional(), billingSettlementResult: z.enum(billingSettlementResults).optional(), + evidenceCard: evidenceCardTelemetrySchema.optional(), }).strict().refine( (event) => Boolean(event.runId || event.requestId || event.sessionId || event.caseId), { message: "at least one controlled identifier is required" }, diff --git a/frontend/src/mastra/consultation-tools.ts b/frontend/src/mastra/consultation-tools.ts index c5d91e6f..7dc17c4a 100644 --- a/frontend/src/mastra/consultation-tools.ts +++ b/frontend/src/mastra/consultation-tools.ts @@ -14,12 +14,14 @@ import { runV9RangeReading } from "../lib/rectification-agentic/v9/engine-client import { createConsultationPlan, type ConsultationPlan } from "../lib/consultation-plan.ts"; import type { TechniqueAuditRow, WorkflowReceipt } from "../lib/consultation-agent-events.ts"; import { normalizeTechniqueAuditRows } from "../lib/consultation-technique-audit.ts"; -import type { AgentModelFinishReason } from "../lib/agent-observability.ts"; +import type { AgentModelFinishReason, EvidenceCardTelemetry } from "../lib/agent-observability.ts"; import { agentGenerationSettings, AGENT_SLICE_ANSWER_OUTPUT_TOKENS, AGENT_SLICE_THINKING_OUTPUT_TOKENS } from "../lib/agent-generation-settings.ts"; import { chartCalculationProgressLabel, evidenceLookupActivityLabel } from "../lib/consultation-activity-labels.ts"; import { buildEvidenceCard, + citedEvidenceCardFields, EVIDENCE_LOOKUP_SECTIONS, + evidenceCardTokenEstimate, type EvidenceCard, type EvidenceLookupSection, } from "../lib/consultation-evidence-card.ts"; @@ -338,6 +340,27 @@ export function consultationModelStepTelemetry(state: ConsultationRuntimeState) }; } +/** + * The six-field evidence-card record for the observability log (D7). The + * answer is read only to find which card fields it quoted (ISO date, degree, + * planet-in-sign phrase); neither the answer nor the matched text is kept. + */ +export function consultationEvidenceCardTelemetry( + state: ConsultationRuntimeState, + answer: string, +): EvidenceCardTelemetry | undefined { + const card = state.evidenceCard; + if (!card || state.evidenceCardChars === undefined) return undefined; + return { + domains: card.domains.slice(0, 2), + cardVersion: card.card_version, + cardChars: state.evidenceCardChars, + cardTokenEstimate: evidenceCardTokenEstimate(state.evidenceCardChars), + citedFieldIds: citedEvidenceCardFields(card, answer), + feedback: "none", + }; +} + /** * Public receipts carry only the fields the client contract allows. Building * the list from an explicit allowlist keeps internal diagnostics, such as the diff --git a/frontend/tests/consult-evidence-card-telemetry-20260927.test.ts b/frontend/tests/consult-evidence-card-telemetry-20260927.test.ts new file mode 100644 index 00000000..0cafab9d --- /dev/null +++ b/frontend/tests/consult-evidence-card-telemetry-20260927.test.ts @@ -0,0 +1,91 @@ +// TASK-consult-evidence-card-20260927 T6 (D7): per-turn evidence-card record +// in [agent-observability], six fields only, no user text. +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import test from "node:test"; + +import { + createAgentObservabilityLogger, + EVIDENCE_CARD_TELEMETRY_FIELDS, + evidenceCardTelemetrySchema, + type AgentObservabilityEvent, +} from "../src/lib/agent-observability.ts"; +import { + consultationEvidenceCardTelemetry, + createConsultationRuntimeState, + createConsultationTools, +} from "../src/mastra/consultation-tools.ts"; + +type Json = Record; +const golden = JSON.parse(readFileSync( + new URL("./fixtures/consult-evidence-card-golden.json", import.meta.url), + "utf8", +)) as { charts: Array<{ id: string; workflow: Json }> }; +const workflow = golden.charts[0]!.workflow; +const route = readFileSync(new URL("../src/app/api/consult/route.ts", import.meta.url), "utf8"); + +async function stateAfterCalculation() { + const state = createConsultationRuntimeState(); + const tools = createConsultationTools({ + userId: "user-telemetry", sessionId: "session-telemetry", requestId: "req-telemetry", consultationMode: "verified_chart", + serverChart: { name: "public", toolInput: { year: 2000, month: 1, day: 1, hour: 0, minute: 0, city: "fixture", lat: 0, lon: 0, tz: 0, ayanamsa: "raman" }, truth: { birthTimeSource: "reported" } } as never, + state, + runWorkflow: async () => structuredClone(workflow) as never, + }); + await tools["run-jyotish-consultation"].execute!({ question: "q", domains: ["parents"] } as never, { writer: { custom: async () => {} } } as never); + return state; +} + +test("the record has exactly the six approved fields", async () => { + assert.deepEqual([...EVIDENCE_CARD_TELEMETRY_FIELDS].sort(), Object.keys(evidenceCardTelemetrySchema.unwrap().shape).sort()); + assert.deepEqual([...EVIDENCE_CARD_TELEMETRY_FIELDS], ["domains", "cardVersion", "cardChars", "cardTokenEstimate", "citedFieldIds", "feedback"]); + const state = await stateAfterCalculation(); + const record = consultationEvidenceCardTelemetry(state, ""); + assert.ok(record); + assert.deepEqual(Object.keys(record).sort(), [...EVIDENCE_CARD_TELEMETRY_FIELDS].sort()); + assert.deepEqual(record.domains, ["parents"]); + assert.equal(record.cardVersion, "evidence-card-v1"); + assert.equal(record.cardChars, state.evidenceCardChars); + assert.equal(record.cardTokenEstimate, Math.round(record.cardChars / 3.5)); + assert.deepEqual(record.citedFieldIds, []); + assert.equal(record.feedback, "none"); +}); + +test("the schema is strict: no extra field, no free text in field ids", () => { + const valid = { domains: ["parents"], cardVersion: "evidence-card-v1", cardChars: 10, cardTokenEstimate: 3, citedFieldIds: ["base.natal.ascendant.sign"], feedback: "up" }; + assert.equal(evidenceCardTelemetrySchema.safeParse(valid).success, true); + for (const extra of ["question", "answer", "sessionId", "userId", "birthDate", "matched"]) { + assert.equal(evidenceCardTelemetrySchema.safeParse({ ...valid, [extra]: "x" }).success, false, extra); + } + assert.equal(evidenceCardTelemetrySchema.safeParse({ ...valid, citedFieldIds: ["我和父母关系如何"] }).success, false); + assert.equal(evidenceCardTelemetrySchema.safeParse({ ...valid, feedback: "great" }).success, false); +}); + +test("the logged line keeps field ids and drops every word of the user's text", async () => { + const state = await stateAfterCalculation(); + const card = state.evidenceCard!; + const pdStart = String((card.base.timing.vimshottari.pratyantardasha as Json).start); + // Fictional sensitive values, not a real person. + const secrets = ["虚构名字王小明", "fictional.person@example.invalid", "1990-01-01", "我和父母关系如何"]; + const answer = `${secrets[0]}(${secrets[1]},生日 ${secrets[2]})问:${secrets[3]}。子运从 ${pdStart} 开始。`; + const lines: AgentObservabilityEvent[] = []; + const log = createAgentObservabilityLogger((event) => lines.push(event)); + log({ runId: "req-telemetry", agentVersion: "consultation-evidence-card-v1", evidenceCard: consultationEvidenceCardTelemetry(state, answer) }); + const text = JSON.stringify(lines); + for (const secret of secrets) assert.equal(text.includes(secret), false, secret); + assert.equal(text.includes(pdStart), false, "the matched date is not kept"); + assert.equal(text.includes("session-telemetry"), false); + assert.equal(text.includes("user-telemetry"), false); + assert.ok(lines[0]?.evidenceCard?.citedFieldIds.includes("base.timing.vimshottari.pratyantardasha.start")); +}); + +test("the consult route logs the record once per natal turn, without session or user id", () => { + assert.match(route, /const evidenceCard = consultationEvidenceCardTelemetry\(state, answerForTelemetry\);/); + const block = route.slice(route.indexOf("const evidenceCard = consultationEvidenceCardTelemetry"), route.indexOf("const settleRun = async")); + assert.match(block, /agentVersion: "consultation-evidence-card-v1"/); + assert.doesNotMatch(block, /sessionId|userId/); + const natal = route.slice(route.indexOf("const agent = getJyotishAgent(selectedModel, agentContext);")); + assert.match(natal, /answerForTelemetry = output;[\s\S]*return settleRun\(\(\) => completeResponse\(/); + // Only a finished answer counts citations; a failed run logs the card size with no citations. + assert.doesNotMatch(natal, /onError: \(error, _emitted, output\)/); +});