Unify reader cleanup rules, lock writer table guards, and register the exact fictional timestamp collision. Preserve existing ordinary-report safety contracts and source-data gaps. Validation: report Node 165/165, final safety 29/29, Python 101/101, Chrome 28/28; both PDFs retain all 130 rows. Full Node 3704 tests with the same 91 baseline failures. Privacy test: 62 passed, 1 failed due to 17 protected-file READ_ERRORs; not a green gate. Build, DB, manual checklist and controlled-login gaps remain documented. User explicitly authorized staging push with these gaps disclosed. Co-Authored-By: Claude Code <noreply@anthropic.com>
33 lines
1.2 KiB
TypeScript
33 lines
1.2 KiB
TypeScript
import vocabulary from "../../../scripts/reader_appendix_language.rules.json";
|
|
|
|
type VocabularyRule = Readonly<{
|
|
pattern: string;
|
|
ignoreCase?: boolean;
|
|
replacement?: string;
|
|
labels?: Readonly<Record<string, string>>;
|
|
labelGroup?: number;
|
|
prefix?: string;
|
|
beforeGroups?: readonly number[];
|
|
afterGroups?: readonly number[];
|
|
}>;
|
|
|
|
// One rule source for Python and TS; the JSON ships in both existing images.
|
|
const chartFence = new RegExp(vocabulary.chartFencePattern, "m");
|
|
const replacements = (vocabulary.rules as readonly VocabularyRule[]).map((rule) => ({
|
|
rule,
|
|
pattern: new RegExp(rule.pattern, rule.ignoreCase ? "gi" : "g"),
|
|
}));
|
|
|
|
export function cleanReaderAppendixMarkdown(markdown: string): string {
|
|
return markdown.split(chartFence).map((part, index) => {
|
|
if (index % 2) return part;
|
|
return replacements.reduce((value, { pattern, rule }) => value.replace(pattern, (...match) => {
|
|
if (rule.replacement !== undefined) return rule.replacement;
|
|
return (rule.prefix ?? "")
|
|
+ (rule.beforeGroups ?? []).map((group) => match[group]).join("")
|
|
+ rule.labels![match[rule.labelGroup!]]
|
|
+ (rule.afterGroups ?? []).map((group) => match[group]).join("");
|
|
}), part);
|
|
}).join("");
|
|
}
|