From 4b2bfee6cfb41e191b6ff68073b9d0bcc15e4a0d Mon Sep 17 00:00:00 2001 From: Kuba Sunderland-Ober Date: Sun, 27 Sep 2026 21:36:49 +0200 Subject: [PATCH 1/4] book, eval, wisdom: parse through lib/cli.mjs --- book/render-book.mjs | 35 +++++++------- builder/PLAN-TOOLING-REVIEW.md | 51 ++++++++++++++++++++ docs/Documentation/Tools.md | 4 +- eval/build_corpus.mjs | 33 ++++++++----- eval/nav_hops.mjs | 31 +++++++----- eval/run_case.mjs | 53 ++++++++++++-------- eval/search_quality.mjs | 48 +++++++++---------- eval/site_search.mjs | 34 ++++++++----- eval/transcript.mjs | 28 +++++++---- scripts/check_cli.mjs | 88 ++++++++++++++++++++++++++++++++++ wisdom/wisdom.mjs | 63 ++++++++++++++---------- 11 files changed, 332 insertions(+), 136 deletions(-) diff --git a/book/render-book.mjs b/book/render-book.mjs index c2fd8231..6f9c53c3 100644 --- a/book/render-book.mjs +++ b/book/render-book.mjs @@ -32,6 +32,7 @@ import { dirname, resolve } from 'node:path'; import { writeFileSync, existsSync } from 'node:fs'; import puppeteer from 'puppeteer'; import { PDFDocument } from 'pdf-lib'; +import { parseCli, withUsageError } from '../lib/cli.mjs'; // Side-effecting imports. Mutate pdf-lib's live module exports // before any pdf-lib operation -- order doesn't matter. See // perf/notes/08-pdf-lib.md. @@ -201,24 +202,22 @@ const __dirname = dirname(fileURLToPath(import.meta.url)); // --- arg parsing -------------------------------------------------------- -const args = process.argv.slice(2); -let inputArg = null; -let outputArg = null; -let outlineTagsArg = 'h1,h2,h3,h4'; -let timeoutMs = 0; -const additionalScripts = []; -for (let i = 0; i < args.length; i++) { - const a = args[i]; - if (a === '-o' || a === '--output') outputArg = args[++i]; - else if (a === '--outline-tags') outlineTagsArg = args[++i]; - else if (a === '-t' || a === '--timeout') timeoutMs = parseInt(args[++i], 10); - else if (a === '--additional-script') additionalScripts.push(args[++i]); - else if (!inputArg && !a.startsWith('-')) inputArg = a; - else { - console.error(`unknown arg: ${a}`); - process.exit(2); - } -} +const { values, positionals } = withUsageError(() => parseCli(process.argv.slice(2), { + options: { + output: { type: 'string', short: 'o' }, + 'outline-tags': { type: 'string', default: 'h1,h2,h3,h4' }, + timeout: { type: 'string', short: 't', default: '0' }, + 'additional-script': { type: 'string', multiple: true }, + }, + positionals: { max: 1 }, + unknown: 'error', + acceptsValue: () => true, +}), { format: (err) => `unknown arg: ${err.arg}`, exitCode: 2 }); +const inputArg = positionals[0]; +const outputArg = values.output; +const outlineTagsArg = values.outlineTags; +const timeoutMs = parseInt(values.timeout, 10); +const additionalScripts = values.additionalScript; if (!inputArg || !outputArg) { console.error('usage: node render-book.mjs -o [--outline-tags ...] [-t ms] [--additional-script path]...'); process.exit(2); diff --git a/builder/PLAN-TOOLING-REVIEW.md b/builder/PLAN-TOOLING-REVIEW.md index d02fe794..5e1c5af9 100644 --- a/builder/PLAN-TOOLING-REVIEW.md +++ b/builder/PLAN-TOOLING-REVIEW.md @@ -1682,6 +1682,57 @@ otherwise stays as it is with a comment saying why. **Verify.** `check_cli.mjs`'s cases; `book.bat` renders; each `eval/` script's cheapest mode unchanged. +**Landed.** All seven tools the entry names, and `eval/search_quality.mjs`, parse through +`parseCli`. `search_quality` came in with the merge of PR #210, after this entry was written, +and the owner added it to this commit (2026-09-27). `wisdom.mjs` fits cleanly, so it migrated: +its command is its first argument, whatever that is, and the rest is parsed against one table +for all three commands, as its loop did. Every tool takes `acceptsValue: () => true`, since each +value flag took whatever followed it, and converts a value (`path.resolve`, `Number`, +`parseInt`, `parseFloat`) only when one was given, so a trailing value flag still fails where +it did: a path `TypeError`, a `NaN`, a `split` of undefined. `render-book` (`unknown arg: X`, +exit 2), `build_corpus` (`unknown argument: X`, exit 1), `run_case` (`unknown argument: X`, +exit 2), `search_quality` (`unrecognised argument: X`, exit 1) and `wisdom` (`Unknown option: +X`, exit 1) refuse through `withUsageError`. `nav_hops`, `site_search` and `transcript` take +`unknown: "positional"`, because an unknown flag was a pattern, a search term or an ignored +argument to them. `printHelpAndExit` replaces the six usage-then-exit copies (five in `eval/`, +and `wisdom`'s dispatch default, on stderr), each keeping its exit-code condition, and prints +`search_quality`'s help. `transcript` declares `--help` without `-h`, because a lone `-h` was +its file argument: until C71, `-h` alone exits 0 and `--help` alone exits 1. `build_corpus` +threw on an unknown argument, so Node printed its stack; it now prints the line, still exiting +1. `render-book`'s usage line for a missing input or output is an error, not one of the six, +and is unchanged. The edit was a Sonnet agent's (49 calls, ~236k, 22 min), reviewed line by +line; one comment was rewritten. + +The cases were recorded from the unedited tools first: 73 across the eight. They cover +`render-book`'s refusals, `--help` among them, and a value flag taking a dash-led value. For +each `eval/` tool they cover its usage both ways, its refusals or its taking an unknown flag, +and a value flag given last. They also cover `transcript`'s exit codes for `--help` and `-h`, +and `wisdom`'s usage for no command, `--help` and an unknown command, and its refusals after a +command. Every `wisdom` case gives the command `bogus`, so a broken parse can only print the +usage, never start an export. A crash's stream is pinned by the line that names the problem, +allowing `\r?\n` for Node's own report. Tools.md's `check_cli` section says so now, adds a +file to what a case may stop at, and says the gate takes a few seconds (2.8 s before this +commit, 4.5 s after). The `site_search` cases were recorded before PR #210's merge rewrote +much of that file and pass on both. `check_cli` makes 218 checks: 39 probes and 179 cases. +The kit's `c51-tools.mjs` runs 26 real invocations through HEAD's copies and the migrated +tools, and compares the exit code, the masked output and every file written. The +invocations: `render-book` stopping at a missing input and a missing extra script, and one full +`book.bat` render per side, compared by page count and `pdftotext`; `build_corpus` over a +fixture and over the whole repository; `run_case --prompt-only` for both protocols and +`--smoke`, and its refusal of a corpus holding `CLAUDE.md`; `nav_hops` three ways, one over +that corpus; `site_search` four ways; `search_quality` over the full query set with `--save` +and with `--compare` (not `--sample`, which draws its queries at random, so no two runs +agree); `transcript` over a made-up session; `wisdom` with no command, `process` into scratch +whole and filtered by `--since`, `--force` and two `--channel`s, and `extract --dry-run` three +ways. All are the same. The kit's `c27-compare.mjs` reports all 21 of its export cases the +same. `build.bat`, `check.bat` (the a11y line unchanged) and `test.bat` exit 0. + +What differs, none of it a recorded case: `--name=value` is accepted for a known flag. After +`--`, an argument is a positional and `--` is not one, so `wisdom extract --` runs `extract` +and `build_corpus --dest x --` builds, where both refused the `--`. A lone `-` is +`render-book`'s input. A short group is split, so `nav_hops -hx` prints the usage. And +`render-book -ofile` and `-t5` take the attached value. + ### C52 — `builder: tbdocs parses through lib/cli.mjs` **A1-8 (R3), last, as decision (e) says.** `tbdocs.mjs`'s parser (`:91-194`) is neither diff --git a/docs/Documentation/Tools.md b/docs/Documentation/Tools.md index 8f09aba9..319b0c9c 100644 --- a/docs/Documentation/Tools.md +++ b/docs/Documentation/Tools.md @@ -523,11 +523,11 @@ Exits 1 on any failed probe, 2 if it cannot run. node scripts/check_cli.mjs -Verifies `lib/cli.mjs`, the module the tools read their command lines through, and each tool's recorded command-line errors. Nothing else tests how a tool reads its command line, which is how a value flag given no value came to be read as `NaN` or as the next flag. No built tree, no browser, no twinBASIC install; about a second. +Verifies `lib/cli.mjs`, the module the tools read their command lines through, and each tool's recorded command-line errors. Nothing else tests how a tool reads its command line, which is how a value flag given no value came to be read as `NaN` or as the next flag. No built tree, no browser, no twinBASIC install; a few seconds. The module's probes cover what `parseCli` returns and refuses, with a comparison against a strict `node:util` `parseArgs` over the same argument lists, and what `numberOption`, `withUsageError` and `printHelpAndExit` do. A tool that is more lenient than a strict parse today --- one that ignores an unknown flag, say --- keeps its leniency through two of `parseCli`'s parameters, and the probes cover those too. -The recorded cases are invocations that stop while the tool reads its command line, or at its first check of the project, folder or install the command line names, each with its exit code and what it prints on each stream: the tool's own words for the error exactly, and the opening of a usage text printed after it. A tool's cases are recorded before it moves onto `lib/cli.mjs`, so the move has to keep them. Each case runs the tool as a child process, in an empty folder of its own and with `TB_IDE` and `PUPPETEER_EXECUTABLE_PATH` naming files that do not exist, so a case that gets past the command line fails on a different message rather than starting a twinBASIC IDE or a browser. A case belongs here only if the tool stops before doing any work. +The recorded cases are invocations that stop while the tool reads its command line, or at its first check of the project, folder, file or install the command line names, each with its exit code and what it prints on each stream: the tool's own words for the error exactly, a crash's only by the line that names the problem, and the opening of a usage text printed after it. A tool's cases are recorded before it moves onto `lib/cli.mjs`, so the move has to keep them. Each case runs the tool as a child process, in an empty folder of its own and with `TB_IDE` and `PUPPETEER_EXECUTABLE_PATH` naming files that do not exist, so a case that gets past the command line fails on a different message rather than starting a twinBASIC IDE or a browser. A case belongs here only if the tool stops before doing any work. Exits 1 on any failed probe or case, 2 if it cannot run. diff --git a/eval/build_corpus.mjs b/eval/build_corpus.mjs index 9a31f718..5223ece3 100644 --- a/eval/build_corpus.mjs +++ b/eval/build_corpus.mjs @@ -15,6 +15,7 @@ import fs from "node:fs"; import path from "node:path"; +import { parseCli, printHelpAndExit, withUsageError } from "../lib/cli.mjs"; import { isOutputTree } from "../lib/markdown-files.mjs"; import { REPO_ROOT } from "../lib/repo-paths.mjs"; @@ -94,15 +95,23 @@ const WITHHELD = [ const STUB = "/* [ source withheld for this exercise -- treat this file as unreadable ] */\n"; function parseArgs(argv) { - const o = { src: REPO_ROOT, dest: null, quiet: false }; - for (let i = 0; i < argv.length; i++) { - if (argv[i] === "--src") o.src = path.resolve(argv[++i]); - else if (argv[i] === "--dest") o.dest = path.resolve(argv[++i]); - else if (argv[i] === "--quiet") o.quiet = true; - else if (argv[i] === "--help" || argv[i] === "-h") o.help = true; - else throw new Error(`unknown argument: ${argv[i]}`); - } - return o; + const { values } = withUsageError(() => parseCli(argv, { + options: { + src: { type: "string" }, + dest: { type: "string" }, + quiet: { type: "boolean", default: false }, + help: { type: "boolean", short: "h" }, + }, + positionals: 0, + unknown: "error", + acceptsValue: () => true, + }), { format: (err) => `unknown argument: ${err.arg}`, exitCode: 1 }); + return { + src: "src" in values ? path.resolve(values.src) : REPO_ROOT, + dest: "dest" in values ? path.resolve(values.dest) : null, + quiet: values.quiet, + help: values.help, + }; } function isExcluded(rel) { @@ -196,12 +205,12 @@ function report(dest, counts) { const opts = parseArgs(process.argv.slice(2)); if (opts.help || !opts.dest) { - console.log( + printHelpAndExit( "Usage: node eval/build_corpus.mjs --dest [--src ] [--quiet]\n\n" + "Mirrors the repository with every non-prose file replaced by an unreadable\n" + "stub, so a documentation evaluation cannot silently read the implementation.\n" + - "See eval/README.md." + "See eval/README.md.", + { exitCode: opts.help ? 0 : 1 }, ); - process.exit(opts.help ? 0 : 1); } build(opts); diff --git a/eval/nav_hops.mjs b/eval/nav_hops.mjs index e8a4ca8c..2566534f 100644 --- a/eval/nav_hops.mjs +++ b/eval/nav_hops.mjs @@ -32,6 +32,7 @@ import fs from "node:fs"; import path from "node:path"; import { pathToFileURL } from "node:url"; +import { parseCli, printHelpAndExit } from "../lib/cli.mjs"; import { parseFrontmatter } from "../lib/frontmatter.mjs"; import { blockRegions } from "../lib/markdown.mjs"; import { REPO_ROOT } from "../lib/repo-paths.mjs"; @@ -44,15 +45,22 @@ const USAGE = "whose permalink matches each regex. See eval/README.md."; function parseArgs(argv) { - const o = { src: REPO_ROOT, from: "docs/index.md", targets: [] }; - for (let i = 0; i < argv.length; i++) { - const a = argv[i]; - if (a === "--from") o.from = argv[++i]; - else if (a === "--src") o.src = path.resolve(argv[++i]); - else if (a === "--help" || a === "-h") o.help = true; - else o.targets.push(a); - } - return o; + const { values, positionals } = parseCli(argv, { + options: { + from: { type: "string", default: "docs/index.md" }, + src: { type: "string" }, + help: { type: "boolean", short: "h" }, + }, + positionals: { max: Infinity }, + unknown: "positional", + acceptsValue: () => true, + }); + return { + from: values.from, + src: "src" in values ? path.resolve(values.src) : REPO_ROOT, + help: values.help, + targets: positionals, + }; } /** A URL reduced to what identifies a page: no fragment, query, extension or trailing slash. */ @@ -114,10 +122,7 @@ function resolve(pages, from, href) { async function main(argv) { const o = parseArgs(argv); - if (o.help || !o.targets.length) { - console.log(USAGE); - return o.help ? 0 : 2; - } + if (o.help || !o.targets.length) return printHelpAndExit(USAGE, { exitCode: o.help ? 0 : 2 }); // Git Bash turns an argument that looks like a POSIX path into a Windows one, // so '^/tB/Core/Open$' arrives as '^C:/Program Files/Git/tB/Core/Open$', and // every target then reports as unreachable, which reads as a finding. diff --git a/eval/run_case.mjs b/eval/run_case.mjs index ba3fc565..8ef63d8a 100644 --- a/eval/run_case.mjs +++ b/eval/run_case.mjs @@ -58,6 +58,7 @@ import { spawn } from "node:child_process"; import fs from "node:fs"; import path from "node:path"; +import { parseCli, printHelpAndExit, withUsageError } from "../lib/cli.mjs"; import { blockRegions } from "../lib/markdown.mjs"; import { REPO_ROOT } from "../lib/repo-paths.mjs"; import { printDigest, readTranscript, summarize } from "./transcript.mjs"; @@ -76,22 +77,37 @@ const MEMORY_FILES = ["CLAUDE.md", "CLAUDE.local.md", "AGENTS.md"]; const fwd = (p) => p.split(path.sep).join("/"); function parseArgs(argv) { - const o = { claude: process.env.EVAL_CLAUDE || "claude", model: "sonnet", timeout: 20 }; - for (let i = 0; i < argv.length; i++) { - const a = argv[i]; - if (a === "--corpus") o.corpus = path.resolve(argv[++i]); - else if (a === "--site") o.site = path.resolve(argv[++i]); - else if (a === "--protocol") o.protocol = argv[++i]; - else if (a === "--goal") o.goal = path.resolve(argv[++i]); - else if (a === "--out") o.out = path.resolve(argv[++i]); - else if (a === "--claude") o.claude = argv[++i]; - else if (a === "--model") o.model = argv[++i]; - else if (a === "--timeout") o.timeout = Number(argv[++i]); - else if (a === "--smoke") o.smoke = true; - else if (a === "--prompt-only") o.promptOnly = true; - else if (a === "--help" || a === "-h") o.help = true; - else throw new Error(`unknown argument: ${a}`); - } + const { values } = withUsageError(() => parseCli(argv, { + options: { + corpus: { type: "string" }, + site: { type: "string" }, + goal: { type: "string" }, + out: { type: "string" }, + protocol: { type: "string" }, + claude: { type: "string", default: process.env.EVAL_CLAUDE || "claude" }, + model: { type: "string", default: "sonnet" }, + timeout: { type: "string" }, + smoke: { type: "boolean" }, + "prompt-only": { type: "boolean" }, + help: { type: "boolean", short: "h" }, + }, + positionals: 0, + unknown: "error", + acceptsValue: () => true, + }), { format: (err) => `unknown argument: ${err.arg}`, exitCode: 2 }); + const o = { + corpus: "corpus" in values ? path.resolve(values.corpus) : undefined, + site: "site" in values ? path.resolve(values.site) : undefined, + goal: "goal" in values ? path.resolve(values.goal) : undefined, + out: "out" in values ? path.resolve(values.out) : undefined, + protocol: values.protocol, + claude: values.claude, + model: values.model, + timeout: "timeout" in values ? Number(values.timeout) : 20, + smoke: values.smoke, + promptOnly: values.promptOnly, + help: values.help, + }; if (o.smoke) o.protocol = "site"; return o; } @@ -245,10 +261,7 @@ async function main(argv) { const o = parseArgs(argv); const complete = o.corpus && o.site && o.out && (o.smoke || (o.goal && ["repo", "site"].includes(o.protocol))); - if (o.help || !complete) { - console.log(USAGE); - return o.help ? 0 : 2; - } + if (o.help || !complete) return printHelpAndExit(USAGE, { exitCode: o.help ? 0 : 2 }); const cwd = o.protocol === "site" ? path.join(o.corpus, "docs") : o.corpus; const needed = [cwd, path.join(o.site, "assets/js/search-data.json"), path.join(o.site, "assets/js/vendor/lunr.min.js")]; diff --git a/eval/search_quality.mjs b/eval/search_quality.mjs index 856a2f20..3383e4c2 100644 --- a/eval/search_quality.mjs +++ b/eval/search_quality.mjs @@ -125,6 +125,7 @@ import fs from "node:fs"; import path from "node:path"; import zlib from "node:zlib"; import { performance } from "node:perf_hooks"; +import { parseCli, printHelpAndExit, withUsageError } from "../lib/cli.mjs"; import { REPO_ROOT } from "../lib/repo-paths.mjs"; import { load, buildIndex, search, KIND_WORDS } from "./site_search.mjs"; @@ -132,29 +133,29 @@ import { load, buildIndex, search, KIND_WORDS } from "./site_search.mjs"; // ---------------------------------------------------------------- arg parsing function parseArgs(argv) { - const o = { - site: path.join(REPO_ROOT, "docs/_site"), - save: null, - compare: null, - sample: null, - worstN: 15, - failures: 0, + const { values } = withUsageError(() => parseCli(argv, { + options: { + site: { type: "string" }, + save: { type: "string" }, + compare: { type: "string" }, + sample: { type: "string" }, + worst: { type: "string" }, + failures: { type: "string" }, + help: { type: "boolean", short: "h" }, + }, + positionals: 0, + unknown: "error", + acceptsValue: () => true, + }), { format: (err) => `unrecognised argument: ${err.arg}`, exitCode: 1 }); + return { + site: "site" in values ? path.resolve(values.site) : path.join(REPO_ROOT, "docs/_site"), + save: "save" in values ? path.resolve(values.save) : null, + compare: "compare" in values ? path.resolve(values.compare) : null, + sample: "sample" in values ? Number(values.sample) : null, + worstN: "worst" in values ? Number(values.worst) : 15, + failures: "failures" in values ? Number(values.failures) : 0, + help: values.help, }; - for (let i = 0; i < argv.length; i++) { - const a = argv[i]; - if (a === "--site") o.site = path.resolve(argv[++i]); - else if (a === "--save") o.save = path.resolve(argv[++i]); - else if (a === "--compare") o.compare = path.resolve(argv[++i]); - else if (a === "--sample") o.sample = Number(argv[++i]); - else if (a === "--worst") o.worstN = Number(argv[++i]); - else if (a === "--failures") o.failures = Number(argv[++i]); - else if (a === "--help" || a === "-h") o.help = true; - else { - console.error(`unrecognised argument: ${a}`); - process.exit(1); - } - } - return o; } // ------------------------------------------------------------- URL normalize @@ -693,11 +694,10 @@ function printCompare(current, saved, worstN) { function main() { const opts = parseArgs(process.argv.slice(2)); if (opts.help) { - console.log( + printHelpAndExit( "Usage: node eval/search_quality.mjs [--site docs/_site] [--save file] " + "[--compare file] [--worst N] [--sample N] [--failures N]\n\nSee the header comment in this file." ); - process.exit(0); } const ctx = load(opts.site); diff --git a/eval/site_search.mjs b/eval/site_search.mjs index 2d51fcbb..c68a2123 100644 --- a/eval/site_search.mjs +++ b/eval/site_search.mjs @@ -25,20 +25,30 @@ import { createRequire } from "node:module"; import { pathToFileURL } from "node:url"; import fs from "node:fs"; import path from "node:path"; +import { parseCli, printHelpAndExit } from "../lib/cli.mjs"; import { REPO_ROOT } from "../lib/repo-paths.mjs"; const require = createRequire(import.meta.url); function parseArgs(argv) { - const o = { site: path.join(REPO_ROOT, "docs/_site"), n: 8, terms: [] }; - for (let i = 0; i < argv.length; i++) { - if (argv[i] === "--site") o.site = path.resolve(argv[++i]); - else if (argv[i] === "--n") o.n = Number(argv[++i]); - else if (argv[i] === "--composition") o.composition = true; - else if (argv[i] === "--help" || argv[i] === "-h") o.help = true; - else o.terms.push(argv[i]); - } - return o; + const { values, positionals } = parseCli(argv, { + options: { + site: { type: "string" }, + n: { type: "string" }, + composition: { type: "boolean" }, + help: { type: "boolean", short: "h" }, + }, + positionals: { max: Infinity }, + unknown: "positional", + acceptsValue: () => true, + }); + return { + site: "site" in values ? path.resolve(values.site) : path.join(REPO_ROOT, "docs/_site"), + n: "n" in values ? Number(values.n) : 8, + composition: values.composition, + help: values.help, + terms: positionals, + }; } export function resolvePaths(site) { @@ -517,13 +527,13 @@ function composition(docs) { if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { const opts = parseArgs(process.argv.slice(2)); if (opts.help || (!opts.composition && !opts.terms.length)) { - console.log( + printHelpAndExit( 'Usage: node eval/site_search.mjs "" [--n ] [--site ]\n' + " node eval/site_search.mjs --composition\n\n" + "Queries the built site's real lunr index with the real query logic.\n" + - "See eval/README.md." + "See eval/README.md.", + { exitCode: opts.help ? 0 : 1 }, ); - process.exit(opts.help ? 0 : 1); } const ctx = load(opts.site); diff --git a/eval/transcript.mjs b/eval/transcript.mjs index 638482cb..43c4e09e 100644 --- a/eval/transcript.mjs +++ b/eval/transcript.mjs @@ -24,6 +24,7 @@ import fs from "node:fs"; import path from "node:path"; import { fileURLToPath } from "node:url"; +import { parseCli, printHelpAndExit } from "../lib/cli.mjs"; /** Every event in a stream-json session, in order. */ export function readTranscript(file) { @@ -188,19 +189,28 @@ export function printDigest(s, { calls = false, report = false } = {}) { } function main(argv) { - const file = argv.find((a) => !a.startsWith("--")); - if (!file || argv.includes("--help") || argv.includes("-h")) { - console.log( + const { values, positionals } = parseCli(argv, { + options: { + calls: { type: "boolean", default: false }, + report: { type: "boolean", default: false }, + // No short "h": a lone -h is taken as the file, so it prints the usage + // and exits 0, where --help alone exits 1. C71 makes both exit 0. + help: { type: "boolean" }, + }, + positionals: { max: Infinity }, + unknown: "positional", + }); + const file = positionals.find((a) => !a.startsWith("--")); + const help = values.help || positionals.includes("-h"); + if (!file || help) { + printHelpAndExit( "Usage: node eval/transcript.mjs [--calls] [--report]\n\n" + "Summarises an evaluator's session and audits the order of its channels.\n" + - "See eval/README.md." + "See eval/README.md.", + { exitCode: file ? 0 : 1 }, ); - process.exit(file ? 0 : 1); } - printDigest(summarize(readTranscript(file)), { - calls: argv.includes("--calls"), - report: argv.includes("--report"), - }); + printDigest(summarize(readTranscript(file)), { calls: values.calls, report: values.report }); } if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { diff --git a/scripts/check_cli.mjs b/scripts/check_cli.mjs index 8707853e..350ee0e7 100644 --- a/scripts/check_cli.mjs +++ b/scripts/check_cli.mjs @@ -390,6 +390,94 @@ const CASES = [ { tool: "scripts/compare_trees.mjs", args: ["--bogus", "--help"], exit: 2, stderr: /^compare_trees: unknown argument "--bogus"\n\nusage: node scripts\/compare_trees\.mjs / }, { tool: "scripts/compare_trees.mjs", args: ["--before", "--", "x"], exit: 2, stderr: /^compare_trees: --before needs a value\n\nusage: node scripts\/compare_trees\.mjs / }, { tool: "scripts/compare_trees.mjs", args: ["--keep=1"], exit: 2, stderr: /^compare_trees: unknown argument "--keep=1"\n\nusage: node scripts\/compare_trees\.mjs / }, + + // Recorded in C51, before render-book, eval/ and wisdom moved onto + // lib/cli.mjs. In every one of them a value flag takes whatever follows it, + // so a missing value shows only where the value is used. render-book + // refuses --help, an unknown flag and a second input with "unknown arg". + // build_corpus threw on an unknown argument, so only its message is pinned, + // as are the other crashes here; run_case and search_quality refuse one in + // their own words. nav_hops and site_search take an unknown + // flag as a pattern or a search term. transcript's file is its first + // argument that does not start with --, -h included, and its exit code + // follows whether there is one, so a bare --help exits 1. wisdom takes its + // first argument as the command and answers --help there as an unknown + // command; after it, an unknown option or a stray argument is refused + // before any command runs, and the command in these cases is never a real + // one, so that none can start an export. + { tool: "book/render-book.mjs", args: ["--help"], exit: 2, stderr: "unknown arg: --help\n" }, + { tool: "book/render-book.mjs", args: [], exit: 2, stderr: "usage: node render-book.mjs -o [--outline-tags ...] [-t ms] [--additional-script path]...\n" }, + { tool: "book/render-book.mjs", args: ["a.html", "b.html"], exit: 2, stderr: "unknown arg: b.html\n" }, + { tool: "book/render-book.mjs", args: ["a.html", "-o"], exit: 2, stderr: /^usage: node render-book\.mjs / }, + { tool: "book/render-book.mjs", args: ["-o", "--bogus", "a.html"], exit: 1, stderr: /^input not found: .*a\.html\n$/ }, + { tool: "book/render-book.mjs", args: ["a.html", "-o", "out.pdf", "--outline-tags"], exit: 1, stderr: /TypeError: Cannot read properties of undefined \(reading 'split'\)\r?\n/ }, + { tool: "book/render-book.mjs", args: ["a.html", "-o", "out.pdf", "-t", "abc"], exit: 1, stderr: /^input not found: .*a\.html\n$/ }, + { tool: "book/render-book.mjs", args: ["-x"], exit: 2, stderr: "unknown arg: -x\n" }, + { tool: "eval/build_corpus.mjs", args: ["--help"], exit: 0, stdout: /^Usage: node eval\/build_corpus\.mjs --dest / }, + { tool: "eval/build_corpus.mjs", args: [], exit: 1, stdout: /^Usage: node eval\/build_corpus\.mjs --dest / }, + { tool: "eval/build_corpus.mjs", args: ["--bogus"], exit: 1, stderr: /(^|\n)(Error: )?unknown argument: --bogus\r?\n/ }, + { tool: "eval/build_corpus.mjs", args: ["stray"], exit: 1, stderr: /(^|\n)(Error: )?unknown argument: stray\r?\n/ }, + { tool: "eval/build_corpus.mjs", args: ["--help", "--bogus"], exit: 1, stderr: /(^|\n)(Error: )?unknown argument: --bogus\r?\n/ }, + { tool: "eval/build_corpus.mjs", args: ["--quiet=1"], exit: 1, stderr: /(^|\n)(Error: )?unknown argument: --quiet=1\r?\n/ }, + { tool: "eval/build_corpus.mjs", args: ["-hq"], exit: 1, stderr: /(^|\n)(Error: )?unknown argument: -hq\r?\n/ }, + { tool: "eval/build_corpus.mjs", args: ["--src"], exit: 1, stderr: /TypeError \[ERR_INVALID_ARG_TYPE\]: The "paths\[0\]" argument must be of type string\. Received undefined\r?\n/ }, + { tool: "eval/nav_hops.mjs", args: ["--help"], exit: 0, stdout: /^Usage: node eval\/nav_hops\.mjs \[--from \] / }, + { tool: "eval/nav_hops.mjs", args: [], exit: 2, stdout: /^Usage: node eval\/nav_hops\.mjs \[--from \] / }, + { tool: "eval/nav_hops.mjs", args: ["--from", "nope.md", "x"], exit: 2, stderr: /^no start page: .*[\\/]nope\.md\n$/ }, + { tool: "eval/nav_hops.mjs", args: ["--src", "nowhere", "--bogus"], exit: 2, stderr: /^no start page: .*[\\/]nowhere[\\/]docs[\\/]index\.md\n$/ }, + { tool: "eval/nav_hops.mjs", args: ["--help=1", "--src", "nowhere"], exit: 2, stderr: /^no start page: .*[\\/]nowhere[\\/]docs[\\/]index\.md\n$/ }, + { tool: "eval/nav_hops.mjs", args: ["--from", "--src", "x"], exit: 2, stderr: /^no start page: .*[\\/]--src\n$/ }, + { tool: "eval/nav_hops.mjs", args: ["--bogus", "C:/x"], exit: 2, stderr: "these patterns arrived as Windows paths: C:/x\nGit Bash converted them. Run with MSYS_NO_PATHCONV=1 set, or from another shell.\n" }, + { tool: "eval/nav_hops.mjs", args: ["--src"], exit: 2, stderr: /^TypeError \[ERR_INVALID_ARG_TYPE\]: The "paths\[0\]" argument must be of type string\. Received undefined\n/ }, + { tool: "eval/nav_hops.mjs", args: ["x", "--from"], exit: 2, stderr: /^TypeError \[ERR_INVALID_ARG_TYPE\]: The "paths\[1\]" argument must be of type string\. Received undefined\n/ }, + { tool: "eval/run_case.mjs", args: ["--help"], exit: 0, stdout: /^Usage: node eval\/run_case\.mjs --corpus / }, + { tool: "eval/run_case.mjs", args: [], exit: 2, stdout: /^Usage: node eval\/run_case\.mjs --corpus / }, + { tool: "eval/run_case.mjs", args: ["--bogus"], exit: 2, stderr: "unknown argument: --bogus\n" }, + { tool: "eval/run_case.mjs", args: ["stray"], exit: 2, stderr: "unknown argument: stray\n" }, + { tool: "eval/run_case.mjs", args: ["--help", "--bogus"], exit: 2, stderr: "unknown argument: --bogus\n" }, + { tool: "eval/run_case.mjs", args: ["--prompt-only=1"], exit: 2, stderr: "unknown argument: --prompt-only=1\n" }, + { tool: "eval/run_case.mjs", args: ["--corpus"], exit: 2, stderr: 'The "paths[0]" argument must be of type string. Received undefined\n' }, + { tool: "eval/run_case.mjs", args: ["--smoke", "--corpus", "c", "--site", "s", "--out", "o"], exit: 2, stderr: /^missing: .*[\\/]c[\\/]docs, .*search-data\.json, .*lunr\.min\.js\n$/ }, + { tool: "eval/run_case.mjs", args: ["--smoke", "--corpus", "c", "--site", "s", "--out", "o", "--timeout", "abc"], exit: 2, stderr: /^missing: .*[\\/]c[\\/]docs, / }, + { tool: "eval/run_case.mjs", args: ["--corpus", "c", "--site", "s", "--out", "o", "--protocol", "--smoke"], exit: 2, stdout: /^Usage: node eval\/run_case\.mjs --corpus / }, + { tool: "eval/site_search.mjs", args: ["--help"], exit: 0, stdout: /^Usage: node eval\/site_search\.mjs "" / }, + { tool: "eval/site_search.mjs", args: [], exit: 1, stdout: /^Usage: node eval\/site_search\.mjs "" / }, + { tool: "eval/site_search.mjs", args: ["--site", "nowhere", "--bogus"], exit: 1, stderr: /^missing .*search-data\.json\nRun build\.bat / }, + { tool: "eval/site_search.mjs", args: ["--help=1", "--site", "nowhere"], exit: 1, stderr: /^missing .*search-data\.json\nRun build\.bat / }, + { tool: "eval/site_search.mjs", args: ["--composition", "--site", "nowhere"], exit: 1, stderr: /^missing .*search-data\.json\nRun build\.bat / }, + { tool: "eval/site_search.mjs", args: ["--site"], exit: 1, stderr: /TypeError \[ERR_INVALID_ARG_TYPE\]: The "paths\[0\]" argument must be of type string\. Received undefined\r?\n/ }, + { tool: "eval/search_quality.mjs", args: ["--help"], exit: 0, stdout: /^Usage: node eval\/search_quality\.mjs \[--site docs\/_site\] / }, + { tool: "eval/search_quality.mjs", args: ["--bogus"], exit: 1, stderr: "unrecognised argument: --bogus\n" }, + { tool: "eval/search_quality.mjs", args: ["stray"], exit: 1, stderr: "unrecognised argument: stray\n" }, + { tool: "eval/search_quality.mjs", args: ["--help", "--bogus"], exit: 1, stderr: "unrecognised argument: --bogus\n" }, + { tool: "eval/search_quality.mjs", args: ["--help=1"], exit: 1, stderr: "unrecognised argument: --help=1\n" }, + { tool: "eval/search_quality.mjs", args: ["-x"], exit: 1, stderr: "unrecognised argument: -x\n" }, + { tool: "eval/search_quality.mjs", args: ["--site", "nowhere"], exit: 1, stderr: /^missing .*search-data\.json\nRun build\.bat / }, + { tool: "eval/search_quality.mjs", args: ["--site", "nowhere", "--sample", "abc"], exit: 1, stderr: /^missing .*search-data\.json\nRun build\.bat / }, + { tool: "eval/search_quality.mjs", args: ["--site", "--help"], exit: 1, stderr: /^missing .*[\\/]--help[\\/]assets[\\/]js[\\/]search-data\.json\nRun build\.bat / }, + { tool: "eval/search_quality.mjs", args: ["--site"], exit: 1, stderr: /TypeError \[ERR_INVALID_ARG_TYPE\]: The "paths\[0\]" argument must be of type string\. Received undefined\r?\n/ }, + { tool: "eval/search_quality.mjs", args: ["--site", "nowhere", "--save"], exit: 1, stderr: /TypeError \[ERR_INVALID_ARG_TYPE\]: The "paths\[0\]" argument must be of type string\. Received undefined\r?\n/ }, + { tool: "eval/transcript.mjs", args: ["--help"], exit: 1, stdout: /^Usage: node eval\/transcript\.mjs / }, + { tool: "eval/transcript.mjs", args: ["-h"], exit: 0, stdout: /^Usage: node eval\/transcript\.mjs / }, + { tool: "eval/transcript.mjs", args: [], exit: 1, stdout: /^Usage: node eval\/transcript\.mjs / }, + { tool: "eval/transcript.mjs", args: ["nope.jsonl", "--help"], exit: 0, stdout: /^Usage: node eval\/transcript\.mjs / }, + { tool: "eval/transcript.mjs", args: ["--bogus"], exit: 1, stdout: /^Usage: node eval\/transcript\.mjs / }, + { tool: "eval/transcript.mjs", args: ["--help=1"], exit: 1, stdout: /^Usage: node eval\/transcript\.mjs / }, + { tool: "eval/transcript.mjs", args: ["nope.jsonl"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*nope\.jsonl'\r?\n/ }, + { tool: "eval/transcript.mjs", args: ["--bogus", "nope.jsonl"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*nope\.jsonl'\r?\n/ }, + { tool: "eval/transcript.mjs", args: ["-x"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*[\\/]-x'\r?\n/ }, + { tool: "eval/transcript.mjs", args: ["a.jsonl", "b.jsonl"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*a\.jsonl'\r?\n/ }, + { tool: "wisdom/wisdom.mjs", args: [], exit: 0, stderr: /^Usage: node wisdom\/wisdom\.mjs \[options\]\n/ }, + { tool: "wisdom/wisdom.mjs", args: ["--help"], exit: 1, stderr: /^Usage: node wisdom\/wisdom\.mjs \[options\]\n/ }, + { tool: "wisdom/wisdom.mjs", args: ["bogus"], exit: 1, stderr: /^Usage: node wisdom\/wisdom\.mjs \[options\]\n/ }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "--guild", "--bogus"], exit: 1, stderr: /^Usage: node wisdom\/wisdom\.mjs \[options\]\n/ }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "--cap"], exit: 1, stderr: /^Usage: node wisdom\/wisdom\.mjs \[options\]\n/ }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "--bogus"], exit: 1, stderr: "Unknown option: --bogus\n" }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "stray"], exit: 1, stderr: "Unknown option: stray\n" }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "--help"], exit: 1, stderr: "Unknown option: --help\n" }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "--force=1"], exit: 1, stderr: "Unknown option: --force=1\n" }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "-x"], exit: 1, stderr: "Unknown option: -x\n" }, + { tool: "wisdom/wisdom.mjs", args: ["bogus", "--guild", "x", "--bogus"], exit: 1, stderr: "Unknown option: --bogus\n" }, ]; const TIMEOUT_MS = 30_000; diff --git a/wisdom/wisdom.mjs b/wisdom/wisdom.mjs index d36e8c86..bc8051ea 100644 --- a/wisdom/wisdom.mjs +++ b/wisdom/wisdom.mjs @@ -3,6 +3,7 @@ import { mkdirSync, existsSync } from 'node:fs' import { join, dirname } from 'node:path' import { fileURLToPath } from 'node:url' +import { parseCli, printHelpAndExit, withUsageError } from '../lib/cli.mjs' import { loadConfig } from './config.mjs' import { readJsonFile, writeFileAtomic } from './files.mjs' import { createClient, CapReachedError, timestampToSnowflake, EXIT_CAP_REACHED } from './discord/api.mjs' @@ -14,30 +15,41 @@ import { runExtract, runMerge } from './extract/prep.mjs' const __dirname = dirname(fileURLToPath(import.meta.url)) function parseArgs(argv) { - const args = argv.slice(2) - const command = args[0] - const flags = { channels: [] } - - for (let i = 1; i < args.length; i++) { - switch (args[i]) { - case '--guild': flags.guild = args[++i]; break - case '--channel': flags.channels.push(args[++i]); break - case '--since': flags.since = args[++i]; break - case '--force': flags.force = true; break - case '--in': flags.in = args[++i]; break - case '--out': flags.out = args[++i]; break - case '--concurrency': flags.concurrency = parseInt(args[++i], 10); break - case '--rate-limit': flags.rateLimit = parseFloat(args[++i]); break - case '--cap': flags.cap = parseInt(args[++i], 10); break - case '--dry-run': flags.dryRun = true; break - case '--merge': flags.merge = true; break - case '--all': flags.all = true; break - case '--min-confidence': flags.minConfidence = args[++i]; break - default: - process.stderr.write(`Unknown option: ${args[i]}\n`) - process.exit(1) - } - } + const [command, ...rest] = argv.slice(2) + const { values } = withUsageError(() => parseCli(rest, { + options: { + guild: { type: 'string' }, + channel: { type: 'string', multiple: true }, + since: { type: 'string' }, + in: { type: 'string' }, + out: { type: 'string' }, + concurrency: { type: 'string' }, + 'rate-limit': { type: 'string' }, + cap: { type: 'string' }, + 'min-confidence': { type: 'string' }, + force: { type: 'boolean' }, + 'dry-run': { type: 'boolean' }, + merge: { type: 'boolean' }, + all: { type: 'boolean' }, + }, + positionals: 0, + unknown: 'error', + acceptsValue: () => true, + }), { format: (err) => `Unknown option: ${err.arg}`, exitCode: 1 }) + + const flags = { channels: values.channel } + if ('guild' in values) flags.guild = values.guild + if ('since' in values) flags.since = values.since + if ('in' in values) flags.in = values.in + if ('out' in values) flags.out = values.out + if ('concurrency' in values) flags.concurrency = parseInt(values.concurrency, 10) + if ('rateLimit' in values) flags.rateLimit = parseFloat(values.rateLimit) + if ('cap' in values) flags.cap = parseInt(values.cap, 10) + if ('minConfidence' in values) flags.minConfidence = values.minConfidence + if (values.force) flags.force = true + if (values.dryRun) flags.dryRun = true + if (values.merge) flags.merge = true + if (values.all) flags.all = true return { command, flags } } @@ -258,6 +270,5 @@ switch (command) { else await runExtract(flags) break default: - process.stderr.write(USAGE) - process.exit(command ? 1 : 0) + printHelpAndExit(USAGE, { stream: 'stderr', exitCode: command ? 1 : 0 }) } From 12d6c9c2e27f95a090e791311a8445d0ece20d04 Mon Sep 17 00:00:00 2001 From: Kuba Sunderland-Ober Date: Sun, 27 Sep 2026 21:57:42 +0200 Subject: [PATCH 2/4] search: build the index in slices, so the box takes keystrokes while it loads --- WIP.Search.md | 10 +- builder/offline.mjs | 40 ++-- builder/vendor/just-the-docs/README.md | 50 ++++- .../just-the-docs/assets/js/just-the-docs.js | 176 +++++++++++++++--- test/search.test.mjs | 71 +++++++ 5 files changed, 295 insertions(+), 52 deletions(-) diff --git a/WIP.Search.md b/WIP.Search.md index 308272a9..1b04e1f1 100644 --- a/WIP.Search.md +++ b/WIP.Search.md @@ -618,8 +618,14 @@ found" (`.search-no-result`, reused rather than adding a class, since visually it's the same single centred message) and the same text in the `a11y-status` live region, then yields (a `requestAnimationFrame` raced by a 100 ms timer, since frames never fire in a hidden tab, then -`setTimeout(fn, 0)`) so that message actually paints before the -synchronous, comparatively expensive index build runs on the main thread. +`setTimeout(fn, 0)`) so that message actually paints before the index +build begins on the main thread. The build runs 25 ms at a time and yields +between slices (`buildIndexInSlices()`, lunr's own work in pieces: an +entry added, a hundred fields' vectors, a term put into the token set), so +the reader can keep typing while the message shows; the longest task left +is about 50 ms. Built in one piece, the index held the main thread for +about 1.6 s, the box took no keystrokes, and the first search was for the +text typed before the build began. A keystroke during the load leaves the message in place. Checked in a browser: the message stays up until results replace it, with no empty panel in between, both online and in the offline tree. diff --git a/builder/offline.mjs b/builder/offline.mjs index 36f0e706..63ceebc5 100644 --- a/builder/offline.mjs +++ b/builder/offline.mjs @@ -364,7 +364,10 @@ const JTD_INITSEARCH_FN_REPLACEMENT = `function initSearch() { // Mirrors the online build's stem-twin patch (stemTwins() and // qualifiedField() are the online copy's). var twins = stemTwins(docs); - var index = lunr(function(){ + // Mirrors the online build's sliced build (buildIndexInSlices() is + // the online copy's), so the search box takes keystrokes while the + // index builds. + buildIndexInSlices(function(){ this.ref('id'); this.field('title', { boost: 200 }); this.field('content', { boost: 2 }); @@ -394,24 +397,25 @@ const JTD_INITSEARCH_FN_REPLACEMENT = `function initSearch() { // twinBASIC keywords (Do, For, If, Is, On, With, Each...) that the // search pipeline's query side never dropped. this.pipeline.remove(lunr.stopWordFilter); - - for (var i in docs) { - this.add({ - id: i, - title: docs[i].title, - content: indexedContent(docs[i]), - names: docs[i].names || '', - qualified: qualifiedField(docs[i], twins), - exact: (docs[i].names || '').split(/\\s+/).filter(Boolean).map(exactName).join(' '), - primary: (docs[i].primary || '').split(/\\s+/).filter(Boolean).map(exactName).join(' '), - page: docs[i].doc || '', - index: indexField(docs[i]), - relUrl: docs[i].relUrl - }); - } + }, docs, function(i) { + return { + id: i, + title: docs[i].title, + content: indexedContent(docs[i]), + names: docs[i].names || '', + qualified: qualifiedField(docs[i], twins), + exact: (docs[i].names || '').split(/\\s+/).filter(Boolean).map(exactName).join(' '), + primary: (docs[i].primary || '').split(/\\s+/).filter(Boolean).map(exactName).join(' '), + page: docs[i].doc || '', + index: indexField(docs[i]), + relUrl: docs[i].relUrl + }; + }, function(index) { + onSuccess(index, docs); + }, function(e) { + console.log('Error building search index: ' + e); + onError(); }); - - onSuccess(index, docs); } catch (e) { console.log('Error building search index: ' + e); onError(); diff --git a/builder/vendor/just-the-docs/README.md b/builder/vendor/just-the-docs/README.md index 76adb7ee..c9d15612 100644 --- a/builder/vendor/just-the-docs/README.md +++ b/builder/vendor/just-the-docs/README.md @@ -402,7 +402,7 @@ that leaves the box non-empty. That first keystroke shows a loading message (reusing `.search-no-result`) and the matching `a11y-status` text, then yields: a frame (`requestAnimationFrame`) raced by a 100 ms timer, since frames never fire in a hidden tab, then `setTimeout(fn, 0)`, so the message -paints before the synchronous build runs. It then searches whatever is in +paints before the build begins. It then searches whatever is in the box *when the build finishes*, which may differ from what triggered it, since the reader may keep typing. Only one load is ever in flight; a keystroke mid-load leaves the loading message in place (`update()` checks @@ -417,6 +417,43 @@ stop-word and dot-run-split patches above from inside their own `loadIndex()`. `eval/site_search.mjs` has no lazy build to mirror -- the CLI always wants an index built up front. +**The build held the main thread for about 1.6 s, so the search box took no +keystrokes while "Loading search index..." showed**, and the first search +was for the text typed before the build began. `buildIndexInSlices(config, +docs, entry, onSuccess, onError)` does lunr 2.3.9's own work, in its own +order, `INDEX_SLICE_MS` (25 ms) at a time, and yields between slices +through a `MessageChannel` message, which a browser does not clamp to 4 ms +as it does nested timeouts. `lunr(config)` gives a new builder two +pipelines, calls `config` on it and calls `build()`, which works out the +average field lengths, each field's vector (`createFieldVectors()`) and +the token set (`TokenSet.fromArray()` over the sorted terms), then makes +the `Index`. The pieces are an entry added (`config` no longer adds them; +`entry(id)` returns the one for `docs[id]`), `FIELDS_PER_PIECE` (100) +fields' vectors, and a term put into a `TokenSet.Builder`. The vectors are +lunr's own `createFieldVectors()`, called on a view of the builder +(`Object.create(builder)`) whose `fieldTermFrequencies` holds only that +piece's fields. That method caches each term's idf for one call only, so +each piece would count the documents of every common term again, which +made the vectors 6 to 13 times as slow; while a piece runs, `lunr.idf`, +which lunr looks up at call time, answers from a cache kept for the whole +build, by the term's number (`posting._index`), and is put back in a +`finally`. A slice that throws reports through `onError`, as a failed +fetch does: "Search is unavailable", and the next keystroke retries. +`test/search.test.mjs`'s sliced-build guard checks that the index comes +out the same as `lunr(config)`'s, sliced one piece at a time and as the +site slices it, and that `lunr.idf` is put back after a throw. + +Measured in headless Chrome on the built site, typing ahead of the build, +each version three times over: in one piece, the longest main-thread task +took 1.6 to 1.7 s, the keystrokes waited for it, and the first query +searched `f`; with only the entries sliced, 0.35 to 0.38 s (the unsliced +`build()`); sliced throughout, 52 to 54 ms, and the first query searched +`form`. The whole load took 1.46 to 1.73 s in one piece and 1.55 to 1.83 +s sliced. Both copies of `initSearch()` call `buildIndexInSlices()`; it +sits outside them, so `offline.mjs` carries it unchanged. `loadIndexNow()` +also shows the loading message again when the box was cleared mid-load and +the reader types again. + ## Licence just-the-docs is MIT-licensed, and `LICENSE.txt` beside this file is the @@ -504,16 +541,21 @@ Bumping the just-the-docs version is a deliberate operation. Procedure: 5. Re-apply the copy-button patch, the edit-distance cap, the asterisk guard and query-token trim, the `names`/`qualified` fields, the smart dot split, the stop-word removal, the dot-run-split tokenizer wrapper, the - lazy index build, the `exact`/`primary`/`page` fields with the + lazy index build and its slices, the `exact`/`primary`/`page` fields with the exact-name and all-words-first query, the `index` field with its helpers and query clauses, the stem twins held whole in `qualified`, and the separated token-set keys in `assets/js/just-the-docs.js` (see above). On a lunr upgrade, check `test/search.test.mjs`'s token-set key - guard: it says whether the new lunr still collides. Diffing against + guard: it says whether the new lunr still collides. Check too that the + new `lunr()` and `Builder#build()` still work as `buildIndexInSlices()` + repeats them, and that `createFieldVectors()` still reads + `fieldTermFrequencies` and calls `lunr.idf` through the namespace; the + sliced-build guard fails if the index comes out different. Diffing against the previous vendored copy via `git diff` is the easiest way to spot what needs to come back. Then re-check `offline.mjs`'s `JTD_INITSEARCH_FN_REPLACEMENT` still carries the same six extra fields - at the same boosts and the same stop-word/dot-run-split/lazy-build patches, and + at the same boosts and the same stop-word/dot-run-split/lazy-build patches, + and still builds through `buildIndexInSlices()`, and run `test/search.test.mjs`'s field-list drift guard and its stop-word/ dot-run-split sibling guard -- both fail loudly if the re-vendor left the copies out of step. If upstream's `initSearch()`/`searchLoaded()` split diff --git a/builder/vendor/just-the-docs/assets/js/just-the-docs.js b/builder/vendor/just-the-docs/assets/js/just-the-docs.js index 433883ac..d2b1e3e7 100644 --- a/builder/vendor/just-the-docs/assets/js/just-the-docs.js +++ b/builder/vendor/just-the-docs/assets/js/just-the-docs.js @@ -143,7 +143,7 @@ function initSearch() { accumulateSetUnions(); var twins = stemTwins(docs); - var index = lunr(function(){ + buildIndexInSlices(function(){ this.ref('id'); this.field('title', { boost: 200 }); this.field('content', { boost: 2 }); @@ -185,25 +185,25 @@ function initSearch() { // many are twinBASIC keywords (Do, For, If, Is, On, With, Each...). // See WIP.Search.md's "Design" section. this.pipeline.remove(lunr.stopWordFilter); - - for (var i in docs) { - - this.add({ - id: i, - title: docs[i].title, - content: indexedContent(docs[i]), - names: docs[i].names || '', - qualified: qualifiedField(docs[i], twins), - exact: (docs[i].names || '').split(/\s+/).filter(Boolean).map(exactName).join(' '), - primary: (docs[i].primary || '').split(/\s+/).filter(Boolean).map(exactName).join(' '), - page: docs[i].doc || '', - index: indexField(docs[i]), - relUrl: docs[i].relUrl - }); - } + }, docs, function(i) { + return { + id: i, + title: docs[i].title, + content: indexedContent(docs[i]), + names: docs[i].names || '', + qualified: qualifiedField(docs[i], twins), + exact: (docs[i].names || '').split(/\s+/).filter(Boolean).map(exactName).join(' '), + primary: (docs[i].primary || '').split(/\s+/).filter(Boolean).map(exactName).join(' '), + page: docs[i].doc || '', + index: indexField(docs[i]), + relUrl: docs[i].relUrl + }; + }, function(index) { + onSuccess(index, docs); + }, function(e) { + console.log('Error building search index: ' + e); + onError(); }); - - onSuccess(index, docs); } catch (e) { console.log('Error building search index: ' + e); onError(); @@ -432,6 +432,120 @@ function pinIndexFieldLengths(builder) { }; } +// Patched: builds the index as lunr(config) does, but INDEX_SLICE_MS worth at +// a time, handing the main thread back to the browser between slices, so the +// search box takes keystrokes while the index builds. Built in one piece, the +// index held the main thread for about 1.6 s on a desktop, the search box +// took no keystrokes until it was done, and the first search was for the +// text typed before the build began. `entry(id)` gives the entry to add for +// docs[id]. The work is lunr 2.3.9's own, in its own order: lunr(config) +// gives a new builder two pipelines and calls `config` on it, and build() +// works out the average field lengths, then each field's vector +// (createFieldVectors()), then the token set (TokenSet.fromArray() over the +// sorted terms), then makes the Index. Only the pieces are new: an entry +// added, FIELDS_PER_PIECE fields' vectors, a term put into the token set. +// test/search.test.mjs checks that the index comes out the same as +// lunr(config)'s. Called from initSearch() above and from offline.mjs's copy +// of it. +var INDEX_SLICE_MS = 25; +var FIELDS_PER_PIECE = 100; +function buildIndexInSlices(config, docs, entry, onSuccess, onError) { + var builder = new lunr.Builder(); + builder.pipeline.add(lunr.trimmer, lunr.stopWordFilter, lunr.stemmer); + builder.searchPipeline.add(lunr.stemmer); + config.call(builder, builder); + + var ids = Object.keys(docs); + var added = 0; + var fields = null; // the field refs, once every entry is in + var vectored = 0; + var vectors = {}; + var idfs = []; + var terms = null; // the sorted terms, once every vector is made + var tokens = null; + var inserted = 0; + + // Does one piece of the build, and returns false once none is left. + function piece() { + if (added < ids.length) { + builder.add(entry(ids[added++])); + } else if (!fields) { + builder.calculateAverageFieldLengths(); + fields = Object.keys(builder.fieldTermFrequencies); + } else if (vectored < fields.length) { + addFieldVectors(fields.slice(vectored, vectored + FIELDS_PER_PIECE)); + vectored += FIELDS_PER_PIECE; + } else if (!terms) { + terms = Object.keys(builder.invertedIndex).sort(); + tokens = new lunr.TokenSet.Builder(); + } else if (inserted < terms.length) { + tokens.insert(terms[inserted++]); + } else { + return false; + } + return true; + } + + // lunr's own createFieldVectors(), on a view of the builder that holds + // only `refs`' term frequencies. That method caches each term's idf for + // the one call, so a piece would count the documents of every common term + // again, which made the vectors 6 to 13 times as slow; lunr.idf answers from + // `idfs` meanwhile, by the term's number, which each posting holds. + function addFieldVectors(refs) { + var view = Object.create(builder); + view.fieldTermFrequencies = {}; + for (var i = 0; i < refs.length; i++) { + view.fieldTermFrequencies[refs[i]] = builder.fieldTermFrequencies[refs[i]]; + } + var idf = lunr.idf; + lunr.idf = function(posting, documentCount) { + if (idfs[posting._index] === undefined) idfs[posting._index] = idf(posting, documentCount); + return idfs[posting._index]; + }; + try { + view.createFieldVectors(); + } finally { + lunr.idf = idf; + } + for (var ref in view.fieldVectors) vectors[ref] = view.fieldVectors[ref]; + } + + // A message rather than setTimeout(fn, 0), which a browser delays by at + // least 4 ms once timeouts are nested five deep. + var channel = new MessageChannel(); + channel.port1.onmessage = slice; + channel.port2.postMessage(null); + + function slice() { + var index; + try { + var until = Date.now() + INDEX_SLICE_MS; + var more; + do { + more = piece(); + } while (more && Date.now() < until); + if (more) { + channel.port2.postMessage(null); + return; + } + tokens.finish(); + index = new lunr.Index({ + invertedIndex: builder.invertedIndex, + fieldVectors: vectors, + tokenSet: tokens.root, + fields: Object.keys(builder._fields), + pipeline: builder.searchPipeline + }); + } catch (e) { + channel.port1.close(); + onError(e); + return; + } + channel.port1.close(); + onSuccess(index); + } +} + // Patched: a query of two or more words that reads the same as a result's // whole title, or its page title and title together, names that result: // `Return Syntax`, `DTPicker Properties`. lunr alone ranks a one-word entry @@ -511,18 +625,24 @@ function searchLoaded(loadIndex) { // Patched: starts the deferred fetch + build (step 5C). Only one load // ever runs at a time -- a keystroke that lands while `indexLoading` is - // true just returns below in update(), and finishLoad() re-reads the - // search box once the build finishes, so it searches whatever is in it - // by then, not whatever triggered the load. + // true just returns here, and finishLoad() re-reads the search box once + // the build finishes, so it searches whatever is in it by then, not + // whatever triggered the load. The build hands the main thread back + // between slices (buildIndexInSlices()), so the reader can keep typing. function loadIndexNow() { - if (indexLoading) return; + if (indexLoading) { + // Clearing the box mid-load empties the panel; typing again says so + // again. + if (!searchResults.firstChild) showStatusMessage('Loading search index\u2026'); + return; + } indexLoading = true; showStatusMessage('Loading search index\u2026'); - // Yield so the loading message above actually paints before the - // synchronous (and comparatively expensive) index build runs on the - // main thread: a frame, then a task. requestAnimationFrame never fires - // while the tab is hidden, so a 100 ms timer races it -- whichever - // comes first starts the load, and the other does nothing. + // Yield so the loading message above actually paints before the index + // fetch and build begin on the main thread: a frame, then a task. + // requestAnimationFrame never fires while the tab is hidden, so a + // 100 ms timer races it -- whichever comes first starts the load, and + // the other does nothing. var started = false; function start() { if (started) return; diff --git a/test/search.test.mjs b/test/search.test.mjs index 5b6e19e5..3b681df2 100644 --- a/test/search.test.mjs +++ b/test/search.test.mjs @@ -1286,3 +1286,74 @@ describe("kind-word guard: online client, eval replica", () => { assert.deepEqual(urls("Mid function"), ["/Strings/Mid#mid"]); }); }); + +// Sliced build: both clients build the index through buildIndexInSlices(), +// which does lunr 2.3.9's own build a piece at a time, reaching into +// lunr.Builder's internals and swapping lunr.idf while it makes the vectors. +// The index must come out the same as lunr(config)'s, sliced as finely as it +// can be (one piece a slice, three fields' vectors a piece) and as the site +// slices it, and lunr.idf must be put back, even when the build throws. +describe("sliced-build guard: online client against lunr()", () => { + const read = (rel) => fs.readFileSync(path.join(REPO_ROOT, rel), "utf8"); + const lunr = loadLunr(path.join(REPO_ROOT, "builder/vendor/just-the-docs/assets/js/vendor/lunr.min.js")); + const src = read("builder/vendor/just-the-docs/assets/js/just-the-docs.js"); + const fnSrc = src.match(/function buildIndexInSlices\([^)]*\) \{[\s\S]*?\r?\n\}/); + const constant = (name) => { + const m = src.match(new RegExp(`var ${name} = (\\d+);`)); + assert.ok(m, `just-the-docs.js has no ${name}`); + return Number(m[1]); + }; + const sliced = (sliceMs, fieldsPerPiece) => { + assert.ok(fnSrc, "just-the-docs.js has no buildIndexInSlices()"); + return new Function("lunr", `var INDEX_SLICE_MS = ${sliceMs};\nvar FIELDS_PER_PIECE = ${fieldsPerPiece};\n${fnSrc[0]}\nreturn buildIndexInSlices;`)(lunr); + }; + const build = (buildIndexInSlices, config, docs, entry) => + new Promise((resolve, reject) => buildIndexInSlices(config, docs, entry, resolve, reject)); + + const docs = [ + { title: "Form", body: "A form is a window. Forms hold controls." }, + { title: "Form.Show", body: "Shows the form. A modal form waits." }, + { title: "Debug.Print", body: "Prints to the debug window." }, + { title: "Do...Loop", body: "Repeats a block while a condition holds." }, + { title: "Print", body: "Prints a line. See also Debug.Print and Form.Print." }, + { title: "Window", body: "" }, + ]; + const fields = function () { + this.ref("id"); + this.field("title", { boost: 10 }); + this.field("body"); + this.metadataWhitelist = ["position"]; + }; + const entry = (i) => ({ id: i, title: docs[i].title, body: docs[i].body }); + const whole = lunr(function () { + fields.call(this); + for (const i of Object.keys(docs)) this.add(entry(i)); + }); + const same = (index, label) => { + assert.equal(JSON.stringify(index.toJSON()), JSON.stringify(whole.toJSON()), `${label}: the index differs from lunr()'s`); + assert.deepEqual(index.tokenSet.toArray(), whole.tokenSet.toArray(), `${label}: the token set differs from lunr()'s`); + for (const q of ["form", "f*", "window print", "debug"]) { + assert.deepEqual(index.search(q), whole.search(q), `${label}: ${JSON.stringify(q)} finds other results`); + } + }; + + test("the index is lunr()'s, sliced finely and as the site slices it", async () => { + const idf = lunr.idf; + same(await build(sliced(0, 3), fields, docs, entry), "one piece a slice"); + same(await build(sliced(constant("INDEX_SLICE_MS"), constant("FIELDS_PER_PIECE")), fields, docs, entry), "the site's slices"); + assert.equal(lunr.idf, idf, "lunr.idf was not put back"); + }); + + test("a build that throws reports it and puts lunr.idf back", async () => { + const idf = lunr.idf; + // `boost` is read only while the vectors are made, so this throws with + // lunr.idf swapped. + const throwing = function () { + fields.call(this); + Object.defineProperty(this._fields.title, "boost", { get: () => { throw new Error("boost"); } }); + }; + await assert.rejects(build(sliced(0, 3), throwing, docs, entry), /boost/); + assert.equal(lunr.idf, idf, "lunr.idf was not put back after a throw"); + await assert.rejects(build(sliced(0, 3), fields, docs, (i) => { if (i === "2") throw new Error("entry"); return entry(i); }), /entry/); + }); +}); From 191c4f0090670c054bf0a4c4214907f94468f1f8 Mon Sep 17 00:00:00 2001 From: Kuba Sunderland-Ober Date: Sun, 27 Sep 2026 22:18:19 +0200 Subject: [PATCH 3/4] scripts: check_cli's transcript -x case passes on Linux --- builder/PLAN-TOOLING-REVIEW.md | 19 +++++++++++++++++++ scripts/check_cli.mjs | 6 ++++-- 2 files changed, 23 insertions(+), 2 deletions(-) diff --git a/builder/PLAN-TOOLING-REVIEW.md b/builder/PLAN-TOOLING-REVIEW.md index 5e1c5af9..08bc2778 100644 --- a/builder/PLAN-TOOLING-REVIEW.md +++ b/builder/PLAN-TOOLING-REVIEW.md @@ -1733,6 +1733,22 @@ and `build_corpus --dest x --` builds, where both refused the `--`. A lone `-` i `render-book`'s input. A short group is split, so `nav_hops -hx` prints the usage. And `render-book -ofile` and `-t5` take the attached value. +### C51a — `scripts: check_cli's transcript -x case passes on Linux` + +**Found by CI after C51.** The fork's deploy runs of C51 (36345344000) and of the search +commit after it (36347252812) failed at `check_cli`, 1 of 218, on `transcript -x`. The case +required a folder before the file name in Node's `ENOENT` line, which Node prints on Windows, +where it resolves the path, and not on Linux, where it prints `open '-x'` as given. + +**Change.** The folder is optional in the case's pattern, and the C51 block's comment says +why. + +**Landed.** As the entry says. The new pattern matches CI's line and a resolved Windows or +POSIX path, and refuses `--x` and `a-x`; `check_cli` makes 218 checks, all passing, and lint +is clean. The other 217 passed on Linux in both runs, so this is the whole of what CI found. +The steps after `check_cli` in the composite action did not run in either, so +`check_dot_fit`, `check_axe_patch_equiv` and the accessibility steps wait for the next push. + ### C52 — `builder: tbdocs parses through lib/cli.mjs` **A1-8 (R3), last, as decision (e) says.** `tbdocs.mjs`'s parser (`:91-194`) is neither @@ -2659,6 +2675,9 @@ Defects the review did not have, found by building something this plan asks for. read `Default/`, so that pages of one title in two packages hid all but one. No extract run has used it: the one on disk predates the move. Fixed in `wisdom: group reference pages by package, below Default/ and Built-In/`. +- **`check_cli`'s `transcript -x` case failed on Linux**, found by CI after C51: it required + a folder in a file name that Node prints with one only on Windows. Fixed in `scripts: + check_cli's transcript -x case passes on Linux`. ## Open questions diff --git a/scripts/check_cli.mjs b/scripts/check_cli.mjs index 350ee0e7..710c24c7 100644 --- a/scripts/check_cli.mjs +++ b/scripts/check_cli.mjs @@ -396,7 +396,9 @@ const CASES = [ // so a missing value shows only where the value is used. render-book // refuses --help, an unknown flag and a second input with "unknown arg". // build_corpus threw on an unknown argument, so only its message is pinned, - // as are the other crashes here; run_case and search_quality refuse one in + // as are the other crashes here. Node names a file it cannot open with its + // folder on Windows and as given on Linux, so a crash's file name may + // follow a folder or stand alone. run_case and search_quality refuse one in // their own words. nav_hops and site_search take an unknown // flag as a pattern or a search term. transcript's file is its first // argument that does not start with --, -h included, and its exit code @@ -465,7 +467,7 @@ const CASES = [ { tool: "eval/transcript.mjs", args: ["--help=1"], exit: 1, stdout: /^Usage: node eval\/transcript\.mjs <case\.jsonl> / }, { tool: "eval/transcript.mjs", args: ["nope.jsonl"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*nope\.jsonl'\r?\n/ }, { tool: "eval/transcript.mjs", args: ["--bogus", "nope.jsonl"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*nope\.jsonl'\r?\n/ }, - { tool: "eval/transcript.mjs", args: ["-x"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*[\\/]-x'\r?\n/ }, + { tool: "eval/transcript.mjs", args: ["-x"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '(?:[^']*[\\/])?-x'\r?\n/ }, { tool: "eval/transcript.mjs", args: ["a.jsonl", "b.jsonl"], exit: 1, stderr: /Error: ENOENT: no such file or directory, open '[^']*a\.jsonl'\r?\n/ }, { tool: "wisdom/wisdom.mjs", args: [], exit: 0, stderr: /^Usage: node wisdom\/wisdom\.mjs <command> \[options\]\n/ }, { tool: "wisdom/wisdom.mjs", args: ["--help"], exit: 1, stderr: /^Usage: node wisdom\/wisdom\.mjs <command> \[options\]\n/ }, From 955bbfb6c70b1e8efaf46e66d30f797bbc8b97ab Mon Sep 17 00:00:00 2001 From: Kuba Sunderland-Ober <kuba@mareimbrium.org> Date: Sun, 27 Sep 2026 22:25:52 +0200 Subject: [PATCH 4/4] scripts: the gate roster reads node --test lines; CI runs the search tests --- .github/actions/run-gates/action.yml | 8 ++++ WIP.md | 5 ++- builder/PLAN-TOOLING-REVIEW.md | 41 ++++++++++++++++++++ docs/Documentation/Building.md | 1 + docs/Documentation/Tools.md | 29 +++++++++----- scripts/check_ci_workflows.mjs | 20 ++++++++-- scripts/check_gate_lists.mjs | 58 +++++++++++++++++++--------- scripts/lib/gate-roster.mjs | 21 ++++++---- 8 files changed, 141 insertions(+), 42 deletions(-) diff --git a/.github/actions/run-gates/action.yml b/.github/actions/run-gates/action.yml index e3155e31..35ce1c93 100644 --- a/.github/actions/run-gates/action.yml +++ b/.github/actions/run-gates/action.yml @@ -64,6 +64,14 @@ runs: - name: Lint the tooling (check_lint.mjs) shell: bash run: node scripts/check_lint.mjs + # Unit tests for the site search, run by Node's own test runner: what the + # entries builder/search.mjs writes hold, which the build's check does not + # read, and guards that the online client, the offline client and + # eval/site_search.mjs's replica still agree. No browser, no built tree, + # well under a second. + - name: Unit-test the site search (test/search.test.mjs) + shell: bash + run: node --test test/search.test.mjs # A regex that backtracks exponentially does not fail a build, it stops # one: the corpus passes until a page happens to contain the trigger, and # then a render worker sits inside String.replace forever. VOID_TAGS_RE diff --git a/WIP.md b/WIP.md index 611e1a11..39658e26 100644 --- a/WIP.md +++ b/WIP.md @@ -451,7 +451,7 @@ Why the report separates the wedged task from the merely blocked ones, and why - `build.bat` — runs `node builder\tbdocs.mjs --src docs --check-audit-index` (which implies `--check`) and produces three trees in one pass: the online copy at `_site/`, a `file://`-browsable copy at `_site-offline/`, and the sparse pagedjs source at `_site-pdf/`. The offline pass adds ~700 ms and the PDF pass adds ~150 ms on top of the ~2 s online build. Toggle `also_build_offline` / `also_build_pdf` in `_config.yml` (or pass `--no-offline` / `--no-pdf`) to skip a sibling output. `--check` adds ~1.7 s and runs the link + integrity check over the HTML while it is still in worker memory; `build.bat --no-check` gets a plain build. - `serve.bat` — runs `tbdocs --serve`: initial build, then a long-lived process with watcher, debounced rebuilds, and SSE-driven browser auto-reload. Writes to `docs/_serve/` (disjoint from `build.bat`'s `_site*/`) and skips the offline + PDF passes — so a one-off `build.bat` for the PDF or offline mirror doesn't disturb the live preview. Ctrl+C to stop. - `check.bat` — the gates that read the built site: a freshness check that refuses a stale tree (`scripts/check_tree_fresh.mjs`), the DOT diagram fit check (`scripts/check_dot_fit.mjs`), the a11y sample-coverage check (`scripts/pick_a11y_sample.mjs --check`), then the accessibility check (`scripts/check_a11y.mjs`). The link + integrity check moved into `build.bat`. ~37 s. -- `test.bat` — the tests the *toolchain* has to pass: the publish-allowlist self-test (`scripts/check_publish_policy.mjs`), the gate-list check (`scripts/check_gate_lists.mjs`), the CI-workflow roster check (`scripts/check_ci_workflows.mjs`), the lint gate (`scripts/check_lint.mjs`), the regex-safety gate (`scripts/check_regex_safety.mjs`), the code-region gate (`scripts/check_code_regions.mjs`), the page-count drift-guard probes (`scripts/check_page_baseline.mjs`), the book-coverage probes (`scripts/check_book_coverage.mjs`), the symbol-index probes (`scripts/check_symbol_index.mjs`), the command-line probes and cases (`scripts/check_cli.mjs`), and the axe source-patch verification (`scripts/check_axe_patch_equiv.mjs`). ~9 s. See [What belongs in test.bat rather than check.bat](WIP.Build.md#what-belongs-in-testbat-rather-than-checkbat). +- `test.bat` — the tests the *toolchain* has to pass: the publish-allowlist self-test (`scripts/check_publish_policy.mjs`), the gate-list check (`scripts/check_gate_lists.mjs`), the CI-workflow roster check (`scripts/check_ci_workflows.mjs`), the lint gate (`scripts/check_lint.mjs`), the site-search unit tests (`node --test test/search.test.mjs`), the regex-safety gate (`scripts/check_regex_safety.mjs`), the code-region gate (`scripts/check_code_regions.mjs`), the page-count drift-guard probes (`scripts/check_page_baseline.mjs`), the book-coverage probes (`scripts/check_book_coverage.mjs`), the symbol-index probes (`scripts/check_symbol_index.mjs`), the command-line probes and cases (`scripts/check_cli.mjs`), and the axe source-patch verification (`scripts/check_axe_patch_equiv.mjs`). ~9 s. See [What belongs in test.bat rather than check.bat](WIP.Build.md#what-belongs-in-testbat-rather-than-checkbat). - `book.bat` — renders the PDF from `docs\_site-pdf\book.html` via `node book\render-book.mjs` into `docs\_pdf\twinBASIC Book.pdf`. Run `build.bat` first to populate `_site-pdf/`; `book.bat` refuses a tree older than its sources rather than rendering the previous book (see [The book refuses a stale source tree](WIP.Build.md#the-book-refuses-a-stale-source-tree)). - `examples.bat` — compiles the documentation's own twinBASIC code samples, every `tb` fence marked `check_build`, and reports the ones the compiler refuses against the line in the page they came from. Needs a twinBASIC install and Windows, so it is outside every gate and outside CI; ~120 s over the 1,129 samples marked as of 2026-09-25. Two modes need no compiler at all: `--census` classifies every fence and says how many classifiable ones are still unmarked, and `--report <survey.json>` groups a saved `--propose --json` survey by diagnostic, section and unresolved name. `--propose` itself does compile. See [Compiling the reference's own code samples](#compiling-the-references-own-code-samples) and [WIP.ExamplesBuild.md](WIP.ExamplesBuild.md). @@ -469,7 +469,7 @@ build.bat && check.bat On the dev box that is ~4 s of build against ~37 s of check, of which the axe scan is ~20 s. [builder/PLAN-checks.md](builder/PLAN-checks.md) records how the link checker got folded into the build's task graph, what it cost and what it saved; the axe follow-ons are designed there but not implemented. -**If the change touched `builder/`, `scripts/`, `lib/`, `book/`, `eval/`, `wisdom/`, `test/`, the site's scripts in `docs/assets/js/`, a wrapper or a workflow, run `test.bat` as well** --- another ~9 s. Eight of its eleven gates cannot be affected by an edit under `docs/` at all. **Three can.** `check_lint.mjs` lints the site's two scripts in `docs/assets/js/` along with the tooling. `check_gate_lists.mjs` is the easy one to predict: it reads `README.md` and every page under `docs/Documentation/`, so an edit to any developer page that states a gate count can fail it. **`check_code_regions.mjs` is the one worth understanding**, and which half of it a content edit reaches is worth keeping straight. Its corpus sweep reads `DOCS_DIR`, which is `<repo>/docs`, and tokenises all 906 markdown files, so a page that provokes a rewrite into *altering* a code region fails it --- that half is content-dependent. Its fixed probes are not: they run against their own sources whatever the tree holds, and they cover the **mirror** fault, where a rewrite silently stops firing. The sweep structurally cannot see that one, because text the rewrite skipped is stashed and restored unchanged and every region still matches. So run `test.bat` after adding an unusual code construct --- a fence whose contents include a fence marker, a 4-space indented block, an admonition wrapping a fence --- and read the built page as well, because for the mirror fault the gate is asserting that the stasher still works rather than checking your page: +**If the change touched `builder/`, `scripts/`, `lib/`, `book/`, `eval/`, `wisdom/`, `test/`, the site's scripts in `docs/assets/js/`, a wrapper or a workflow, run `test.bat` as well** --- another ~9 s. Nine of its twelve gates cannot be affected by an edit under `docs/` at all. **Three can.** `check_lint.mjs` lints the site's two scripts in `docs/assets/js/` along with the tooling. `check_gate_lists.mjs` is the easy one to predict: it reads `README.md` and every page under `docs/Documentation/`, so an edit to any developer page that states a gate count can fail it. **`check_code_regions.mjs` is the one worth understanding**, and which half of it a content edit reaches is worth keeping straight. Its corpus sweep reads `DOCS_DIR`, which is `<repo>/docs`, and tokenises all 906 markdown files, so a page that provokes a rewrite into *altering* a code region fails it --- that half is content-dependent. Its fixed probes are not: they run against their own sources whatever the tree holds, and they cover the **mirror** fault, where a rewrite silently stops firing. The sweep structurally cannot see that one, because text the rewrite skipped is stashed and restored unchanged and every region still matches. So run `test.bat` after adding an unusual code construct --- a fence whose contents include a fence marker, a 4-space indented block, an admonition wrapping a fence --- and read the built page as well, because for the mirror fault the gate is asserting that the stasher still works rather than checking your page: ```sh build.bat && check.bat && test.bat @@ -506,6 +506,7 @@ wrapper: | `test.bat` | `check_cli` | `lib/cli.mjs` parses as a strict `parseArgs` does, and keeps a lenient tool's leniency where asked; each tool's recorded command-line errors still exit and print as recorded, run with an IDE and a browser that do not exist | | `test.bat` | `check_ci_workflows` | both CI workflows run every wrapper gate, with the same arguments and order, and build with `build.bat`'s flags | | `test.bat` | `check_lint` | Biome finds nothing in the tooling, warnings included, and checked at least one script | +| `test.bat` | `test/search.test.mjs` | the search entries `builder/search.mjs` writes hold what they should, and the copies of the search client still agree. Run by `node --test`; the gate roster reads such a line as a gate, named by its path | | `test.bat` | `check_publish_policy`, `check_gate_lists`, `check_page_baseline`, `check_book_coverage`, `check_axe_patch_equiv` | the gates on the gates | **A gate belongs in `test.bat` rather than `check.bat` if it would still mean diff --git a/builder/PLAN-TOOLING-REVIEW.md b/builder/PLAN-TOOLING-REVIEW.md index 08bc2778..df40406b 100644 --- a/builder/PLAN-TOOLING-REVIEW.md +++ b/builder/PLAN-TOOLING-REVIEW.md @@ -1749,6 +1749,43 @@ is clean. The other 217 passed on Linux in both runs, so this is the whole of wh The steps after `check_cli` in the composite action did not run in either, so `check_dot_fit`, `check_axe_patch_equiv` and the accessibility steps wait for the next push. +### C51b — `scripts: the gate roster reads node --test lines; CI runs the search tests` + +**Found while landing C51a.** PR #210 put `node --test test/search.test.mjs` into `test.bat`, +and CI has never run it: `scripts/lib/gate-roster.mjs` read only `node scripts/<name>.mjs` +lines, so `check_ci_workflows` and `check_gate_lists` did not see the step, and neither the +composite action nor Tools.md's list had it. + +**Change.** The roster reads `node --test test/<name>.mjs` as a gate too, named by its path +from the repository root (`gateName`), and `check_gate_lists` reads such a name in Tools.md's +list (a `test/` link) and in a POSIX block. The composite action runs the tests after +`check_lint`, as `test.bat` does; Tools.md lists them as `test.bat`'s fifth step, with a +section of their own, and Building.md's POSIX block and WIP.md's bullet and gate table have +them. + +**Landed.** As the entry says, at the owner's choice of registering the step fully over +adding it to CI alone. Before the action and the pages had it, both gates failed on the real +tree: `check_ci_workflows` with a `missing` finding for `test/search.test.mjs` in each +workflow, and `check_gate_lists` with six disagreements (the list, the stated count, both +POSIX blocks, and Tools.md's "Eleven steps" and "of the eleven"). After, `check_ci_workflows` +passes with 19 probes and 15 gates, and `check_gate_lists` with 21 probes and `test.bat +(12)`. The new probes: in `check_ci_workflows`, a test file in `test.bat` and not in CI, and +one in both; in `check_gate_lists`, a test file the docs do not list, with a count that agrees +with the list unless the file is read, and a test file listed by its path, CRLF and a +backslash in the wrapper. Each fault through the kit's `c43-fault.mjs` fails: a roster that +reads no test line fails a probe in both gates; a doc list that reads no `test/` link fails +the new negative; POSIX blocks that read no test line split Building.md's and Tools.md's +blocks in two. The last is caught by the real tree only, as every POSIX-block defect is. + +Found in passing, and fixed here at the owner's choice: Tools.md said eight of `test.bat`'s +eleven gates could not be affected by an edit under `docs/` and named two exceptions; +`check_lint`, which lints `docs/assets/js/`, was the third. It now says nine of twelve, and +names all three. The search tests read `builder/`, `builder/vendor/` and `eval/` only. + +**CI must show**, on the owner's next push: `check_ci_workflows: 19 probes, all pass` and +`both workflows run the wrappers' 15 gates`, and the new step passing on Linux with `tests +67` and `pass 67`. + ### C52 — `builder: tbdocs parses through lib/cli.mjs` **A1-8 (R3), last, as decision (e) says.** `tbdocs.mjs`'s parser (`:91-194`) is neither @@ -2678,6 +2715,10 @@ Defects the review did not have, found by building something this plan asks for. - **`check_cli`'s `transcript -x` case failed on Linux**, found by CI after C51: it required a folder in a file name that Node prints with one only on Windows. Fixed in `scripts: check_cli's transcript -x case passes on Linux`. +- **CI never ran `test/search.test.mjs`**, found while landing C51a: the gate roster read only + `node scripts/` lines, so the two roster gates could not see a `node --test` step in + `test.bat`. Fixed in `scripts: the gate roster reads node --test lines; CI runs the search + tests`. ## Open questions diff --git a/docs/Documentation/Building.md b/docs/Documentation/Building.md index dc66134b..fade06ea 100644 --- a/docs/Documentation/Building.md +++ b/docs/Documentation/Building.md @@ -72,6 +72,7 @@ Each `.bat` opens with `@pushd "%~dp0"`, which is what lets it be invoked from a && node scripts/check_gate_lists.mjs \ && node scripts/check_ci_workflows.mjs \ && node scripts/check_lint.mjs \ + && node --test test/search.test.mjs \ && node scripts/check_regex_safety.mjs \ && node scripts/check_code_regions.mjs \ && node scripts/check_page_baseline.mjs \ diff --git a/docs/Documentation/Tools.md b/docs/Documentation/Tools.md index 319b0c9c..0865a170 100644 --- a/docs/Documentation/Tools.md +++ b/docs/Documentation/Tools.md @@ -65,19 +65,20 @@ One of the four does not mean the same thing locally as it does in CI, on any pl test.bat -The tests the toolchain has to pass. Eleven steps, each stopping the run if it fails: +The tests the toolchain has to pass. Twelve steps, each stopping the run if it fails: 1. [`scripts/check_publish_policy.mjs`](#check-publish-policy) --- verifies the publish allowlist still refuses the types it is meant to. Needs neither a browser nor a built tree, so it goes first. 2. [`scripts/check_gate_lists.mjs`](#check-gate-lists) --- verifies the two gate lists on this page still match the wrappers that run them. 3. [`scripts/check_ci_workflows.mjs`](#check-ci-workflows) --- verifies both CI workflows run the gates the wrappers run, and build as `build.bat` does. 4. [`scripts/check_lint.mjs`](#check-lint) --- runs Biome over the tooling and fails on any finding, warnings included. -5. [`scripts/check_regex_safety.mjs`](#check-regex-safety) --- refuses a regex that can backtrack exponentially, written as a literal or built from constants. -6. [`scripts/check_code_regions.mjs`](#check-code-regions) --- verifies no pre-render rewrite alters the contents of a code fence or code span, and that `lib/markdown.mjs` and `lib/frontmatter.mjs` pass their probes. -7. [`scripts/check_page_baseline.mjs`](#check-page-baseline) --- verifies the page-count drift guard still refuses a fall. -8. [`scripts/check_book_coverage.mjs`](#check-book-coverage) --- verifies the build still warns about a page `docs/_book.yml` does not mention. -9. [`scripts/check_symbol_index.mjs`](#check-symbol-index) --- verifies the symbol index still places each kind of symbol, and its drift guard still refuses a lost URL. -10. [`scripts/check_cli.mjs`](#check-cli) --- verifies `lib/cli.mjs`, the command-line parser, and each tool's recorded command-line errors. -11. [`scripts/check_axe_patch_equiv.mjs`](#check-axe-patch-equiv) --- verifies the vendored axe source patch still produces identical colour values. +5. [`test/search.test.mjs`](#search-test) --- unit tests for the site search: what the search entries hold, and that the copies of the search client agree. +6. [`scripts/check_regex_safety.mjs`](#check-regex-safety) --- refuses a regex that can backtrack exponentially, written as a literal or built from constants. +7. [`scripts/check_code_regions.mjs`](#check-code-regions) --- verifies no pre-render rewrite alters the contents of a code fence or code span, and that `lib/markdown.mjs` and `lib/frontmatter.mjs` pass their probes. +8. [`scripts/check_page_baseline.mjs`](#check-page-baseline) --- verifies the page-count drift guard still refuses a fall. +9. [`scripts/check_book_coverage.mjs`](#check-book-coverage) --- verifies the build still warns about a page `docs/_book.yml` does not mention. +10. [`scripts/check_symbol_index.mjs`](#check-symbol-index) --- verifies the symbol index still places each kind of symbol, and its drift guard still refuses a lost URL. +11. [`scripts/check_cli.mjs`](#check-cli) --- verifies `lib/cli.mjs`, the command-line parser, and each tool's recorded command-line errors. +12. [`scripts/check_axe_patch_equiv.mjs`](#check-axe-patch-equiv) --- verifies the vendored axe source patch still produces identical colour values. POSIX: @@ -85,6 +86,7 @@ POSIX: && node scripts/check_gate_lists.mjs \ && node scripts/check_ci_workflows.mjs \ && node scripts/check_lint.mjs \ + && node --test test/search.test.mjs \ && node scripts/check_regex_safety.mjs \ && node scripts/check_code_regions.mjs \ && node scripts/check_page_baseline.mjs \ @@ -93,9 +95,9 @@ POSIX: && node scripts/check_cli.mjs \ && node scripts/check_axe_patch_equiv.mjs -**Eight of the eleven cannot be affected by an edit confined to `docs/`**, which is why they are separate from `check.bat`. Run this one when the change touches `builder/`, `scripts/`, `lib/`, `book/`, `eval/`, `wisdom/` or `test/`, the site's scripts in `docs/assets/js/`, a wrapper, or a workflow. Both CI workflows run all eleven unconditionally, as they always did, so skipping it locally cannot let a tooling regression reach `staging`. +**Nine of the twelve cannot be affected by an edit confined to `docs/`**, which is why they are separate from `check.bat`. Run this one when the change touches `builder/`, `scripts/`, `lib/`, `book/`, `eval/`, `wisdom/` or `test/`, the site's scripts in `docs/assets/js/`, a wrapper, or a workflow. Both CI workflows run all twelve unconditionally, so skipping it locally cannot let a tooling regression reach `staging`. -The two exceptions are [`check_code_regions.mjs`](#check-code-regions) and [`check_gate_lists.mjs`](#check-gate-lists), which reads this page. The first is worth knowing in detail. Its corpus sweep tokenises every markdown file under `docs/`, so a page that provokes a rewrite into altering a code region fails it. Its fixed probes are a different matter: they run against their own sources whatever the tree holds, and they cover the *mirror* fault, where a rewrite silently stops firing. The sweep cannot see that one --- text the rewrite skipped is stashed and restored unchanged, so every region still matches. Add a page with an unusual code construct and run `test.bat`, but read the built page too. +The three exceptions are [`check_code_regions.mjs`](#check-code-regions), [`check_gate_lists.mjs`](#check-gate-lists), which reads this page, and [`check_lint.mjs`](#check-lint), which lints the site's scripts in `docs/assets/js/`. The first is worth knowing in detail. Its corpus sweep tokenises every markdown file under `docs/`, so a page that provokes a rewrite into altering a code region fails it. Its fixed probes are a different matter: they run against their own sources whatever the tree holds, and they cover the *mirror* fault, where a rewrite silently stops firing. The sweep cannot see that one --- text the rewrite skipped is stashed and restored unchanged, so every region still matches. Add a page with an unusual code construct and run `test.bat`, but read the built page too. The split is by what a gate **interrogates**, not by what it happens to open. `check_axe_patch_equiv.mjs` loads a built page, so it does want `build.bat` to have run and it does want Chromium --- but only because its probe needs some document to run inside; what it tests is the axe patch. The test for where a new gate belongs is whether it would still mean something against an empty `docs/`. @@ -481,6 +483,13 @@ Lint before every commit that touches one of those folders, or let the pre-commi A clone without the hook is still checked, because `test.bat` and both CI workflows run this gate over the whole scope. `npx biome lint --write` applies the fixes Biome marks safe. The fixes it offers for an unused import or variable are marked unsafe and need `--unsafe` as well, so read the diff after applying them. +### search.test.mjs +{: #search-test } + + node --test test/search.test.mjs + +Unit tests for the site search, run by Node's own test runner rather than as a script under `scripts/`. The first group builds search entries from small synthetic pages through `builder/search.mjs` and checks what each entry holds: the split at headings, the folding of generic sections such as See Also into the member they belong to, index marks, the join with the symbol index, and output that is the same byte for byte from one build to the next. The build's own check sees only which URLs the index covers. The rest are guards that the copies of the search client's query code still agree --- the online client under `builder/vendor/just-the-docs/`, the offline client in `builder/offline.mjs` and the replica in `eval/site_search.mjs`, all three or two of them --- and that the online client's index, built in slices, is the index lunr builds in one call. No browser, no built tree, well under a second. Exits 1 when a test fails. + ### check_page_baseline.mjs {: #check-page-baseline } diff --git a/scripts/check_ci_workflows.mjs b/scripts/check_ci_workflows.mjs index 78f3bdd9..c342e8b2 100644 --- a/scripts/check_ci_workflows.mjs +++ b/scripts/check_ci_workflows.mjs @@ -195,11 +195,16 @@ const P_ALLOWED = { const BUILD_ONE = "node builder/tbdocs.mjs --src docs --no-fetch-assets --check-audit-index"; const BUILD_TWO = "node builder/tbdocs.mjs --src docs --url '${{ steps.pages.outputs.origin }}' --no-fetch-assets --check-audit-index"; const GOOD = ["a.mjs", "b.mjs", "c.mjs --check"]; +const TEST_FILE = "--test test/d.test.mjs"; + +// A gate as the probes spell it: a script's file name and arguments, or +// `--test <path>` for a test file. +const runOf = (g) => (g.startsWith("--test ") ? `node ${g}` : `node scripts/${g}`); function wf(gates, build) { const steps = [{ name: "Checkout", uses: "actions/checkout@v5" }]; if (build) steps.push({ name: "Build", run: build }); - for (const g of gates) steps.push({ name: g, run: `node scripts/${g}` }); + for (const g of gates) steps.push({ name: g, run: runOf(g) }); return { jobs: { [JOB]: { steps } } }; } @@ -211,12 +216,12 @@ function pair(one, two) { function wfAction(build, extra = []) { const steps = [{ name: "Checkout", uses: "actions/checkout@v5" }, { name: "Build", run: build }]; steps.push({ name: "Run the gates", uses: "./gates" }); - for (const g of extra) steps.push({ name: g, run: `node scripts/${g}` }); + for (const g of extra) steps.push({ name: g, run: runOf(g) }); return { jobs: { [JOB]: { steps } } }; } function actionOf(gates) { - return { runs: { using: "composite", steps: gates.map((g) => ({ name: g, shell: "bash", run: `node scripts/${g}` })) } }; + return { runs: { using: "composite", steps: gates.map((g) => ({ name: g, shell: "bash", run: runOf(g) })) } }; } const PROBES = [ @@ -242,6 +247,15 @@ const PROBES = [ ["a gate added to test.bat and not to CI", { testBat: `${P_TEST}\r\nnode scripts/d.mjs\r\n` }, ["missing"]], + // test/search.test.mjs ran in test.bat and not in CI for as long as the + // roster read only scripts/. + ["a test file added to test.bat and not to CI", + { testBat: `${P_TEST}\r\nnode ${TEST_FILE}\r\n` }, + ["missing"]], + ["a test file in test.bat and in both workflows", + { testBat: `${P_TEST}\r\nnode ${TEST_FILE}\r\n`, + workflows: pair(wf([...GOOD, TEST_FILE, "ci.mjs"], BUILD_ONE), wf([...GOOD, TEST_FILE], BUILD_TWO)) }, + []], ["a build without --check-audit-index", { workflows: pair(wf([...GOOD, "ci.mjs"], BUILD_ONE), wf(GOOD, BUILD_TWO.replace(" --check-audit-index", ""))) }, ["build"]], diff --git a/scripts/check_gate_lists.mjs b/scripts/check_gate_lists.mjs index cce1a9d7..c3b0a988 100644 --- a/scripts/check_gate_lists.mjs +++ b/scripts/check_gate_lists.mjs @@ -84,7 +84,7 @@ import { createMarkdownIt } from "../builder/render.mjs"; import { parseCli } from "../lib/cli.mjs"; import { splitOnMarker } from "../lib/markdown.mjs"; import { REPO_ROOT } from "../lib/repo-paths.mjs"; -import { gatesFromBat } from "./lib/gate-roster.mjs"; +import { gateName, gatesFromBat } from "./lib/gate-roster.mjs"; const TOOLS_MD = "docs/Documentation/Tools.md"; @@ -116,18 +116,19 @@ function sectionBody(src, heading) { } /** - * The gate scripts a documented numbered list names, in order. + * The gates a documented numbered list names, in order. * * Deliberately narrow: an ordered-list item whose text opens with a link to - * `scripts/<name>.mjs`. Prose elsewhere in the section may mention a script - * without being a claim about the list -- Tools.md's check.bat entry names - * test.bat in its opening paragraph, and that is a cross-reference, not a step. + * `scripts/<name>.mjs`, or to `test/<name>.mjs` for a test file. Prose + * elsewhere in the section may mention a script without being a claim about + * the list -- Tools.md's check.bat entry names test.bat in its opening + * paragraph, and that is a cross-reference, not a step. */ function gatesFromDoc(body) { const out = []; for (const line of body.split(/\r?\n/)) { - const m = /^\s*\d+\.\s+\[`scripts[\\/]([A-Za-z0-9_]+\.mjs)/.exec(line); - if (m) out.push(m[1]); + const m = /^\s*\d+\.\s+\[`(?:scripts[\\/]([A-Za-z0-9_]+\.mjs)|test[\\/]([A-Za-z0-9_.]+\.mjs))/.exec(line); + if (m) out.push(gateName(m[1], m[2])); } return out; } @@ -157,11 +158,11 @@ function commandRuns(src, file) { cur = null; }; src.split(/\r?\n/).forEach((line, i) => { - const m = /^\s*(&&\s*)?node\s+scripts[\\/]([A-Za-z0-9_]+\.mjs)/.exec(line); + const m = /^\s*(&&\s*)?node\s+(?:scripts[\\/]([A-Za-z0-9_]+\.mjs)|--test\s+test[\\/]([A-Za-z0-9_.]+\.mjs))/.exec(line); if (!m) { flush(); return; } if (!cur) cur = { file, line: i + 1, gates: [], chained: true }; else if (!m[1]) cur.chained = false; - cur.gates.push(m[2]); + cur.gates.push(gateName(m[2], m[3])); }); flush(); return runs; @@ -428,16 +429,33 @@ const PROBES = [ bat: "node scripts/a.mjs\n", doc: "### x.bat\n\nIt runs:\n\n1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n", }, + { + // test/search.test.mjs ran in test.bat for as long as the roster read only + // scripts/, so neither this gate nor check_ci_workflows saw it, and CI + // never ran it. The count here agrees with the list unless the test file + // is read, so this fails when the roster stops reading one. + name: "a test file the docs do not list", + bat: "node scripts/a.mjs\nnode --test test/a.test.mjs\n", + doc: "### x.bat\n\nOne step:\n\n1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n", + }, ]; -// Must NOT fire: a section whose prose mentions another wrapper's gate outside -// the numbered list, which is what Tools.md's real entries do. -const NEGATIVE = { - name: "a cross-reference in prose is not a step", - bat: "node scripts/a.mjs\n", - doc: "### x.bat\n\nTests of the toolchain are [`scripts/zz.mjs`](#zz), not these. One step:\n\n" + - "1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n", -}; +// Must NOT fire. +const NEGATIVES = [ + { + // What Tools.md's real entries do: prose that mentions another wrapper's + // gate outside the numbered list. + name: "a cross-reference in prose is not a step", + bat: "node scripts/a.mjs\n", + doc: "### x.bat\n\nTests of the toolchain are [`scripts/zz.mjs`](#zz), not these. One step:\n\n" + + "1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n", + }, + { + name: "a test file listed by its path is a step", + bat: "node scripts/a.mjs\r\nnode --test test\\a.test.mjs\r\n", + doc: "### x.bat\n\nTwo steps:\n\n1. [`scripts/a.mjs`](#a) --- a.\n2. [`test/a.test.mjs`](#a-test) --- tests.\n\n### next\n", + }, +]; // Probes for the prose sweep. Every positive is a sentence that was on a // published page at 4f97bac, against the counts that were true at the time @@ -516,8 +534,10 @@ function selfTest() { const found = compareWrapper({ bat: "x.bat", heading: "### x.bat" }, p.bat, p.doc); results.push([found.length > 0, p.name]); } - const neg = compareWrapper({ bat: "x.bat", heading: "### x.bat" }, NEGATIVE.bat, NEGATIVE.doc); - results.push([neg.length === 0, NEGATIVE.name]); + for (const n of NEGATIVES) { + const found = compareWrapper({ bat: "x.bat", heading: "### x.bat" }, n.bat, n.doc); + results.push([found.length === 0, n.name]); + } for (const p of PROSE_PROBES) { results.push([proseFindings(p.md, "<probe>", REAL_COUNTS).length > 0, `prose: ${p.name}`]); diff --git a/scripts/lib/gate-roster.mjs b/scripts/lib/gate-roster.mjs index 6ce119be..ec274803 100644 --- a/scripts/lib/gate-roster.mjs +++ b/scripts/lib/gate-roster.mjs @@ -1,16 +1,21 @@ // The gates a wrapper or a CI workflow runs, read from its own text. // -// A wrapper runs one `node scripts/<name>.mjs` per line, chained with -// `@if errorlevel` rather than `&&`; a workflow runs one per step's `run:`. -// Both are read the same way: an invocation at the start of a line, so a -// commented line (`@rem`, `rem`, `#`) or a continuation never counts. +// A wrapper runs one gate per line, chained with `@if errorlevel` rather than +// `&&`; a workflow runs one per step's `run:`. Both are read the same way: an +// invocation at the start of a line, so a commented line (`@rem`, `rem`, `#`) +// or a continuation never counts. A gate is a script, `node scripts/<name>.mjs`, +// named by its file name, or a test file that Node's test runner runs, `node +// --test test/<name>.mjs`, named by its path from the repository root. // // check_gate_lists.mjs compares the wrappers with Tools.md's numbered lists, // and check_ci_workflows.mjs compares them with the two workflows. -const GATE_LINE = /^\s*@?node\s+scripts[\\/]([A-Za-z0-9_]+\.mjs)[ \t]*(.*?)\s*$/; +const GATE_LINE = /^\s*@?node\s+(?:scripts[\\/]([A-Za-z0-9_]+\.mjs)|--test\s+test[\\/]([A-Za-z0-9_.]+\.mjs))[ \t]*(.*?)\s*$/; const BUILD_LINE = /^\s*@?node\s+builder[\\/]tbdocs\.mjs[ \t]*(.*?)\s*$/; +/** A gate's name from the two captures every gate pattern has: a script's file name, or a test file's. */ +export const gateName = (script, testFile) => script ?? `test/${testFile}`; + function matchLines(text, re) { const out = []; for (const line of String(text).split(/\r?\n/)) { @@ -20,12 +25,12 @@ function matchLines(text, re) { return out; } -/** The gate scripts a text invokes, in order, as `{script, args}`. */ +/** The gates a text runs, in order, as `{script, args}`, `script` the gate's name. */ export function gateSteps(text) { - return matchLines(text, GATE_LINE).map((m) => ({ script: m[1], args: m[2] })); + return matchLines(text, GATE_LINE).map((m) => ({ script: gateName(m[1], m[2]), args: m[3] })); } -/** The gate scripts a batch file invokes, in order: the names alone. */ +/** The gates a batch file runs, in order: the names alone. */ export function gatesFromBat(src) { return gateSteps(src).map((s) => s.script); }