Repository navigation
Expand file tree
/
Copy pathcheck_gate_lists.mjs
More file actions
702 lines (652 loc) · 29.4 KB
/
Copy pathcheck_gate_lists.mjs
File metadata and controls
702 lines (652 loc) · 29.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
#!/usr/bin/env node
// Verify that check.bat and test.bat still match Tools.md's two gate lists,
// and that no developer page states a gate count that disagrees with them.
//
// node scripts/check_gate_lists.mjs
// node scripts/check_gate_lists.mjs --verbose
// node scripts/check_gate_lists.mjs --self-test
//
// ---------------------------------------------------------------- why
//
// A gate count or list restated by hand in prose drifts: the wrong number
// breaks no link, fails no gate, and reads no differently from a right one.
// That is the same argument `{{tbdocs:...}}` makes for page counts, and the
// remedy is the same in spirit: stop asserting by hand what can be derived
// from the artifact. A page that says test.bat reads no page of documentation
// would also route an author past a gate that does (check_code_regions.mjs
// reads every markdown file).
//
// -------------------------------------------------------------- the design
//
// **One page owns the lists and the others cite it.** Tools.md carries the two
// numbered lists; Building.md and Extending.md link to its entries instead of
// restating them.
//
// **That convention is not self-enforcing.** A gate scoped to one page is a
// guard against one file, not against a class of restatement -- so the prose
// sweep below reads README.md and every page under docs/Documentation/ and
// checks every gate count they state, wherever it is stated.
//
// **The batch file is the source of truth, not the documentation.** A gate
// that compared the two pages against each other would be satisfied by two
// pages that agree and are both wrong.
//
// **It checks order as well as membership.** Both wrappers document their
// steps as numbered lists with a stated reason for the order (cheapest first,
// so a five-second failure does not wait on a twenty-second scan), so a
// reordering that the prose no longer matches is a real defect.
//
// ------------------------------------------------- what the sweep does not see
//
// Stated plainly, because a gate whose limits are not written down gets read
// as covering more than it does.
//
// **An ordinal.** "check_code_regions.mjs is the fifth" sat two paragraphs
// from "Five gates" and is not reported; a rule for ordinals would have to
// decide whether "the fifth of six" is wrong, and it usually is not. In
// practice the section total above it fires and a reader fixing that one is
// looking straight at this one.
//
// **A count in a wrapper section that is a subset claim.** Only the *first*
// bare `N gates` in a wrapper's own section is read as its total, because
// later ones ("two of them read docs/") are legitimate. The cost is the other
// way round: a section that opens with a subset claim is reported. That is
// deliberate rather than tolerated -- the remedy is to delete the number, and
// the failure message says so, because a subset count restated in prose is
// the same thing that drifts.
//
// **Anything outside README.md and docs/Documentation/.** builder/*.md are
// design notes and frozen audit snapshots, and rewriting one to match a later
// change destroys the only thing it is for.
//
// Exit codes: 0 clean, 1 a list or a stated count disagrees or a probe failed, 2 the
// gate could not run (a refused command line, or a crash).
import { readFile, readdir } from "node:fs/promises";
import path from "node:path";
import { createMarkdownIt } from "../builder/render.mjs";
import { parseCli, printHelpAndExit, withUsageError } from "../lib/cli.mjs";
import { splitOnMarker } from "../lib/markdown.mjs";
import { REPO_ROOT } from "../lib/repo-paths.mjs";
import { gateName, gatesFromBat } from "./lib/gate-roster.mjs";
const TOOLS_MD = "docs/Documentation/Tools.md";
// The site's own parser, so that what is code is what the renderer will make
// of it. Built on first use, inside main, so a failure exits 2.
let siteMd;
const siteParser = () =>
(siteMd ??= createMarkdownIt({ highlighter: null, linkTables: null, baseurl: "", staticFiles: new Set() }));
// The wrappers this gate covers, and the heading each one is documented under.
// book.bat and build.bat are deliberately absent: neither runs a list of gates,
// so there is nothing here to drift.
const WRAPPERS = [
{ bat: "check.bat", heading: "### check.bat" },
{ bat: "test.bat", heading: "### test.bat" },
];
// Tools.md spells step counts as words, which is house style for a small
// number in prose. Only as far as we could plausibly grow.
// biome-ignore format: a table, one entry per line
const NUMBER_WORDS = [
"zero", "one", "two", "three", "four", "five",
"six", "seven", "eight", "nine", "ten", "eleven", "twelve",
"thirteen", "fourteen", "fifteen", "sixteen", "seventeen", "eighteen", "nineteen", "twenty",
"twenty-one", "twenty-two", "twenty-three", "twenty-four", "twenty-five",
];
// Longest first, so that "twenty-one" is not read as "twenty" where a word
// boundary follows the number, as in "of the twenty-one".
const NUMBER_ALTERNATION = [...NUMBER_WORDS].sort((a, b) => b.length - a.length).join("|");
/** The body of a `### <name>` section: up to the next heading of any level. */
function sectionBody(src, heading) {
const sec = splitSections(src).find((s) => s.heading === heading);
return sec ? sec.lines.slice(1).join("\n") : null;
}
/**
* The gates a documented numbered list names, in order.
*
* Deliberately narrow: an ordered-list item whose text opens with a link to
* `scripts/<name>.mjs`, or to `test/<name>.mjs` for a test file. Prose
* elsewhere in the section may mention a script without being a claim about
* the list -- Tools.md's check.bat entry names test.bat in its opening
* paragraph, and that is a cross-reference, not a step.
*/
function gatesFromDoc(body) {
const out = [];
for (const line of body.split(/\r?\n/)) {
const m = /^\s*\d+\.\s+\[`(?:scripts[\\/]([A-Za-z0-9_-]+\.mjs)|test[\\/]([A-Za-z0-9_.-]+\.mjs))/.exec(line);
if (m) out.push(gateName(m[1], m[2]));
}
return out;
}
/**
* Every multi-command run of gate scripts spelled out anywhere in the
* developer documentation, as {file, line, gates}.
*
* Tools.md's numbered lists are one way to restate a wrapper; a command block
* is the other, and Building.md has two of them for a good reason -- they are
* the POSIX equivalents of the `.bat` files, which is the whole point of that
* section. So they are checked rather than forbidden.
*
* **`&&` is what makes a run a claim about a wrapper**, and requiring it is
* not a detail. Consecutive `node scripts/...` lines are far more often a
* list of a single script's usage forms -- `check_links_diff.mjs` has three
* such blocks, `impexp.mjs` another -- and reading those as a wrapper
* sequence produced eight false findings on the first run. A wrapper's POSIX
* equivalent is a single shell command chained with `&&`, so every line after
* the first begins with it; a usage block never does.
*/
function commandRuns(src, file) {
const runs = [];
let cur = null;
const flush = () => {
if (cur && cur.gates.length > 1 && cur.chained) runs.push(cur);
cur = null;
};
src.split(/\r?\n/).forEach((line, i) => {
const m = /^\s*(&&\s*)?node\s+(?:scripts[\\/]([A-Za-z0-9_-]+\.mjs)|--test\s+test[\\/]([A-Za-z0-9_.-]+\.mjs))/.exec(
line,
);
if (!m) {
flush();
return;
}
if (!cur) cur = { file, line: i + 1, gates: [], chained: true };
else if (!m[1]) cur.chained = false;
cur.gates.push(gateName(m[2], m[3]));
});
flush();
return runs;
}
/** The count a section claims in prose, as a number, or null if it makes none. */
function statedCount(body) {
// `steps?` because a one-gate wrapper would correctly write "One step".
const m = new RegExp(`\\b(${NUMBER_ALTERNATION})\\s+steps?\\b`, "i").exec(body);
return m ? NUMBER_WORDS.indexOf(m[1].toLowerCase()) : null;
}
// ------------------------------------------------------- the prose sweep
//
// Everything above reads Tools.md's two numbered lists. Everything below
// reads what the other developer pages *say* about those lists, because that
// is where the drift actually went both times it recurred.
//
// Four shapes, and each one is a site the round-4 review found wrong:
//
// possessive "two of `check.bat`'s four steps"
// verb "`test.bat` is six more" "`check.bat` runs six further gates"
// line-initial "check.bat # six more gates" "| `check.bat` | four scripts |"
// section total a wrapper's own section opening "Five gates that ..."
//
// The fourth is the one a per-line scan cannot see, and it is the one that
// went wrong most often: `Building.md` states the count in a section whose
// only mention of the wrapper is the command line under the heading. So the
// sweep is per section, and a section's subject wrapper is the one named in
// its heading or on the command line directly beneath it.
const NUM = `(?:${NUMBER_ALTERNATION}|\\d+)`;
const WRAP = "`?(check|test)\\.bat`?";
// Deliberately not `checks?`: a count of checks is ordinary English in the
// corpus ("Two checks enforce the registration", "two more checks"), not a
// count of gates.
const NOUN = "(?:gates?|steps?|scripts?)";
const QUAL = "(?:more|further|other|separate|cheaper|local|remaining)\\s+";
// "`test.bat` is six more" names no noun at all, and that sentence is one of
// the seven sites. Where the wrapper is already named, a bare qualifier is
// enough; BARE below is not given the same latitude, because "one more is
// worth knowing about" is ordinary prose.
const TAIL = `(?:(?:${QUAL})?${NOUN}|more\\b|further\\b)`;
const POSSESSIVE = new RegExp(`${WRAP}'s\\s+(?:own\\s+)?(${NUM})\\b`, "gi");
// An explicit verb, not proximity. `Tools.md` narrates the defects this gate
// exists for -- "found `test.bat` documented as three gates when it had four"
// -- and a proximity rule reports that true sentence as a false one.
const VERBAL = new RegExp(`${WRAP}\\s+(?:is|are|runs?|has|have|adds)\\s+(${NUM})\\s+${TAIL}`, "gi");
// Line-initial, so a table cell or a command comment counts and a mention in
// the middle of a paragraph does not.
//
// Two steps rather than one regex, and that is not a style choice. Written as
// `^...WRAP\b[^\n]{0,80}?\b(NUM)\s+TAIL` it is what recheck calls polynomial
// degree 3: the lazy gap and the count can divide the same text. Nothing in
// this file is a regex *literal*, so check_regex_safety.mjs cannot see it --
// its documented blind spot -- and a gate that goes quadratic-and-worse on a
// long table row is exactly the shape that file exists to refuse. Anchoring
// the wrapper first and searching a bounded slice of what follows leaves no
// division to try.
// One character class rather than `^[ \t]*(?:\|[ \t]*)?`, which recheck rates
// polynomial degree 2 -- two groups that can each consume the same leading
// space. Nothing here needs to tell an indent from a table pipe.
const LINE_HEAD = new RegExp(`^[ \\t|]*${WRAP}\\b`, "i");
const COUNT_ON_LINE = new RegExp(`\\b(${NUM})\\s+${TAIL}`, "gi");
const LINE_WINDOW = 80;
const BARE = new RegExp(`\\b(${NUM})\\s+(?:${QUAL})?${NOUN}\\b`, "gi");
const ANAPHORA = new RegExp(`\\bof the\\s+(${NUM})\\b`, "gi");
const asNumber = (w) => {
const i = NUMBER_WORDS.indexOf(String(w).toLowerCase());
return i === -1 ? Number(w) : i;
};
/**
* Split markdown into sections: a heading and everything up to the next one.
* A heading-shaped line inside a fence, code block or HTML block starts none;
* Wisdom.md's `staging.md` example holds one. A section's `lines` begin with
* its heading, and `start` is that line's 1-based number.
*/
function splitSections(src) {
return splitOnMarker(src, (line) => /^#{1,6}\s/.test(line), { md: siteParser() }).map((s) =>
s.marker === null
? { heading: "(top of file)", start: 1, lines: s.lines }
: { heading: s.marker.trim(), start: s.start + 1, lines: [s.marker, ...s.lines] },
);
}
/**
* The wrapper a section is *about*, or null.
*
* Its heading (`### check.bat`), else the first command line under it -- which
* is how Building.md marks its wrapper sections, and the reason a per-line
* scan of that file finds nothing.
*/
function subjectWrapper(sec) {
const h = /\b(check|test)\.bat\b/.exec(sec.heading);
if (h) return h[1];
for (const line of sec.lines.slice(1)) {
if (!line.trim()) continue;
// A kramdown attribute block (`{: #tests-of-the-toolchain }`) sits between
// the heading and the command under it. Skipping it is not a detail: that
// one line is what stopped this rule seeing Building.md's own section, and
// Building.md's own section is the defect the rule was written for.
if (/^\{:/.test(line.trim())) continue;
if (!/^(?: {4}|\t|```|~~~)/.test(line)) return null; // prose, not a command
const m = /^[ \t`~]*(check|test)\.bat\b/.exec(line);
if (m) return m[1];
if (!/^(?:```|~~~)/.test(line)) return null;
}
return null;
}
/**
* Every gate count a page states, as {file, line, wrapper, count, quote}.
*
* @param {string} src the page
* @param {string} file its repo-relative path, for the report
*/
function proseClaims(src, file) {
const claims = [];
const at = (sec, body, idx) => sec.start + body.slice(0, idx).split("\n").length - 1;
const push = (sec, body, m, wrapper, raw, rule) =>
claims.push({
file,
line: at(sec, body, m.index),
wrapper: `${wrapper}.bat`,
count: asNumber(raw),
quote: m[0].trim().replace(/\s+/g, " ").slice(0, 80),
rule,
});
for (const sec of splitSections(src)) {
const body = sec.lines.join("\n");
for (const m of body.matchAll(POSSESSIVE)) push(sec, body, m, m[1], m[2], "possessive");
for (const m of body.matchAll(VERBAL)) push(sec, body, m, m[1], m[2], "verb");
sec.lines.forEach((line, i) => {
const head = LINE_HEAD.exec(line);
if (!head) return;
const rest = line.slice(head[0].length, head[0].length + LINE_WINDOW);
for (const m of rest.matchAll(COUNT_ON_LINE)) {
claims.push({
file,
line: sec.start + i,
wrapper: `${head[1]}.bat`,
count: asNumber(m[1]),
quote: line.trim().replace(/\s+/g, " ").slice(0, 80),
rule: "line",
});
}
});
const subject = subjectWrapper(sec);
if (!subject) continue;
// Only the FIRST bare count in a wrapper's own section is read as that
// wrapper's total. Later ones are subset claims -- "two of them read
// docs/", "why one gate does read your pages" -- and both are legitimate
// sentences the wrapper sections actually contain.
const [total] = [...body.matchAll(BARE)];
if (!total) continue;
push(sec, body, total, subject, total[1], "section total");
// "Four of the five cannot be affected", "Another of the five" -- a
// back-reference to the total just stated, and the shape a grep for
// "five gates" misses. Only counted when it matches that total, so
// "the first of the three" elsewhere is not a claim about a wrapper.
for (const m of body.matchAll(ANAPHORA)) {
if (asNumber(m[1]) !== asNumber(total[1])) continue;
push(sec, body, m, subject, m[1], "back-reference");
}
}
// The rules overlap by design; report one finding per site.
const seen = new Set();
return claims.filter((c) => {
const key = `${c.line}:${c.wrapper}:${c.count}`;
if (seen.has(key)) return false;
seen.add(key);
return true;
});
}
/** Findings for one page, given the real gate counts. */
function proseFindings(src, file, counts) {
const out = [];
for (const c of proseClaims(src, file)) {
const actual = counts.get(c.wrapper);
if (actual === undefined || c.count === actual) continue;
out.push(
`${c.file}:${c.line}: says ${c.wrapper} runs ${c.count}; it runs ${actual}.\n` + ` ${c.rule}: "${c.quote}"`,
);
}
return out;
}
/**
* Compare one wrapper against its documentation.
* @returns {string[]} findings, empty when they agree
*/
function compareWrapper({ bat, heading }, batSrc, toolsMd) {
const actual = gatesFromBat(batSrc);
const findings = [];
if (actual.length === 0) {
return [`${bat}: no \`node scripts/*.mjs\` invocation found -- has the wrapper's shape changed?`];
}
const body = sectionBody(toolsMd, heading);
if (body === null) {
return [`${TOOLS_MD}: no \`${heading}\` section -- the gate lists have no documented home.`];
}
const documented = gatesFromDoc(body);
if (documented.join("\0") !== actual.join("\0")) {
findings.push(
`${bat}: documented list does not match the wrapper.\n` +
` ${bat} : ${actual.join(", ")}\n` +
` ${TOOLS_MD}: ${documented.join(", ") || "(none found)"}`,
);
}
// A stated count that disagrees with its own list is a common drift, so it
// is reported separately from the membership failure -- the two have
// different fixes.
const stated = statedCount(body);
if (stated === null) {
findings.push(
`${bat}: the \`${heading}\` section states no step count. ` + `It should, so that this gate can check it.`,
);
} else if (stated !== actual.length) {
findings.push(`${bat}: documented as "${NUMBER_WORDS[stated]} steps", but the wrapper runs ${actual.length}.`);
}
return findings;
}
// ------------------------------------------------------------- self-test
//
// The probes ride along in the normal run rather than hiding behind a flag,
// because on a healthy tree a gate that has stopped detecting prints exactly
// what a working one prints. Each probe is a real shape of drift.
const PROBES = [
{
name: "a wrapper gaining a gate the docs do not list",
bat: "node scripts/a.mjs\nnode scripts/b.mjs\n",
doc: "### x.bat\n\nOne step:\n\n1. [`scripts/a.mjs`](#a) --- does a thing.\n\n### next\n",
},
{
name: "the docs naming a gate the wrapper does not run",
bat: "node scripts/a.mjs\n",
doc: "### x.bat\n\nTwo steps:\n\n1. [`scripts/a.mjs`](#a) --- a.\n2. [`scripts/zz.mjs`](#zz) --- not run.\n\n### next\n",
},
{
name: "a reordering the prose no longer matches",
bat: "node scripts/b.mjs\nnode scripts/a.mjs\n",
doc: "### x.bat\n\nTwo steps:\n\n1. [`scripts/a.mjs`](#a) --- a.\n2. [`scripts/b.mjs`](#b) --- b.\n\n### next\n",
},
{
// Building.md's real defect: the right scripts, the wrong number.
name: "a stated count that disagrees with its own list",
bat: "node scripts/a.mjs\nnode scripts/b.mjs\n",
doc: "### x.bat\n\nThree steps:\n\n1. [`scripts/a.mjs`](#a) --- a.\n2. [`scripts/b.mjs`](#b) --- b.\n\n### next\n",
},
{
name: "a section that states no count at all",
bat: "node scripts/a.mjs\n",
doc: "### x.bat\n\nIt runs:\n\n1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n",
},
{
// test/search.test.mjs ran in test.bat for as long as the roster read only
// scripts/, so neither this gate nor check_ci_workflows saw it, and CI
// never ran it. The count here agrees with the list unless the test file
// is read, so this fails when the roster stops reading one.
name: "a test file the docs do not list",
bat: "node scripts/a.mjs\nnode --test test/a.test.mjs\n",
doc: "### x.bat\n\nOne step:\n\n1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n",
},
{
// A name with a hyphen was not read at all, in a wrapper or a list, so
// a hyphenated gate could leave a wrapper, or CI, with both roster gates
// green. These two fail when one is not read.
name: "a hyphenated test file the docs do not list",
bat: "node scripts/a.mjs\nnode --test test/a-b.test.mjs\n",
doc: "### x.bat\n\nOne step:\n\n1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n",
},
{
name: "a hyphenated script the docs do not list",
bat: "node scripts/a.mjs\nnode scripts/a-b.mjs\n",
doc: "### x.bat\n\nOne step:\n\n1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n",
},
];
// Must NOT fire.
const NEGATIVES = [
{
// What Tools.md's real entries do: prose that mentions another wrapper's
// gate outside the numbered list.
name: "a cross-reference in prose is not a step",
bat: "node scripts/a.mjs\n",
doc:
"### x.bat\n\nTests of the toolchain are [`scripts/zz.mjs`](#zz), not these. One step:\n\n" +
"1. [`scripts/a.mjs`](#a) --- a.\n\n### next\n",
},
{
name: "a test file listed by its path is a step",
bat: "node scripts/a.mjs\r\nnode --test test\\a.test.mjs\r\n",
doc: "### x.bat\n\nTwo steps:\n\n1. [`scripts/a.mjs`](#a) --- a.\n2. [`test/a.test.mjs`](#a-test) --- tests.\n\n### next\n",
},
{
name: "hyphenated names listed by their paths are steps",
bat: "node scripts/a-b.mjs\nnode --test test/a-b.test.mjs\n",
doc: "### x.bat\n\nTwo steps:\n\n1. [`scripts/a-b.mjs`](#a-b) --- a.\n2. [`test/a-b.test.mjs`](#a-b-test) --- tests.\n\n### next\n",
},
];
// Probes for the prose sweep. Every positive is a sentence of the kind a
// published page states, against fixed counts (check.bat four, test.bat six).
const REAL_COUNTS = new Map([
["check.bat", 4],
["test.bat", 6],
]);
const PROSE_PROBES = [
{
name: "README's command block understating check.bat",
md: "## Building the site\n\n```\nnpm ci # once\ncheck.bat # six more gates, from the publish allowlist to the accessibility scan\n```\n",
},
{
name: "a table row restating a wrapper's step count",
md: "| Wrapper | Runs |\n|---|---|\n| `check.bat` | six scripts in a fixed order, below |\n",
},
{
name: "a wrapper named with an explicit verb",
md: "`test.bat` is four more, in the same cheapest-first order:\n",
},
{
name: "a possessive count in another page's prose",
md: "Chromium is needed for two of `check.bat`'s five steps and one of `test.bat`'s four.\n",
},
{
// Building.md:248 exactly. No line in this section names the wrapper
// except the command under the heading, which is why a per-line grep
// found four of seven sites and called the sweep thorough.
name: "a section total stated under a bare heading",
md: "## Tests of the toolchain\n\n test.bat\n\nFive gates that test the build system rather than the site.\n",
},
{
name: "a back-reference to a wrong total",
md: "## Tests of the toolchain\n\n test.bat\n\nSeven gates test the build system.\n\nFour of the seven cannot be affected by an edit confined to `docs/`.\n",
},
{
name: "a count in the sub-page list of an index page",
md: "`build.bat` produces three output trees; `check.bat` runs six further gates, from the publish allowlist to the accessibility scan.\n",
},
{
// Not a published sentence: Wisdom.md's `staging.md` example has a fenced
// `## ` line, and read as a heading it cut the section it sits in, so a
// count after it belonged to a section about no wrapper.
name: "a section total after a fenced heading-shaped line",
md: "## Tests of the toolchain\n\n test.bat\n\n```\n## not a heading\n```\n\nFive gates that test the build system rather than the site.\n",
},
];
const PROSE_NEGATIVES = [
{
name: "a correct possessive is not a finding",
md: "Chromium is needed for two of `check.bat`'s four steps and one of `test.bat`'s six.\n",
},
{
// Tools.md narrates this gate's own history, including the numbers that
// were wrong. A proximity rule reported the true sentence as a false one.
name: "a past wrong number, quoted as history",
md: "Round 2 found `test.bat` documented as three gates when it had four, and that was fixed here.\n",
},
{
name: "a subset claim after a correct total",
md: "## Tests of the toolchain\n\n test.bat\n\nSix steps, each stopping the run if it fails.\n\nTwo gates do read `docs/`, and that is why one gate can fail on a content edit.\n",
},
{
name: "a count in a section with no wrapper subject",
md: "### scripts/check_a11y.mjs\n\n node scripts/check_a11y.mjs\n\nThree cheaper gates run first and stop the run if they fail.\n",
},
{
name: "a POSIX command block is not a count",
md: " node scripts/check_publish_policy.mjs \\\n && node scripts/check_gate_lists.mjs\n",
},
];
function selfTest() {
const results = [];
for (const p of PROBES) {
const found = compareWrapper({ bat: "x.bat", heading: "### x.bat" }, p.bat, p.doc);
results.push([found.length > 0, p.name]);
}
for (const n of NEGATIVES) {
const found = compareWrapper({ bat: "x.bat", heading: "### x.bat" }, n.bat, n.doc);
results.push([found.length === 0, n.name]);
}
for (const p of PROSE_PROBES) {
results.push([proseFindings(p.md, "<probe>", REAL_COUNTS).length > 0, `prose: ${p.name}`]);
}
for (const p of PROSE_NEGATIVES) {
results.push([proseFindings(p.md, "<probe>", REAL_COUNTS).length === 0, `prose: ${p.name}`]);
}
return results;
}
// ------------------------------------------------------------------ main
const USAGE = `usage: node scripts/check_gate_lists.mjs [--verbose] [--self-test] [-h, --help]
Checks that check.bat and test.bat still match the two gate lists in Tools.md,
and that no developer page states a gate count that disagrees with them.
--verbose print every probe and every wrapper that agrees
--self-test run the probes only, to prove the check still detects a wrong list
-h, --help print this text and exit
Exit codes:
0 the wrappers match the gate lists, every stated count agrees, and every probe passed
1 a list or a stated count disagrees, or a probe failed
2 the gate could not run: a refused command line, or a crash`;
async function main(argv) {
const { values } = withUsageError(() =>
parseCli(argv, {
options: {
verbose: { type: "boolean" },
"self-test": { type: "boolean" },
help: { type: "boolean", short: "h" },
},
stopAt: ["help"],
}),
);
if (values.help) printHelpAndExit(USAGE);
const verbose = values.verbose;
const onlySelfTest = values.selfTest;
const probes = selfTest();
const probesFailed = probes.filter(([ok]) => !ok);
for (const [ok, name] of probes) {
if (!ok) console.error(` FAIL probe: ${name}`);
else if (verbose || onlySelfTest) console.log(` ok probe: ${name}`);
}
if (!verbose && !onlySelfTest && !probesFailed.length) {
console.log(`ok ${probes.length} probes: a wrong gate list is detected`);
}
if (onlySelfTest) {
console.log(
probesFailed.length
? `check_gate_lists: ${probesFailed.length} of ${probes.length} probes failed`
: `check_gate_lists: ${probes.length} probes, all pass`,
);
return probesFailed.length ? 1 : 0;
}
const toolsMd = await readFile(path.join(REPO_ROOT, TOOLS_MD), "utf8");
const findings = [];
const wrapperGates = new Map();
for (const w of WRAPPERS) {
const batSrc = await readFile(path.join(REPO_ROOT, w.bat), "utf8");
wrapperGates.set(w.bat, gatesFromBat(batSrc));
const found = compareWrapper(w, batSrc, toolsMd);
findings.push(...found);
if (verbose && !found.length) {
console.log(` ok ${w.bat}: ${gatesFromBat(batSrc).join(", ")}`);
}
}
// Command blocks and stated counts anywhere a developer page can carry
// them. Building.md restates both wrappers as POSIX command blocks, which
// is legitimate and is exactly the kind of second copy that drifts;
// README.md is here because it states counts too.
const docsDir = path.join(REPO_ROOT, "docs/Documentation");
const wanted = [...wrapperGates.values()].map((g) => g.join("\0"));
const counts = new Map([...wrapperGates].map(([bat, g]) => [bat, g.length]));
const rels = [
"README.md",
...(await readdir(docsDir)).filter((n) => n.endsWith(".md")).map((n) => `docs/Documentation/${n}`),
];
let claimsSeen = 0;
for (const rel of rels) {
const src = await readFile(path.join(REPO_ROOT, rel), "utf8");
for (const run of commandRuns(src, rel)) {
if (wanted.includes(run.gates.join("\0"))) {
if (verbose) console.log(` ok ${rel}:${run.line}: matches a wrapper`);
continue;
}
findings.push(
`${rel}:${run.line}: a run of ${run.gates.length} gate scripts matches no wrapper.\n` +
` documented : ${run.gates.join(", ")}\n` +
[...wrapperGates].map(([b, g]) => ` ${b.padEnd(11)}: ${g.join(", ")}`).join("\n"),
);
}
for (const c of proseClaims(src, rel)) {
claimsSeen++;
if (verbose) console.log(` ok ${rel}:${c.line}: ${c.wrapper} = ${c.count} (${c.rule})`);
}
findings.push(...proseFindings(src, rel, counts));
}
for (const f of findings) console.error(` FAIL ${f}`);
// Reported separately: a failed probe means this gate has stopped detecting,
// which is a different problem from a documented list having drifted, and
// printing "0 disagreements" beside a non-zero exit helps nobody.
if (probesFailed.length) {
console.error(
`\ncheck_gate_lists: ${probesFailed.length} of ${probes.length} self-test probes failed.\n` +
` The gate itself is not detecting what it claims to; fix that before trusting a pass.`,
);
return 1;
}
if (findings.length) {
console.error(
`\ncheck_gate_lists: ${findings.length} disagreement(s) with the wrappers.\n` +
` ${TOOLS_MD} owns these lists; every other page cites it rather than restating it.\n` +
` Fix the list there -- and where a page states a count in prose, prefer deleting the\n` +
` number over correcting it. The command block or the linked list carries it already.`,
);
return 1;
}
console.log(
`check_gate_lists: ${WRAPPERS.map((w) => `${w.bat} (${wrapperGates.get(w.bat).length})`).join(" + ")} ` +
`match ${TOOLS_MD}; ${claimsSeen} stated count(s) across ${rels.length} pages agree -- clean`,
);
return 0;
}
main(process.argv.slice(2))
.then((code) => {
process.exitCode = code;
})
.catch((err) => {
console.error(err);
process.exitCode = 2;
});