Repository navigation
Expand file tree
/
Copy pathcheck.mjs
More file actions
575 lines (528 loc) · 22.2 KB
/
Copy pathcheck.mjs
File metadata and controls
575 lines (528 loc) · 22.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
// The link check, over pages already rendered.
//
// It has two front ends. `tbdocs --src docs --check` runs it on the HTML
// the build already has in worker memory, rather than on ~270 MB read
// back off disk. scripts/check_links.mjs runs the same functions over a
// tree it reads from disk, for a tree the build did not produce. Both
// sit on builder/link-check.mjs, the pure core:
//
// deriveTreeRels() what each output tree receives, from the build's
// own records rather than a directory walk. Build
// only.
// checkChunk() checks a chunk of pages against one tree. The
// build runs it on a worker, inside flush(), on HTML
// already decoded in the lane's memory; the script
// runs it once, over the whole tree.
// joinChunks() merges the chunks, settles the fragment references
// no single chunk could decide, and runs the three
// cross-file checks.
// formatReport() the build's human-readable report and exit code.
// The script prints its own summary lines around the
// same two link-check.mjs reporters.
// findingsFor() the conclusions as sorted strings, which
// scripts/check_links_diff.mjs compares.
//
// In the build, two properties make replacing the filesystem with an
// index sound rather than merely faster:
//
// 1. prepareDestinations() wipes _site/, _site-offline/ and _site-pdf/
// before the build writes a byte, so nothing survives from a
// previous run. An index built from what the build emitted is
// complete, not an approximation.
// 2. Both CI workflows run the checks immediately after the build, in
// the same job. There is no build that would start paying for a
// check it does not already pay for.
//
// What a check task must never do is abort the graph. A broken link
// still produces a valid site you want on disk to look at, so failures
// are collected and reported; only the exit code changes.
import * as path from "node:path";
import { performance } from "node:perf_hooks";
import {
extractFromHtml,
resolveOccurrences,
settleFragments,
buildTreeIndex,
IndexOracle,
normalizeBasePath,
checkSitemap,
checkSearch,
checkCanonical,
formatLinkReport,
formatIntegrityReport,
} from "./link-check.mjs";
import { posix } from "./check-tree.mjs";
export { deriveTreeRels } from "./check-tree.mjs";
export { normalizeBasePath };
// The three passes, verbatim from check.bat. Fusion must not quietly
// unify them: the online tree has a sitemap and a search index and the
// offline tree has neither, a live-site link is a fault in the offline
// tree and an expected, listed one in the book, and the book pass is
// informational.
export const FALLBACK_EXTS = ["html"];
export const INDEX_FILES = ["index.html", "."];
export const TREES = {
online: {
suffix: "",
label: "_site",
checkOpts: {
checkHtml: true,
checkA11y: true,
checkIds: true,
checkRemoteAssets: true,
checkCanonical: true,
captureRedirectStub: false, // the build knows its own stubs
},
forbid: null,
crossFile: { sitemap: true, search: true, canonical: true },
fallbackExts: FALLBACK_EXTS,
indexFiles: INDEX_FILES,
includeFragments: true,
},
offline: {
suffix: "-offline",
label: "_site-offline",
checkOpts: {
checkHtml: true,
checkA11y: true,
checkIds: true,
checkRemoteAssets: true,
checkCanonical: false,
captureRedirectStub: false,
},
// Catches live-site links the offlinify rewrite missed.
forbid: ["https://docs.twinbasic.com"],
crossFile: { sitemap: false, search: false, canonical: false },
fallbackExts: FALLBACK_EXTS,
indexFiles: INDEX_FILES,
includeFragments: true,
},
pdf: {
suffix: "-pdf",
label: "_site-pdf",
checkOpts: null,
// Every link book.mjs sent to the website because the page it names
// is not in the book. They are collected the way the offline tree
// collects its forbidden links, and reported under a name of their
// own: here they are expected, and the list says which pages the
// book leaves out and still links to.
forbid: ["https://docs.twinbasic.com"],
forbidReport: { tag: "OUT OF BOOK", reason: "not in the book; opens the website", noun: "out of book" },
crossFile: { sitemap: false, search: false, canonical: false },
// book.html is one flattened document whose links are almost
// entirely internal fragments; there is no directory structure to
// fall back through.
fallbackExts: [],
indexFiles: [],
includeFragments: true,
noFail: true,
},
};
// ── Per-chunk work (worker side) ────────────────────────────────────
// Check one chunk of already-rendered pages against one tree.
//
// `docs` is [{ destPath, html }, ...]; `env` carries the tree root, the
// prebuilt index and the tree's options. Returns a reduction sized by
// the number of *findings*, not by the 793k link occurrences: the
// occurrences never cross a thread boundary, and on a clean build the
// only bulk in the payload is the per-page id sets the join needs to
// settle cross-page fragments.
//
// Fragments pointing into this chunk's own pages are settled here --
// 92 % of fragment references are same-page -- and the rest come back
// as `pending` for joinChunks().
export function checkChunk(docs, env) {
const { root, index, tree, basePath = "" } = env;
// The oracle and the two resolution caches live on `env`, which is
// per worker per tree, not per chunk. A lane checks ~10 chunks of
// ~6 pages each, and every one of them walks the same nav and footer
// links; rebuilding the caches per chunk threw that reuse away and
// cost roughly 3x on this stage.
env.oracle ??= IndexOracle(index);
env.caches ??= { resolution: new Map(), path: new Map() };
const occurrences = [];
const localIds = new Map();
const ids = [];
const integrity = [];
const canonicals = [];
const forbidden = [];
const stubs = [];
const t0 = performance.now();
for (const doc of docs) {
const abs = path.join(root, doc.destPath);
const r = extractFromHtml(doc.html, tree.includeFragments, tree.forbid, tree.checkOpts);
const srcDir = path.dirname(abs);
for (const h of r.links) occurrences.push(abs, srcDir, h);
if (r.ids) {
localIds.set(abs, r.ids);
ids.push([abs, [...r.ids]]);
}
if (r.forbidden && r.forbidden.length) {
for (const f of r.forbidden) forbidden.push(doc.destPath, f.url, f.prefix);
}
if (r.htmlErrors?.length || r.a11yErrors?.length || r.dupIds?.length || r.remoteAssets?.length) {
integrity.push([
doc.destPath,
{
htmlErrors: r.htmlErrors,
a11yErrors: r.a11yErrors,
dupIds: r.dupIds,
remoteAssets: r.remoteAssets,
},
]);
}
if (r.canonicalHref) canonicals.push([doc.destPath, r.canonicalHref]);
if (r.isRedirectStub) stubs.push(doc.destPath);
}
const extract = performance.now() - t0;
const res = resolveOccurrences(occurrences, env.oracle, {
rootStr: root,
basePath,
fallbackExts: tree.fallbackExts,
indexFiles: tree.indexFiles,
includeFragments: tree.includeFragments,
localIds,
deferFragments: true,
caches: env.caches,
});
// Report sources tree-relative so the payload does not carry the
// absolute root 793k times over, and so findings compare equal
// regardless of where the tree lives.
const broken = [];
for (let i = 0; i < res.broken.length; i += 3) {
broken.push(relTo(root, res.broken[i]), res.broken[i + 1], res.broken[i + 2]);
}
return {
files: docs.length,
occurrences: occurrences.length / 3,
uniqueLocal: res.uniqueCount,
broken,
brokenKeys: res.brokenKeys,
forbidden,
pending: res.pendingFragments,
ids,
integrity,
canonicals,
// Only the standalone script uses these three. The build knows its
// own stubs (captureRedirectStub is off in every TREES entry, so
// `stubs` stays empty), and the script checks a whole tree as one
// chunk, so its -v figures are the tree's.
stubs,
fragmentTargets: res.fragmentTargets.size,
stages: { extract, ...res.stages },
};
}
function relTo(root, p) {
return path.relative(root, p).replaceAll("\\", "/");
}
// ── Join (main thread) ──────────────────────────────────────────────
// Merge the per-chunk reductions, settle the fragment references no
// single chunk could decide, and run the three cross-file checks.
//
// `aux` also carries `sitemapOptOut` / `searchOptOut`: the rel paths
// the two generators were asked to skip. Without them the first page to
// use `sitemap: false` or `search_exclude: true` fails --check, even
// though both keys are supported and one is documented.
//
// `aux` carries the content the cross-file checks need -- sitemap.xml
// and search-data.json as the build wrote them, not as re-read from
// disk. Checking the string the build emitted is checking the same
// bytes the tree received.
export function joinChunks(chunks, { root, tree, basePath = "", relFiles = [], stubRels = null, aux = {} }) {
let occurrences = 0,
files = 0;
const broken = [];
const brokenKeys = new Set();
const forbiddenBySource = tree.forbid ? new Map() : null;
const integrityByFile = new Map();
const canonicalByRel = new Map();
const idsByTarget = new Map();
const pending = [];
const errors = [];
for (const c of chunks) {
// On the chunk-merge path "the piece is missing" is a bug, not a
// case to handle. A skipped chunk means the check examined part of
// the tree and reported a pass over the whole of it, which is the
// exact failure this design exists to prevent -- and the one that
// dropped six pages from search-data.json on about one build in
// three. An `error` chunk is different: that piece ran and failed,
// and saying so is the point.
if (!c) {
throw new Error(
`joinChunks(${tree.label}): a chunk produced no result; the check ` +
`would have covered only part of the tree`,
);
}
if (c.error) {
errors.push(c.error);
continue;
}
occurrences += c.occurrences;
files += c.files;
for (const x of c.broken) broken.push(x);
for (const k of c.brokenKeys) brokenKeys.add(k);
for (const [k, v] of c.ids) idsByTarget.set(k, new Set(v));
for (const [k, v] of c.integrity) integrityByFile.set(k, v);
for (const [k, v] of c.canonicals) canonicalByRel.set(k, v);
for (const p of c.pending) pending.push(p);
if (forbiddenBySource) {
for (let i = 0; i < c.forbidden.length; i += 3) {
const src = c.forbidden[i];
let list = forbiddenBySource.get(src);
if (!list) {
list = [];
forbiddenBySource.set(src, list);
}
list.push({ url: c.forbidden[i + 1], prefix: c.forbidden[i + 2] });
}
}
}
// Cross-page fragments: 1 457 site-wide, because the chunks settled
// the other 92 % without leaving their lanes.
const settled = settleFragments(pending, idsByTarget);
for (let i = 0; i < settled.broken.length; i += 3) {
broken.push(relTo(root, settled.broken[i]), settled.broken[i + 1], settled.broken[i + 2]);
}
for (const k of settled.brokenKeys) brokenKeys.add(k);
// Redirect stubs are excluded from all three cross-file checks. The
// standalone script detects them by sniffing for a meta refresh; the
// build simply knows which pages it generated as stubs.
const contentRels = stubRels ? relFiles.filter((r) => !stubRels.has(r)) : relFiles;
// Each of the three is `null` when it did not run, and `[]` when it
// ran and found nothing. formatReport cannot tell those apart -- it
// prints `0 integrity` for both -- so a precondition that quietly
// stopped being met would take a check dark and still read as a pass.
// Every `null` therefore gets a reason, printed in the same shape the
// standalone script uses for the same situation.
let sitemapIssues = null,
searchIssues = null,
canonicalIssues = null;
const skipped = [];
if (!tree.crossFile.sitemap) {
// Not a skip: this tree's definition does not include the check.
} else if (aux.sitemapXml == null) {
skipped.push("sitemap: sitemap.xml was not generated, skipping");
} else {
sitemapIssues = checkSitemap(aux.sitemapXml, contentRels, basePath, aux.sitemapOptOut).sort();
}
if (!tree.crossFile.search) {
// As above.
} else if (aux.searchJson == null) {
skipped.push("search: search-data.json was not generated, skipping");
} else {
let data = null;
try {
data = JSON.parse(aux.searchJson);
} catch {
data = null;
}
if (data) {
searchIssues = checkSearch(data, contentRels, basePath, aux.searchOptOut).sort();
} else {
skipped.push("search: search-data.json did not parse as JSON, skipping");
}
}
if (tree.crossFile.canonical) {
const forCheck = new Map();
for (const [rel, href] of canonicalByRel) {
if (stubRels && stubRels.has(rel)) continue;
forCheck.set(rel, href);
}
if (forCheck.size) {
canonicalIssues = checkCanonical(forCheck, basePath).sort();
} else {
skipped.push('canonical: no <link rel="canonical"> in any page, skipping');
}
}
return {
label: tree.label,
tree,
noFail: tree.noFail === true,
files,
occurrences,
brokenUnique: brokenKeys.size,
broken,
forbiddenBySource,
integrityByFile,
sitemapIssues,
searchIssues,
canonicalIssues,
skipped,
errors,
};
}
// ── Reporting ───────────────────────────────────────────────────────
// Render one tree's result. Returns { text, linksFailed, integrityFailed }.
// Nothing here throws: a failing check reports, it does not take the
// build's output down with it.
export function formatReport(r) {
const out = [];
for (const e of r.errors) out.push(` ERROR ${e}\n`);
// A cross-file check that did not run says so. Without this, a
// precondition quietly ceasing to hold reads as `0 integrity` -- the
// same thing the check prints when it ran and passed.
for (const s of r.skipped ?? []) out.push(` WARN ${r.label}: ${s}\n`);
// The script prints bare walk paths; prefix the tree so a fused run
// covering three trees says which one each finding came from.
const forbidReport = r.tree.forbidReport;
const linkReport = formatLinkReport(r.broken, r.forbiddenBySource, forbidReport ?? {});
if (linkReport) out.push(prefixPaths(linkReport, r.label));
const integrity = formatIntegrityReport(r.integrityByFile);
if (integrity.text) out.push(prefixPaths(integrity.text, r.label));
let integrityCount = integrity.count;
for (const issues of [r.sitemapIssues, r.searchIssues, r.canonicalIssues]) {
if (!issues || !issues.length) continue;
out.push("\n" + issues.map((i) => `${r.label}/${i}`).join("\n") + "\n");
integrityCount += issues.length;
}
let forbiddenCount = 0;
if (r.forbiddenBySource) {
for (const hits of r.forbiddenBySource.values()) forbiddenCount += hits.length;
}
const linksFailed = r.broken.length > 0 || forbiddenCount > 0;
// An error means the check did not complete, which has to fail the
// run: a pass reported over a partial examination is worse than no
// pass at all. Printing it and exiting 0 was the earlier behaviour
// and exactly the wrong shape.
const integrityFailed = integrityCount > 0 || r.errors.length > 0;
// The book's live-site links are counted the way the report lists
// them, once per target, as broken links are. The offline tree keeps
// its count of occurrences, which scripts/check_links.mjs prints too.
let forbidNote = "";
if (r.forbiddenBySource && forbidReport) {
const targets = new Set();
for (const hits of r.forbiddenBySource.values()) for (const h of hits) targets.add(h.url);
forbidNote = `, ${targets.size} ${forbidReport.noun}`;
} else if (r.forbiddenBySource) {
forbidNote = `, ${forbiddenCount} forbidden`;
}
const failNote = r.noFail && (linksFailed || integrityFailed) ? " (informational)" : "";
out.push(
` ${r.label.padEnd(14)} ${String(r.occurrences).padStart(7)} occurrences -- ` +
`${r.brokenUnique} broken${forbidNote}, ${integrityCount} integrity${failNote}\n`,
);
return {
text: out.join(""),
linksFailed: r.noFail ? false : linksFailed,
integrityFailed: r.noFail ? false : integrityFailed,
};
}
// Both reporters start a path at column 0 and indent everything else
// -- the link report's per-source group headers, the integrity
// report's one-line findings -- so prefixing every line that begins
// with a non-space qualifies exactly the paths and nothing else.
// Blank lines have no non-space to match.
function prefixPaths(text, label) {
return text.replace(/^(?=\S)/gm, `${label}/`);
}
// ── Structured findings ─────────────────────────────────────────────
// The same conclusions as sorted arrays of tree-relative strings, one
// category per check. Both front ends produce them here -- the script
// in its `structured` mode -- so scripts/check_links_diff.mjs compares
// one implementation reading a tree two ways: from disk, and held in
// the build's memory through its index and its chunks. That comparison
// is the only thing that says the build's pass still looks at every
// page and link a read from disk finds; check.bat going green does not.
//
// `unique` is null here. The script checks its tree as one chunk and
// fills in that chunk's own count; the build resolves in 160 chunks and
// reconstructing a global figure would mean shipping every unique
// target key back from every chunk -- hundreds of thousands of strings
// for a number that appears in a summary line and is not a finding.
// The harness skips a count either side reports as null.
export function findingsFor(r) {
const on = r.tree.checkOpts ?? {};
const gate = (enabled, a) => (enabled ? a.sort() : null);
const brokenOut = [];
for (let i = 0; i < r.broken.length; i += 3) {
brokenOut.push(`${r.broken[i]}\t${r.broken[i + 1]}\t${r.broken[i + 2]}`);
}
const forbiddenOut = [];
if (r.forbiddenBySource) {
for (const [src, hits] of r.forbiddenBySource) {
for (const h of hits) forbiddenOut.push(`${posix(src)}\t${h.url}\t${h.prefix}`);
}
}
const html = [],
a11y = [],
dupIds = [],
remoteAssets = [];
for (const [src, rec] of r.integrityByFile) {
const s = posix(src);
for (const e of rec.htmlErrors ?? []) html.push(`${s}: html-${e.type}: <${e.tag}>`);
for (const e of rec.a11yErrors ?? []) {
if (e.type === "img-missing-alt") a11y.push(`${s}: a11y-img-missing-alt: src=${e.src}`);
else if (e.type === "empty-anchor") a11y.push(`${s}: a11y-empty-anchor`);
else if (e.type === "empty-href") a11y.push(`${s}: a11y-empty-href: <${e.tag}>`);
}
for (const e of rec.dupIds ?? []) dupIds.push(`${s}: duplicate-id: '${e.id}' appears ${e.count} times`);
for (const e of rec.remoteAssets ?? []) remoteAssets.push(`${s}: remote-asset: <${e.tag} src="${e.src}">`);
}
const crossFileCount =
(r.sitemapIssues?.length ?? 0) + (r.searchIssues?.length ?? 0) + (r.canonicalIssues?.length ?? 0);
const integrityCount = html.length + a11y.length + dupIds.length + remoteAssets.length + crossFileCount;
let forbiddenCount = 0;
if (r.forbiddenBySource) {
for (const hits of r.forbiddenBySource.values()) forbiddenCount += hits.length;
}
return {
broken: brokenOut.sort(),
forbidden: gate(r.tree.forbid !== null, forbiddenOut),
html: gate(on.checkHtml, html),
a11y: gate(on.checkA11y, a11y),
dupIds: gate(on.checkIds, dupIds),
remoteAssets: gate(on.checkRemoteAssets, remoteAssets),
sitemap: r.sitemapIssues,
search: r.searchIssues,
canonical: r.canonicalIssues,
counts: {
files: r.files,
occurrences: r.occurrences,
unique: null,
brokenUnique: r.brokenUnique,
forbidden: forbiddenCount,
integrity: integrityCount,
},
// Skips ride along so --check-findings can assert a cross-file
// check ran, rather than only that it found nothing.
skipped: r.skipped ?? [],
// Raw, before --no-fail is applied: the book pass reports its ten
// broken links as failures here even though it never fails a build.
linksFailed: r.broken.length > 0 || forbiddenCount > 0,
// An error means the check did not complete. formatReport counts it
// as an integrity failure and the process exits 1; omitting it here
// let --check-findings report `false` for a run that exited 1.
integrityFailed: integrityCount > 0 || r.errors.length > 0,
};
}
// ── Index audit (development aid) ───────────────────────────────────
// Diff the derived index against the tree on disk. An entry missing
// from the index turns a working link into a reported break, which is
// loud; a spurious entry masks a real break, which is silent. Only the
// second needs a dedicated check, and this is it.
export async function auditIndex(root, rels) {
const { promises: fsP } = await import("node:fs");
const onDisk = new Set();
let entries;
try {
entries = await fsP.readdir(root, { recursive: true, withFileTypes: true });
} catch {
return { missing: [], spurious: [], unreadable: true };
}
for (const e of entries) {
if (!e.isFile()) continue;
const abs = path.join(e.parentPath || root, e.name);
onDisk.add(path.relative(root, abs).replaceAll("\\", "/"));
}
const derived = new Set(rels.map(posix));
return {
missing: [...onDisk].filter((r) => !derived.has(r)).sort(),
spurious: [...derived].filter((r) => !onDisk.has(r)).sort(),
unreadable: false,
};
}
// Build a tree index the same way both sides do, so callers do not have
// to remember that the root string's shape is load-bearing.
export function treeIndexFor(root, rels) {
return buildTreeIndex(root, rels);
}