From cf931bda77c4ab90cf5621e61c24a954156d80e0 Mon Sep 17 00:00:00 2001 From: quality Date: Wed, 7 Oct 2026 18:14:59 -0400 Subject: [PATCH] test: gate redirect pages on what a browser decodes from make-redirects.sh output make-redirects.sh interpolates each MAP url and label into the refresh url=, canonical href, , and link text unescaped. check-redirects.sh proves the page matches the generator and check-links.sh curls the raw text, so a url ending in a legacy no-semicolon entity (¶, ®, ©, ...) or a label with a quote or angle bracket regenerates cleanly and ships a broken redirect. Add scripts/redirect-targets.test.mjs: parse the MAP, reject uninterpolatable url/label characters, decode every committed redirect page's attributes under HTML attribute rules and its title/link text under text rules, require each to resolve to the MAP entry, and report any other ambiguous ampersand. Fixture self-tests for each rule; README Checks entry. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Signed-off-by: quality <quality@hive.kubestellar.io> --- README.md | 9 + scripts/redirect-targets.test.mjs | 348 ++++++++++++++++++++++++++++++ 2 files changed, 357 insertions(+) create mode 100644 scripts/redirect-targets.test.mjs diff --git a/README.md b/README.md index 9e061a9..c474cd6 100644 --- a/README.md +++ b/README.md @@ -132,6 +132,15 @@ red PR job against. must be at `https://<CNAME>`; and no committed page, stylesheet or text asset may link the `hivecommons.github.io` fallback host (which bypasses the custom domain). Each rule has a fixture self-test. Zero dependencies. +- `node --test scripts/redirect-targets.test.mjs` — static gate over what a browser makes + of each shortcut-redirect page. `make-redirects.sh` interpolates every `MAP` url and + label into HTML unescaped, so the gate parses the `MAP`, rejects any url or label that + carries `"`, `<` or `>` or a label that decodes as a character reference, and then + decodes each committed page's refresh `url=`, canonical `href` and `<a href>` under HTML + attribute rules (legacy no-semicolon entities such as `¶`, `®`, `©` included) + and its `<title>`/link text under text rules, requiring each to resolve to exactly the + `MAP` url/label; any other ambiguous ampersand on the page is reported (`&` is the + one accepted escape). Each rule has a fixture self-test. Zero dependencies. - `scripts/check-redirects.sh` — redirect-page drift gate described above. - `scripts/check-redirects.test.sh` — fixture-based self-test of the drift gate; runs offline against throwaway sites with a two-entry generator. diff --git a/scripts/redirect-targets.test.mjs b/scripts/redirect-targets.test.mjs new file mode 100644 index 0000000..702f748 --- /dev/null +++ b/scripts/redirect-targets.test.mjs @@ -0,0 +1,348 @@ +#!/usr/bin/env node +// Gate for what a browser actually does with the shortcut-redirect pages. +// +// make-redirects.sh interpolates each MAP url and label into HTML verbatim: +// the refresh url=, the canonical href, the <a href>, the <title> and the link +// text. Nothing is escaped. check-redirects.sh proves the committed page is +// byte-identical to the generator's output, and check-links.sh curls the raw +// url text — neither asks what the HTML parser makes of it. A url that ends in +// "¶" or "®", a label with "©" in it, a "<" or a '"' anywhere, all +// regenerate cleanly and curl fine, and then redirect visitors somewhere else +// (or nowhere) because the parser decoded a character reference or closed the +// attribute early. Today's MAP already carries raw "&" separators, so the +// hazard is one MAP edit away. +// +// Rules, each with a fixture self-test (zero dependencies): +// 1. The MAP parses, every entry has a url and a label, and neither contains +// a character that the unescaped template cannot carry ('"', '<', '>'). +// 2. For every committed redirect page, the refresh url, canonical href and +// <a href> decode (HTML attribute-value rules) to exactly the MAP url, and +// the <title> and link text decode (text rules) to exactly the label. +// 3. No attribute or text node in a redirect page contains an ambiguous +// ampersand: "&" followed by "#" or by a named reference that any +// browser resolves. Escaped "&" is the one accepted form. +// Usage: node --test scripts/redirect-targets.test.mjs +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync, readdirSync, statSync, existsSync } from "node:fs"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = join(HERE, ".."); +const GENERATOR = join(ROOT, "make-redirects.sh"); +const SKIP_DIRS = new Set([".git", ".github", "node_modules", "scripts", ".linkcheck-work"]); + +// The HTML Standard's named character references that browsers resolve even +// without a trailing ";" (the Latin-1 set plus AMP/COPY/GT/LT/QUOT/REG). +export const LEGACY_REFS = new Set(` +AElig AMP Aacute Acirc Agrave Aring Atilde Auml COPY Ccedil ETH Eacute Ecirc Egrave Euml GT +Iacute Icirc Igrave Iuml LT Ntilde Oacute Ocirc Ograve Oslash Otilde Ouml QUOT REG THORN +Uacute Ucirc Ugrave Uuml Yacute aacute acirc acute aelig agrave amp aring atilde auml brvbar +ccedil cedil cent copy curren deg divide eacute ecirc egrave eth euml frac12 frac14 frac34 gt +iacute icirc iexcl igrave iquest iuml laquo lt macr micro middot nbsp not ntilde oacute ocirc +ograve ordf ordm oslash otilde ouml para plusmn pound quot raquo reg sect shy sup1 sup2 sup3 +szlig thorn times uacute ucirc ugrave uml uuml yacute yen yuml +`.trim().split(/\s+/)); + +const FIVE = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", AMP: "&", LT: "<", GT: ">", QUOT: '"' }; + +// Decode character references the way the HTML tokenizer does for an attribute +// value (inAttribute=true) or a text node (inAttribute=false). Returns the +// decoded string plus the raw references that were consumed, so a caller can +// name exactly what the browser would rewrite. References outside the five +// predefined ones decode to U+FFFD: the test only needs to know that the +// browser would change the text, not what to. +export function decodeHtml(raw, { inAttribute }) { + let out = ""; + const refs = []; + for (let i = 0; i < raw.length; ) { + if (raw[i] !== "&") { + out += raw[i++]; + continue; + } + const rest = raw.slice(i); + const numeric = /^&#(?:[xX][0-9a-fA-F]+|[0-9]+);?/.exec(rest); + if (numeric) { + refs.push(numeric[0]); + out += "\uFFFD"; + i += numeric[0].length; + continue; + } + const named = /^&([A-Za-z][A-Za-z0-9]*)(;?)/.exec(rest); + if (named && named[2] === ";") { + refs.push(named[0]); + out += FIVE[named[1]] ?? "\uFFFD"; + i += named[0].length; + continue; + } + if (named) { + // No semicolon: only a legacy name matches, by longest prefix. + let legacy = null; + for (let len = named[1].length; len > 0; len--) { + const cand = named[1].slice(0, len); + if (LEGACY_REFS.has(cand)) { legacy = cand; break; } + } + if (legacy) { + const next = named[1][legacy.length] ?? raw[i + 1 + legacy.length] ?? ""; + const quirk = inAttribute && (next === "=" || /^[A-Za-z0-9]$/.test(next)); + if (!quirk) { + refs.push("&" + legacy); + out += FIVE[legacy] ?? "\uFFFD"; + i += 1 + legacy.length; + continue; + } + } + } + out += "&"; + i++; + } + return { value: out, refs }; +} + +// --- make-redirects.sh MAP ---------------------------------------------------- + +export function parseMap(script) { + const entries = []; + const problems = []; + const block = /declare -A MAP=\(([\s\S]*?)^\)/m.exec(script); + if (!block) return { entries, problems: ["make-redirects.sh has no `declare -A MAP=( ... )` block"] }; + for (const line of block[1].split("\n")) { + const m = /^\s*\[([^\]]+)\]="(.*)"\s*$/.exec(line); + if (!m) { + if (line.trim()) problems.push(`unparseable MAP line: ${line.trim()}`); + continue; + } + const [path, value] = [m[1], m[2]]; + const bar = value.indexOf("|"); + const url = bar === -1 ? value : value.slice(0, bar); + const label = bar === -1 ? "" : value.slice(bar + 1); + if (!/^https?:\/\/\S+$/.test(url)) problems.push(`MAP[${path}]: url "${url}" is not an absolute http(s) URL without whitespace`); + if (!label.trim()) problems.push(`MAP[${path}]: missing label after "|"`); + for (const [what, text] of [["url", url], ["label", label]]) { + const bad = [...new Set(text.match(/["<>]/g) ?? [])]; + if (bad.length) problems.push(`MAP[${path}]: ${what} contains ${bad.map((c) => JSON.stringify(c)).join(", ")}, which the unescaped template cannot carry`); + if (what === "label" && /&(?:#|[A-Za-z])/.test(text)) { + const { refs } = decodeHtml(text, { inAttribute: false }); + if (refs.length) problems.push(`MAP[${path}]: label would be rewritten by the browser (${refs.join(", ")} decode in text)`); + } + } + entries.push({ path, url, label }); + } + return { entries, problems }; +} + +// --- committed redirect pages ------------------------------------------------ + +function attr(tag, name) { + const m = new RegExp(`\\b${name}\\s*=\\s*(?:"([^"]*)"|'([^']*)')`, "i").exec(tag); + return m ? (m[1] ?? m[2]) : undefined; +} + +export function isRedirectPage(html) { + return /<meta[^>]+http-equiv\s*=\s*["']refresh["']/i.test(html); +} + +// Pull every place the page repeats the target or the label. +export function redirectParts(html) { + const refreshTag = [...html.matchAll(/<meta\b[^>]*>/gi)].map((m) => m[0]).find((t) => /http-equiv\s*=\s*["']refresh["']/i.test(t)); + const content = refreshTag ? attr(refreshTag, "content") : undefined; + const refreshUrl = content === undefined ? undefined : (/^\s*\d+\s*;\s*url\s*=\s*(.*)$/i.exec(content)?.[1] ?? ""); + const canonicalTag = [...html.matchAll(/<link\b[^>]*>/gi)].map((m) => m[0]).find((t) => /\brel\s*=\s*["']canonical["']/i.test(t)); + const canonical = canonicalTag ? attr(canonicalTag, "href") : undefined; + const anchor = /<a\b([^>]*)>([\s\S]*?)<\/a>/i.exec(html); + const anchorHref = anchor ? attr(anchor[1], "href") : undefined; + const anchorText = anchor ? anchor[2] : undefined; + const title = /<title\b[^>]*>([\s\S]*?)<\/title>/i.exec(html)?.[1]; + return { refreshUrl, canonical, anchorHref, anchorText, title }; +} + +export function redirectPageProblems(html, entry) { + const problems = []; + const parts = redirectParts(html); + const attrChecks = [["refresh url=", parts.refreshUrl], ["canonical href", parts.canonical], ["<a href>", parts.anchorHref]]; + for (const [what, raw] of attrChecks) { + if (raw === undefined) { problems.push(`${what} is missing`); continue; } + const { value, refs } = decodeHtml(raw, { inAttribute: true }); + if (value !== entry.url) { + const why = refs.length ? `browser decodes ${refs.join(", ")}` : "text differs"; + problems.push(`${what} resolves to "${value}", MAP says "${entry.url}" (${why})`); + } + } + const textChecks = [["<title>", parts.title, `Hive Commons — ${entry.label}`], ["link text", parts.anchorText, entry.label]]; + for (const [what, raw, want] of textChecks) { + if (raw === undefined) { problems.push(`${what} is missing`); continue; } + const { value, refs } = decodeHtml(raw, { inAttribute: false }); + if (value !== want) { + const why = refs.length ? `browser decodes ${refs.join(", ")}` : "text differs"; + problems.push(`${what} reads "${value}", expected "${want}" (${why})`); + } + } + // Ambiguous ampersands anywhere on the page: "&" that is not "&" yet + // starts something a parser may consume as a character reference. + for (const m of html.matchAll(/&(#(?:[xX][0-9a-fA-F]+|[0-9]+);?|[A-Za-z][A-Za-z0-9]*;?)/g)) { + if (m[0] === "&") continue; + const name = m[1].replace(/;$/, ""); + const ambiguous = m[1].startsWith("#") || m[1].endsWith(";") || [...name].some((_, k) => LEGACY_REFS.has(name.slice(0, k + 1))); + if (ambiguous) problems.push(`ambiguous ampersand "${m[0]}" at offset ${m.index}; write "&" or change the MAP entry`); + } + return [...new Set(problems)]; +} + +export function committedRedirectPages(root = ROOT) { + const out = []; + for (const name of readdirSync(root)) { + const dir = join(root, name); + if (SKIP_DIRS.has(name) || !statSync(dir).isDirectory()) continue; + const file = join(dir, "index.html"); + if (existsSync(file) && isRedirectPage(readFileSync(file, "utf8"))) out.push(name); + } + return out.sort(); +} + +// --- fixtures --------------------------------------------------------------- + +function page(url, label) { + return `<!doctype html><html lang="en"><head><meta charset="utf-8"> +<meta http-equiv="refresh" content="0; url=${url}"> +<meta name="robots" content="noindex"> +<link rel="canonical" href="${url}"><title>Hive Commons — ${label} +

Redirecting to ${label}…

+`; +} + +const MAP_SCRIPT = `#!/usr/bin/env bash +set -euo pipefail +declare -A MAP=( + [tv]="https://example.test/tv|the test channel" + [cal]="https://example.test/c?a=1&b=2|the calendar" +) +for path in "\${!MAP[@]}"; do :; done +`; + +test("fixture: decodeHtml follows attribute vs text rules for legacy references", () => { + assert.deepEqual(decodeHtml("a=1&b=2", { inAttribute: true }), { value: "a=1&b=2", refs: [] }); + assert.deepEqual(decodeHtml("x&y", { inAttribute: true }), { value: "x&y", refs: ["&"] }); + assert.deepEqual(decodeHtml("x&y", { inAttribute: false }), { value: "x&y", refs: ["&"] }); + // Legacy name followed by "=" or alnum: left alone in attributes, decoded in text. + assert.deepEqual(decodeHtml("?x=1¬=2", { inAttribute: true }), { value: "?x=1¬=2", refs: [] }); + assert.deepEqual(decodeHtml("?x=1¬ify=2", { inAttribute: true }), { value: "?x=1¬ify=2", refs: [] }); + assert.equal(decodeHtml("Q¬ify", { inAttribute: false }).refs[0], "¬"); + // Legacy name at the end of the value or before punctuation decodes everywhere. + assert.deepEqual(decodeHtml("?x=1¶", { inAttribute: true }).refs, ["¶"]); + assert.deepEqual(decodeHtml("?x=1®/", { inAttribute: true }).refs, ["®"]); + assert.deepEqual(decodeHtml("?x=1©&z", { inAttribute: true }).refs, ["©"]); + // Semicolon-terminated and numeric references always decode. + assert.deepEqual(decodeHtml("a<b", { inAttribute: true }), { value: "a { + const { entries, problems } = parseMap(MAP_SCRIPT); + assert.deepEqual(problems, []); + assert.deepEqual(entries, [ + { path: "tv", url: "https://example.test/tv", label: "the test channel" }, + { path: "cal", url: "https://example.test/c?a=1&b=2", label: "the calendar" }, + ]); + assert.deepEqual(parseMap("echo no map").problems, ["make-redirects.sh has no `declare -A MAP=( ... )` block"]); +}); + +test("fixture: MAP entries with quotes, angle brackets, bad urls, missing labels or decodable labels are reported", () => { + const bad = MAP_SCRIPT.replace( + ' [cal]="https://example.test/c?a=1&b=2|the calendar"', + [ + ' [q]="https://example.test/a?x=\\"1|the quote"', + ' [lt]="https://example.test/b|bold label"', + ' [rel]="/relative|a path"', + ' [sp]="https://example.test/c d|spaced"', + ' [nolabel]="https://example.test/e"', + ' [ent]="https://example.test/f|R© by us"', + " garbage line", + ].join("\n"), + ); + const { problems } = parseMap(bad); + assert.ok(problems.some((p) => p.startsWith("MAP[q]: url contains '\"'") || p.includes('MAP[q]: url contains "\\""')), problems.join("\n")); + assert.ok(problems.some((p) => p.startsWith("MAP[lt]: label contains"))); + assert.ok(problems.some((p) => p.startsWith("MAP[rel]: url"))); + assert.ok(problems.some((p) => p.startsWith("MAP[sp]: url"))); + assert.ok(problems.some((p) => p === "MAP[nolabel]: missing label after \"|\"")); + assert.ok(problems.some((p) => p.includes("MAP[ent]: label would be rewritten") && p.includes("©"))); + assert.ok(problems.some((p) => p === "unparseable MAP line: garbage line")); +}); + +test("fixture: a page whose attributes and text decode to the MAP entry passes, raw '&' separators included", () => { + const entry = { path: "cal", url: "https://example.test/c?a=1&b=2&tmsrc=x", label: "the calendar" }; + assert.deepEqual(redirectPageProblems(page(entry.url, entry.label), entry), []); + // The escaped form is equally acceptable. + assert.deepEqual(redirectPageProblems(page(entry.url.replace(/&/g, "&"), entry.label), entry), []); +}); + +test("fixture: a url ending in a legacy reference is reported on every attribute that carries it", () => { + const entry = { path: "x", url: "https://example.test/c?a=1¶", label: "the page" }; + const problems = redirectPageProblems(page(entry.url, entry.label), entry); + for (const what of ["refresh url=", "canonical href", ""]) { + assert.ok(problems.some((p) => p.startsWith(what) && p.includes("browser decodes ¶")), `${what}: ${problems.join("\n")}`); + } + assert.ok(problems.some((p) => p.startsWith('ambiguous ampersand "¶"'))); +}); + +test("fixture: a label with a legacy reference is reported for the title and link text", () => { + const entry = { path: "x", url: "https://example.test/c", label: "Q¬ify" }; + const problems = redirectPageProblems(page(entry.url, entry.label), entry); + assert.ok(problems.some((p) => p.startsWith(" reads") && p.includes("¬"))); + assert.ok(problems.some((p) => p.startsWith("link text reads") && p.includes("¬"))); +}); + +test("fixture: a page that drifted from the MAP url, or lost a part, is reported", () => { + const entry = { path: "x", url: "https://example.test/new", label: "the page" }; + const stale = redirectPageProblems(page("https://example.test/old", entry.label), entry); + assert.equal(stale.filter((p) => p.includes("(text differs)")).length, 3); + const noCanonical = page(entry.url, entry.label).replace(/<link rel="canonical"[^>]*>/, ""); + assert.ok(redirectPageProblems(noCanonical, entry).includes("canonical href is missing")); + const noRefresh = page(entry.url, entry.label).replace(/<meta http-equiv="refresh"[^>]*>/, ""); + assert.ok(redirectPageProblems(noRefresh, entry).includes("refresh url= is missing")); + const badContent = page(entry.url, entry.label).replace('content="0; url=', 'content="'); + assert.ok(redirectPageProblems(badContent, entry).some((p) => p.startsWith('refresh url= resolves to ""'))); +}); + +test("fixture: numeric and semicolon-terminated references anywhere on the page are ambiguous ampersands", () => { + const entry = { path: "x", url: "https://example.test/c", label: "the page" }; + const numeric = page(entry.url, entry.label).replace("Redirecting", "Redirecting…"); + assert.ok(redirectPageProblems(numeric, entry).some((p) => p.startsWith('ambiguous ampersand "…"'))); + const named = page(entry.url, entry.label).replace("Redirecting", "Redirecting…"); + assert.ok(redirectPageProblems(named, entry).some((p) => p.startsWith('ambiguous ampersand "…"'))); + const plain = page(entry.url + "?a=1&tmeid=2", entry.label); + assert.ok(!redirectPageProblems(plain, { ...entry, url: entry.url + "?a=1&tmeid=2" }).some((p) => p.startsWith("ambiguous"))); +}); + +// --- live checks over the committed tree ------------------------------------ + +const script = readFileSync(GENERATOR, "utf8"); +const { entries, problems: mapProblems } = parseMap(script); + +test("make-redirects.sh: the MAP parses and every url/label is safe to interpolate unescaped", () => { + assert.deepEqual(mapProblems, []); + assert.ok(entries.length > 0, "MAP has no entries"); +}); + +test("make-redirects.sh: the template still interpolates url and label where this gate expects them", () => { + assert.match(script, /content="0; url=\$url"/); + assert.match(script, /<link rel="canonical" href="\$url"><title>Hive Commons — \$label<\/title>/); + assert.match(script, /<a href="\$url">\$label<\/a>/); +}); + +test("every committed redirect page is in the MAP, and every MAP entry has a committed page", () => { + assert.deepEqual(committedRedirectPages(), entries.map((e) => e.path).sort()); +}); + +for (const entry of entries) { + test(`${entry.path}/index.html: refresh, canonical and link resolve to the MAP url; title and link text to the label; no ambiguous ampersands`, () => { + const file = join(ROOT, entry.path, "index.html"); + assert.ok(existsSync(file), `${entry.path}/index.html missing (run ./make-redirects.sh)`); + assert.deepEqual(redirectPageProblems(readFileSync(file, "utf8"), entry), []); + }); +}