diff --git a/plugins/english/wetriedtls.ts b/plugins/english/wetriedtls.ts index b00eb3c6e..27e3b824f 100644 --- a/plugins/english/wetriedtls.ts +++ b/plugins/english/wetriedtls.ts @@ -133,14 +133,88 @@ function paragraphText(p: string): string { return decodeEntities(p.replace(/<[^>]+>/g, '')).trim(); } -function isTitleRepeat(p: string): boolean { - // A paragraph that is nothing but bold text, e.g. the repeated - // series / chapter title the site prepends to every chapter body. +/** Lowercase and drop everything but letters and digits, for loose matching. */ +function normalizeText(s: string): string { + return s + .toLowerCase() + .replace( + /[\s\u2000-\u2bff\u3000-\u303f\uff01-\uff0f!-/:-@[-`{-~\xa0-\xbf]+/g, + '', + ); +} + +/** + * True when a block is the site's repeated series / chapter title header, + * e.g. `◈ Series Name`, `Chapter 12: Title`, `Series
Chapter 12`. The + * titles come from the chapter page itself, so this needs no state from + * parseNovel. A block only counts when nothing but known titles (plus a + * bare "Chapter N") is left after removing them, so a genuine bold line + * such as a POV label or scene header is never stripped. + */ +function isTitleRepeat(p: string, knownTitles: string[]): boolean { + let rest = normalizeText(paragraphText(p.replace(//gi, ' '))); + if (!rest) return false; + const titles = knownTitles + .map(normalizeText) + .filter(t => t.length > 0) + .sort((a, b) => b.length - a.length); + let removed = false; + for (const t of titles) { + if (rest.indexOf(t) !== -1) { + rest = rest.split(t).join(''); + removed = true; + } + } + rest = rest.replace(/(?:volume|vol|chapter|ch|episode|ep)\d+/g, ''); + // Catalog titles often omit a leading article the header keeps. + return removed && /^(?:the|an?)?$/.test(rest); +} + +/** + * Translator / editor credits and the "Discord:" / "Ko-Fi:" lines that sit + * in the chapter header next to the banner are site chrome, not story + * content. Covers `Translator: X`, `Editors: A, B`, `Translator/Editor: X` + * and `[Translator – X]`. Narrow on purpose: it must start with a role + * word followed by a separator (a dash only when spaced, so prose such as + * "Editor-in-chief Kim ..." never matches). + */ +function isCreditLine(p: string): boolean { + const t = paragraphText(p); + return ( + /^\[?\s*(?:(?:translat(?:or|ors|ion)|editors?|proofreaders?|typesetters?|tlc?|qc)\s*[/&,]?\s*)+\s*(?:[:\]]|[–—-]\s|by\b)/i.test( + t, + ) || /^(?:discord|ko-?fi|patreon)\s*:/i.test(t) + ); +} + +/** + * splitTopLevel, with plain wrapper elements (div / section / article) + * unwrapped so a header and the story inside one wrapper are judged block + * by block instead of as a single unit. + */ +function splitBlocks(html: string): string[] { + const out: string[] = []; + for (const b of splitTopLevel(html)) { + const w = /^<(div|section|article)\b[^>]*>([\s\S]*)<\/\1\s*>$/i.exec(b); + if (w) for (const inner of splitBlocks(w[2])) out.push(inner); + else out.push(b); + } + return out; +} + +/** A block that is nothing but bold text, e.g. a header line. */ +function isBoldOnly(p: string): boolean { const inner = p - .replace(/^]*>/i, '') - .replace(/<\/p>$/i, '') + .replace(/<\/?(?:p|span)\b[^>]*>/gi, '') + .replace(//gi, '') .trim(); - return /^[\s\S]*<\/strong>$/.test(inner); + return /^<(strong|b)\b[^>]*>[\s\S]*<\/\1>$/i.test(inner); +} + +/** A divider row (`──────` or a horizontal rule) in the chapter header. */ +function isDivider(p: string): boolean { + if (//]/i.test(p) && !paragraphText(p)) return true; + return /^[─━—–_=~-]{3,}$/.test(paragraphText(p).replace(/\s+/g, '')); } function isPromoParagraph(p: string): boolean { @@ -149,12 +223,189 @@ function isPromoParagraph(p: string): boolean { if (/]/i.test(p)) return false; const t = paragraphText(p).toLowerCase(); if (!t || t === '= = =') return true; - if (t.indexOf('we tried translations') !== -1) return true; - if (t.indexOf('dsc.gg') !== -1 || t.indexOf('join our discord') !== -1) + if (/we\s*tried\s*translations/.test(t) || /^we\s*tried\s*tls$/.test(t)) + return true; + if (t.indexOf('dsc.gg') !== -1 || /join (our|the) discord/.test(t)) return true; return false; } +/** + * Decode the HTML entities that can appear inside an attribute value — + * decimal (j), hex (j) and the named entities that can smuggle a + * scheme past a prefix check (:) — so the scheme test sees what the + * reader will actually navigate to. + */ +function decodeAttrEntities(s: string): string { + return s + .replace(/&#x([0-9a-f]+);?/gi, (_m, h: string) => + String.fromCharCode(parseInt(h, 16)), + ) + .replace(/&#(\d+);?/g, (_m, n: string) => + String.fromCharCode(parseInt(n, 10)), + ) + .replace(/:?/gi, ':') + .replace(/&tab;?/gi, '\t') + .replace(/&newline;?/gi, '\n'); +} + +/** True for URLs that would execute script when followed from the reader. */ +function isScriptUrl(url: string): boolean { + // Browsers ignore whitespace and control characters inside a scheme. + const norm = decodeAttrEntities(url) + .split('') + .filter(ch => ch.charCodeAt(0) > 0x20 && ch.charCodeAt(0) !== 0x7f) + .join('') + .replace(/\s+/g, ''); + return /^(javascript|vbscript):/i.test(norm); +} + +// Dangerous elements, plus the raw-text elements (title, xmp, ...) whose +// content a browser reads as text: tags inside them must not be judged as +// real tags by the attribute pass below. +const UNSAFE_TAGS = + 'script|iframe|object|embed|form|input|textarea|select|button|style|link|meta|base|noscript|title|xmp|noembed|noframes|plaintext'; + +/** Attributes whose value the reader may navigate to. */ +const URL_ATTRS = /^(?:xlink:)?(?:href|src|action|formaction)$/; + +/** SVG animation attributes that can write a script URL into an href. */ +const ANIMATION_ATTRS = /^(?:values|to|from)$/; + +function isUnsafeAttr(name: string, value: string): boolean { + const n = name.toLowerCase(); + if (n.indexOf('on') === 0) return true; + if (URL_ATTRS.test(n)) return isScriptUrl(value); + if (ANIMATION_ATTRS.test(n)) return value.split(';').some(isScriptUrl); + return false; +} + +/** + * Strip anything that could execute code from chapter HTML: dangerous + * elements, event handler attributes (), and script URLs in + * link-like or SVG animation attributes. The reader sanitizes too; this + * keeps the plugin's own output safe. Everything else is preserved as-is. + */ +function sanitizeHtml(html: string): string { + // Comments and other markup declarations hold no story text, and a quote + // inside one must not hide a following tag from the attribute pass. + // Repeat until stable so removing one tag cannot splice a new one together. + let s = html; + for (let prev = ''; prev !== s; ) { + prev = s; + s = s + .replace(/