diff --git a/plugins/english/wetriedtls.ts b/plugins/english/wetriedtls.ts
index b00eb3c6e..27e3b824f 100644
--- a/plugins/english/wetriedtls.ts
+++ b/plugins/english/wetriedtls.ts
@@ -133,14 +133,88 @@ function paragraphText(p: string): string {
return decodeEntities(p.replace(/<[^>]+>/g, '')).trim();
}
-function isTitleRepeat(p: string): boolean {
- // A paragraph that is nothing but bold text, e.g. the repeated
- // series / chapter title the site prepends to every chapter body.
+/** Lowercase and drop everything but letters and digits, for loose matching. */
+function normalizeText(s: string): string {
+ return s
+ .toLowerCase()
+ .replace(
+ /[\s\u2000-\u2bff\u3000-\u303f\uff01-\uff0f!-/:-@[-`{-~\xa0-\xbf]+/g,
+ '',
+ );
+}
+
+/**
+ * True when a block is the site's repeated series / chapter title header,
+ * e.g. `◈ Series Name`, `Chapter 12: Title`, `Series
Chapter 12`. The
+ * titles come from the chapter page itself, so this needs no state from
+ * parseNovel. A block only counts when nothing but known titles (plus a
+ * bare "Chapter N") is left after removing them, so a genuine bold line
+ * such as a POV label or scene header is never stripped.
+ */
+function isTitleRepeat(p: string, knownTitles: string[]): boolean {
+ let rest = normalizeText(paragraphText(p.replace(/
/gi, ' ')));
+ if (!rest) return false;
+ const titles = knownTitles
+ .map(normalizeText)
+ .filter(t => t.length > 0)
+ .sort((a, b) => b.length - a.length);
+ let removed = false;
+ for (const t of titles) {
+ if (rest.indexOf(t) !== -1) {
+ rest = rest.split(t).join('');
+ removed = true;
+ }
+ }
+ rest = rest.replace(/(?:volume|vol|chapter|ch|episode|ep)\d+/g, '');
+ // Catalog titles often omit a leading article the header keeps.
+ return removed && /^(?:the|an?)?$/.test(rest);
+}
+
+/**
+ * Translator / editor credits and the "Discord:" / "Ko-Fi:" lines that sit
+ * in the chapter header next to the banner are site chrome, not story
+ * content. Covers `Translator: X`, `Editors: A, B`, `Translator/Editor: X`
+ * and `[Translator – X]`. Narrow on purpose: it must start with a role
+ * word followed by a separator (a dash only when spaced, so prose such as
+ * "Editor-in-chief Kim ..." never matches).
+ */
+function isCreditLine(p: string): boolean {
+ const t = paragraphText(p);
+ return (
+ /^\[?\s*(?:(?:translat(?:or|ors|ion)|editors?|proofreaders?|typesetters?|tlc?|qc)\s*[/&,]?\s*)+\s*(?:[:\]]|[–—-]\s|by\b)/i.test(
+ t,
+ ) || /^(?:discord|ko-?fi|patreon)\s*:/i.test(t)
+ );
+}
+
+/**
+ * splitTopLevel, with plain wrapper elements (div / section / article)
+ * unwrapped so a header and the story inside one wrapper are judged block
+ * by block instead of as a single unit.
+ */
+function splitBlocks(html: string): string[] {
+ const out: string[] = [];
+ for (const b of splitTopLevel(html)) {
+ const w = /^<(div|section|article)\b[^>]*>([\s\S]*)<\/\1\s*>$/i.exec(b);
+ if (w) for (const inner of splitBlocks(w[2])) out.push(inner);
+ else out.push(b);
+ }
+ return out;
+}
+
+/** A block that is nothing but bold text, e.g. a header line. */
+function isBoldOnly(p: string): boolean {
const inner = p
- .replace(/^
]*>/i, '')
- .replace(/<\/p>$/i, '')
+ .replace(/<\/?(?:p|span)\b[^>]*>/gi, '')
+ .replace(/
/gi, '')
.trim();
- return /^[\s\S]*<\/strong>$/.test(inner);
+ return /^<(strong|b)\b[^>]*>[\s\S]*<\/\1>$/i.test(inner);
+}
+
+/** A divider row (`──────` or a horizontal rule) in the chapter header. */
+function isDivider(p: string): boolean {
+ if (/
/]/i.test(p) && !paragraphText(p)) return true;
+ return /^[─━—–_=~-]{3,}$/.test(paragraphText(p).replace(/\s+/g, ''));
}
function isPromoParagraph(p: string): boolean {
@@ -149,12 +223,189 @@ function isPromoParagraph(p: string): boolean {
if (/]/i.test(p)) return false;
const t = paragraphText(p).toLowerCase();
if (!t || t === '= = =') return true;
- if (t.indexOf('we tried translations') !== -1) return true;
- if (t.indexOf('dsc.gg') !== -1 || t.indexOf('join our discord') !== -1)
+ if (/we\s*tried\s*translations/.test(t) || /^we\s*tried\s*tls$/.test(t))
+ return true;
+ if (t.indexOf('dsc.gg') !== -1 || /join (our|the) discord/.test(t))
return true;
return false;
}
+/**
+ * Decode the HTML entities that can appear inside an attribute value —
+ * decimal (j), hex (j) and the named entities that can smuggle a
+ * scheme past a prefix check (:) — so the scheme test sees what the
+ * reader will actually navigate to.
+ */
+function decodeAttrEntities(s: string): string {
+ return s
+ .replace(/([0-9a-f]+);?/gi, (_m, h: string) =>
+ String.fromCharCode(parseInt(h, 16)),
+ )
+ .replace(/(\d+);?/g, (_m, n: string) =>
+ String.fromCharCode(parseInt(n, 10)),
+ )
+ .replace(/:?/gi, ':')
+ .replace(/&tab;?/gi, '\t')
+ .replace(/&newline;?/gi, '\n');
+}
+
+/** True for URLs that would execute script when followed from the reader. */
+function isScriptUrl(url: string): boolean {
+ // Browsers ignore whitespace and control characters inside a scheme.
+ const norm = decodeAttrEntities(url)
+ .split('')
+ .filter(ch => ch.charCodeAt(0) > 0x20 && ch.charCodeAt(0) !== 0x7f)
+ .join('')
+ .replace(/\s+/g, '');
+ return /^(javascript|vbscript):/i.test(norm);
+}
+
+// Dangerous elements, plus the raw-text elements (title, xmp, ...) whose
+// content a browser reads as text: tags inside them must not be judged as
+// real tags by the attribute pass below.
+const UNSAFE_TAGS =
+ 'script|iframe|object|embed|form|input|textarea|select|button|style|link|meta|base|noscript|title|xmp|noembed|noframes|plaintext';
+
+/** Attributes whose value the reader may navigate to. */
+const URL_ATTRS = /^(?:xlink:)?(?:href|src|action|formaction)$/;
+
+/** SVG animation attributes that can write a script URL into an href. */
+const ANIMATION_ATTRS = /^(?:values|to|from)$/;
+
+function isUnsafeAttr(name: string, value: string): boolean {
+ const n = name.toLowerCase();
+ if (n.indexOf('on') === 0) return true;
+ if (URL_ATTRS.test(n)) return isScriptUrl(value);
+ if (ANIMATION_ATTRS.test(n)) return value.split(';').some(isScriptUrl);
+ return false;
+}
+
+/**
+ * Strip anything that could execute code from chapter HTML: dangerous
+ * elements, event handler attributes (
), and script URLs in
+ * link-like or SVG animation attributes. The reader sanitizes too; this
+ * keeps the plugin's own output safe. Everything else is preserved as-is.
+ */
+function sanitizeHtml(html: string): string {
+ // Comments and other markup declarations hold no story text, and a quote
+ // inside one must not hide a following tag from the attribute pass.
+ // Repeat until stable so removing one tag cannot splice a new one together.
+ let s = html;
+ for (let prev = ''; prev !== s; ) {
+ prev = s;
+ s = s
+ .replace(/