From bdd2df7821f77033e62598421dd65ed23f99528a Mon Sep 17 00:00:00 2001 From: Raiyn Aydin Date: Tue, 29 Sep 2026 06:43:08 +0800 Subject: [PATCH 1/3] fix(english/wetriedtls): strip site header chrome, keep container content, fail loudly on bad chapter lists --- plugins/english/wetriedtls.ts | 304 ++++++++++++++++++++++++++++------ 1 file changed, 256 insertions(+), 48 deletions(-) diff --git a/plugins/english/wetriedtls.ts b/plugins/english/wetriedtls.ts index b00eb3c6e..117e3f2df 100644 --- a/plugins/english/wetriedtls.ts +++ b/plugins/english/wetriedtls.ts @@ -133,14 +133,72 @@ function paragraphText(p: string): string { return decodeEntities(p.replace(/<[^>]+>/g, '')).trim(); } -function isTitleRepeat(p: string): boolean { - // A paragraph that is nothing but bold text, e.g. the repeated - // series / chapter title the site prepends to every chapter body. +/** Lowercase and drop everything but letters and digits, for loose matching. */ +function normalizeText(s: string): string { + return s + .toLowerCase() + .replace( + /[\s\u2000-\u2bff\u3000-\u303f\uff01-\uff0f!-/:-@[-`{-~\xa0-\xbf]+/g, + '', + ); +} + +/** + * True when a block is the site's repeated series / chapter title header, + * e.g. `◈ Series Name`, `Chapter 12: Title`, `Series
Chapter 12`. The + * titles come from the chapter page itself, so this needs no state from + * parseNovel. A block only counts when nothing but known titles (plus a + * bare "Chapter N") is left after removing them, so a genuine bold line + * such as a POV label or scene header is never stripped. + */ +function isTitleRepeat(p: string, knownTitles: string[]): boolean { + let rest = normalizeText(paragraphText(p.replace(//gi, ' '))); + if (!rest) return false; + const titles = knownTitles + .map(normalizeText) + .filter(t => t.length > 0) + .sort((a, b) => b.length - a.length); + let removed = false; + for (const t of titles) { + if (rest.indexOf(t) !== -1) { + rest = rest.split(t).join(''); + removed = true; + } + } + rest = rest.replace(/(?:volume|vol|chapter|ch|episode|ep)\d+/g, ''); + // Catalog titles often omit a leading article the header keeps. + return removed && /^(?:the|an?)?$/.test(rest); +} + +/** + * Translator / editor credits and the "Discord:" / "Ko-Fi:" lines that sit + * in the chapter header next to the banner are site chrome, not story + * content. Covers `Translator: X`, `Editors: A, B`, `Translator/Editor: X` + * and `[Translator – X]`. Narrow on purpose: it must start with a role + * word followed by a separator, so ordinary prose never matches. + */ +function isCreditLine(p: string): boolean { + const t = paragraphText(p); + return ( + /^\[?\s*(?:(?:translat(?:or|ors|ion)|editors?|proofreaders?|typesetters?|tlc?|qc)\s*[/&,]?\s*)+\s*(?:[:\]–—-]|by\b)/i.test( + t, + ) || /^(?:discord|ko-?fi|patreon)\s*:/i.test(t) + ); +} + +/** A block that is nothing but bold text, e.g. a header line. */ +function isBoldOnly(p: string): boolean { const inner = p - .replace(/^]*>/i, '') - .replace(/<\/p>$/i, '') + .replace(/<\/?(?:p|span)\b[^>]*>/gi, '') + .replace(//gi, '') .trim(); - return /^[\s\S]*<\/strong>$/.test(inner); + return /^<(strong|b)\b[^>]*>[\s\S]*<\/\1>$/i.test(inner); +} + +/** A divider row (`──────` or a horizontal rule) in the chapter header. */ +function isDivider(p: string): boolean { + if (//]/i.test(p) && !paragraphText(p)) return true; + return /^[─━—–_=~-]{3,}$/.test(paragraphText(p).replace(/\s+/g, '')); } function isPromoParagraph(p: string): boolean { @@ -149,12 +207,119 @@ function isPromoParagraph(p: string): boolean { if (/]/i.test(p)) return false; const t = paragraphText(p).toLowerCase(); if (!t || t === '= = =') return true; - if (t.indexOf('we tried translations') !== -1) return true; - if (t.indexOf('dsc.gg') !== -1 || t.indexOf('join our discord') !== -1) + if (/we\s*tried\s*translations/.test(t) || /^we\s*tried\s*tls$/.test(t)) + return true; + if (t.indexOf('dsc.gg') !== -1 || /join (our|the) discord/.test(t)) return true; return false; } +/** + * Decode the HTML entities that can appear inside an attribute value — + * decimal (j), hex (j) and the named entities that can smuggle a + * scheme past a prefix check (:) — so the scheme test sees what the + * reader will actually navigate to. + */ +function decodeAttrEntities(s: string): string { + return s + .replace(/&#x([0-9a-f]+);?/gi, (_m, h: string) => + String.fromCharCode(parseInt(h, 16)), + ) + .replace(/&#(\d+);?/g, (_m, n: string) => + String.fromCharCode(parseInt(n, 10)), + ) + .replace(/:?/gi, ':') + .replace(/&tab;?/gi, '\t') + .replace(/&newline;?/gi, '\n'); +} + +/** True for URLs that would execute script when followed from the reader. */ +function isScriptUrl(url: string): boolean { + // Browsers ignore whitespace and control characters inside a scheme. + const norm = decodeAttrEntities(url) + .split('') + .filter(ch => ch.charCodeAt(0) > 0x20 && ch.charCodeAt(0) !== 0x7f) + .join('') + .replace(/\s+/g, ''); + return /^(javascript|vbscript):/i.test(norm); +} + +const UNSAFE_TAGS = + 'script|iframe|object|embed|form|input|textarea|select|button|style|link|meta|base|noscript'; + +/** + * Strip anything that could execute code from chapter HTML before it + * reaches the reader, which renders it unsanitized: dangerous elements, + * event handler attributes () and script URLs in any + * link-like attribute, including SVG's xlink:href. Everything else is + * preserved as-is. + */ +function sanitizeHtml(html: string): string { + return html + .replace( + new RegExp('<(' + UNSAFE_TAGS + ')[\\s>/][\\s\\S]*?', 'gi'), + '', + ) + .replace(new RegExp(']*>', 'gi'), '') + .replace(/[\s/]on[a-z]+\s*=\s*("[^"]*"|'[^']*'|[^\s"'=<>`]+)/gi, '') + .replace( + /\s(?:xlink:)?(?:href|src|action|formaction)\s*=\s*("[^"]*"|'[^']*'|[^\s"'=<>`]+)/gi, + (m, val: string) => + isScriptUrl(val.replace(/^['"]|['"]$/g, '')) ? '' : m, + ); +} + +const VOID_TAGS = + /^(area|base|br|col|embed|hr|img|input|link|meta|param|source|track|wbr)$/i; + +/** + * Split chapter HTML into its top-level nodes in document order: each + * element (with everything nested in it, so lists, tables, blockquotes and + * divs stay intact) or run of loose text becomes one block. An unclosed + *

is ended by the next

, as an HTML parser would. + */ +function splitBlocks(html: string): string[] { + const blocks: string[] = []; + const tagRe = /<(\/?)([a-z][a-z0-9]*)\b[^>]*?(\/?)>/gi; + let depth = 0; + let openName = ''; + let blockStart = 0; + let m: RegExpExecArray | null; + const push = (from: number, to: number) => { + const b = html.slice(from, to).trim(); + if (b) blocks.push(b); + }; + while ((m = tagRe.exec(html)) !== null) { + const closing = m[1] === '/'; + const name = m[2].toLowerCase(); + const selfClosed = m[3] === '/' || VOID_TAGS.test(name); + if (depth === 0) { + // A stray close tag, or a
/ inside loose text, is not a block. + if (closing || /^(br|wbr)$/.test(name)) continue; + push(blockStart, m.index); // loose text before this element + blockStart = m.index; + if (selfClosed) { + push(blockStart, m.index + m[0].length); + blockStart = m.index + m[0].length; + } else { + depth = 1; + openName = name; + } + } else if (!closing && name === 'p' && depth === 1 && openName === 'p') { + push(blockStart, m.index); // unclosed

: end it, start a new one + blockStart = m.index; + } else if (!selfClosed) { + depth += closing ? -1 : 1; + if (depth === 0) { + push(blockStart, m.index + m[0].length); + blockStart = m.index + m[0].length; + } + } + } + push(blockStart, html.length); + return blocks; +} + /** Parse the /query API response (catalog + search share the shape). */ function parseQueryResults(jsonText: string): { items: NovelCard[]; @@ -210,31 +375,32 @@ function parseChapterList( jsonText: string, locked = false, ): { items: ChapterInfo[]; lastPage: number } { + // fetchText returns '' for a failed request. That must never read as a + // valid empty page, or parseNovel would take a failed fetch for the end + // of the list and present a truncated chapter list as complete. const root = safeJson(jsonText); + if (!isRecord(root) || !Array.isArray(root.data)) + throw new Error('Failed to load the chapter list (unexpected response)'); + const meta = root.meta; + if (!isRecord(meta) || typeof meta.last_page !== 'number') + throw new Error('Failed to load the chapter list (no pagination info)'); const items: ChapterInfo[] = []; - let lastPage = 1; - if (isRecord(root)) { - const meta = root.meta; - if (isRecord(meta) && typeof meta.last_page === 'number') - lastPage = meta.last_page; - const data = Array.isArray(root.data) ? root.data : []; - for (const c of data) { - if (!isRecord(c)) continue; - const slug = str(c.chapter_slug).trim(); - const name = str(c.chapter_name).trim(); - if (!slug || !name) continue; - const title = str(c.chapter_title).trim(); - const idx = parseFloat(str(c.index)); - items.push({ - slug, - name: title ? name + ': ' + decodeEntities(title) : name, - number: isNaN(idx) ? 0 : idx, - publishedAt: str(c.created_at), - locked, - }); - } + for (const c of root.data) { + const slug = isRecord(c) ? str(c.chapter_slug).trim() : ''; + const name = isRecord(c) ? str(c.chapter_name).trim() : ''; + if (!isRecord(c) || !slug || !name) + throw new Error('Failed to load the chapter list (malformed chapter)'); + const title = str(c.chapter_title).trim(); + const idx = parseFloat(str(c.index)); + items.push({ + slug, + name: title ? name + ': ' + decodeEntities(title) : name, + number: isNaN(idx) ? 0 : idx, + publishedAt: str(c.created_at), + locked, + }); } - return { items, lastPage }; + return { items, lastPage: meta.last_page }; } /** @@ -267,15 +433,34 @@ function shrinkIllustrations(html: string): string { } /** - * Route a cover through a fast image proxy at a list-friendly size. - * The site's own covers are up to ~1 MB (they load painfully slowly in - * the app's novel list) and the CDN offers no smaller variant, so we - * request a 400px-wide webp instead. Non-URL values pass through. + * Covers are served straight from the site's CDN. Routing them through an + * image proxy cannot carry a fallback in a single URL field, so a blocked + * or down proxy would break every cover while the CDN still works. + * Non-URL values pass through. */ function coverUrl(thumbnail: string): string { - const t = (thumbnail || '').trim(); - if (!/^https?:\/\//i.test(t)) return t; - return proxiedImageUrl(t, 400); + return (thumbnail || '').trim(); +} + +/** + * The series title, chapter name and chapter title recorded in the chapter + * page's own data, used to recognise the repeated title header. + */ +function pageTitles(flight: string): string[] { + const titles: string[] = []; + const add = (re: RegExp) => { + const m = re.exec(flight); + if (!m) return; + try { + titles.push(decodeEntities(JSON.parse('"' + m[1] + '"'))); + } catch { + /* ignore an unreadable title */ + } + }; + add(/"series":\{[^}]*?"title":"((?:[^"\\]|\\.)*)"/); + add(/"chapter_name":"((?:[^"\\]|\\.)*)"/); + add(/"chapter_title":"((?:[^"\\]|\\.)*)"/); + return titles; } type ChapterContentResult = @@ -346,20 +531,43 @@ function parseChapterContent(html: string): ChapterContentResult { .trim(); if (!body) return { status: 'empty' }; - // Split into top-level blocks — paragraphs, headings, figures and - // standalone images, in document order — and trim the site's promo - // header / footer (banner, series/chapter title repeats, discord plug). - const blocks = body.match( - /|||]*>/gi, - ) || [body]; + // Split into top-level blocks in document order and trim the site's + // promo header / footer (banner, credits, title repeats, discord plug) + // from the edges. Blocks the parser does not recognize (lists, tables, + // blockquotes) are kept whole: dropping them would lose chapter text. + const blocks = splitBlocks(body); + if (blocks.length === 0) return { status: 'empty' }; + const titles = pageTitles(flight); + const isEdgeJunk = (p: string) => + isPromoParagraph(p) || + isCreditLine(p) || + isDivider(p) || + isTitleRepeat(p, titles); let start = 0; let end = blocks.length; - const isEdgeJunk = (p: string) => isPromoParagraph(p) || isTitleRepeat(p); - while (start < end && isEdgeJunk(blocks[start])) start++; + // The site's header is a bold series / chapter title followed by a divider + // row, and those lines do not always match the catalog title exactly. A + // bold-only lead-in directly followed by a divider is header; a bold line + // without one is kept as content. + for (;;) { + while (start < end && isEdgeJunk(blocks[start])) start++; + let divider = -1; + for (let i = start; i < Math.min(end, start + 8) && divider < 0; i++) { + if (isDivider(blocks[i])) divider = i; + } + let header = divider >= 0; + for (let i = start; header && i < divider; i++) { + header = isEdgeJunk(blocks[i]) || isBoldOnly(blocks[i]); + } + if (!header) break; + start = divider + 1; + } while (end > start && isEdgeJunk(blocks[end - 1])) end--; const cleaned = blocks.slice(start, end).join('\n'); - if (!paragraphText(cleaned)) return { status: 'empty' }; - return { status: 'ok', html: shrinkIllustrations(cleaned) }; + // An image-only chapter (illustrations with no text) is still content. + if (!paragraphText(cleaned) && !/]/i.test(cleaned)) + return { status: 'empty' }; + return { status: 'ok', html: sanitizeHtml(shrinkIllustrations(cleaned)) }; } function mapStatus(s: string): string { @@ -399,7 +607,7 @@ class WeTriedTLS implements Plugin.PluginBase { name = 'We Tried TLS'; icon = 'src/en/wetriedtls/icon.png'; site = SITE; - version = '1.0.4'; + version = '1.1.0'; filters = { status: { From 9c8e3bb45cac52beb013fbf564be16109a4542b5 Mon Sep 17 00:00:00 2001 From: Raiyn Aydin Date: Tue, 29 Sep 2026 06:54:49 +0800 Subject: [PATCH 2/3] fix(english/wetriedtls): keep story inside wrappers and bold scene openings --- plugins/english/wetriedtls.ts | 39 +++++++++++++++++++++++++++++------ 1 file changed, 33 insertions(+), 6 deletions(-) diff --git a/plugins/english/wetriedtls.ts b/plugins/english/wetriedtls.ts index 117e3f2df..6cb1e3379 100644 --- a/plugins/english/wetriedtls.ts +++ b/plugins/english/wetriedtls.ts @@ -186,6 +186,21 @@ function isCreditLine(p: string): boolean { ); } +/** + * splitTopLevel, with plain wrapper elements (div / section / article) + * unwrapped so a header and the story inside one wrapper are judged block + * by block instead of as a single unit. + */ +function splitBlocks(html: string): string[] { + const out: string[] = []; + for (const b of splitTopLevel(html)) { + const w = /^<(div|section|article)\b[^>]*>([\s\S]*)<\/\1\s*>$/i.exec(b); + if (w) for (const inner of splitBlocks(w[2])) out.push(inner); + else out.push(b); + } + return out; +} + /** A block that is nothing but bold text, e.g. a header line. */ function isBoldOnly(p: string): boolean { const inner = p @@ -278,7 +293,7 @@ const VOID_TAGS = * divs stay intact) or run of loose text becomes one block. An unclosed *

is ended by the next

, as an HTML parser would. */ -function splitBlocks(html: string): string[] { +function splitTopLevel(html: string): string[] { const blocks: string[] = []; const tagRe = /<(\/?)([a-z][a-z0-9]*)\b[^>]*?(\/?)>/gi; let depth = 0; @@ -538,11 +553,23 @@ function parseChapterContent(html: string): ChapterContentResult { const blocks = splitBlocks(body); if (blocks.length === 0) return { status: 'empty' }; const titles = pageTitles(flight); + // Site chrome is a short plain line. A long block, or a list / table / + // blockquote that merely contains a banner or credit, is story text and + // must never be trimmed as junk. const isEdgeJunk = (p: string) => - isPromoParagraph(p) || - isCreditLine(p) || - isDivider(p) || - isTitleRepeat(p, titles); + !/^<(?:blockquote|ul|ol|table|pre)\b/i.test(p) && + paragraphText(p).length <= 250 && + (isPromoParagraph(p) || + isCreditLine(p) || + isDivider(p) || + isTitleRepeat(p, titles)); + // A bold lead-in only counts as header when it reads like a title line. + const isHeaderLine = (p: string) => + isEdgeJunk(p) || + (isBoldOnly(p) && + /^[◈◆■●\s]*(?:chapter|ch\.?|vol\.?|volume|episode|ep\.?|prologue|epilogue|side story|extra)\b|^[◈◆■●]|[◈◆■●]$/i.test( + paragraphText(p), + )); let start = 0; let end = blocks.length; // The site's header is a bold series / chapter title followed by a divider @@ -557,7 +584,7 @@ function parseChapterContent(html: string): ChapterContentResult { } let header = divider >= 0; for (let i = start; header && i < divider; i++) { - header = isEdgeJunk(blocks[i]) || isBoldOnly(blocks[i]); + header = isHeaderLine(blocks[i]); } if (!header) break; start = divider + 1; From 724c5af694bb827d33fbafc3d683af85247a81ab Mon Sep 17 00:00:00 2001 From: Raiyn Aydin Date: Tue, 29 Sep 2026 09:51:57 +0800 Subject: [PATCH 3/3] fix(english/wetriedtls): harden sanitizer and strip untitled chapter headers - sanitize attributes by tokenizing tags as a browser does, so prose such as "one = 1" is no longer deleted and handlers or script URLs that directly follow a quoted value are no longer missed; also drop comments, raw-text elements and script URLs in SVG animation attributes - treat a short header line that opens with a known title as header, so "Series
Chapter 88: ..." and "Title (1)" above a divider are stripped when the page data lacks the chapter title - only accept a dash after a credit role word when it is spaced, so "Editor-in-chief ..." is not trimmed as a credit line --- plugins/english/wetriedtls.ts | 120 ++++++++++++++++++++++++++++------ 1 file changed, 100 insertions(+), 20 deletions(-) diff --git a/plugins/english/wetriedtls.ts b/plugins/english/wetriedtls.ts index 6cb1e3379..27e3b824f 100644 --- a/plugins/english/wetriedtls.ts +++ b/plugins/english/wetriedtls.ts @@ -175,12 +175,13 @@ function isTitleRepeat(p: string, knownTitles: string[]): boolean { * in the chapter header next to the banner are site chrome, not story * content. Covers `Translator: X`, `Editors: A, B`, `Translator/Editor: X` * and `[Translator – X]`. Narrow on purpose: it must start with a role - * word followed by a separator, so ordinary prose never matches. + * word followed by a separator (a dash only when spaced, so prose such as + * "Editor-in-chief Kim ..." never matches). */ function isCreditLine(p: string): boolean { const t = paragraphText(p); return ( - /^\[?\s*(?:(?:translat(?:or|ors|ion)|editors?|proofreaders?|typesetters?|tlc?|qc)\s*[/&,]?\s*)+\s*(?:[:\]–—-]|by\b)/i.test( + /^\[?\s*(?:(?:translat(?:or|ors|ion)|editors?|proofreaders?|typesetters?|tlc?|qc)\s*[/&,]?\s*)+\s*(?:[:\]]|[–—-]\s|by\b)/i.test( t, ) || /^(?:discord|ko-?fi|patreon)\s*:/i.test(t) ); @@ -259,29 +260,99 @@ function isScriptUrl(url: string): boolean { return /^(javascript|vbscript):/i.test(norm); } +// Dangerous elements, plus the raw-text elements (title, xmp, ...) whose +// content a browser reads as text: tags inside them must not be judged as +// real tags by the attribute pass below. const UNSAFE_TAGS = - 'script|iframe|object|embed|form|input|textarea|select|button|style|link|meta|base|noscript'; + 'script|iframe|object|embed|form|input|textarea|select|button|style|link|meta|base|noscript|title|xmp|noembed|noframes|plaintext'; + +/** Attributes whose value the reader may navigate to. */ +const URL_ATTRS = /^(?:xlink:)?(?:href|src|action|formaction)$/; + +/** SVG animation attributes that can write a script URL into an href. */ +const ANIMATION_ATTRS = /^(?:values|to|from)$/; + +function isUnsafeAttr(name: string, value: string): boolean { + const n = name.toLowerCase(); + if (n.indexOf('on') === 0) return true; + if (URL_ATTRS.test(n)) return isScriptUrl(value); + if (ANIMATION_ATTRS.test(n)) return value.split(';').some(isScriptUrl); + return false; +} /** - * Strip anything that could execute code from chapter HTML before it - * reaches the reader, which renders it unsanitized: dangerous elements, - * event handler attributes () and script URLs in any - * link-like attribute, including SVG's xlink:href. Everything else is - * preserved as-is. + * Strip anything that could execute code from chapter HTML: dangerous + * elements, event handler attributes (), and script URLs in + * link-like or SVG animation attributes. The reader sanitizes too; this + * keeps the plugin's own output safe. Everything else is preserved as-is. */ function sanitizeHtml(html: string): string { - return html - .replace( - new RegExp('<(' + UNSAFE_TAGS + ')[\\s>/][\\s\\S]*?', 'gi'), - '', - ) - .replace(new RegExp(']*>', 'gi'), '') - .replace(/[\s/]on[a-z]+\s*=\s*("[^"]*"|'[^']*'|[^\s"'=<>`]+)/gi, '') - .replace( - /\s(?:xlink:)?(?:href|src|action|formaction)\s*=\s*("[^"]*"|'[^']*'|[^\s"'=<>`]+)/gi, - (m, val: string) => - isScriptUrl(val.replace(/^['"]|['"]$/g, '')) ? '' : m, - ); + // Comments and other markup declarations hold no story text, and a quote + // inside one must not hide a following tag from the attribute pass. + // Repeat until stable so removing one tag cannot splice a new one together. + let s = html; + for (let prev = ''; prev !== s; ) { + prev = s; + s = s + .replace(/