Skip to content
Open
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
115 changes: 63 additions & 52 deletions plugins/multisrc/lightnovelwp/template.ts
Original file line number Diff line number Diff line change
Expand Up @@ -42,7 +42,7 @@ export class LightNovelWPPlugin implements Plugin.PluginBase {
this.icon = `multisrc/lightnovelwp/${metadata.id.toLowerCase()}/icon.png`;
this.site = metadata.sourceSite;
const versionIncrements = metadata.options?.versionIncrements || 0;
this.version = `1.1.${10 + versionIncrements}`;
this.version = `1.1.${11 + versionIncrements}`;
this.options = metadata.options ?? ({} as LightNovelWPOptions);
this.filters = metadata.filters satisfies Filters;

Expand Down Expand Up @@ -438,64 +438,75 @@ export class LightNovelWPPlugin implements Plugin.PluginBase {
throw error;
}
}
// The chapter text lives in the `epcontent` block. Older child themes
// closed it with a `bottomnav` div, but that wrapper is gone on current
// ones, so anchoring on it silently returned an empty chapter.
//
// Walk the markup with the parser rather than cutting the document with a
// regex: the block has to end at the element that closes it, and any
// landmark guessed from the surrounding text is wrong on some site. A
// <script> inside the block is prose-adjacent, not the end of it, so
// treating the first one as a boundary truncates the chapter.
let depth = 0;
let inside = false;
let found = false;
let isReadingParagraph = false;
let paragraph = '';
const paragraphs: string[] = [];
// The chapter lives in the `epcontent` block. Older child themes closed
// it with a `bottomnav` div, but that wrapper is gone on current ones, so
// select the element itself rather than cutting the document between
// landmarks: a <script> inside the block then cannot end it early either.
const $ = load(data);
const content = $('div.epcontent').first();

const parser = new Parser({
onopentag(name, attribs) {
if (name === 'div' && !inside) {
const className = attribs['class'] || '';
if (/(^|\s)epcontent(\s|$)/.test(className)) {
inside = true;
found = true;
depth = 1;
}
// No epcontent block means this was not a chapter page (a login wall or
// an error page served with 200). Returning the whole document here would
// show that page's text as the chapter, so report nothing instead.
if (!content.length) return '';

content.find('script, style, noscript').remove();
const pageUrl = this.site + chapterPath;
content.find('img').each((_, el) => {
const img = $(el);
let src = img.attr('src');
if (!src || src.startsWith('data:')) {
src = img.attr('data-lazy-src') || img.attr('data-src') || src;
}
if (!src) return;
// Resolve relative paths against the chapter page, as a browser would.
if (!/^[a-z][a-z\d+.-]*:/i.test(src)) {
try {
src = new URL(src, pageUrl).href;
} catch {
return;
}
if (!inside) return;
if (name === 'div') depth++;
else if (name === 'p') isReadingParagraph = true;
},
ontext(text) {
if (inside && isReadingParagraph && text.trim()) {
paragraph += text;
}
},
onclosetag(name) {
if (!inside) return;
if (name === 'p') {
if (paragraph.trim()) paragraphs.push(paragraph.trim());
paragraph = '';
isReadingParagraph = false;
} else if (name === 'div') {
depth--;
if (depth === 0) inside = false;
}
},
}
img.attr('src', src);
});

parser.write(data);
parser.end();
// The app renders this HTML as is, so drop event handlers and URLs that
// would run script.
content.find('*').each((_, el) => {
const node = $(el);
for (const name of Object.keys(el.attribs)) {
if (/^on/i.test(name)) {
node.removeAttr(name);
} else if (name === 'href' || name === 'src') {
const value = el.attribs[name]
.split('')
.filter(c => c.charCodeAt(0) > 32)
.join('')
.toLowerCase();
if (
/^(javascript|vbscript):/.test(value) ||
(value.startsWith('data:') && !value.startsWith('data:image/'))
) {
node.removeAttr(name);
}
}
}
});

// No epcontent block means this was not a chapter page (a login wall or
// an error page served with 200). Returning the whole document here would
// show that page's text as the chapter, so report nothing instead.
if (!found) return '';
// Keep paragraphs as HTML so inline formatting and images inside them
// survive, plus images outside any paragraph: illustration chapters hold
// only images, sometimes in bare <div>s with no <p> at all.
const blocks: string[] = [];
content.find('p, img').each((_, el) => {
const node = $(el);
if (node.is('img')) {
if (!node.parents('p').length) blocks.push($.html(node));
} else if (node.text().trim() || node.find('img').length) {
blocks.push($.html(node));
Comment on lines +503 to +505

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 security Chapter HTML can run code

A chapter page can put an onerror attribute on an image or a script URL in a link. The new parser keeps those values, and the in-repo chapter viewer inserts the returned HTML without cleaning it. Strip executable attributes and unsafe URLs before returning the HTML.

How this was verified: Source-controlled image attributes pass through $.html(node) into the viewer’s unsanitized HTML insertion.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Fixed in 2baad3b. Before returning, every element in the kept block has its on* attributes removed. Any href/src using a javascript:, vbscript: or non-image data: scheme is also removed; the check ignores case, whitespace and control characters. I tested it with a synthetic page (onerror/ONLOAD/onclick, a tab-obfuscated JaVaScript: href, vbscript:, a data:text/html href, a javascript: img src). All were stripped, and https: links and data:image/* images were kept.

}
});

return paragraphs.join('\n');
return blocks.join('\n');
}
async searchNovels(
searchTerm: string,
Expand Down
Loading