feat: fetch-novel builds an EPUB from any plugin; Kavita TOC fix
Publish Plugins / Publish Plugins (push) Has been cancelled
Publish Plugins / Publish Plugins (push) Has been cancelled
- scripts/load-plugins.js: shared plugin loader for the novel CLI scripts - scripts/fetch-novel.js: download a novel via its plugin into a clean EPUB 3 - find-novel prints the plugin id/path for each match - kavita 1.0.1: innermost TOC entry names the chapter Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,178 @@
|
||||
// Download a whole novel through its plugin and write it as an EPUB 3 file.
|
||||
// Usage: node scripts/fetch-novel.js <plugin-id> <novel-path> <out-dir>
|
||||
// Prints the written file path as the last stdout line.
|
||||
import JSZip from 'jszip';
|
||||
import { load } from 'cheerio';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import { loadPlugins, print, ROOT } from './load-plugins.js';
|
||||
|
||||
// Be gentle with the source site: few parallel requests, retries with backoff.
|
||||
const CONCURRENCY = 3;
|
||||
const RETRIES = 4;
|
||||
|
||||
const [pluginId, novelPath, outDir] = process.argv.slice(2);
|
||||
if (!pluginId || !novelPath || !outDir) {
|
||||
console.error(
|
||||
'Usage: node scripts/fetch-novel.js <plugin-id> <novel-path> <out-dir>',
|
||||
);
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
const langs = fs
|
||||
.readdirSync(path.join(ROOT, 'plugins'), { withFileTypes: true })
|
||||
.filter(d => d.isDirectory() && d.name !== 'multisrc')
|
||||
.map(d => d.name);
|
||||
const plugin = (await loadPlugins(langs)).find(p => p.id === pluginId);
|
||||
if (!plugin) throw new Error(`No plugin with id "${pluginId}"`);
|
||||
|
||||
const retry = async (label, fn) => {
|
||||
for (let attempt = 1; ; attempt++) {
|
||||
try {
|
||||
return await fn();
|
||||
} catch (e) {
|
||||
if (attempt >= RETRIES) throw new Error(`${label}: ${e.message}`);
|
||||
await new Promise(r => setTimeout(r, 1000 * 2 ** attempt));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const novel = await retry('novel', () => plugin.parseNovel(novelPath));
|
||||
const chapters = novel.chapters || [];
|
||||
if (!chapters.length) throw new Error('Novel has no chapters');
|
||||
process.stderr.write(
|
||||
`${novel.name}: ${chapters.length} chapters from ${plugin.name}\n`,
|
||||
);
|
||||
|
||||
const esc = s =>
|
||||
String(s ?? '').replace(
|
||||
/[&<>"]/g,
|
||||
c => ({ '&': '&', '<': '<', '>': '>', '"': '"' })[c],
|
||||
);
|
||||
|
||||
// Chapter HTML is tag soup; re-serialise it as well-formed XHTML without
|
||||
// scripts or remote resources (an EPUB must be self-contained).
|
||||
const toXhtml = html => {
|
||||
const $ = load(html, null, false);
|
||||
$('script, style, iframe, link, meta, img, video, audio, source').remove();
|
||||
return $.xml();
|
||||
};
|
||||
|
||||
const bodies = new Array(chapters.length);
|
||||
let done = 0;
|
||||
const queue = chapters.map((c, i) => i);
|
||||
await Promise.all(
|
||||
Array.from({ length: CONCURRENCY }, async () => {
|
||||
for (let i; (i = queue.shift()) !== undefined;) {
|
||||
const html = await retry(chapters[i].name, () =>
|
||||
plugin.parseChapter(chapters[i].path),
|
||||
);
|
||||
bodies[i] = toXhtml(html);
|
||||
if (++done % 50 === 0 || done === chapters.length) {
|
||||
process.stderr.write(` ${done}/${chapters.length}\n`);
|
||||
}
|
||||
}
|
||||
}),
|
||||
);
|
||||
|
||||
const zip = new JSZip();
|
||||
zip.file('mimetype', 'application/epub+zip', { compression: 'STORE' });
|
||||
zip.file(
|
||||
'META-INF/container.xml',
|
||||
`<?xml version="1.0" encoding="UTF-8"?>
|
||||
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
|
||||
<rootfiles><rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/></rootfiles>
|
||||
</container>`,
|
||||
);
|
||||
|
||||
const page = (title, body) => `<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html>
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops" lang="en" xml:lang="en">
|
||||
<head><meta charset="UTF-8"/><title>${esc(title)}</title></head>
|
||||
<body>${body}</body>
|
||||
</html>`;
|
||||
|
||||
const ids = chapters.map((_, i) => `ch${String(i + 1).padStart(4, '0')}`);
|
||||
chapters.forEach((c, i) => {
|
||||
zip.file(
|
||||
`OEBPS/${ids[i]}.xhtml`,
|
||||
page(c.name, `<h2>${esc(c.name)}</h2>\n${bodies[i]}`),
|
||||
);
|
||||
});
|
||||
|
||||
const toc = chapters
|
||||
.map((c, i) => `<li><a href="${ids[i]}.xhtml">${esc(c.name)}</a></li>`)
|
||||
.join('\n');
|
||||
zip.file(
|
||||
'OEBPS/nav.xhtml',
|
||||
page('Contents', `<nav epub:type="toc"><ol>\n${toc}\n</ol></nav>`),
|
||||
);
|
||||
|
||||
let cover = '';
|
||||
if (novel.cover && /^https?:/.test(novel.cover)) {
|
||||
try {
|
||||
const res = await fetch(novel.cover, { headers: { Referer: plugin.site } });
|
||||
const type = (res.headers.get('content-type') || '').split(';')[0];
|
||||
const ext = {
|
||||
'image/jpeg': 'jpg',
|
||||
'image/png': 'png',
|
||||
'image/webp': 'webp',
|
||||
'image/gif': 'gif',
|
||||
}[type];
|
||||
if (res.ok && ext) {
|
||||
zip.file(`OEBPS/cover.${ext}`, Buffer.from(await res.arrayBuffer()));
|
||||
cover = `<item id="cover" href="cover.${ext}" media-type="${type}" properties="cover-image"/>`;
|
||||
}
|
||||
} catch {
|
||||
// A missing cover is not worth failing the book for.
|
||||
}
|
||||
}
|
||||
|
||||
const uid = `urn:lnreader:${plugin.id}:${novelPath}`;
|
||||
const subjects = (novel.genres || '')
|
||||
.split(',')
|
||||
.map(g => g.trim())
|
||||
.filter(Boolean)
|
||||
.map(g => `<dc:subject>${esc(g)}</dc:subject>`)
|
||||
.join('\n ');
|
||||
zip.file(
|
||||
'OEBPS/content.opf',
|
||||
`<?xml version="1.0" encoding="UTF-8"?>
|
||||
<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="uid">
|
||||
<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
|
||||
<dc:identifier id="uid">${esc(uid)}</dc:identifier>
|
||||
<dc:title>${esc(novel.name)}</dc:title>
|
||||
<dc:language>en</dc:language>
|
||||
${novel.author ? `<dc:creator>${esc(novel.author)}</dc:creator>` : ''}
|
||||
${novel.summary ? `<dc:description>${esc(novel.summary)}</dc:description>` : ''}
|
||||
<dc:publisher>${esc(plugin.name)}</dc:publisher>
|
||||
<dc:source>${esc(plugin.resolveUrl ? plugin.resolveUrl(novelPath, true) : plugin.site)}</dc:source>
|
||||
${subjects}
|
||||
<meta property="dcterms:modified">${new Date().toISOString().replace(/\.\d+Z$/, 'Z')}</meta>
|
||||
</metadata>
|
||||
<manifest>
|
||||
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" properties="nav"/>
|
||||
${cover}
|
||||
${ids.map(id => `<item id="${id}" href="${id}.xhtml" media-type="application/xhtml+xml"/>`).join('\n ')}
|
||||
</manifest>
|
||||
<spine>
|
||||
${ids.map(id => `<itemref idref="${id}"/>`).join('\n ')}
|
||||
</spine>
|
||||
</package>`,
|
||||
);
|
||||
|
||||
fs.mkdirSync(outDir, { recursive: true });
|
||||
const file = path.join(
|
||||
outDir,
|
||||
`${novel.name.replace(/[/\\:*?"<>|]/g, '-')}.epub`,
|
||||
);
|
||||
fs.writeFileSync(
|
||||
file,
|
||||
await zip.generateAsync({
|
||||
type: 'nodebuffer',
|
||||
compression: 'DEFLATE',
|
||||
mimeType: 'application/epub+zip',
|
||||
}),
|
||||
);
|
||||
print(file);
|
||||
process.exit(0);
|
||||
Reference in New Issue
Block a user