feat(scripts): strip translator credits and promo lines from fetched and existing EPUBs

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
masi
2026-09-25 17:18:05 -03:00
parent fba9a35075
commit 750c2824ce
3 changed files with 82 additions and 1 deletions
+38
View File
@@ -0,0 +1,38 @@
// Translator credits and promo lines that sites paste into every chapter
// ("Translator: …", "Discord: https://…", "Join the Discord…", ko-fi/Patreon).
// A paragraph is dropped only when it is short and the whole thing is promo, so
// story text that happens to mention a word like "discord" stays.
// The illustrator's reader.js carries the same rule for in-app reading.
const MAX_PROMO_LENGTH = 200;
const URL = /https?:\/\/\S+|\b(dsc|discord)\.gg\/\S*/gi;
const MAX_LINK_LABEL = 60; // "Discord: <url>", "Ko-Fi: <url>", but not prose with a link
const PROMO = [
/^(discord|patreon|ko-?fi)\b\s*:?/i,
/\b(join|support)\b.{0,40}\b(discord|patreon|ko-?fi)\b/i,
/\bko-?fi\b|\bpatreon\b/i,
/^(translators?|translated by|translation|editors?|edited by|editing|proofreaders?|proofread by|tl|tln|pr|ed)(\s*\/\s*\w+)*\s*[:-]/i,
/\bread (the )?(latest|advanced?|more) chapters?\b/i,
];
export const isPromo = text => {
const t = text.replace(/\s+/g, ' ').trim();
if (!t || t.length > MAX_PROMO_LENGTH) return false;
const label = t.replace(URL, '').trim();
if (label.length < t.length && label.length <= MAX_LINK_LABEL) return true;
return PROMO.some(r => r.test(t));
};
/** Remove promo paragraphs from a cheerio document; returns how many. */
export const stripPromo = $ => {
let removed = 0;
$('p, div, h3, h4, h5, h6, li').each((_, el) => {
const node = $(el);
// Innermost blocks only, so a wrapper holding the whole chapter is never dropped.
if (node.find('p, div, li').length) return;
if (isPromo(node.text())) {
node.remove();
removed++;
}
});
return removed;
};
+41
View File
@@ -0,0 +1,41 @@
// Strip translator credits and promo lines from an existing EPUB, in place.
// Usage: node scripts/clean-epub.js <file.epub>...
import JSZip from 'jszip';
import { load } from 'cheerio';
import fs from 'fs';
import { stripPromo } from './clean-chapter.js';
const files = process.argv.slice(2);
if (!files.length) {
console.error('Usage: node scripts/clean-epub.js <file.epub>...');
process.exit(2);
}
for (const file of files) {
const zip = await JSZip.loadAsync(fs.readFileSync(file));
let removed = 0;
for (const name of Object.keys(zip.files)) {
if (!/\.x?html?$/i.test(name) || /(^|\/)nav\.xhtml$/i.test(name)) continue;
const $ = load(await zip.file(name).async('string'), { xml: true });
const n = stripPromo($);
if (n) {
zip.file(name, $.xml());
removed += n;
}
}
if (removed) {
const tmp = `${file}.tmp`;
// EPUB requires mimetype stored uncompressed; loaded entries lose that option.
zip.file('mimetype', 'application/epub+zip', { compression: 'STORE' });
fs.writeFileSync(
tmp,
await zip.generateAsync({
type: 'nodebuffer',
compression: 'DEFLATE',
mimeType: 'application/epub+zip',
}),
);
fs.renameSync(tmp, file);
}
process.stdout.write(`${file}: removed ${removed} promo paragraphs\n`);
}
+3 -1
View File
@@ -6,6 +6,7 @@ import { load } from 'cheerio';
import fs from 'fs';
import path from 'path';
import { loadPlugins, print, ROOT } from './load-plugins.js';
import { stripPromo } from './clean-chapter.js';
// Be gentle with the source site: few parallel requests, retries with backoff.
const CONCURRENCY = 3;
@@ -51,10 +52,11 @@ const esc = s =>
);
// Chapter HTML is tag soup; re-serialise it as well-formed XHTML without
// scripts or remote resources (an EPUB must be self-contained).
// scripts, remote resources (an EPUB must be self-contained) or promo lines.
const toXhtml = html => {
const $ = load(html, null, false);
$('script, style, iframe, link, meta, img, video, audio, source').remove();
stripPromo($);
return $.xml();
};