From 750c2824ce70f7afe0f6659ed9760fbe489b33f0 Mon Sep 17 00:00:00 2001 From: masi Date: Fri, 25 Sep 2026 17:18:05 -0300 Subject: [PATCH] feat(scripts): strip translator credits and promo lines from fetched and existing EPUBs Co-Authored-By: Claude Opus 5.5 --- scripts/clean-chapter.js | 38 +++++++++++++++++++++++++++++++++++++ scripts/clean-epub.js | 41 ++++++++++++++++++++++++++++++++++++++++ scripts/fetch-novel.js | 4 +++- 3 files changed, 82 insertions(+), 1 deletion(-) create mode 100644 scripts/clean-chapter.js create mode 100644 scripts/clean-epub.js diff --git a/scripts/clean-chapter.js b/scripts/clean-chapter.js new file mode 100644 index 0000000..950b412 --- /dev/null +++ b/scripts/clean-chapter.js @@ -0,0 +1,38 @@ +// Translator credits and promo lines that sites paste into every chapter +// ("Translator: …", "Discord: https://…", "Join the Discord…", ko-fi/Patreon). +// A paragraph is dropped only when it is short and the whole thing is promo, so +// story text that happens to mention a word like "discord" stays. +// The illustrator's reader.js carries the same rule for in-app reading. +const MAX_PROMO_LENGTH = 200; +const URL = /https?:\/\/\S+|\b(dsc|discord)\.gg\/\S*/gi; +const MAX_LINK_LABEL = 60; // "Discord: ", "Ko-Fi: ", but not prose with a link +const PROMO = [ + /^(discord|patreon|ko-?fi)\b\s*:?/i, + /\b(join|support)\b.{0,40}\b(discord|patreon|ko-?fi)\b/i, + /\bko-?fi\b|\bpatreon\b/i, + /^(translators?|translated by|translation|editors?|edited by|editing|proofreaders?|proofread by|tl|tln|pr|ed)(\s*\/\s*\w+)*\s*[:-]/i, + /\bread (the )?(latest|advanced?|more) chapters?\b/i, +]; + +export const isPromo = text => { + const t = text.replace(/\s+/g, ' ').trim(); + if (!t || t.length > MAX_PROMO_LENGTH) return false; + const label = t.replace(URL, '').trim(); + if (label.length < t.length && label.length <= MAX_LINK_LABEL) return true; + return PROMO.some(r => r.test(t)); +}; + +/** Remove promo paragraphs from a cheerio document; returns how many. */ +export const stripPromo = $ => { + let removed = 0; + $('p, div, h3, h4, h5, h6, li').each((_, el) => { + const node = $(el); + // Innermost blocks only, so a wrapper holding the whole chapter is never dropped. + if (node.find('p, div, li').length) return; + if (isPromo(node.text())) { + node.remove(); + removed++; + } + }); + return removed; +}; diff --git a/scripts/clean-epub.js b/scripts/clean-epub.js new file mode 100644 index 0000000..90da805 --- /dev/null +++ b/scripts/clean-epub.js @@ -0,0 +1,41 @@ +// Strip translator credits and promo lines from an existing EPUB, in place. +// Usage: node scripts/clean-epub.js ... +import JSZip from 'jszip'; +import { load } from 'cheerio'; +import fs from 'fs'; +import { stripPromo } from './clean-chapter.js'; + +const files = process.argv.slice(2); +if (!files.length) { + console.error('Usage: node scripts/clean-epub.js ...'); + process.exit(2); +} + +for (const file of files) { + const zip = await JSZip.loadAsync(fs.readFileSync(file)); + let removed = 0; + for (const name of Object.keys(zip.files)) { + if (!/\.x?html?$/i.test(name) || /(^|\/)nav\.xhtml$/i.test(name)) continue; + const $ = load(await zip.file(name).async('string'), { xml: true }); + const n = stripPromo($); + if (n) { + zip.file(name, $.xml()); + removed += n; + } + } + if (removed) { + const tmp = `${file}.tmp`; + // EPUB requires mimetype stored uncompressed; loaded entries lose that option. + zip.file('mimetype', 'application/epub+zip', { compression: 'STORE' }); + fs.writeFileSync( + tmp, + await zip.generateAsync({ + type: 'nodebuffer', + compression: 'DEFLATE', + mimeType: 'application/epub+zip', + }), + ); + fs.renameSync(tmp, file); + } + process.stdout.write(`${file}: removed ${removed} promo paragraphs\n`); +} diff --git a/scripts/fetch-novel.js b/scripts/fetch-novel.js index 94bf022..7921b82 100644 --- a/scripts/fetch-novel.js +++ b/scripts/fetch-novel.js @@ -6,6 +6,7 @@ import { load } from 'cheerio'; import fs from 'fs'; import path from 'path'; import { loadPlugins, print, ROOT } from './load-plugins.js'; +import { stripPromo } from './clean-chapter.js'; // Be gentle with the source site: few parallel requests, retries with backoff. const CONCURRENCY = 3; @@ -51,10 +52,11 @@ const esc = s => ); // Chapter HTML is tag soup; re-serialise it as well-formed XHTML without -// scripts or remote resources (an EPUB must be self-contained). +// scripts, remote resources (an EPUB must be self-contained) or promo lines. const toXhtml = html => { const $ = load(html, null, false); $('script, style, iframe, link, meta, img, video, audio, source').remove(); + stripPromo($); return $.xml(); };