fix(inoveltranslation): Move Icon to Static Path and Update Lexical JSON Extraction Algorithm (#2160)
* fix(inoveltranslation): move icon to correct static path * bump: inoveltranslation version to 1.0.1 * fix(inoveltranslation): robust lexical extraction and html fallback --------- Co-authored-by: 7ui77 <root@localhost.localdomain>
This commit is contained in:
@@ -11,7 +11,7 @@ class INovelTranslation implements Plugin.PluginBase {
|
|||||||
name = 'iNovelTranslation';
|
name = 'iNovelTranslation';
|
||||||
icon = 'src/en/inoveltranslation/icon.png';
|
icon = 'src/en/inoveltranslation/icon.png';
|
||||||
site = 'https://inoveltranslation.com';
|
site = 'https://inoveltranslation.com';
|
||||||
version = '1.0.0';
|
version = '1.0.1';
|
||||||
filters: Filters | undefined = undefined;
|
filters: Filters | undefined = undefined;
|
||||||
|
|
||||||
pluginSettings = {
|
pluginSettings = {
|
||||||
@@ -170,35 +170,46 @@ class INovelTranslation implements Plugin.PluginBase {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// ==========================================
|
// ==========================================
|
||||||
// 2. DEEP LEXICAL EXTRACTION ALGORITHM
|
// 2. ROBUST LEXICAL EXTRACTION ALGORITHM
|
||||||
// ==========================================
|
// ==========================================
|
||||||
|
|
||||||
// 2.1. Basic cleanup of escaped characters in the RSC stream
|
// Use a more reliable signature: the start of the root Lexical object
|
||||||
let cleanText = rscText.replace(/\\"/g, '"').replace(/\\\\/g, '\\');
|
const signatures = [
|
||||||
|
'"root":{"type":"root"',
|
||||||
|
'\\"root\\":{\\"type\\":\\"root\\"',
|
||||||
|
'"children":[{"type":"paragraph"',
|
||||||
|
'\\"children\\":[{\\"type\\":\\"paragraph\\"',
|
||||||
|
];
|
||||||
|
|
||||||
// 2.2. Locate the core content signature (first paragraph children)
|
let sigIndex = -1;
|
||||||
const signature = '"children":[{"type":"paragraph"';
|
for (const sig of signatures) {
|
||||||
let sigIndex = cleanText.indexOf(signature);
|
sigIndex = rscText.indexOf(sig);
|
||||||
|
if (sigIndex !== -1) break;
|
||||||
|
}
|
||||||
|
|
||||||
if (sigIndex !== -1) {
|
if (sigIndex !== -1) {
|
||||||
// 2.3. Backtrack to find the opening brace { of the Lexical Object
|
// Backtrack to find the opening brace { of the Lexical Object
|
||||||
let startIndex = cleanText.lastIndexOf('{', sigIndex);
|
let startIndex = rscText.lastIndexOf('{', sigIndex);
|
||||||
|
|
||||||
// Check if it is within a "root": { ... } object to backtrack one level further
|
// Check for "content" or "root" before to find the start of the relevant object
|
||||||
const rootIndex = cleanText.lastIndexOf('"root"', sigIndex);
|
const contextKeys = ['"content"', '\\"content\\"', '"root"', '\\"root\\"'];
|
||||||
if (rootIndex !== -1 && rootIndex > startIndex - 30) {
|
for (const key of contextKeys) {
|
||||||
startIndex = cleanText.lastIndexOf('{', rootIndex);
|
const keyIndex = rscText.lastIndexOf(key, sigIndex);
|
||||||
|
if (keyIndex !== -1 && keyIndex > startIndex - 50) {
|
||||||
|
startIndex = rscText.lastIndexOf('{', keyIndex);
|
||||||
|
break;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (startIndex !== -1) {
|
if (startIndex !== -1) {
|
||||||
// 2.4. High-performance Brace Balancing algorithm
|
|
||||||
let braces = 0;
|
let braces = 0;
|
||||||
let inString = false;
|
let inString = false;
|
||||||
let escape = false;
|
let escape = false;
|
||||||
let jsonStr = '';
|
let jsonStr = '';
|
||||||
|
|
||||||
for (let i = startIndex; i < cleanText.length; i++) {
|
// Perform brace balancing on the raw stream to preserve escaping
|
||||||
const char = cleanText[i];
|
for (let i = startIndex; i < rscText.length; i++) {
|
||||||
|
const char = rscText[i];
|
||||||
if (escape) {
|
if (escape) {
|
||||||
escape = false;
|
escape = false;
|
||||||
continue;
|
continue;
|
||||||
@@ -218,59 +229,70 @@ class INovelTranslation implements Plugin.PluginBase {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if (braces === 0 && i > startIndex) {
|
if (braces === 0 && i > startIndex) {
|
||||||
jsonStr = cleanText.substring(startIndex, i + 1);
|
jsonStr = rscText.substring(startIndex, i + 1);
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (jsonStr) {
|
if (jsonStr) {
|
||||||
try {
|
try {
|
||||||
// 2.5 Standardize and Parse JSON
|
// Attempt to parse directly (if it's unescaped RSC)
|
||||||
// Strip control characters that might break JSON.parse
|
let safeJson = jsonStr.replace(/[\x00-\x1F\x7F-\x9F]/g, '');
|
||||||
const safeJson = jsonStr.replace(/[\x00-\x1F\x7F-\x9F]/g, '');
|
let parsedData;
|
||||||
const parsedData = JSON.parse(safeJson);
|
try {
|
||||||
let lexicalRoot = parsedData.root || parsedData;
|
parsedData = JSON.parse(safeJson);
|
||||||
|
} catch {
|
||||||
|
// If fails, it might be escaped, so clean it up and try again
|
||||||
|
const cleanJson = jsonStr
|
||||||
|
.replace(/\\"/g, '"')
|
||||||
|
.replace(/\\\\/g, '\\')
|
||||||
|
.replace(/[\x00-\x1F\x7F-\x9F]/g, '');
|
||||||
|
parsedData = JSON.parse(cleanJson);
|
||||||
|
}
|
||||||
|
|
||||||
|
let lexicalRoot = parsedData.root || parsedData.content?.root || parsedData;
|
||||||
if (lexicalRoot && lexicalRoot.children) {
|
if (lexicalRoot && lexicalRoot.children) {
|
||||||
return this.lexicalToHtml(lexicalRoot);
|
return this.lexicalToHtml(lexicalRoot);
|
||||||
}
|
}
|
||||||
} catch (e: any) {
|
} catch (e: any) {
|
||||||
// ==========================================
|
// Fallback to regex text extraction if JSON parsing fails
|
||||||
// 3. ULTIMATE FAILSAFE (REGEX TEXT EXTRACTION)
|
|
||||||
// ==========================================
|
|
||||||
// If JSON parsing fails due to corrupted RSC stream segments,
|
|
||||||
// we extract all "text":"..." fragments to reconstruct the story.
|
|
||||||
let fallbackHtml = '';
|
let fallbackHtml = '';
|
||||||
const textMatches = jsonStr.match(/"text":"(.*?)"/g);
|
const textMatches = jsonStr.match(/\\?"text\\?"\s*:\s*\\?"(.*?)\\?"/g);
|
||||||
if (textMatches && textMatches.length > 0) {
|
if (textMatches && textMatches.length > 0) {
|
||||||
textMatches.forEach(m => {
|
textMatches.forEach(m => {
|
||||||
let text = m.substring(8, m.length - 1);
|
let text = m.match(/: ?"?(.*?)"?$/)?.[1] || '';
|
||||||
if (text.trim() && text !== ' ') {
|
text = text.replace(/\\"/g, '"').replace(/\\\\/g, '\\');
|
||||||
|
if (text.trim() && text !== ' ' && !text.startsWith('Ch. ')) {
|
||||||
fallbackHtml += `<p>${text}</p>`;
|
fallbackHtml += `<p>${text}</p>`;
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
return fallbackHtml;
|
if (fallbackHtml) return fallbackHtml;
|
||||||
}
|
}
|
||||||
|
|
||||||
throw new Error(
|
|
||||||
`JSON Parse error: ${e.message}. Data snippet: ${jsonStr.substring(0, 500)}`,
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ==========================================
|
// ==========================================
|
||||||
// 4. Final HTML Scavenger Fallback
|
// 3. HTML SCAVENGER FALLBACK
|
||||||
// ==========================================
|
// ==========================================
|
||||||
const $ = loadCheerio(rscText);
|
// If RSC extraction failed, try fetching the standard HTML page
|
||||||
let htmlContent = $(
|
try {
|
||||||
'main > section[data-sentry-component="RichText"]',
|
const htmlResponse = await fetchApi(this.site + chapterPath, {
|
||||||
).html();
|
headers: this.HEADERS,
|
||||||
if (htmlContent) return htmlContent;
|
});
|
||||||
|
const htmlText = await htmlResponse.text();
|
||||||
|
const $ = loadCheerio(htmlText);
|
||||||
|
const htmlContent = $(
|
||||||
|
'main > section[data-sentry-component="RichText"]',
|
||||||
|
).html();
|
||||||
|
if (htmlContent) return htmlContent;
|
||||||
|
} catch (e) {
|
||||||
|
// Ignore fallback errors and throw the final error below
|
||||||
|
}
|
||||||
|
|
||||||
throw new Error(
|
throw new Error(
|
||||||
'Story content not found. Cloudflare might be blocking the request or the page structure has changed. Please try opening in WebView first.',
|
'Story content not found. The page structure might have changed. Please try opening in WebView to verify.',
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
|
Before Width: | Height: | Size: 10 KiB After Width: | Height: | Size: 10 KiB |
Reference in New Issue
Block a user