From e1567d66153db99268c0afa83310cc96511cfea7 Mon Sep 17 00:00:00 2001 From: Martijn van der Ven Date: Sun, 17 Nov 2024 03:44:18 +0100 Subject: [PATCH] Divine Dao Library: update all data fetching (#1320) * feat: update Divine Dao Library Add novel filters, support both popular and latest novels, get covers, and find the actual last published chapter. * feat: introduce caching to Divine Dao Library --- src/plugins/english/divinedaolibrary.ts | 367 +++++++++++++++++------- 1 file changed, 270 insertions(+), 97 deletions(-) diff --git a/src/plugins/english/divinedaolibrary.ts b/src/plugins/english/divinedaolibrary.ts index d87e405..04f8bed 100644 --- a/src/plugins/english/divinedaolibrary.ts +++ b/src/plugins/english/divinedaolibrary.ts @@ -1,125 +1,298 @@ -import { CheerioAPI, load as parseHTML } from 'cheerio'; -import { fetchApi } from '@libs/fetch'; -import { Plugin } from '@typings/plugin'; import { defaultCover } from '@libs/defaultCover'; +import { fetchApi } from '@libs/fetch'; +import { Filters, FilterTypes, FilterValueWithType } from '@libs/filterInputs'; +import { Plugin } from '@typings/plugin'; +import { load as parseHTML } from 'cheerio'; class DDLPlugin implements Plugin.PluginBase { id = 'DDL.com'; name = 'Divine Dao Library'; site = 'https://www.divinedaolibrary.com/'; - version = '1.0.1'; + version = '1.1.0'; icon = 'src/en/divinedaolibrary/icon.png'; - parseNovels(loadedCheerio: CheerioAPI, searchTerm?: string) { - let novels: Plugin.NovelItem[] = []; + filters = { + category: { + type: FilterTypes.CheckboxGroup, + label: 'State', + value: ['Completed', 'Translating', 'Lost in Voting Poll', 'Dropped'], + options: [ + { label: 'Completed', value: 'Completed' }, + { label: 'Translating', value: 'Translating' }, + { label: 'Lost in Voting Poll', value: 'Lost in Voting Poll' }, + { label: 'Dropped', value: 'Dropped' }, + { label: 'Personally Written', value: 'Personally Written' }, + ], + }, + } satisfies Filters; - loadedCheerio('#main') - .find('li') - .each((i, el) => { - const novelName = loadedCheerio(el).find('a').text(); - const novelCover = defaultCover; - const novelUrl = loadedCheerio(el).find('a').attr('href'); + allNovelsCache: readonly (readonly [string, string, string])[] | undefined; + novelItemCache = new Map>(); - if (!novelUrl) return; - - const novel = { - name: novelName, - cover: novelCover, - path: novelUrl.replace(this.site, ''), - }; - - novels.push(novel); - }); - - if (searchTerm) { - novels = novels.filter(novel => - novel.name.toLowerCase().includes(searchTerm.toLowerCase()), - ); + /** + * Safely extract the pathname from any URL on {@link site}. Check the root + * site as there are novels linking off-site (to Patreon). + * + * @private + */ + getPath(url: string): string | undefined { + if (!url.startsWith(this.site)) { + return undefined; } - return novels; + const trimmed = url.substring(this.site.length).replace(/(^\/+|\/+$)/g, ''); + if (trimmed.length === 0) { + return undefined; + } + return trimmed; } - async popularNovels(): Promise { - const link = this.site + 'novels'; - - const body = await fetchApi(link).then(res => res.text()); - - const loadedCheerio = parseHTML(body); - return this.parseNovels(loadedCheerio); + /** + * Map an array with an asynchronous function and only return the array items + * that successfully were fulfilled with values other than undefined. + * + * @private + */ + async asyncMap( + collection: readonly T[], + callbackfn: (value: T, index: number, array: readonly T[]) => Promise, + ): Promise[]> { + return (await Promise.allSettled(collection.map(callbackfn))) + .filter( + ( + p: PromiseSettledResult, + ): p is PromiseFulfilledResult> => + p.status === 'fulfilled' && p.value !== undefined, + ) + .map(({ value }) => value); } - async parseNovel(novelPath: string): Promise { - const result = await fetchApi(this.site + novelPath); - const body = await result.text(); + /** + * DDL links to future (unpublished) chapters from its novel pages. To be + * able to report updates correctly, try to figure out which chapter was the + * actual latest one to be published. + * + * @private + * @returns the path value of the latest published chapter (or undefined) + */ + async findLatestChapter(novelPath: string): Promise { + const link = `${this.site}wp-json/wp/v2/categories?slug=${novelPath}`; + const guessCategory = await fetchApi(link).then(res => res.json()); + if (guessCategory.length !== 1) { + return undefined; + } + const categoryId = guessCategory[0].id; + const chapterLink = `${this.site}wp-json/wp/v2/posts?categories=${categoryId}&per_page=1`; + const lastChapter = await fetchApi(chapterLink).then(res => res.json()); + if (lastChapter.length !== 1) { + return undefined; + } + return lastChapter[0].slug; + } - const loadedCheerio = parseHTML(body); - - const novel: Plugin.SourceNovel = { - path: novelPath, - name: loadedCheerio('h1.entry-title').text().trim() || 'Untitled', - cover: - loadedCheerio('.entry-content').find('img').attr('data-ezsrc') || - defaultCover, - chapters: [], + /** + * Based on the path, grab all the available information about a novel. Used + * to seed the {@link novelItemCache} as well as to fetch the actual single + * novel views with chapter list. + * + * @private + */ + async grabNovel( + path: string, + getChapters = false, + ): Promise<(Plugin.SourceNovel & Required) | undefined> { + const link = `${this.site}wp-json/wp/v2/pages?slug=${path}`; + const data = await fetchApi(link).then(res => res.json()); + if (data.length !== 1) { + return undefined; + } + const content = parseHTML(data[0].content.rendered); + const excerpt = parseHTML(data[0].excerpt.rendered); + const image = content('img').first(); + let chapters: Plugin.ChapterItem[] = []; + if (getChapters) { + const linkedChapters = content('li > span > a') + .map((_, anchorEl) => { + const chapterPath = this.getPath(anchorEl.attribs['href']); + if (!chapterPath) return; + return { + name: content(anchorEl).text(), + path: chapterPath, + } satisfies Plugin.ChapterItem; + }) + .toArray(); + const lastChapterPath = await this.findLatestChapter(path); + if (lastChapterPath) { + chapters = linkedChapters.slice( + 0, + 1 + + linkedChapters.findIndex( + chapter => chapter.path === lastChapterPath, + ), + ); + } else { + chapters = linkedChapters; + } + } + return { + name: data[0].title.rendered, + path, + cover: image.attr('data-lazy-src') ?? image.attr('src') ?? defaultCover, + author: content('h3') + .first() + .text() + .replace(/^Author:\s*/g, ''), + summary: excerpt('p') + .first() + .text() + .replace(/^.+Description\s*/g, ''), + chapters, }; + } - novel.summary = loadedCheerio('#main > article > div > p:nth-child(6)') - .text() - .trim(); - - novel.author = loadedCheerio('#main > article > div > h3:nth-child(2)') - .text() - .replace(/Author:/g, '') - .trim(); - - const chapter: Plugin.ChapterItem[] = []; - - loadedCheerio('#main') - .find('li > span > a') - .each((i, el) => { - const chapterName = loadedCheerio(el).text().trim(); - const chapterUrl = loadedCheerio(el).attr('href'); - - if (!chapterUrl) return; - - chapter.push({ - name: chapterName, - path: chapterUrl.replace(this.site, ''), - }); - }); - - novel.chapters = chapter; - + /** + * Based on the path, grab the {@link Plugin.NovelItem} information from the + * cache. If not available in the cache, it will fetch the information. + * + * @private + */ + async grabCachedNovel(path: string): Promise { + const fromCache = this.novelItemCache.get(path); + if (fromCache) { + return fromCache; + } + const sourceNovel = await this.grabNovel(path, false); + if (sourceNovel === undefined) { + return undefined; + } + const novel = { + name: sourceNovel.name, + path: sourceNovel.path, + cover: sourceNovel.cover, + }; + this.novelItemCache.set(path, novel); return novel; } - async parseChapter(chapterPath: string): Promise { - const result = await fetchApi(this.site + chapterPath); - const body = await result.text(); - - const loadedCheerio = parseHTML(body); - - const chapterName = loadedCheerio('.entry-title').text().trim(); - - let chapterText = loadedCheerio('.entry-content').html(); - - if (!chapterText) { - chapterText = loadedCheerio('.page-header').html(); + /** + * Get the list of all novels listed on the novels page. Cache it so filters + * do not have to refetch the same HTML again. + * + * @private + */ + async grabCachedNovels(): Promise< + readonly (readonly [string, string, string])[] | undefined + > { + if (this.allNovelsCache) { + return this.allNovelsCache; } - - chapterText = `

${chapterName}

` + chapterText; - - return chapterText; + const body = await fetchApi(this.site + 'novels').then(res => res.text()); + const loadedCheerio = parseHTML(body); + const novels = loadedCheerio('.entry-content ul') + .map((_, listEl) => { + const list = loadedCheerio(listEl); + const category = list.prev().text(); + return list + .find('a') + .map((_, anchorEl) => { + const path = this.getPath(anchorEl.attribs['href']); + if (!path) return; + const name = loadedCheerio(anchorEl).text(); + return [[category, name, path]] as const; + }) + .toArray(); + }) + .toArray(); + this.allNovelsCache = novels; + return novels; } - async searchNovels(searchTerm: string): Promise { - const url = this.site + 'novels'; - - const result = await fetchApi(url); - const body = await result.text(); - + /** + * Parse list of paths of recently updated novels from the homepage. + * + * @private + */ + async latestNovels(): Promise { + const body = await fetchApi(this.site).then(res => res.text()); const loadedCheerio = parseHTML(body); - return this.parseNovels(loadedCheerio, searchTerm); + const novelPaths = loadedCheerio('#main') + .find('a[rel="category tag"]') + .map((_, anchorEl) => { + const path = this.getPath(anchorEl.attribs['href']); + return path; + }) + .toArray(); + const uniqueNovelPaths = new Set(novelPaths); + return Array.from(uniqueNovelPaths); + } + + /** + * Parse list of paths from novels list, optionally filtered. + * + * @private + */ + async allNovels( + categoryFilter: FilterValueWithType, + ) { + const allNovels = await this.grabCachedNovels(); + if (!allNovels) return []; + return allNovels + .filter(([category]) => categoryFilter.value.includes(category)) + .map(([, , path]) => path); + } + + async popularNovels( + pageNo: number, + options: Plugin.PopularNovelsOptions, + ): Promise { + if (pageNo !== 1) { + return []; + } + const novelsList = options.showLatestNovels + ? this.latestNovels() + : this.allNovels(options.filters.category); + return await this.asyncMap( + await novelsList, + this.grabCachedNovel.bind(this), + ); + } + + async parseNovel(novelPath: string): Promise { + const sourceNovel = await this.grabNovel(novelPath, true); + if (sourceNovel === undefined) { + throw new Error(`The path "${novelPath}" could not be resolved.`); + } + return sourceNovel; + } + + async parseChapter(chapterPath: string): Promise { + const chapterLink = `${this.site}wp-json/wp/v2/posts?slug=${chapterPath}`; + const chapter = await fetchApi(chapterLink).then(res => res.json()); + if (chapter.length !== 1) { + return ''; + } + const title = `

${chapter[0].title.rendered}

`; + const content = chapter[0].content.rendered; + return `${title}${content}`; + } + + async searchNovels( + searchTerm: string, + pageNo: number, + ): Promise { + if (pageNo !== 1) { + return []; + } + const allNovels = await this.grabCachedNovels(); + if (!allNovels) return []; + const foundNovels = allNovels + .filter(([, name]) => + name.toLocaleLowerCase().includes(searchTerm.toLocaleLowerCase()), + ) + .map(([, , path]) => path); + return await this.asyncMap( + await foundNovels, + this.grabCachedNovel.bind(this), + ); } }