From df6572b659bf25cabc6a71108b8cd9b70056262c Mon Sep 17 00:00:00 2001 From: LinxHex <92703641+LinxHex@users.noreply.github.com> Date: Fri, 13 Mar 2026 23:57:36 +0800 Subject: [PATCH] fix(routes/miyuki): format news titles with bracketed categories and keep full-text body-only (#21377) * fix(routes/miyuki): format news titles with bracketed categories and keep full-text body-only * fix(routes/miyuki): clean news body extraction and normalize images - Updated lib/routes/miyuki/news.ts to keep item titles as plain article titles and store category information only in the `category` field. - Switched the route to use the canonical Miyuki domain and added the route `url` metadata. - Refined full-text extraction so `description` keeps only the main article body and excludes page header metadata such as section title, article title, date, and category. - Fixed malformed article-body handling for pages like `10217`, where the source HTML structure breaks the original body selector. - Normalized article images by deduplicating `for_pc` / `for_sp` duplicates, converting relative image URLs to absolute URLs, and replacing list-based photo blocks with plain containers so RSS readers do not render black bullets. - Resolved the related CodeFactor/lint issues and verified the file with `oxlint` and `eslint`. * feat(apple): add full-text Apple Newsroom route Add /apple/newsroom for the official Apple Newsroom China feed. Fetch article pages to provide full-text content with cached detail requests, and normalize article metadata and media URLs. Remove gallery navigation, download buttons, video playback controls, empty wrapper nodes, and other non-content UI elements from articles. Format image and video captions as distinct descriptions, and drop duplicate captions when their text matches the article body. * fix(routes/miyuki): simplify image handling in news descriptions * Remove unrelated changes from PR --- lib/routes/miyuki/news.ts | 72 ++++++++++++++++++++++++++++++++++----- 1 file changed, 63 insertions(+), 9 deletions(-) diff --git a/lib/routes/miyuki/news.ts b/lib/routes/miyuki/news.ts index 7ffb135f8..43a933a74 100644 --- a/lib/routes/miyuki/news.ts +++ b/lib/routes/miyuki/news.ts @@ -6,13 +6,21 @@ import ofetch from '@/utils/ofetch'; import { parseDate } from '@/utils/parse-date'; import timezone from '@/utils/timezone'; -const ORIGIN = 'https://miyuki.jp'; +const ORIGIN = 'https://www.miyuki.jp'; const NEWS_LINK = `${ORIGIN}/s/y10/news/list`; +const DETAIL_HEADER_SELECTOR = [ + '.pc__news_detail__title', + '.pc__news_detail__title__japanese', + '.news_detail__date', + '.news_detail__title', + '.news_detail__ganre', +].join(', '); export const route: Route = { path: '/news', example: '/miyuki/news', name: 'News', + url: 'www.miyuki.jp/s/y10/news/list', categories: ['new-media'], maintainers: ['KarasuShin'], features: { @@ -33,21 +41,18 @@ async function handler() { const items = await Promise.all( $('.list__side_border li') .toArray() - .map(async (item) => { + .map((item) => { const $item = $(item); const link = `${ORIGIN}${$item.find('a').attr('href')!}`; - return await cache.tryGet(link, async () => { + return cache.tryGet(link, async () => { const category = $item.find('p span').last().text(); + const title = $item.find('a').text(); return { - title: `${category} - ${$item.find('a').text()}`, + title, link, pubDate: timezone(parseDate($item.find('p span').first().text()), +9), category: [category], - description: await cache.tryGet(link, async () => { - const detailHtml = await ofetch(link); - const $detail = load(detailHtml); - return $detail('.contents_area__inner').html()!; - }), + description: await getDescription(link), } as DataItem; }); }) @@ -59,3 +64,52 @@ async function handler() { item: items, }; } + +async function getDescription(link: string) { + const detailHtml = await ofetch(link); + const $ = load(detailHtml); + const content = $('.contents_area__inner'); + content.children(DETAIL_HEADER_SELECTOR).remove(); + normalizePhotoLists($, content); + content.children().each((_, element) => { + const child = $(element); + if (!hasMeaningfulHtml(child.html())) { + child.remove(); + } + }); + + return content.html()?.trim() ?? ''; +} + +function hasMeaningfulHtml(html?: string | null) { + return Boolean(html?.replaceAll(//g, '').replaceAll(' ', '').trim()); +} + +function normalizePhotoLists($, content) { + content.find('.news_detail__photo_list').each((_, element) => { + const list = $(element); + const items = list + .children('li') + .toArray() + .flatMap((item) => { + const photoItem = $(item); + photoItem.find('img.for_sp').remove(); + photoItem.find('img').each((__, image) => { + const img = $(image); + const src = img.attr('src'); + if (!src) { + img.remove(); + return; + } + + img.attr('src', new URL(src, ORIGIN).href); + img.removeAttr('class'); + }); + + const html = photoItem.html(); + return hasMeaningfulHtml(html) ? [`
${html!.trim()}
`] : []; + }); + + list.replaceWith(items.join('

')); + }); +}