From 874c8db78d36a8cffe7061593f2c2e2ed579f9f7 Mon Sep 17 00:00:00 2001 From: Logan Rockmore Date: Mon, 16 Dec 2019 21:12:02 -0700 Subject: [PATCH] feat: Add the Vulture router. (#3581) --- docs/en/new-media.md | 11 +++++ lib/router.js | 3 ++ lib/routes/vulture/index.js | 8 ++++ lib/routes/vulture/utils.js | 93 +++++++++++++++++++++++++++++++++++++ 4 files changed, 115 insertions(+) create mode 100644 lib/routes/vulture/index.js create mode 100644 lib/routes/vulture/utils.js diff --git a/docs/en/new-media.md b/docs/en/new-media.md index 06dd0e2a8..29e99a45c 100644 --- a/docs/en/new-media.md +++ b/docs/en/new-media.md @@ -85,3 +85,14 @@ Compared to the official one, this feed: Provides a better reading experience (full text articles) over the official one. + +## Vulture + + + +Supported sub-sites: +| TV | Movies | Comedy | Music | TV Recaps | Books | Theater | Art | Awards | Video | +| ----- | ------ | ------ | ------ | ------ | ------ | ------ | ------ | ------ | ------ | +| tv | movies | comedy | music | tvrecaps | books | theater | art | awards | video | + + diff --git a/lib/router.js b/lib/router.js index 3c7f31dfc..d3efaa602 100644 --- a/lib/router.js +++ b/lib/router.js @@ -2036,6 +2036,9 @@ router.get('/mastodon/timeline/:site/:only_media?', require('./routes/mastodon/t // Kernel Aliyun router.get('/aliyun-kernel/index', require('./routes/aliyun-kernel/index')); +// Vulture +router.get('/vulture/:type', require('./routes/vulture/index')); + // xinwenlianbo router.get('/xinwenlianbo/index', require('./routes/xinwenlianbo/index')); diff --git a/lib/routes/vulture/index.js b/lib/routes/vulture/index.js new file mode 100644 index 000000000..21fefb6ff --- /dev/null +++ b/lib/routes/vulture/index.js @@ -0,0 +1,8 @@ +const utils = require('./utils'); + +module.exports = async (ctx) => { + const url = `https://www.vulture.com/${ctx.params.type}/`; + const title = `Vulture - ${ctx.params.type}`; + + ctx.state.data = await utils.getData(ctx, url, title); +}; diff --git a/lib/routes/vulture/utils.js b/lib/routes/vulture/utils.js new file mode 100644 index 000000000..e84412757 --- /dev/null +++ b/lib/routes/vulture/utils.js @@ -0,0 +1,93 @@ +const got = require('@/utils/got'); +const cheerio = require('cheerio'); + +async function load(link) { + const response = await got.get(link); + const $ = cheerio.load(response.data); + + const description = $('div.article-content'); + + // remove the content that we don't want to show + description.find('aside.related').remove(); + description.find('aside.article-details_heading-with-paragraph').remove(); + description.find('section.package-list').remove(); + description.find('div.source-links h2').remove(); + description.find('div.source-links svg').remove(); + description.find('div.mobile-secondary-area').remove(); + description.find('aside.newsletter-flex-text').remove(); + + return { + description: description.html(), + }; +} + +async function ProcessFeed(list, caches) { + return await Promise.all( + list.map(async (item) => { + const itemUrl = item.canonicalUrl; + + let bylineString = ''; + if (item.byline) { + const byline = item.byline[0]; + const bylineNames = byline.names.map((name) => name.text); + const bylineNamesString = bylineNames.join(', '); + + bylineString = 'by ' + bylineNamesString; + } + + const single = { + title: item.primaryHeadline, + link: itemUrl, + author: bylineString, + guid: itemUrl, + pubDate: item.date, + }; + + const other = await caches.tryGet(itemUrl, async () => await load(itemUrl)); + + return Promise.resolve(Object.assign({}, single, other)); + }) + ); +} + +const getData = async (ctx, url, title) => { + const htmlResponse = await got({ + method: 'get', + url: url, + headers: { + Referer: url, + }, + }); + + const htmlData = htmlResponse.data; + + const $ = cheerio.load(htmlData); + const dataUri = $('section.paginated-feed').attr('data-uri'); + + // get the raw data + const response = await got({ + method: 'get', + url: dataUri, + headers: { + Referer: dataUri, + }, + }); + + const data = response.data; + + // limit the list to only 25 articles, to make sure that load times remain reasonable + const list = data.articles.slice(0, 25); + + const result = await ProcessFeed(list, ctx.cache); + + return { + title: title, + link: url, + description: $('meta[name="description"]').attr('content'), + item: result, + }; +}; + +module.exports = { + getData, +};