From 44305f529b1f68912491eaa08c03cb4b43afdc2c Mon Sep 17 00:00:00 2001 From: Joshua Peek Date: Thu, 19 Jan 2023 06:15:12 -0800 Subject: [PATCH] fix(route): bloomberg author (#11619) * Switch bloomberg author feed to html scraper * Safely resolve relative href * Prefer WHATWG URL API * Remove list cache * Cache article list * Avoid caching captcha * Add back rss parser as fallback * Load author news list from api * Fallback to rss * docs: add docs for new params --- docs/en/finance.md | 2 +- lib/v2/bloomberg/authors.js | 52 +++++++++++++++++++++++++++++----- lib/v2/bloomberg/maintainer.js | 2 +- lib/v2/bloomberg/router.js | 2 +- 4 files changed, 48 insertions(+), 10 deletions(-) diff --git a/docs/en/finance.md b/docs/en/finance.md index d60061b65b..2ad64dbac4 100644 --- a/docs/en/finance.md +++ b/docs/en/finance.md @@ -29,7 +29,7 @@ pageClass: routes ### Authors - + ## CFD diff --git a/lib/v2/bloomberg/authors.js b/lib/v2/bloomberg/authors.js index e226144461..99d52681e7 100644 --- a/lib/v2/bloomberg/authors.js +++ b/lib/v2/bloomberg/authors.js @@ -1,14 +1,52 @@ +const cheerio = require('cheerio'); +const got = require('@/utils/got'); +const rssParser = require('@/utils/rss-parser'); const { asyncPoolAll, parseArticle } = require('./utils'); -const parser = require('@/utils/rss-parser'); + +const parseAuthorNewsList = async (slug) => { + const baseURL = `https://www.bloomberg.com/authors/${slug}`; + const apiUrl = `https://www.bloomberg.com/lineup/api/lazy_load_author_stories?slug=${slug}&authorType=default&page=1`; + const resp = await got(apiUrl); + // Likely rate limited + if (!resp.data.html) { + return []; + } + const $ = cheerio.load(resp.data.html); + const articles = $('article.story-list-story'); + return articles + .map((index, item) => { + item = $(item); + const headline = item.find('a.story-list-story__info__headline-link'); + return { + title: headline.text(), + pubDate: item.attr('data-updated-at'), + guid: `bloomberg:${item.attr('data-id')}`, + link: new URL(headline.attr('href'), baseURL).href, + }; + }) + .get(); +}; module.exports = async (ctx) => { - const { id, slug } = ctx.params; - const feed = await parser.parseURL(`https://www.bloomberg.com/authors/${id}/${slug}.rss`); - const item = await asyncPoolAll(1, feed.items, (item) => parseArticle(item, ctx)); + const { id, slug, source } = ctx.params; + const link = `https://www.bloomberg.com/authors/${id}/${slug}`; + + let list = []; + if (!source || source === 'api') { + list = await parseAuthorNewsList(`${id}/${slug}`); + } + // Fallback to rss if api failed or requested by param + if (source === 'rss' || list.length === 0) { + list = (await rssParser.parseURL(`${link}.rss`)).items; + } + + const item = await asyncPoolAll(1, list, (item) => parseArticle(item, ctx)); + const authorName = item.find((i) => i.author)?.author ?? 'Unknown'; + ctx.state.data = { - title: `Bloomberg - ${feed.title.split(' - ', 2)[0]}`, - link: feed.link, - language: feed.language, + title: `Bloomberg - ${authorName}`, + link, + language: 'en-us', item, }; }; diff --git a/lib/v2/bloomberg/maintainer.js b/lib/v2/bloomberg/maintainer.js index 17c8230dcc..e47815d06f 100644 --- a/lib/v2/bloomberg/maintainer.js +++ b/lib/v2/bloomberg/maintainer.js @@ -1,5 +1,5 @@ module.exports = { - '/authors/:id/:slug': ['josh'], + '/authors/:id/:slug/:source?': ['josh'], '/:site?': ['bigfei'], '/': ['bigfei'], }; diff --git a/lib/v2/bloomberg/router.js b/lib/v2/bloomberg/router.js index a310c8fd5b..61cc62d573 100644 --- a/lib/v2/bloomberg/router.js +++ b/lib/v2/bloomberg/router.js @@ -1,5 +1,5 @@ module.exports = function (router) { - router.get('/authors/:id/:slug', require('./authors')); + router.get('/authors/:id/:slug/:source?', require('./authors')); router.get('/:site', require('./index')); router.get('/', require('./index')); };