From d2abeb988388eb367d409b8d793a251bbcbdebfd Mon Sep 17 00:00:00 2001 From: mjysci Date: Tue, 2 Mar 2021 14:58:09 +0800 Subject: [PATCH] fix(route): AP News (#7035) --- docs/new-media.md | 5 ++- docs/traditional-media.md | 3 +- lib/routes/apnews/topics.js | 83 ++++++++++++++----------------------- 3 files changed, 37 insertions(+), 54 deletions(-) diff --git a/docs/new-media.md b/docs/new-media.md index 0c184d79bb..7ad170ad04 100644 --- a/docs/new-media.md +++ b/docs/new-media.md @@ -1917,10 +1917,11 @@ column 为 third 时可选的 category: ## 网易号(通用) -优先使用方法一,若是网易号搜索页面搜不到的小众网易号(文章页面不含`data-wemediaid`)则可使用此法。 -触发反爬会只抓取到标题,建议自建。 +优先使用方法一,若是网易号搜索页面搜不到的小众网易号(文章页面不含`data-wemediaid`)则可使用此法。 +触发反爬会只抓取到标题,建议自建。 + ## 网易新闻 ### 排行榜 diff --git a/docs/traditional-media.md b/docs/traditional-media.md index 75afea6d8c..5955668956 100644 --- a/docs/traditional-media.md +++ b/docs/traditional-media.md @@ -28,7 +28,8 @@ pageClass: routes ### 话题 - + +采用了`puppeteer`规避`Project Shield`,无全文抓取,建议自建。 ## BBC diff --git a/lib/routes/apnews/topics.js b/lib/routes/apnews/topics.js index 9a3a5d158e..484c119b91 100644 --- a/lib/routes/apnews/topics.js +++ b/lib/routes/apnews/topics.js @@ -1,62 +1,43 @@ -const got = require('@/utils/got'); const cheerio = require('cheerio'); -const url = require('url'); + +const HOME_PAGE = 'https://apnews.com'; module.exports = async (ctx) => { const topic = ctx.params.topic; - const response = await got({ - method: 'get', - url: `https://apnews.com/${topic}`, - }); + const browser = await require('@/utils/puppeteer')(); + const page = await browser.newPage(); + const url_link = `${HOME_PAGE}/hub/${topic}`; + await page.goto(url_link); + const html = await page.evaluate(() => document.documentElement.innerHTML); + browser.close(); - const data = response.data; + const $ = cheerio.load(html); - const $ = cheerio.load(data); - // const list = $('div.FeedCard'); - const list = []; - $('div.FeedCard').each(function (index, item) { - if ($(item).find('a[class^=Component-headline]').attr('href') !== undefined) { - list.push(item); - } - }); - - const out = await Promise.all( - list.map(async (article) => { - const link = url.resolve('https://apnews.com', $(article).find('a[class^=Component-headline]').attr('href')); - - const [title, author, pubDate, description] = await ctx.cache.tryGet(link, async () => { - const result = await got.get(link); - - const $ = cheerio.load(result.data); - - const head = JSON.parse($('script[type="application/ld+json"]').html()); - - const title = head.headline; - const author = head.author.join(' & '); - const pubDate = head.datePublished; - - const text = $('div.Article').html(); - const imageUrl = head.image; - const description = `` + text; - - return [title, author, pubDate, description]; - }); - - const item = { - title: title, - description: description, - pubDate: pubDate, - link: link, - author: author, - }; - return Promise.resolve(item); - }) - ); + const list = $('div.FeedCard'); ctx.state.data = { - title: 'AP News - ' + $('title').text(), - link: `https://www.apnews.com/${topic}`, - item: out, + title: $('.Body div').find('h1[class^=hubTitle]').text(), + link: HOME_PAGE, + item: list + .slice(0, 5) + .map((index, item) => { + item = $(item); + const title = item.find('h1[class^=Component-h1]').text(); + const link = item.find('a[class^=Component-headline-]').attr('href'); + const pubDate = item.find('span[class^=Timestamp]').attr('title'); + const text = item.find('div[class^=content]').text(); + const imageUrl = item.find('img[class^=image-]').attr('src'); + const author = item.find('span[class^=Component-bylines-]').text().slice(3); + + return { + title: title, + description: `` + text, + link: `${HOME_PAGE}${link}`, + pubDate: pubDate, + author: author, + }; + }) + .get(), }; };