feat: 增加财新博客全文输出 (#2090)

* feat: 增加财新博客全文输出

同时更改了原财新网的路由

* refactor: 保留旧路由,向下兼容

* fix: tryGet 包含了设置缓存,不需要再次 set

* fix: 使用 config 里的统一 ua 配置

* refactor: 统一使用 cheerio 搜索子节点的方式

* style: fix formatting

* fix: 移除无用参数

* docs: 更新路由
This commit is contained in:
Xiang Li
2019-05-09 18:55:48 +08:00
committed by DIYgod
parent 87beeadb11
commit 79f4399617
3 changed files with 70 additions and 0 deletions
+8
View File
@@ -550,3 +550,11 @@ type 为 all 时,category 参数不支持 cost 和 free
<Route name="首页" author="qiwihui" example="/paidai" path="/paidao" />
<Route name="论坛" author="qiwihui" example="/paidai/bbs" path="/paidao/bbs" />
<Route name="商道" author="qiwihui" example="/paidai/news" path="/paidao/news" />
## 财新博客
<Route name="用户博客" author="Maecenas" example="/caixin/blog/zhangwuchang" path="/caixin/blog/:column" :paramsDesc="['博客名称,可在博客主页的 URL 找到']">
通过提取文章全文,以提供比官方源更佳的阅读体验.
</Route>
+2
View File
@@ -405,6 +405,8 @@ router.get('/mihoyo/bh2/:type', require('./routes/mihoyo/bh2'));
// 央视新闻
router.get('/cctv/:category', require('./routes/cctv/category'));
// 财新博客
router.get('/caixin/blog/:column', require('./routes/caixin/blog'));
// 财新
router.get('/caixin/:column/:category', require('./routes/caixin/category'));
+60
View File
@@ -0,0 +1,60 @@
const axios = require('../../utils/axios');
const cheerio = require('cheerio');
const { ua } = require('../../config');
const Parser = require('rss-parser');
const parser = new Parser({ headers: { 'User-Agent': ua } });
async function load(link, need_feed_description) {
const response = await axios.get(link);
const $ = cheerio.load(response.data);
const article = $('.blog_content').removeAttr('style');
article.find('img').removeAttr('style');
article
.find('div')
// Non-breaking space U+00A0, `&nbsp;` in html
// element.children[0].data === $(element, article).text()
.filter((_, element) => element.children[0].data === String.fromCharCode(160))
.remove();
const description = article.html();
const item = { description };
if (need_feed_description) {
const author = $('div.widget.author_detail p:nth-child(2)');
// workaround to split author name and author article
const author_name = author.find('strong');
author_name.text(author_name.text() + ',');
item.feed_description = author.text();
}
return item;
}
module.exports = async (ctx) => {
const { column } = ctx.params;
const link = `http://${column}.blog.caixin.com`;
const feed_url = `${link}/feed`;
const feed = await parser.parseURL(feed_url);
const title = `财新博客 - ${/[^»]*$/.exec(feed.title)[0]}`;
const items = await Promise.all(
feed.items.slice(0, 10).map(async (item, index) => {
const link = item.link;
const single = {
title: item.title,
pubDate: item.pubDate,
link,
author: item['dc:creator'],
};
const other = await ctx.cache.tryGet(link, async () => await load(link, index === 0));
return Promise.resolve(Object.assign({}, single, other));
})
);
ctx.state.data = {
title,
link,
description: items[0].feed_description,
item: items,
};
};