fix(route): Financial Times HTTPS support (#8765)

This commit is contained in:
Tony
2022-01-22 20:02:27 +00:00
committed by GitHub
parent b1a20c3a17
commit c86782b91b
9 changed files with 99 additions and 69 deletions
-1
View File
@@ -157,7 +157,6 @@ pageClass: routes
::: tip 提示
- 不支持付费文章。
- 由于未知原因 FT 中文网的 SSL 证书不被信任 (参见 [SSL Labs 报告](https://www.ssllabs.com/ssltest/analyze.html?d=www.ftchinese.com&latest)), 所有文章通过 http 协议获取。
:::
+3 -3
View File
@@ -988,9 +988,9 @@ router.get('/nhk/news_web_easy', lazyloadRouteHandler('./routes/nhk/news_web_eas
// BBC
router.get('/bbc/:site?/:channel?', lazyloadRouteHandler('./routes/bbc/index'));
// Financial Times
router.get('/ft/myft/:key', lazyloadRouteHandler('./routes/ft/myft'));
router.get('/ft/:language/:channel?', lazyloadRouteHandler('./routes/ft/channel'));
// Financial Times migrated to v2
// router.get('/ft/myft/:key', lazyloadRouteHandler('./routes/ft/myft'));
// router.get('/ft/:language/:channel?', lazyloadRouteHandler('./routes/ft/channel'));
// The Verge
router.get('/verge', lazyloadRouteHandler('./routes/verge/index'));
-46
View File
@@ -1,46 +0,0 @@
const got = require('@/utils/got');
const parser = require('@/utils/rss-parser');
const cheerio = require('cheerio');
module.exports = async (ctx) => {
const ProcessFeed = (content) => {
// clean up the article
content.find('div.o-share, aside, div.o-ads').remove();
return content.html();
};
const link = `https://www.ft.com/myft/following/${ctx.params.key}.rss`;
const feed = await parser.parseURL(link);
const items = await Promise.all(
feed.items.map(async (item) => {
const response = await got({
method: 'get',
url: item.link,
headers: {
Referer: 'https://www.facebook.com',
},
});
const $ = cheerio.load(response.data);
const single = {
title: item.title,
description: ProcessFeed($('div.article__content-body')),
author: $('a.n-content-tag--author').text(),
pubDate: item.pubDate,
link: item.link,
};
return Promise.resolve(single);
})
);
ctx.state.data = {
title: `FT.com - myFT`,
link,
description: `FT.com - myFT`,
item: items,
};
};
@@ -4,5 +4,6 @@ module.exports = async (ctx) => {
ctx.state.data = await utils.getData({
site: ctx.params.language === 'chinese' ? 'www' : 'big5',
channel: ctx.params.channel,
ctx,
});
};
+4
View File
@@ -0,0 +1,4 @@
module.exports = {
'/ft/myft/:key': ['HenryQW'],
'/ft/:language/:channel?': ['HenryQW', 'xyqfer'],
};
+44
View File
@@ -0,0 +1,44 @@
const got = require('@/utils/got');
const parser = require('@/utils/rss-parser');
const cheerio = require('cheerio');
module.exports = async (ctx) => {
const ProcessFeed = (content) => {
// clean up the article
content.find('div.o-share, aside, div.o-ads').remove();
return content.html();
};
const link = `https://www.ft.com/myft/following/${ctx.params.key}.rss`;
const feed = await parser.parseURL(link);
const items = await Promise.all(
feed.items.map((item) =>
ctx.cache.tryGet(item.link, async () => {
const response = await got({
method: 'get',
url: item.link,
headers: {
Referer: 'https://www.facebook.com',
},
});
const $ = cheerio.load(response.data);
item.description = ProcessFeed($('div.article__content-body'));
item.author = $('a.n-content-tag--author').text();
return item;
})
)
);
ctx.state.data = {
title: `FT.com - myFT`,
link,
description: `FT.com - myFT`,
item: items,
};
};
+24
View File
@@ -0,0 +1,24 @@
module.exports = {
'ftchinese.com': {
_name: 'Financial Times',
'.': [
{
title: 'FT 中文网',
docs: 'https://docs.rsshub.app/traditional-media.html#financial-times',
},
{
title: 'myFT 个人 RSS',
docs: 'https://docs.rsshub.app/traditional-media.html#financial-times',
},
],
},
'ft.com': {
_name: 'Financial Times',
'.': [
{
title: 'myFT personal RSS',
docs: 'https://docs.rsshub.app/en/traditional-media.html#financial-times',
},
],
},
};
+4
View File
@@ -0,0 +1,4 @@
module.exports = function (router) {
router.get('/myft/:key', require('./myft'));
router.get('/:language/:channel?', require('./channel'));
};
+19 -19
View File
@@ -7,21 +7,21 @@ const ProcessFeed = ($, link) => {
let content = $('div.story-container');
// 处理封面图片
content.find('div.story-image > figure').each((i, e) => {
content.find('div.story-image > figure').each((_, e) => {
const src = `https://thumbor.ftacademy.cn/unsafe/1340x754/${e.attribs['data-url']}`;
$(`<img src=${src}>`).insertAfter(content.find('div.story-lead')[0]);
});
// 付费文章跳转
content.find('div#subscribe-now-container').each((i, e) => {
content.find('div#subscribe-now-container').each((_, e) => {
$(`<br/><p>此文章为付费文章,会员<a href='${link}'>请访问网站阅读</a>。</p>`).insertAfter(content.find('div.story-body')[0]);
$(e).remove();
});
// 获取作者
let author = '';
content.find('span.story-author > a').each((i, e) => {
content.find('span.story-author > a').each((_, e) => {
author += `${$(e).text()} `;
});
author = author.trim();
@@ -31,7 +31,7 @@ const ProcessFeed = ($, link) => {
.find(
'div.story-theme, h1.story-headline, div.story-byline, div.mpu-container-instory,script, div#story-action-placeholder, div.copyrightstatement-container, div.clearfloat, div.o-ads, h2.list-title, div.allcomments, div.logincomment, div.nologincomment'
)
.each((i, e) => {
.each((_, e) => {
$(e).remove();
});
content = content.html();
@@ -39,7 +39,7 @@ const ProcessFeed = ($, link) => {
return { content, author, title };
};
const getData = async ({ site = 'www', channel }) => {
const getData = async ({ site = 'www', channel, ctx }) => {
let feed;
if (channel) {
@@ -47,7 +47,7 @@ const getData = async ({ site = 'www', channel }) => {
channel = channel.split('-').join('/');
try {
feed = await parser.parseURL(`http://${site}.ftchinese.com/rss/${channel}`);
feed = await parser.parseURL(`https://${site}.ftchinese.com/rss/${channel}`);
} catch (error) {
return {
title: `FT 中文网 ${channel} 不存在`,
@@ -55,23 +55,23 @@ const getData = async ({ site = 'www', channel }) => {
};
}
} else {
feed = await parser.parseURL(`http://${site}.ftchinese.com/rss/feed`);
feed = await parser.parseURL(`https://${site}.ftchinese.com/rss/feed`);
}
const items = await Promise.all(
feed.items.splice(0, 10).map(async (item) => {
const response = await got.get(`${item.link}?full=y&archive`);
feed.items.splice(0, 10).map((item) => {
item.link = item.link.replace('http://', 'https://');
return ctx.cache.tryGet(item.link, async () => {
const response = await got.get(`${item.link}?full=y&archive`);
const $ = cheerio.load(response.data);
const result = ProcessFeed($, item.link);
const single = {
title: result.title,
description: result.content,
author: result.author,
pubDate: item.pubDate,
link: item.link,
};
return Promise.resolve(single);
const $ = cheerio.load(response.data);
const result = ProcessFeed($, item.link);
item.title = result.title;
item.description = result.content;
item.author = result.author;
return item;
});
})
);