feat: support twreporter (#6071)

This commit is contained in:
Enoch Ma
2020-11-01 15:53:24 +00:00
committed by GitHub
parent b1a817a65f
commit 8af58a2821
6 changed files with 212 additions and 0 deletions
+14
View File
@@ -537,6 +537,20 @@ Supported sub-sites:
<Route author="nwindz" example="/hinatazaka46/blog" path="/hinatazaka46/blog" />
## 報道者
### 最新
<Route author="emdoe" example="/twreporter/newest" path="/twreporter/newest"/>
### 摄影
<Route author="emdoe" example="/twreporter/photography" path="/twreporter/photography"/>
### 分类
<Route author="emdoe" example="/twreporter/category/reviews" path="/twreporter/category/:tid" :paramsDesc="['分类(议题)名称,于主页获取']"/>
## 本地宝
### 焦点资讯
+5
View File
@@ -96,6 +96,11 @@ router.get('/bangumi/group/:id', require('./routes/bangumi/group/topic'));
router.get('/bangumi/subject/:id', require('./routes/bangumi/subject'));
router.get('/bangumi/user/blog/:id', require('./routes/bangumi/user/blog'));
// 報道者
router.get('/twreporter/newest', require('./routes/twreporter/newest'));
router.get('/twreporter/photography', require('./routes/twreporter/photography'));
router.get('/twreporter/category/:cid', require('./routes/twreporter/category'));
// 微博
router.get('/weibo/user/:uid/:routeParams?', require('./routes/weibo/user'));
router.get('/weibo/keyword/:keyword/:routeParams?', require('./routes/weibo/keyword'));
+36
View File
@@ -0,0 +1,36 @@
const cheerio = require('cheerio');
const got = require('@/utils/got');
const fetch = require('./fetch_article');
module.exports = async (ctx) => {
const baseURL = 'https://www.twreporter.org';
const url = baseURL + `/categories/${ctx.params.cid}`;
const res = await got.get(url);
const $ = cheerio.load(res.data);
const list = $('.lnKPLr').get();
const category = $('.kCfkTU').text();
const out = await Promise.all(
list.map(async (item) => {
const $ = cheerio.load(item);
const address = baseURL + $('a').attr('href');
const title = $('.list-item__Title-sc-1dx5lew-5').text();
const cache = await ctx.cache.get(address);
if (cache) {
return Promise.resolve(JSON.parse(cache));
}
const single = await fetch(address);
single.title = title;
ctx.cache.set(address, JSON.stringify(single));
return Promise.resolve(single);
})
);
ctx.state.data = {
title: `報道者 | ${category}`,
link: url,
item: out,
};
};
+66
View File
@@ -0,0 +1,66 @@
const cheerio = require('cheerio');
const got = require('@/utils/got');
module.exports = async function fetch(address) {
const res = await got.get(address);
const capture = cheerio.load(res.data);
capture('.gIMvvS').remove();
let metaInfoBox = capture('.iRdvCH')
.filter((index) => index === 0)
.get();
// For photography
if (metaInfoBox.length === 0) {
metaInfoBox = capture('.dPQBaW')
.filter((index) => index === 0)
.get();
}
const acquire = cheerio.load(metaInfoBox);
let time = acquire('.gimsRe').text();
// For `photography`
if (!time) {
time = acquire('.kHluJP').text();
}
// # Author(s) of the article
//
// There exists two formats for this section.
// 1. 文字 /...` with `摄影 / ..` in separate line, or
// 2. just a simple line starts with `文/ ...`
// For the first condition, we use a array
// to record the list of author(s), and use
// `;` to connect two lines.
const authors = [];
if (acquire('.loxoWO').text() === '') {
for (const item of acquire('.flciyI').get()) {
const $ = cheerio.load(item);
const job = $('.hGsNtm').text();
// An article may have multiple authors
const name = [];
for (const item of $('.cidPTd > a').get()) {
const $ = cheerio.load(item);
name.push($('.fJSaZP').text());
}
const author = job + '/' + name.join(',');
authors.push(author);
}
} else {
authors.push(acquire('.loxoWO').text());
}
const author = authors.join(';');
// contents = cover photo + italic intro + text
const contents = '<em>' + capture('.hxFBKc').html() + '</em><br>' + capture('.irqyDp').html();
return {
author: author,
description: contents,
link: address,
guid: address,
pubDate: new Date(time).toUTCString(),
};
};
+34
View File
@@ -0,0 +1,34 @@
const cheerio = require('cheerio');
const got = require('@/utils/got');
const fetch = require('./fetch_article');
module.exports = async (ctx) => {
const url = 'https://www.twreporter.org';
const res = await got.get(url);
const $ = cheerio.load(res.data);
const list = $('.gKMjSz').get();
const out = await Promise.all(
list.map(async (item) => {
const $ = cheerio.load(item);
const address = url + $('a').attr('href');
const title = $('.dpNivU').text();
const cache = await ctx.cache.get(address);
if (cache) {
return Promise.resolve(JSON.parse(cache));
}
const single = await fetch(address);
single.title = title;
ctx.cache.set(address, JSON.stringify(single));
return Promise.resolve(single);
})
);
ctx.state.data = {
title: `報道者 | 最新`,
link: url,
item: out,
};
};
+57
View File
@@ -0,0 +1,57 @@
const cheerio = require('cheerio');
const got = require('@/utils/got');
const fetch = require('./fetch_article');
module.exports = async (ctx) => {
const baseURL = 'https://www.twreporter.org';
const url = baseURL + `/photography`;
const res = await got.get(url);
const $ = cheerio.load(res.data);
const coverList = $('.WPJvn').get();
const commonList = $('.eVNsZf').get();
const coverView = await Promise.all(
coverList.map(async (item) => {
const $ = cheerio.load(item);
const address = baseURL + $('li > a').attr('href');
const title = $('.gRCDdm').text();
const cache = await ctx.cache.get(address);
if (cache) {
return Promise.resolve(JSON.parse(cache));
}
const single = await fetch(address);
single.title = title;
ctx.cache.set(address, JSON.stringify(single));
return Promise.resolve(single);
})
);
const listView = await Promise.all(
commonList.map(async (item) => {
const $ = cheerio.load(item);
const address = baseURL + $('li > a').attr('href');
const title = $('.etJLWI').text();
const cache = await ctx.cache.get(address);
if (cache) {
return Promise.resolve(JSON.parse(cache));
}
const single = await fetch(address);
single.title = title;
ctx.cache.set(address, JSON.stringify(single));
return Promise.resolve(single);
})
);
ctx.state.data = {
title: `報道者 | 影像`,
link: url,
item: coverView.concat(listView),
};
};