feat: add 武汉大学新闻网 rss (#3688)

This commit is contained in:
SChen00
2020-01-08 15:32:12 +08:00
committed by DIYgod
parent bd7cd285ac
commit fbe0085348
5 changed files with 234 additions and 0 deletions
+15
View File
@@ -111,6 +111,21 @@ pageClass: routes
</Route>
#### 江苏省教育考试院
<Route author="schen1024" example="/gov/jiangsu/eea/zcgd" path="/gov/jiangsu/eea/:type?" :paramsDesc="['分类, 默认为 `wdyw`, 具体参数见下表']">
注意: 其他栏目的内容格式不兼容, 且不便统一, 此处只做了下标的栏目
| 具体栏目 | 参数 |
| :------: | :--: |
| 招考要闻 | zkyw |
| 政策规定 | zcgd |
| 招考信息 | zkxx |
| 招考资料 | zkzl |
| 学习交流 | xxjl |
</Route>
### 山西省人民政府
#### 山西省人社厅
+28
View File
@@ -963,6 +963,34 @@ https://rsshub.app/**nuist**/`bulletin` 或 https://rsshub.app/**nuist**/`bullet
</Route>
### 武汉大学新闻网
<Route author="SChen1024" example="/whu/news/wdyw" path="/whu/news/:type?" :paramsDesc="['分类, 默认为 `wdyw`, 具体参数见下表']">
注意: 除了 `kydt` 代表学术动态,其余页面均是拼音首字母小写.
| **内容** | **参数** |
| :------: | :------: |
| 武大要闻 | wdyw |
| 媒体武大 | mtwd |
| 专题报道 | ztbd |
| 珞珈人物 | ljrw |
| 国际交流 | gjjl |
| 缤纷校园 | bfxy |
| 校友之声 | xyzs |
| 珞珈论坛 | ljlt |
| 新闻热线 | xwrx |
| 头条新闻 | ttxw |
| 综合新闻 | zhxw |
| 珞珈影像 | ljyx |
| 学术动态 | kydt |
| 点击排行 | djpx |
| 珞珈副刊 | ljfk |
| 校史钩沉 | xsgc |
| 来稿选登 | lgxd |
</Route>
## 西安电子科技大学
### 教务处
+10
View File
@@ -668,6 +668,7 @@ router.get('/kmust/job/jobfairs', require('./routes/universities/kmust/job/jobfa
// 武汉大学
router.get('/whu/cs/:type', require('./routes/universities/whu/cs'));
router.get('/whu/news/:type?', require('./routes/universities/whu/news'));
// 华中科技大学
router.get('/hust/auto/notice/:type?', require('./routes/universities/hust/aia/notice'));
@@ -1092,9 +1093,18 @@ router.get('/gov/province/:name/:category', require('./routes/gov/province'));
router.get('/gov/city/:name/:category', require('./routes/gov/city'));
router.get('/gov/statecouncil/briefing', require('./routes/gov/statecouncil/briefing'));
router.get('/gov/news/:uid', require('./routes/gov/news'));
// 苏州
router.get('/gov/suzhou/news/:uid', require('./routes/gov/suzhou/news'));
router.get('/gov/suzhou/doc', require('./routes/gov/suzhou/doc'));
// 江苏
router.get('/gov/jiangsu/eea/:type?', require('./routes/gov/jiangsu/eea'));
// 山西
router.get('/gov/shanxi/rst/:category', require('./routes/gov/shanxi/rst'));
// 湖南
router.get('/gov/hunan/notice/:type', require('./routes/gov/hunan/notice'));
// 中华人民共和国-海关总署
+84
View File
@@ -0,0 +1,84 @@
const got = require('@/utils/got');
const cheerio = require('cheerio');
// 参考 bit/jwc 北京理工大学的的页面写成
const baseUrl = 'http://www.jseea.cn/';
const siteTitle = '--江苏省教育考试院';
const catrgoryMap = {
zkyw: { title: '招考要闻', suffix: 'zkyw/zkyw_channel173_1.html' },
zcgd: { title: '政策规定', suffix: 'plank/plank_index_1.html' },
zkxx: { title: '招考信息', suffix: 'enrollment/enrollment_index_1.html' },
zkzl: { title: '招考资料', suffix: 'enrollmentinfo/enrollmentinfo_index_1.html' },
xxjl: { title: '学习交流', suffix: 'learningexchange/learningexchange_index_1.html' },
};
// 专门定义一个function用于加载文章内容
async function load(link) {
// 异步请求文章
const response = await got.get(link);
// 加载文章内容
const $ = cheerio.load(response.data);
// 提取文章内容
const description = $('#left-container').html();
// 返回解析的结果
return { description };
}
module.exports = async (ctx) => {
// 默认 正常规定 然后获取列表页面
const type = ctx.params.type || 'zcgd';
const listPageUrl = baseUrl + catrgoryMap[type].suffix;
const response = await got({
method: 'get',
url: listPageUrl,
headers: {
Referer: baseUrl,
},
});
const $ = cheerio.load(response.data);
// console.log(type+":"+listPageUrl);
// 获取当前页面的 list
const list = $('td.content_css2>ul>li');
const result = await Promise.all(
// 遍历每一篇文章
list
.map(async (item) => {
const $ = cheerio.load(list[item]); // 将列表项加载成 html
const $rel_url = $('a').attr('href'); // 获取 每一项的url
// 获取绝对路径 招考要闻是绝对路径
const $item_url = type === 'zkyw' ? $rel_url : baseUrl + $rel_url;
const $title = $('a').attr('title'); // 获取每个的标题
const date_txt = $.text().match(/[1-9]\d{3}-(0[1-9]|1[0-2])-(0[1-9]|[1-2][0-9]|3[0-1])/); // 匹配 yyyy-mm-dd格式时间
const $pubdate = new Date(date_txt); // 正则匹配发布时间 然后转换成时间
// console.log(item + ":" + $item_url + ":" + $title + ":" + date_txt);
// 列表上提取到的信息
// 标题 链接
const single = {
title: $title,
pubDate: $pubdate,
link: $item_url,
guid: $item_url,
};
// 对于列表的每一项, 单独获取 时间与详细内容
const other = await ctx.cache.tryGet($item_url, async () => await load($item_url));
// 合并解析后的结果集作为该篇文章最终的输出结果
return Promise.resolve(Object.assign({}, single, other));
})
.get()
);
ctx.state.data = {
title: catrgoryMap[type].title + siteTitle,
link: baseUrl,
description: catrgoryMap[type].title + siteTitle,
item: result,
};
};
+97
View File
@@ -0,0 +1,97 @@
const got = require('@/utils/got');
const cheerio = require('cheerio');
// 参考 bit/jwc 北京理工大学的的页面写成
const baseUrl = 'https://news.whu.edu.cn/';
const sizeTitle = '武汉大学新闻网';
const catrgoryMap = {
wdyw: '武大要闻',
mtwd: '媒体武大',
ztbd: '专题报道',
ljrw: '珞珈人物',
gjjl: '国际交流',
bfxy: '缤纷校园',
xyzs: '校友之声',
ljlt: '珞珈论坛',
xwrx: '新闻热线',
ttxw: '头条新闻',
zhxw: '综合新闻',
ljyx: '珞珈影像',
kydt: '学术动态',
djpx: '点击排行',
ljfk: '珞珈副刊',
xsgc: '校史钩沉',
lgxd: '来稿选登',
};
// 专门定义一个function用于加载文章内容
async function load(link) {
// 异步请求文章
const response = await got.get(link);
// 加载文章内容
const $ = cheerio.load(response.data);
// 正则匹配发布时间 然后转换成时间
const date_txt = $('div.news_attrib')
.text()
.match(/[1-9]\d{3}-(0[1-9]|1[0-2])-(0[1-9]|[1-2][0-9]|3[0-1])\s+(20|21|22|23|[0-1]\d):[0-5]\d/); // 匹配 yyyy-mm-dd hh:MM 格式时间
const pubDate = new Date(date_txt[0]);
// 提取文章内容
const description = $('div.v_news_content').html();
// 返回解析的结果
return { description, pubDate };
}
module.exports = async (ctx) => {
// 默认 武大要闻 然后获取列表页面
const type = ctx.params.type || 'wdyw';
const listPageUrl = baseUrl + type + '.htm';
const response = await got({
method: 'get',
url: listPageUrl,
headers: {
Referer: baseUrl,
},
});
const $ = cheerio.load(response.data);
// 获取当前页面的 list
const list = $('div.list>div>ul>li');
list.splice(0, 1); // 删除第一个元素 标题栏
const result = await Promise.all(
// 遍历每一篇文章
list
.map(async (item) => {
const $ = cheerio.load(list[item]); // 将列表项加载成 html
const $rel_url = $('div.infotitle>a').attr('href'); // 获取 每一项的url
const $item_url = baseUrl + $rel_url; // 获取绝对路径
const $title = $('div>a').attr('title'); // 获取每个的标题
// 列表上提取到的信息
// 标题 链接
const single = {
title: $title,
// pubDate: $pubdate,
link: $item_url,
guid: $item_url,
};
// 对于列表的每一项, 单独获取 时间与详细内容
const other = await ctx.cache.tryGet($item_url, async () => await load($item_url));
// 合并解析后的结果集作为该篇文章最终的输出结果
return Promise.resolve(Object.assign({}, single, other));
})
.get()
);
ctx.state.data = {
title: catrgoryMap[type] + sizeTitle,
link: baseUrl,
description: catrgoryMap[type] + sizeTitle,
item: result,
};
};