fix(route): harden njxzc, gxmzu and jou routes (#22435)

* fix(route): harden njxzc, gxmzu and jou routes

- parse all observed date formats (ISO, slash, Chinese with optional
  time) instead of a fixed YYYY-MM-DD pattern
- fall back to list data when an article page fails or lacks content,
  instead of crashing the whole feed on cheerio load(null)
- skip content fetching for off-site links (e.g. WeChat posts)
- only override title/pubDate when found on the detail page
- resolve relative links and images against the article URL
- deduplicate the list/detail scaffold into per-namespace utils
- remove dead response.status checks, unused ctx params and noise
  comments; fix namespace URLs and antiCrawler flags

* refactor(route): address review feedback for gxmzu, jou and njxzc

- translate code comments to English (AGENTS.md rule 53)
- select the date cell via $(selector, context) instead of .find(selector)
  to avoid unicorn/no-array-callback-reference false positives

* refactor(route): simplify gxmzu, jou and njxzc per review

- drop the unused type exports, unnecessary .first() calls and the
  unreachable no-content early returns
- restore the original selector object order in gxmzu/jou call sites
- gxmzu/jou: drop try-catch in resolveArticles (no failing article
  observed); keep the same-host guard for off-site list entries
- njxzc: drop the same-host guard (list links are same-host /_redirect
  URLs); keep try-catch for redirects to intranet-only subdomains
- gxmzu/lib: read the link text directly (library rows have no title
  attribute)

* refactor(route): drop jou host guard and njxzc intranet handling

- jou: remove the same-host guard — no off-site rows on the current
  first pages, which are all the route reads
- njxzc: remove the intranet-notice branch and the try-catch — every
  article on the current first pages loads fine from off-campus; drop
  the now-stale route descriptions as well

---------

Co-authored-by: real-jiakai <real.jiakai@gmail.com>
This commit is contained in:
Jiakai Gu
2026-07-20 13:08:00 +08:00
committed by GitHub
co-authored by real-jiakai
parent 9d486c3bda
commit 39aa7af02c
12 changed files with 371 additions and 352 deletions
+14 -8
View File
@@ -1,9 +1,11 @@
import { load } from 'cheerio';
import type { Route } from '@/types';
import ofetch from '@/utils/ofetch';
import { getNoticeList } from './utils';
import { parseNoticeList, resolveArticles } from './utils';
const url = 'https://ai.gxmzu.edu.cn/index/tzgg.htm';
const host = 'https://ai.gxmzu.edu.cn';
const pageUrl = 'https://ai.gxmzu.edu.cn/index/tzgg.htm';
export const route: Route = {
path: '/aitzgg',
@@ -13,7 +15,7 @@ export const route: Route = {
features: {
requireConfig: false,
requirePuppeteer: false,
antiCrawler: true,
antiCrawler: false,
supportBT: false,
supportPodcast: false,
supportScihub: false,
@@ -29,8 +31,12 @@ export const route: Route = {
url: 'ai.gxmzu.edu.cn/index/tzgg.htm',
};
async function handler(ctx) {
const out = await getNoticeList(ctx, url, host, 'a', '.timestyle55267', {
async function handler() {
const response = await ofetch(pageUrl);
const $ = load(response);
const list = parseNoticeList($, pageUrl, 'table.winstyle55267 tr[height="20"]', '.timestyle55267');
const items = await resolveArticles(list, pageUrl, {
title: '.titlestyle55269',
content: '#vsb_newscontent',
date: '.timestyle55269',
@@ -38,7 +44,7 @@ async function handler(ctx) {
return {
title: '广西民族大学人工智能学院 -- 通知公告',
link: url,
item: out,
link: pageUrl,
item: items,
};
}
+25 -39
View File
@@ -1,13 +1,11 @@
import { load } from 'cheerio';
import type { Route } from '@/types';
import cache from '@/utils/cache';
import ofetch from '@/utils/ofetch'; // 使用ofetch库代替got
import { parseDate } from '@/utils/parse-date';
import timezone from '@/utils/timezone';
import ofetch from '@/utils/ofetch';
const url = 'https://library.gxmzu.edu.cn/news/news_list.jsp?urltype=tree.TreeTempUrl&wbtreeid=1010';
const host = 'https://library.gxmzu.edu.cn';
import { parsePubDate, resolveArticles } from './utils';
const pageUrl = 'https://library.gxmzu.edu.cn/news/news_list.jsp?urltype=tree.TreeTempUrl&wbtreeid=1010';
export const route: Route = {
path: '/libzxxx',
@@ -17,7 +15,7 @@ export const route: Route = {
features: {
requireConfig: false,
requirePuppeteer: false,
antiCrawler: true,
antiCrawler: false,
supportBT: false,
supportPodcast: false,
supportScihub: false,
@@ -31,50 +29,38 @@ export const route: Route = {
maintainers: ['real-jiakai'],
handler,
url: 'library.gxmzu.edu.cn/news/news_list.jsp',
description: '部分消息发布于微信公众号等站外页面,此类消息仅输出标题与原文链接。',
};
async function handler() {
const response = await ofetch(url);
if (!response) {
return;
}
const response = await ofetch(pageUrl);
const $ = load(response);
const list = $('#newslist ul li')
.toArray()
.map((item) => {
item = $(item);
.map((el) => {
const $item = $(el);
const $link = $item.find('a');
const href = $link.attr('href');
if (!href) {
return null;
}
return {
title: item.find('a').text(),
link: new URL(item.find('a').attr('href'), host).href,
pubDate: timezone(parseDate(item.find('span').text(), 'YYYY-MM-DD'), 8),
title: $link.text().trim(),
link: new URL(href, pageUrl).href,
pubDate: parsePubDate($item.find('span').text()),
};
});
})
.filter((item) => item !== null);
const out = await Promise.all(
list.map((item) =>
cache.tryGet(item.link, async () => {
if (item.link && !item.link.startsWith('https://library.gxmzu.edu.cn/')) {
item.description = '该通知无法直接预览,请点击原文链接↑查看';
return item;
}
const response = await ofetch(item.link);
if (!response || (response.status >= 300 && response.status < 400)) {
item.description = '该通知无法直接预览,请点击原文链接↑查看';
} else {
const $ = load(response);
item.title = $('h2').text();
item.description = $('.v_news_content').html();
}
return item;
})
)
);
const items = await resolveArticles(list, pageUrl, {
title: 'h2',
content: '.v_news_content',
});
return {
title: '广西民族大学图书馆 -- 最新消息',
link: url,
item: out,
link: pageUrl,
item: items,
};
}
+1 -1
View File
@@ -2,6 +2,6 @@ import type { Namespace } from '@/types';
export const namespace: Namespace = {
name: '广西民族大学',
url: 'ai.gxmzu.edu.cn',
url: 'www.gxmzu.edu.cn',
lang: 'zh-CN',
};
+91 -64
View File
@@ -1,72 +1,99 @@
import { load } from 'cheerio';
import { type CheerioAPI, load } from 'cheerio';
import type { DataItem } from '@/types';
import cache from '@/utils/cache';
import ofetch from '@/utils/ofetch'; // 使用ofetch库代替got
import ofetch from '@/utils/ofetch';
import { parseDate } from '@/utils/parse-date';
import timezone from '@/utils/timezone';
async function getNoticeList(ctx, url, host, titleSelector, dateSelector, contentSelector) {
const response = await ofetch(url);
if (!response) {
return [];
}
const $ = load(response);
const FALLBACK_DESCRIPTION = '该通知无法直接预览,请点击原文链接↑查看';
const list = $('tr[height=20]')
.toArray()
.map((item) => {
item = $(item);
return {
title: item.find(titleSelector).attr('title'),
link: new URL(item.find(titleSelector).attr('href'), host).href,
pubDate: timezone(parseDate(item.find(dateSelector).text().trim(), 'YYYY-MM-DD'), 8),
};
});
const out = await Promise.all(
list.map((item) =>
cache.tryGet(item.link, async () => {
if (item.link.includes('.jsp')) {
// 特殊处理.jsp文件,直接显示消息而不尝试爬取
return {
...item,
description: '该通知无法直接预览,请点击原文链接↑查看',
};
}
const response = await ofetch(item.link);
if (!response || (response.status >= 300 && response.status < 400)) {
item.description = '该通知无法直接预览,请点击原文链接↑查看';
} else {
const $ = load(response);
item.title = $(contentSelector.title).text();
const hasEmbeddedPDFScript = $('script:contains("showVsbpdfIframe")').length > 0;
if (hasEmbeddedPDFScript) {
item.description = '该通知无法直接预览,请点击原文链接↑查看';
} else {
const $content = load($(contentSelector.content).html());
$content('a').each((_, el) => {
const a = $(el);
const href = a.attr('href');
if (href && !href.startsWith('http')) {
a.attr('href', new URL(href, host).href);
}
});
item.description = $content.html();
}
const preDate = $(contentSelector.date)
.text()
.replaceAll(/年|月/g, '-')
.replaceAll('日', '');
item.pubDate = timezone(parseDate(preDate), 8);
}
return item;
})
)
);
return out;
interface NoticeItem extends DataItem {
link: string;
}
export { getNoticeList };
interface DetailSelectors {
title: string;
content: string;
date?: string;
}
// Date formats vary across sites: 2026-05-28, 2026/06/02 or 2026年05月28日 14:53, often prefixed with a label such as "发布日期:"
export function parsePubDate(text?: string): Date | undefined {
const match = text?.match(/(\d{4})[-/.年]\s*(\d{1,2})[-/.月]\s*(\d{1,2})/);
if (!match) {
return undefined;
}
const [, year, month, day] = match;
const time = text?.match(/(\d{1,2}:\d{2}(?::\d{2})?)/)?.[1];
return timezone(parseDate(`${year}-${month.padStart(2, '0')}-${day.padStart(2, '0')}${time ? ` ${time}` : ''}`), 8);
}
export function parseNoticeList($: CheerioAPI, pageUrl: string, rowSelector: string, dateSelector: string): NoticeItem[] {
return $(rowSelector)
.toArray()
.map((el) => {
const $row = $(el);
const $link = $row.find('a');
const href = $link.attr('href');
if (!href) {
return null;
}
return {
title: $link.attr('title') || $link.text().trim(),
link: new URL(href, pageUrl).href,
pubDate: parsePubDate($(dateSelector, $row).text()),
};
})
.filter((item) => item !== null);
}
async function fetchArticle(item: NoticeItem, selectors: DetailSelectors): Promise<DataItem> {
const response = await ofetch(item.link);
const $ = load(response);
// Pages whose body is an embedded PDF have no extractable content
if ($('script:contains("showVsbpdfIframe")').length > 0) {
return { ...item, description: FALLBACK_DESCRIPTION };
}
const $content = $(selectors.content);
$content.find('a').each((_, el) => {
const $a = $(el);
const href = $a.attr('href');
if (href) {
$a.attr('href', new URL(href, item.link).href);
}
});
$content.find('img').each((_, el) => {
const $img = $(el);
const src = $img.attr('src');
if (src) {
$img.attr('src', new URL(src, item.link).href);
}
});
const title = $(selectors.title).text().trim();
const pubDate = selectors.date ? parsePubDate($(selectors.date).text()) : undefined;
return {
...item,
title: title || item.title,
pubDate: pubDate ?? item.pubDate,
description: $content.html() ?? item.description,
};
}
// Content of off-site links (e.g. WeChat posts or sibling-subdomain sites) is not fetched
export function resolveArticles(list: NoticeItem[], pageUrl: string, selectors: DetailSelectors): Promise<DataItem[]> {
const pageHost = new URL(pageUrl).host;
return Promise.all(
list.map((item) => {
if (new URL(item.link).host !== pageHost) {
return { ...item, description: FALLBACK_DESCRIPTION };
}
return cache.tryGet(item.link, () => fetchArticle(item, selectors)) as Promise<DataItem>;
})
);
}
+15 -8
View File
@@ -1,9 +1,11 @@
import { load } from 'cheerio';
import type { Route } from '@/types';
import ofetch from '@/utils/ofetch';
import { getNoticeList } from './utils';
import { parseNoticeList, resolveArticles } from './utils';
const url = 'https://yjs.gxmzu.edu.cn/tzgg/zsgg.htm';
const host = 'https://yjs.gxmzu.edu.cn';
const pageUrl = 'https://yjs.gxmzu.edu.cn/tzgg/zsgg.htm';
export const route: Route = {
path: '/yjszsgg',
@@ -13,7 +15,7 @@ export const route: Route = {
features: {
requireConfig: false,
requirePuppeteer: false,
antiCrawler: true,
antiCrawler: false,
supportBT: false,
supportPodcast: false,
supportScihub: false,
@@ -29,8 +31,13 @@ export const route: Route = {
url: 'yjs.gxmzu.edu.cn/tzgg/zsgg.htm',
};
async function handler(ctx) {
const out = await getNoticeList(ctx, url, host, 'a', '.timestyle55267', {
async function handler() {
const response = await ofetch(pageUrl);
const $ = load(response);
// The graduate school shares the same Boda CMS template with the AI college, so the list and article style IDs are identical
const list = parseNoticeList($, pageUrl, 'table.winstyle55267 tr[height="20"]', '.timestyle55267');
const items = await resolveArticles(list, pageUrl, {
title: '.titlestyle55269',
content: '#vsb_newscontent',
date: '.timestyle55269',
@@ -38,7 +45,7 @@ async function handler(ctx) {
return {
title: '广西民族大学研究生院 -- 招生公告',
link: url,
item: out,
link: pageUrl,
item: items,
};
}
+17 -11
View File
@@ -1,9 +1,11 @@
import { load } from 'cheerio';
import type { Route } from '@/types';
import ofetch from '@/utils/ofetch';
import { getItems } from './utils';
import { parseNoticeList, resolveArticles } from './utils';
const url = 'https://www.jou.edu.cn/index/tzgg.htm';
const host = 'https://www.jou.edu.cn';
const pageUrl = 'https://www.jou.edu.cn/index/tzgg.htm';
export const route: Route = {
path: '/tzgg',
@@ -29,16 +31,20 @@ export const route: Route = {
url: 'www.jou.edu.cn/index/tzgg.htm',
};
async function handler(ctx) {
const out = await getItems(ctx, url, host, 'winstyle106390', 'timestyle106390', 'titlestyle106402', 'timestyle106402');
async function handler() {
const response = await ofetch(pageUrl);
const $ = load(response);
const list = parseNoticeList($, pageUrl, 'table.winstyle106390 tr[height="20"]', '.timestyle106390');
const items = await resolveArticles(list, {
title: '.titlestyle106402',
content: '.v_news_content',
date: '.timestyle106402',
});
// 生成RSS源
return {
// 项目标题
title: '江苏海洋大学 -- 通知公告',
// 项目链接
link: url,
// items的内容
item: out,
link: pageUrl,
item: items,
};
}
+82 -62
View File
@@ -1,70 +1,90 @@
import { load } from 'cheerio';
import { type CheerioAPI, load } from 'cheerio';
import type { DataItem } from '@/types';
import cache from '@/utils/cache';
import ofetch from '@/utils/ofetch'; // 使用ofetch库
import ofetch from '@/utils/ofetch';
import { parseDate } from '@/utils/parse-date';
import timezone from '@/utils/timezone';
async function getItems(ctx, url, host, tableClass, timeStyleClass1, titleStyleClass, timeStyleClass2) {
const response = await ofetch(url);
if (!response) {
return [];
}
const $ = load(response);
const FALLBACK_DESCRIPTION = '该通知无法直接预览,请点击原文链接↑查看';
const list = $(`table.${tableClass} > tbody > tr[height=20]`)
.toArray()
.map((item) => {
const currentItem = $(item);
const item1 = currentItem.find('td:eq(1)');
const item2 = currentItem.find('td:eq(2)');
const link = new URL(item1.find('a').attr('href'), host).href;
return {
title: item1.find('a').attr('title'),
link,
pubDate: timezone(parseDate(item2.find(`.${timeStyleClass1}`).text(), 'YYYY-MM-DD'), 8),
};
});
const out = await Promise.all(
list.map((item) =>
cache.tryGet(item.link, async () => {
const response = await ofetch(item.link);
if (!response || (response.status >= 300 && response.status < 400)) {
// 响应为空或状态码表明发生了重定向
return {
...item,
description: '该通知无法直接预览,请点击原文链接↑查看',
};
}
const $ = load(response);
item.title = $(`.${titleStyleClass}`).text();
const hasEmbeddedPDFScript = $('script:contains("showVsbpdfIframe")').length > 0;
if (hasEmbeddedPDFScript) {
item.description = '该通知无法直接预览,请点击原文链接↑查看';
} else {
const contentHtml = $('.v_news_content').html();
const $content = load(contentHtml);
$content('a').each((_, el) => {
const a = $(el);
const href = a.attr('href');
if (href && !href.startsWith('http')) {
a.attr('href', new URL(href, host).href);
}
});
item.description = $content.html();
}
item.pubDate = timezone(parseDate($(`.${timeStyleClass2}`).text().replace('发布时间:', '')), 8);
return item;
})
)
);
return out;
interface NoticeItem extends DataItem {
link: string;
}
export { getItems };
interface DetailSelectors {
title: string;
content: string;
date?: string;
}
// Date formats vary across sites: 2026-03-30, 2026/05/06 or 2026年03月30日 14:37, often prefixed with a label such as "发布时间:"
export function parsePubDate(text?: string): Date | undefined {
const match = text?.match(/(\d{4})[-/.年]\s*(\d{1,2})[-/.月]\s*(\d{1,2})/);
if (!match) {
return undefined;
}
const [, year, month, day] = match;
const time = text?.match(/(\d{1,2}:\d{2}(?::\d{2})?)/)?.[1];
return timezone(parseDate(`${year}-${month.padStart(2, '0')}-${day.padStart(2, '0')}${time ? ` ${time}` : ''}`), 8);
}
export function parseNoticeList($: CheerioAPI, pageUrl: string, rowSelector: string, dateSelector: string): NoticeItem[] {
return $(rowSelector)
.toArray()
.map((el) => {
const $row = $(el);
const $link = $row.find('a');
const href = $link.attr('href');
if (!href) {
return null;
}
return {
title: $link.attr('title') || $link.text().trim(),
link: new URL(href, pageUrl).href,
pubDate: parsePubDate($(dateSelector, $row).text()),
};
})
.filter((item) => item !== null);
}
async function fetchArticle(item: NoticeItem, selectors: DetailSelectors): Promise<DataItem> {
const response = await ofetch(item.link);
const $ = load(response);
// Pages whose body is an embedded PDF have no extractable content
if ($('script:contains("showVsbpdfIframe")').length > 0) {
return { ...item, description: FALLBACK_DESCRIPTION };
}
const $content = $(selectors.content);
$content.find('a').each((_, el) => {
const $a = $(el);
const href = $a.attr('href');
if (href) {
$a.attr('href', new URL(href, item.link).href);
}
});
$content.find('img').each((_, el) => {
const $img = $(el);
const src = $img.attr('src');
if (src) {
$img.attr('src', new URL(src, item.link).href);
}
});
const title = $(selectors.title).text().trim();
const pubDate = selectors.date ? parsePubDate($(selectors.date).text()) : undefined;
return {
...item,
title: title || item.title,
pubDate: pubDate ?? item.pubDate,
description: $content.html() ?? item.description,
};
}
export function resolveArticles(list: NoticeItem[], selectors: DetailSelectors): Promise<DataItem[]> {
return Promise.all(list.map((item) => cache.tryGet(item.link, () => fetchArticle(item, selectors)) as Promise<DataItem>));
}
+17 -11
View File
@@ -1,9 +1,11 @@
import { load } from 'cheerio';
import type { Route } from '@/types';
import ofetch from '@/utils/ofetch';
import { getItems } from './utils';
import { parseNoticeList, resolveArticles } from './utils';
const url = 'https://yz.jou.edu.cn/index/zxgg.htm';
const host = 'https://yz.jou.edu.cn';
const pageUrl = 'https://yz.jou.edu.cn/index/zxgg.htm';
export const route: Route = {
path: '/yztzgg',
@@ -29,16 +31,20 @@ export const route: Route = {
url: 'yz.jou.edu.cn/index/zxgg.htm',
};
async function handler(ctx) {
const out = await getItems(ctx, url, host, 'winstyle207638', 'timestyle207638', 'titlestyle207543', 'timestyle207543');
async function handler() {
const response = await ofetch(pageUrl);
const $ = load(response);
const list = parseNoticeList($, pageUrl, 'table.winstyle207638 tr[height="20"]', '.timestyle207638');
const items = await resolveArticles(list, {
title: '.titlestyle207543',
content: '.v_news_content',
date: '.timestyle207543',
});
// 生成RSS源
return {
// 项目标题
title: '江苏海洋大学 -- 研招通知公告',
// 项目链接
link: url,
// items的内容
item: out,
link: pageUrl,
item: items,
};
}
+28 -20
View File
@@ -1,9 +1,11 @@
import { load } from 'cheerio';
import type { Route } from '@/types';
import ofetch from '@/utils/ofetch';
import { getNoticeList } from './utils';
import { parsePubDate, resolveArticles } from './utils';
const url = 'https://www.njxzc.edu.cn/89/list.htm';
const host = 'https://www.njxzc.edu.cn';
const pageUrl = 'https://www.njxzc.edu.cn/89/list.htm';
export const route: Route = {
path: '/tzgg',
@@ -13,7 +15,7 @@ export const route: Route = {
features: {
requireConfig: false,
requirePuppeteer: false,
antiCrawler: true,
antiCrawler: false,
supportBT: false,
supportPodcast: false,
supportScihub: false,
@@ -29,24 +31,30 @@ export const route: Route = {
url: 'www.njxzc.edu.cn/89/list.htm',
};
async function handler(ctx) {
const out = await getNoticeList(
ctx,
url,
host,
'a',
'.news_meta',
{
title: '.arti_title',
content: '.wp_articlecontent',
date: '.arti_update',
},
'.news_list .news'
);
async function handler() {
const response = await ofetch(pageUrl);
const $ = load(response);
const list = $('.news_list .news')
.toArray()
.map((el) => {
const $item = $(el);
const $link = $item.find('a');
const href = $link.attr('href');
if (!href) {
return null;
}
return {
title: $link.attr('title') || $link.text().trim(),
link: new URL(href, pageUrl).href,
pubDate: parsePubDate($item.find('.news_meta').text()),
};
})
.filter((item) => item !== null);
return {
title: '南京晓庄学院 -- 通知公告',
link: url,
item: out,
link: pageUrl,
item: await resolveArticles(list),
};
}
+20 -68
View File
@@ -1,13 +1,11 @@
import { load } from 'cheerio';
import type { Route } from '@/types';
import cache from '@/utils/cache';
import ofetch from '@/utils/ofetch';
import { parseDate } from '@/utils/parse-date';
import timezone from '@/utils/timezone';
const url = 'https://lib.njxzc.edu.cn/pxyhd/list.htm';
const host = 'https://lib.njxzc.edu.cn';
import { parsePubDate, resolveArticles } from './utils';
const pageUrl = 'https://lib.njxzc.edu.cn/pxyhd/list.htm';
export const route: Route = {
path: '/libtzgg',
@@ -17,7 +15,7 @@ export const route: Route = {
features: {
requireConfig: false,
requirePuppeteer: false,
antiCrawler: true,
antiCrawler: false,
supportBT: false,
supportPodcast: false,
supportScihub: false,
@@ -34,76 +32,30 @@ export const route: Route = {
};
async function handler() {
const response = await ofetch(url);
if (!response) {
return {
title: '南京晓庄学院 -- 图书馆通知公告',
link: url,
item: [],
};
}
const response = await ofetch(pageUrl);
const $ = load(response);
const list = $('a.btt-2')
.toArray()
.map((item) => {
const $item = $(item);
const href = $item.attr('href') || '';
const link = href.startsWith('http') ? href : new URL(href, host).href;
const day = $item.find('.tm-1').text().trim();
const yearMonth = $item.find('.tm-2').text().trim();
const dateStr = `${yearMonth}-${day}`;
.map((el) => {
const $link = $(el);
const href = $link.attr('href');
if (!href) {
return null;
}
const day = $link.find('.tm-1').text().trim();
const yearMonth = $link.find('.tm-2').text().trim();
return {
title: $item.find('.btt-4').text().trim(),
link,
pubDate: timezone(parseDate(dateStr, 'YYYY-MM-DD'), 8),
title: $link.find('.btt-4').text().trim(),
link: new URL(href, pageUrl).href,
pubDate: parsePubDate(`${yearMonth}-${day}`),
};
});
const out = await Promise.all(
list.map((item) =>
cache.tryGet(item.link, async () => {
const response = await ofetch(item.link);
const $ = load(response);
if ($('.wp_error_msg').length > 0) {
item.description = '您当前ip并非校内地址,该信息仅允许校内地址访问';
} else {
const $content = $('.wp_articlecontent');
// Convert wp_pdf_player iframes to download links
$content.find('.wp_pdf_player').each((_, el) => {
const $iframe = $(el);
const pdfSrc = $iframe.attr('pdfsrc') || '';
const pdfUrl = pdfSrc.startsWith('http') ? pdfSrc : new URL(pdfSrc, host).href;
$iframe.replaceWith(`<p><a href="${pdfUrl}">附件下载</a></p>`);
});
// Fix relative URLs
$content.find('a').each((_, el) => {
const $a = $(el);
const href = $a.attr('href');
if (href && !href.startsWith('http')) {
$a.attr('href', new URL(href, host).href);
}
});
item.description = $content.html() || '';
const title = $('.arti_title').text().trim();
if (title) {
item.title = title;
}
const dateText = $('.arti_update').text().replace('发布时间:', '').trim();
if (dateText) {
item.pubDate = timezone(parseDate(dateText, 'YYYY-MM-DD'), 8);
}
}
return item;
})
)
);
})
.filter((item) => item !== null);
return {
title: '南京晓庄学院 -- 图书馆通知公告',
link: url,
item: out,
link: pageUrl,
item: await resolveArticles(list),
};
}
+2 -1
View File
@@ -2,6 +2,7 @@ import type { Namespace } from '@/types';
export const namespace: Namespace = {
name: '南京晓庄学院',
url: 'lib.njxzc.edu.cn',
url: 'www.njxzc.edu.cn',
description: '部分文章仅限校内 IP 访问,此类文章仅输出标题与原文链接',
lang: 'zh-CN',
};
+59 -59
View File
@@ -1,67 +1,67 @@
import { load } from 'cheerio';
import type { DataItem } from '@/types';
import cache from '@/utils/cache';
import ofetch from '@/utils/ofetch'; // 使用默认导出的方式导入ofetch
import ofetch from '@/utils/ofetch';
import { parseDate } from '@/utils/parse-date';
import timezone from '@/utils/timezone';
async function getNoticeList(ctx, url, host, titleSelector, dateSelector, contentSelector, listSelector) {
const response = await ofetch(url);
if (!response) {
return [];
}
const $ = load(response);
const list = $(listSelector)
.toArray()
.map((item) => {
item = $(item);
const href = item.find(titleSelector).attr('href') || '';
const link = href.startsWith('http') ? href : new URL(href, host).href;
return {
title: item.find(titleSelector).attr('title'),
link,
pubDate: timezone(parseDate(item.find(dateSelector).text(), 'YYYY-MM-DD'), 8),
};
});
const out = await Promise.all(
list.map((item) =>
cache.tryGet(item.link, async () => {
const response = await ofetch(item.link);
const $ = load(response);
if ($('.wp_error_msg').length > 0) {
item.description = '您当前ip并非校内地址,该信息仅允许校内地址访问';
} else {
const $content = $(contentSelector.content);
// Convert wp_pdf_player iframes to download links
$content.find('.wp_pdf_player').each((_, el) => {
const $iframe = $(el);
const pdfSrc = $iframe.attr('pdfsrc') || '';
const pdfUrl = pdfSrc.startsWith('http') ? pdfSrc : new URL(pdfSrc, host).href;
$iframe.replaceWith(`<p><a href="${pdfUrl}">附件下载</a></p>`);
});
// Fix relative URLs
$content.find('a').each((_, el) => {
const $a = $(el);
const href = $a.attr('href');
if (href && !href.startsWith('http')) {
$a.attr('href', new URL(href, host).href);
}
});
item.description = $content.html() || '';
item.title = $(contentSelector.title).text();
const dateText = $(contentSelector.date).text().replace('编辑:', '').replace('发布日期:', '').replace('发布时间:', '');
item.pubDate = timezone(parseDate(dateText, 'YYYY-MM-DD'), 8);
}
return item;
})
)
);
return out;
interface NoticeItem extends DataItem {
link: string;
}
export { getNoticeList };
// Date formats vary across the site: 2026-07-01, 2026/07/01 or 2026年07月01日 14:53, often prefixed with a label such as "发布时间:"
export function parsePubDate(text?: string): Date | undefined {
const match = text?.match(/(\d{4})[-/.年]\s*(\d{1,2})[-/.月]\s*(\d{1,2})/);
if (!match) {
return undefined;
}
const [, year, month, day] = match;
const time = text?.match(/(\d{1,2}:\d{2}(?::\d{2})?)/)?.[1];
return timezone(parseDate(`${year}-${month.padStart(2, '0')}-${day.padStart(2, '0')}${time ? ` ${time}` : ''}`), 8);
}
async function fetchArticle(item: NoticeItem): Promise<DataItem> {
const response = await ofetch(item.link);
const $ = load(response);
const $content = $('.wp_articlecontent');
$content.find('.wp_pdf_player').each((_, el) => {
const $player = $(el);
const pdfSrc = $player.attr('pdfsrc');
if (pdfSrc) {
$player.replaceWith(`<p><a href="${new URL(pdfSrc, item.link).href}">附件下载</a></p>`);
} else {
$player.remove();
}
});
$content.find('a').each((_, el) => {
const $a = $(el);
const href = $a.attr('href');
if (href) {
$a.attr('href', new URL(href, item.link).href);
}
});
$content.find('img').each((_, el) => {
const $img = $(el);
const src = $img.attr('src');
if (src) {
$img.attr('src', new URL(src, item.link).href);
}
});
const title = $('.arti_title').text().trim();
const pubDate = parsePubDate($('.arti_update').text());
return {
...item,
title: title || item.title,
pubDate: pubDate ?? item.pubDate,
description: $content.html() ?? item.description,
};
}
export function resolveArticles(list: NoticeItem[]): Promise<DataItem[]> {
return Promise.all(list.map((item) => cache.tryGet(item.link, () => fetchArticle(item)) as Promise<DataItem>));
}