From 60f3cd5799309cc9ead6841a9acd8d0edd47befe Mon Sep 17 00:00:00 2001 From: "t@123654" <1327444968@qq.com> Date: Wed, 11 Jun 2025 19:44:08 +0800 Subject: [PATCH] v1.4.1 --- core/models/article.py | 2 +- core/wx/wx1.py | 59 ++++++++++++++++++++------------------- core/wx/wx2.py | 59 ++++++++++++++++++++------------------- db.db | Bin 53248 -> 0 bytes init_sys.py | 1 + jobs/fetch_no_article.py | 12 ++++---- jobs/webhook.py | 3 +- publish.bat | 15 ++++++---- 8 files changed, 81 insertions(+), 70 deletions(-) delete mode 100644 db.db diff --git a/core/models/article.py b/core/models/article.py index e26ccae6..5e249e96 100644 --- a/core/models/article.py +++ b/core/models/article.py @@ -6,7 +6,7 @@ class Article(Base): title = Column(String(500)) pic_url = Column(String(500)) url=Column(String(500)) - content = Column(MEDIUMTEXT) + content = Column(Text) description=Column(String(800)) status = Column(Integer,default=1) publish_time = Column(Integer) diff --git a/core/wx/wx1.py b/core/wx/wx1.py index a61c59c8..150942a0 100644 --- a/core/wx/wx1.py +++ b/core/wx/wx1.py @@ -6,39 +6,42 @@ import yaml import re from bs4 import BeautifulSoup from .base import WxGather - +from core.log import logger # 继承 BaseGather 类 class MpsApi(WxGather): # 重写 content_extract 方法 def content_extract(self, url): - session=self.session - r = session.get(url, headers=self.headers) - if r.status_code == 200: - text = r.text - if text is None: - return - soup = BeautifulSoup(text, 'html.parser') - # 找到内容 - js_content_div = soup.find('div', {'id': 'js_content'}) - # 移除style属性中的visibility: hidden; - if js_content_div is None: - return - js_content_div.attrs.pop('style', None) - # 找到所有的img标签 - img_tags = js_content_div.find_all('img') - # 遍历每个img标签并修改属性,设置宽度为1080p - for img_tag in img_tags: - if 'data-src' in img_tag.attrs: - img_tag['src'] = img_tag['data-src'] - del img_tag['data-src'] - if 'style' in img_tag.attrs: - style = img_tag['style'] - # 使用正则表达式替换width属性 - style = re.sub(r'width\s*:\s*\d+\s*px', 'width: 1080px', style) - img_tag['style'] = style - return js_content_div.prettify() - return None + try: + session=self.session + r = session.get(url, headers=self.headers) + if r.status_code == 200: + text = r.text + if text is None: + return + soup = BeautifulSoup(text, 'html.parser') + # 找到内容 + js_content_div = soup.find('div', {'id': 'js_content'}) + # 移除style属性中的visibility: hidden; + if js_content_div is None: + return + js_content_div.attrs.pop('style', None) + # 找到所有的img标签 + img_tags = js_content_div.find_all('img') + # 遍历每个img标签并修改属性,设置宽度为1080p + for img_tag in img_tags: + if 'data-src' in img_tag.attrs: + img_tag['src'] = img_tag['data-src'] + del img_tag['data-src'] + if 'style' in img_tag.attrs: + style = img_tag['style'] + # 使用正则表达式替换width属性 + style = re.sub(r'width\s*:\s*\d+\s*px', 'width: 1080px', style) + img_tag['style'] = style + return js_content_div.prettify() + except Exception as e: + logger.error(e) + return "" # 重写 get_Articles 方法 def get_Articles(self, faker_id:str=None,Mps_id:str=None,Mps_title="",CallBack=None,begin=0,MaxPage:int=1,interval=1,Gather_Content=False,Item_Over_CallBack=None,Over_CallBack=None): super().Start(mp_id=Mps_id) diff --git a/core/wx/wx2.py b/core/wx/wx2.py index 096a69f0..de82428b 100644 --- a/core/wx/wx2.py +++ b/core/wx/wx2.py @@ -6,39 +6,42 @@ import yaml import re from bs4 import BeautifulSoup from .base import WxGather - +from core.log import logger # 继承 BaseGather 类 class MpsWeb(WxGather): # 重写 content_extract 方法 def content_extract(self, url): - session=self.session - r = session.get(url, headers=self.headers) - if r.status_code == 200: - text = r.text - if text is None: - return - soup = BeautifulSoup(text, 'html.parser') - # 找到内容 - js_content_div = soup.find('div', {'id': 'js_content'}) - # 移除style属性中的visibility: hidden; - if js_content_div is None: - return - js_content_div.attrs.pop('style', None) - # 找到所有的img标签 - img_tags = js_content_div.find_all('img') - # 遍历每个img标签并修改属性,设置宽度为1080p - for img_tag in img_tags: - if 'data-src' in img_tag.attrs: - img_tag['src'] = img_tag['data-src'] - del img_tag['data-src'] - if 'style' in img_tag.attrs: - style = img_tag['style'] - # 使用正则表达式替换width属性 - style = re.sub(r'width\s*:\s*\d+\s*px', 'width: 1080px', style) - img_tag['style'] = style - return js_content_div.prettify() - return None + try: + session=self.session + r = session.get(url, headers=self.headers) + if r.status_code == 200: + text = r.text + if text is None: + return + soup = BeautifulSoup(text, 'html.parser') + # 找到内容 + js_content_div = soup.find('div', {'id': 'js_content'}) + # 移除style属性中的visibility: hidden; + if js_content_div is None: + return + js_content_div.attrs.pop('style', None) + # 找到所有的img标签 + img_tags = js_content_div.find_all('img') + # 遍历每个img标签并修改属性,设置宽度为1080p + for img_tag in img_tags: + if 'data-src' in img_tag.attrs: + img_tag['src'] = img_tag['data-src'] + del img_tag['data-src'] + if 'style' in img_tag.attrs: + style = img_tag['style'] + # 使用正则表达式替换width属性 + style = re.sub(r'width\s*:\s*\d+\s*px', 'width: 1080px', style) + img_tag['style'] = style + return js_content_div.prettify() + except Exception as e: + logger.error(e) + return "" # 重写 get_Articles 方法 def get_Articles(self, faker_id:str=None,Mps_id:str=None,Mps_title="",CallBack=None,begin:int=0,MaxPage:int=1,interval=1,Gather_Content=False,Item_Over_CallBack=None,Over_CallBack=None): super().Start(mp_id=Mps_id) diff --git a/db.db b/db.db deleted file mode 100644 index 988252bd9d1fc577e0a9c8d11006e5c0a7132875..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 53248 zcmeI)&u-&H90zcF+vZ=>&4NHaRq~-y8w8ssrR;J*BHNlaG@Iu;1M@pwF)`TTyK*mC6U=XJxU za>sT};>#<-13?spk7Zd9gbe$eV}Icz%|1+r3+!DS`+VKUjPS7k<6PlSA)j6o?%Rc5 z=cN1Vxu5fYXBK%jV_Js;#xnYHO*yyj+wU&9>awsn_3?M|Jw7 z+oN(@*=1|S*6P<*L7V+Zr`dVNLKYpS5aFtF1kGOWBKzBv-7?rlg&cCUk_Ss!zNlPc`(&a&Aa53PofstIhc%W%81t z7tcGHw6wS=em5JCM=C@XvJvXYDtRfFW$o7*ZDm7gMeIiPP91t>1&f$;q#6s%Xme^h zUBV9eV!=74`|6=>AE`aJJ8EZjGQg(8z2aMIu5GFG#2IbLd%)xB`NZ#eVUL3xaVhJ7 zuCfEpb(Z<<+EaxW_Z;i*LPjXss1G@Es$+TbH{?e1knDgzFOx~AcI zY!C3{Pq&*vJI^$nI@@p#3{wSN`f4mqMOVb@s3Z%otF*g$fYo9EE$UFl=P8w-Q*DD zfLaOOGR4%*Rm(xu^!YneL5f|h1S!^`RQH0H=}3wNMK`6`{}Y+vl);q-JJKL9W5us1 zE-VDo|Ku7Nspa5p=6pv?OY`&M#kYYgaeYJUQZIbJ7tsw%ZvlKo{`iJJw)gRQjN3Sd z7MtG!vxzy6m$9!ApM7};a~)s4>f7t_wX>Yl+wU6Q;eQMRn^)MRYP1k*sopwmC z>%TX+vFJm`I^fs;#ll|#d*T5B2tWV=5P$##AOHafKmY;|fWYe~Fg;bsE)UNCUkZhn zuipSMA_zbL0uX=z1Rwwb2tWV=5P(3!1u|1oHu!A;?*EtYOu+CV009U<00Izz00bZa z0SG`K5dygXUn0VcK|%lm5P$##AOHafKmY;|fIz|paQ#2w34q~400Izz00bZa0SG_< z0uX>eA_Q>#KM`TZARzz&2tWV=5P$##AOHafKp^1)IR8(00$}(MfB*y_009U<00Izz L00bbA2!X!=MDg9^ diff --git a/init_sys.py b/init_sys.py index c5dae527..6d45d9df 100644 --- a/init_sys.py +++ b/init_sys.py @@ -26,6 +26,7 @@ def sync_models(): # 同步模型到表结构 from core.data_sync import ModelSync DB.create_tables() + time.sleep(3) sync=ModelSync(eng=DB.get_engine()) sync.sync_all() print_info("模型同步完成") diff --git a/jobs/fetch_no_article.py b/jobs/fetch_no_article.py index d23b5517..d0d26fe3 100644 --- a/jobs/fetch_no_article.py +++ b/jobs/fetch_no_article.py @@ -2,6 +2,7 @@ from core.models.article import Article from core.db import DB from core.wx.base import WxGather from time import sleep +from core.print import print_success,print_error import random def fetch_articles_without_content(): """ @@ -24,24 +25,21 @@ def fetch_articles_without_content(): else: url = f"https://mp.weixin.qq.com/s/{article.id}" - print(f"正在处理文章: {article.id}, URL: {url}") + print(f"正在处理文章: {article.title}, URL: {url}") # 获取内容 content = ga.content_extract(url) - sleep(random.randint(1,5)) + sleep(random.randint(3,10)) if content: # 更新内容 article.content = content session.commit() - print(f"成功更新文章 {article.title} 的内容") + print_success(f"成功更新文章 {article.title} 的内容") else: - print(f"获取文章 {article.title} 内容失败") + print_error(f"获取文章 {article.title} 内容失败") except Exception as e: print(f"处理过程中发生错误: {e}") - session.rollback() - finally: - session.close() from core.task import TaskScheduler scheduler=TaskScheduler() from core.config import cfg diff --git a/jobs/webhook.py b/jobs/webhook.py index 703afe19..06bfb1a1 100644 --- a/jobs/webhook.py +++ b/jobs/webhook.py @@ -110,7 +110,8 @@ def web_hook(hook:MessageWebHook): # 处理articles参数,兼容Article对象和字典类型 processed_articles = [] if len(hook.articles)<=0: - raise ValueError("没有更新到文章") + # raise ValueError("没有更新到文章") + logger.warning("没有更新到文章") return for article in hook.articles: if isinstance(article, dict): diff --git a/publish.bat b/publish.bat index 0afec43a..de470663 100644 --- a/publish.bat +++ b/publish.bat @@ -33,9 +33,14 @@ if %WEB_FLAG%==1 ( REM 读取Python配置文件中的版本号 for /f "tokens=1 delims==" %%v in ('python -c "from core.ver import VERSION; print(VERSION)"') do set VERSION=%%v -set tag="v%VERSION%" +if "%VERSION%"=="" ( + echo 错误:无法从core.ver.py读取版本号 + exit /b 1 +) +set tag=v%VERSION% echo 当前版本: %VERSION% TAG: %tag% + REM 设置comment if %COMMENT_FLAG%==1 ( set comment=%USER_COMMENT% @@ -45,13 +50,13 @@ if %COMMENT_FLAG%==1 ( if exist %version_file% ( for /f "usebackq delims=" %%a in (%version_file%) do set comment=%%a ) else ( - echo 警告:未找到对应版本号的文件 %version_file% + echo "警告:未找到对应版本号的文件%version_file%" ) ) -echo %comment% +echo "%comment%" git add . -git tag "v%VERSION%" +git tag -a "v%VERSION%" -m "%VERSION% %comment%" git commit -m "%VERSION% %comment%" REM 执行git操作 @@ -60,4 +65,4 @@ if %PUSH_FLAG%==1 ( git push origin %tag% git push -u gitee main git push gitee %tag% -) \ No newline at end of file +) \ No newline at end of file