fix: 改进正文提取,支持政采云 SPA API 和大化县政府网 div.article-con

- 新增 _extract_zcy_content(): 通过政采云隐藏 API 获取公告正文
- 新增 _extract_dahuagov_content(): 大化县专用提取,支持 div.article-con
- 通用提取器新增更多选择器
This commit is contained in:
2026-05-26 15:47:48 +08:00
parent 754214692e
commit 68215aa804
+120 -4
View File
@@ -151,8 +151,21 @@ def parse_dahuagov_detail_pubdate(html: str) -> datetime | None:
async def extract_page_content(url: str, timeout: int = 30) -> str | None:
"""抓取详情页并用 BeautifulSoup 提取正文纯文本"""
import urllib.parse
import httpx
# ---- 站点专用处理 ----
# ① 政采云 SPA (HTTP API, 无需渲染)
if "zfcg.gxzf.gov.cn" in url and "articleId=" in url:
return await _extract_zcy_content(url, timeout)
# ② 大化县政府网 (静态HTML)
if "www.gxdh.gov.cn" in url or "gxdh.gov.cn" in url:
return await _extract_dahuagov_content(url, timeout)
# ---- 通用 HTML 提取 ----
try:
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
@@ -172,15 +185,16 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None:
"div.Custom_UnionStyle", "div.pages_content", "div#content",
"div.main-content", "article", ".article", ".detail-content",
".text-content", ".news-content", ".detail-article",
"div.article-con", ".trs_editor_view", ".TRS_UEDITOR",
".trs_paper_default", ".article-content",
]
for selector in selectors:
container = soup.select_one(selector)
if container:
# 移除脚本和样式
for tag in container.find_all(["script", "style"]):
tag.decompose()
text = container.get_text(separator="\n", strip=True)
if len(text) > 50: # 至少50字才算有效正文
if len(text) > 50:
return text
# 兜底:取 body 内所有文本
@@ -189,9 +203,8 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None:
for tag in body.find_all(["script", "style", "nav", "footer", "header"]):
tag.decompose()
text = body.get_text(separator="\n", strip=True)
# 移除过长的空白行
lines = [l.strip() for l in text.split("\n") if l.strip()]
text = "\n".join(lines[:200]) # 最多取前200行
text = "\n".join(lines[:200])
if len(text) > 50:
return text
@@ -200,6 +213,109 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None:
return None
async def _extract_zcy_content(url: str, timeout: int = 30) -> str | None:
"""从政采云 SPA 隐藏 API 提取公告正文"""
import urllib.parse
import httpx
from bs4 import BeautifulSoup
params = urllib.parse.parse_qs(urllib.parse.urlparse(url).query)
article_id = params.get("articleId", [None])[0]
parent_id = params.get("parentId", [None])[0]
if not article_id:
return None
api_url = "https://zfcg.gxzf.gov.cn/portal/detail"
if parent_id:
api_url += f"?articleId={article_id}&parentId={parent_id}"
else:
api_url += f"?articleId={article_id}"
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
"Accept": "application/json, text/plain, */*",
"Referer": url,
}
try:
async with httpx.AsyncClient(timeout=timeout) as client:
resp = await client.get(api_url, headers=headers)
if resp.status_code != 200:
return None
data = resp.json()
if not data.get("success"):
return None
content_html = data.get("result", {}).get("data", {}).get("content", "")
if not content_html:
return None
soup = BeautifulSoup(content_html, "html.parser")
for tag in soup.find_all(["script", "style"]):
tag.decompose()
text = soup.get_text(separator="\n", strip=True)
# 清理过短行和多余空白
lines = [l.strip() for l in text.split("\n") if len(l.strip()) > 5]
text = "\n".join(lines[:300])
return text if len(text) > 50 else None
except Exception:
return None
async def _extract_dahuagov_content(url: str, timeout: int = 30) -> str | None:
"""从大化县政府网详情页提取正文"""
import httpx
from bs4 import BeautifulSoup
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
"Accept": "text/html,application/xhtml+xml",
"Accept-Language": "zh-CN,zh;q=0.9",
}
try:
async with httpx.AsyncClient(timeout=timeout, follow_redirects=True) as client:
resp = await client.get(url, headers=headers)
if resp.status_code != 200:
return None
soup = BeautifulSoup(resp.text, "html.parser")
# 大化县政府网正文容器
selectors = [
"div.article-con", ".trs_editor_view", ".TRS_UEDITOR",
".trs_paper_default", "div.content", "div.TRS_Editor",
"div.article-content", "div.Custom_UnionStyle",
"div#content", "div.main-content", "article", ".detail-content",
]
for selector in selectors:
container = soup.select_one(selector)
if container:
for tag in container.find_all(["script", "style"]):
tag.decompose()
text = container.get_text(separator="\n", strip=True)
if len(text) > 50:
lines = [l.strip() for l in text.split("\n") if l.strip()]
return "\n".join(lines[:300])
# 兜底
body = soup.find("body")
if body:
for tag in body.find_all(["script", "style", "nav", "footer", "header"]):
tag.decompose()
text = body.get_text(separator="\n", strip=True)
lines = [l.strip() for l in text.split("\n") if l.strip()]
text = "\n".join(lines[:200])
if len(text) > 50:
return text
return None
except Exception:
return None
def _generate_hash(ann: dict[str, Any]) -> str:
content = (
f"{ann['title']}|{ann['publish_date'].strftime('%Y-%m-%d')}"