fix: 改进正文提取,支持政采云 SPA API 和大化县政府网 div.article-con
- 新增 _extract_zcy_content(): 通过政采云隐藏 API 获取公告正文 - 新增 _extract_dahuagov_content(): 大化县专用提取,支持 div.article-con - 通用提取器新增更多选择器
This commit is contained in:
+120
-4
@@ -151,8 +151,21 @@ def parse_dahuagov_detail_pubdate(html: str) -> datetime | None:
|
||||
|
||||
async def extract_page_content(url: str, timeout: int = 30) -> str | None:
|
||||
"""抓取详情页并用 BeautifulSoup 提取正文纯文本"""
|
||||
import urllib.parse
|
||||
|
||||
import httpx
|
||||
|
||||
# ---- 站点专用处理 ----
|
||||
|
||||
# ① 政采云 SPA (HTTP API, 无需渲染)
|
||||
if "zfcg.gxzf.gov.cn" in url and "articleId=" in url:
|
||||
return await _extract_zcy_content(url, timeout)
|
||||
|
||||
# ② 大化县政府网 (静态HTML)
|
||||
if "www.gxdh.gov.cn" in url or "gxdh.gov.cn" in url:
|
||||
return await _extract_dahuagov_content(url, timeout)
|
||||
|
||||
# ---- 通用 HTML 提取 ----
|
||||
try:
|
||||
headers = {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
|
||||
@@ -172,15 +185,16 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None:
|
||||
"div.Custom_UnionStyle", "div.pages_content", "div#content",
|
||||
"div.main-content", "article", ".article", ".detail-content",
|
||||
".text-content", ".news-content", ".detail-article",
|
||||
"div.article-con", ".trs_editor_view", ".TRS_UEDITOR",
|
||||
".trs_paper_default", ".article-content",
|
||||
]
|
||||
for selector in selectors:
|
||||
container = soup.select_one(selector)
|
||||
if container:
|
||||
# 移除脚本和样式
|
||||
for tag in container.find_all(["script", "style"]):
|
||||
tag.decompose()
|
||||
text = container.get_text(separator="\n", strip=True)
|
||||
if len(text) > 50: # 至少50字才算有效正文
|
||||
if len(text) > 50:
|
||||
return text
|
||||
|
||||
# 兜底:取 body 内所有文本
|
||||
@@ -189,9 +203,8 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None:
|
||||
for tag in body.find_all(["script", "style", "nav", "footer", "header"]):
|
||||
tag.decompose()
|
||||
text = body.get_text(separator="\n", strip=True)
|
||||
# 移除过长的空白行
|
||||
lines = [l.strip() for l in text.split("\n") if l.strip()]
|
||||
text = "\n".join(lines[:200]) # 最多取前200行
|
||||
text = "\n".join(lines[:200])
|
||||
if len(text) > 50:
|
||||
return text
|
||||
|
||||
@@ -200,6 +213,109 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
async def _extract_zcy_content(url: str, timeout: int = 30) -> str | None:
|
||||
"""从政采云 SPA 隐藏 API 提取公告正文"""
|
||||
import urllib.parse
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
params = urllib.parse.parse_qs(urllib.parse.urlparse(url).query)
|
||||
article_id = params.get("articleId", [None])[0]
|
||||
parent_id = params.get("parentId", [None])[0]
|
||||
if not article_id:
|
||||
return None
|
||||
|
||||
api_url = "https://zfcg.gxzf.gov.cn/portal/detail"
|
||||
if parent_id:
|
||||
api_url += f"?articleId={article_id}&parentId={parent_id}"
|
||||
else:
|
||||
api_url += f"?articleId={article_id}"
|
||||
|
||||
headers = {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
|
||||
"Accept": "application/json, text/plain, */*",
|
||||
"Referer": url,
|
||||
}
|
||||
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=timeout) as client:
|
||||
resp = await client.get(api_url, headers=headers)
|
||||
if resp.status_code != 200:
|
||||
return None
|
||||
|
||||
data = resp.json()
|
||||
if not data.get("success"):
|
||||
return None
|
||||
|
||||
content_html = data.get("result", {}).get("data", {}).get("content", "")
|
||||
if not content_html:
|
||||
return None
|
||||
|
||||
soup = BeautifulSoup(content_html, "html.parser")
|
||||
for tag in soup.find_all(["script", "style"]):
|
||||
tag.decompose()
|
||||
text = soup.get_text(separator="\n", strip=True)
|
||||
|
||||
# 清理过短行和多余空白
|
||||
lines = [l.strip() for l in text.split("\n") if len(l.strip()) > 5]
|
||||
text = "\n".join(lines[:300])
|
||||
return text if len(text) > 50 else None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
async def _extract_dahuagov_content(url: str, timeout: int = 30) -> str | None:
|
||||
"""从大化县政府网详情页提取正文"""
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
headers = {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
|
||||
"Accept": "text/html,application/xhtml+xml",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9",
|
||||
}
|
||||
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=timeout, follow_redirects=True) as client:
|
||||
resp = await client.get(url, headers=headers)
|
||||
if resp.status_code != 200:
|
||||
return None
|
||||
|
||||
soup = BeautifulSoup(resp.text, "html.parser")
|
||||
|
||||
# 大化县政府网正文容器
|
||||
selectors = [
|
||||
"div.article-con", ".trs_editor_view", ".TRS_UEDITOR",
|
||||
".trs_paper_default", "div.content", "div.TRS_Editor",
|
||||
"div.article-content", "div.Custom_UnionStyle",
|
||||
"div#content", "div.main-content", "article", ".detail-content",
|
||||
]
|
||||
for selector in selectors:
|
||||
container = soup.select_one(selector)
|
||||
if container:
|
||||
for tag in container.find_all(["script", "style"]):
|
||||
tag.decompose()
|
||||
text = container.get_text(separator="\n", strip=True)
|
||||
if len(text) > 50:
|
||||
lines = [l.strip() for l in text.split("\n") if l.strip()]
|
||||
return "\n".join(lines[:300])
|
||||
|
||||
# 兜底
|
||||
body = soup.find("body")
|
||||
if body:
|
||||
for tag in body.find_all(["script", "style", "nav", "footer", "header"]):
|
||||
tag.decompose()
|
||||
text = body.get_text(separator="\n", strip=True)
|
||||
lines = [l.strip() for l in text.split("\n") if l.strip()]
|
||||
text = "\n".join(lines[:200])
|
||||
if len(text) > 50:
|
||||
return text
|
||||
return None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _generate_hash(ann: dict[str, Any]) -> str:
|
||||
content = (
|
||||
f"{ann['title']}|{ann['publish_date'].strftime('%Y-%m-%d')}"
|
||||
|
||||
Reference in New Issue
Block a user