diff --git a/app/crawler/parsers.py b/app/crawler/parsers.py index f79c855..25415ce 100644 --- a/app/crawler/parsers.py +++ b/app/crawler/parsers.py @@ -151,8 +151,21 @@ def parse_dahuagov_detail_pubdate(html: str) -> datetime | None: async def extract_page_content(url: str, timeout: int = 30) -> str | None: """抓取详情页并用 BeautifulSoup 提取正文纯文本""" + import urllib.parse + import httpx + # ---- 站点专用处理 ---- + + # ① 政采云 SPA (HTTP API, 无需渲染) + if "zfcg.gxzf.gov.cn" in url and "articleId=" in url: + return await _extract_zcy_content(url, timeout) + + # ② 大化县政府网 (静态HTML) + if "www.gxdh.gov.cn" in url or "gxdh.gov.cn" in url: + return await _extract_dahuagov_content(url, timeout) + + # ---- 通用 HTML 提取 ---- try: headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36", @@ -172,15 +185,16 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None: "div.Custom_UnionStyle", "div.pages_content", "div#content", "div.main-content", "article", ".article", ".detail-content", ".text-content", ".news-content", ".detail-article", + "div.article-con", ".trs_editor_view", ".TRS_UEDITOR", + ".trs_paper_default", ".article-content", ] for selector in selectors: container = soup.select_one(selector) if container: - # 移除脚本和样式 for tag in container.find_all(["script", "style"]): tag.decompose() text = container.get_text(separator="\n", strip=True) - if len(text) > 50: # 至少50字才算有效正文 + if len(text) > 50: return text # 兜底:取 body 内所有文本 @@ -189,9 +203,8 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None: for tag in body.find_all(["script", "style", "nav", "footer", "header"]): tag.decompose() text = body.get_text(separator="\n", strip=True) - # 移除过长的空白行 lines = [l.strip() for l in text.split("\n") if l.strip()] - text = "\n".join(lines[:200]) # 最多取前200行 + text = "\n".join(lines[:200]) if len(text) > 50: return text @@ -200,6 +213,109 @@ async def extract_page_content(url: str, timeout: int = 30) -> str | None: return None +async def _extract_zcy_content(url: str, timeout: int = 30) -> str | None: + """从政采云 SPA 隐藏 API 提取公告正文""" + import urllib.parse + + import httpx + from bs4 import BeautifulSoup + + params = urllib.parse.parse_qs(urllib.parse.urlparse(url).query) + article_id = params.get("articleId", [None])[0] + parent_id = params.get("parentId", [None])[0] + if not article_id: + return None + + api_url = "https://zfcg.gxzf.gov.cn/portal/detail" + if parent_id: + api_url += f"?articleId={article_id}&parentId={parent_id}" + else: + api_url += f"?articleId={article_id}" + + headers = { + "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36", + "Accept": "application/json, text/plain, */*", + "Referer": url, + } + + try: + async with httpx.AsyncClient(timeout=timeout) as client: + resp = await client.get(api_url, headers=headers) + if resp.status_code != 200: + return None + + data = resp.json() + if not data.get("success"): + return None + + content_html = data.get("result", {}).get("data", {}).get("content", "") + if not content_html: + return None + + soup = BeautifulSoup(content_html, "html.parser") + for tag in soup.find_all(["script", "style"]): + tag.decompose() + text = soup.get_text(separator="\n", strip=True) + + # 清理过短行和多余空白 + lines = [l.strip() for l in text.split("\n") if len(l.strip()) > 5] + text = "\n".join(lines[:300]) + return text if len(text) > 50 else None + except Exception: + return None + + +async def _extract_dahuagov_content(url: str, timeout: int = 30) -> str | None: + """从大化县政府网详情页提取正文""" + import httpx + from bs4 import BeautifulSoup + + headers = { + "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36", + "Accept": "text/html,application/xhtml+xml", + "Accept-Language": "zh-CN,zh;q=0.9", + } + + try: + async with httpx.AsyncClient(timeout=timeout, follow_redirects=True) as client: + resp = await client.get(url, headers=headers) + if resp.status_code != 200: + return None + + soup = BeautifulSoup(resp.text, "html.parser") + + # 大化县政府网正文容器 + selectors = [ + "div.article-con", ".trs_editor_view", ".TRS_UEDITOR", + ".trs_paper_default", "div.content", "div.TRS_Editor", + "div.article-content", "div.Custom_UnionStyle", + "div#content", "div.main-content", "article", ".detail-content", + ] + for selector in selectors: + container = soup.select_one(selector) + if container: + for tag in container.find_all(["script", "style"]): + tag.decompose() + text = container.get_text(separator="\n", strip=True) + if len(text) > 50: + lines = [l.strip() for l in text.split("\n") if l.strip()] + return "\n".join(lines[:300]) + + # 兜底 + body = soup.find("body") + if body: + for tag in body.find_all(["script", "style", "nav", "footer", "header"]): + tag.decompose() + text = body.get_text(separator="\n", strip=True) + lines = [l.strip() for l in text.split("\n") if l.strip()] + text = "\n".join(lines[:200]) + if len(text) > 50: + return text + return None + except Exception: + return None + + def _generate_hash(ann: dict[str, Any]) -> str: content = ( f"{ann['title']}|{ann['publish_date'].strftime('%Y-%m-%d')}"