diff --git a/app/crawler/parsers.py b/app/crawler/parsers.py new file mode 100644 index 0000000..d266d11 --- /dev/null +++ b/app/crawler/parsers.py @@ -0,0 +1,139 @@ +import hashlib +from datetime import datetime +from typing import Any, Dict, List +from urllib.parse import urljoin + +from bs4 import BeautifulSoup + + +def parse_gxgp_api_response( + response_data: Dict[str, Any], + source_code: str, + source_name: str, + crawled_at: datetime, + category_id: int, +) -> List[Dict[str, Any]]: + if not response_data.get("success"): + return [] + data = response_data.get("result", {}).get("data", {}) + records = data.get("data", []) + if not records: + return [] + + results = [] + for record in records: + title = str(record.get("title", "")).strip() + if not title: + continue + + timestamp = record.get("publishDate") + if not timestamp: + continue + try: + publish_date = datetime.fromtimestamp(int(timestamp) / 1000) + except (ValueError, TypeError): + continue + + purchase_name = str(record.get("purchaseName", "")).strip() + article_id = record.get("articleId") + if not article_id: + continue + + content_url = ( + f"https://zfcg.gxzf.gov.cn/site/detail?" + f"parentId={category_id}&articleId={article_id}" + ) + + announce = { + "title": title, + "publish_date": publish_date, + "purchase_name": purchase_name, + "content_url": content_url, + "source_code": source_code, + "source_name": source_name, + "announcement_type": "purchase", + "crawl_mode": "auto", + "is_new": True, + "is_today": publish_date.date() == datetime.now().date(), + } + announce["content_hash"] = _generate_hash(announce) + results.append(announce) + + return results + + +def extract_pagination(response_data: Dict[str, Any]) -> Dict[str, Any]: + data = response_data.get("result", {}).get("data", {}) + return { + "total": data.get("total", 0), + "page_no": data.get("pageNo", 1), + "page_size": data.get("pageSize", 100), + "pages": data.get("pages", 0), + "empty": data.get("empty", True), + "has_next": data.get("hasNext", False), + "has_previous": data.get("hasPrevious", False), + } + + +def parse_dahuagov_html(html: str, crawled_at: datetime) -> List[Dict[str, Any]]: + soup = BeautifulSoup(html, "html.parser") + lists = soup.find_all("ul", class_="more-list") + if not lists: + return [] + + results = [] + base_url = "http://www.gxdh.gov.cn" + base_path = "/xxgk/zdlyxxgk/ggzypzly/zfcgly/cggg/" + + for ul in lists: + for li in ul.find_all("li"): + date_span = li.find("span") + if not date_span: + continue + date_text = date_span.get_text(strip=True) + try: + publish_date = datetime.strptime(date_text, "%Y-%m-%d") + except ValueError: + continue + + link_tag = li.find("a") + if not link_tag: + continue + title = link_tag.get("title", "") or link_tag.get_text(strip=True) + href = link_tag.get("href", "") + if not title or not href: + continue + + if href.startswith("./") or href.startswith("../"): + content_url = urljoin(base_url + base_path, href) + elif href.startswith("/"): + content_url = base_url + href + elif href.startswith("http"): + content_url = href + else: + content_url = urljoin(base_url + base_path, href) + + announce = { + "title": title, + "publish_date": publish_date, + "purchase_name": "大化瑶族自治县", + "content_url": content_url, + "source_code": "dahuagov", + "source_name": "大化县政府网采购公告", + "announcement_type": "purchase", + "crawl_mode": "auto", + "is_new": True, + "is_today": publish_date.date() == datetime.now().date(), + } + announce["content_hash"] = _generate_hash(announce) + results.append(announce) + + return results + + +def _generate_hash(ann: Dict[str, Any]) -> str: + content = ( + f"{ann['title']}|{ann['publish_date'].strftime('%Y-%m-%d')}" + f"|{ann['purchase_name']}|{ann['content_url']}|{ann['source_code']}" + ) + return hashlib.sha256(content.encode("utf-8")).hexdigest() diff --git a/tests/test_crawler/test_parsers.py b/tests/test_crawler/test_parsers.py new file mode 100644 index 0000000..ca0be54 --- /dev/null +++ b/tests/test_crawler/test_parsers.py @@ -0,0 +1,106 @@ +from datetime import datetime +from app.crawler.parsers import ( + parse_gxgp_api_response, + parse_dahuagov_html, + extract_pagination, +) + + +def make_api_response(records_data): + return { + "success": True, + "result": { + "data": { + "data": records_data, + "total": len(records_data), + "pageNo": 1, + "pageSize": 100, + "pages": 1, + "empty": len(records_data) == 0, + "hasNext": False, + "hasPrevious": False, + } + }, + } + + +def test_parse_gxgp_single_record(): + response = make_api_response([ + { + "title": "测试采购公告", + "publishDate": 1746720000000, + "purchaseName": "测试采购单位", + "articleId": 12345, + } + ]) + crawled_at = datetime(2026, 5, 9, 10, 0, 0) + results = parse_gxgp_api_response( + response, + source_code="ZcyAnnouncement1", + source_name="采购公告", + crawled_at=crawled_at, + category_id=66485, + ) + assert len(results) == 1 + assert results[0]["title"] == "测试采购公告" + assert results[0]["source_code"] == "ZcyAnnouncement1" + assert "content_hash" in results[0] + + +def test_parse_gxgp_empty_response(): + response = make_api_response([]) + crawled_at = datetime(2026, 5, 9, 10, 0, 0) + results = parse_gxgp_api_response( + response, "ZcyAnnouncement1", "采购公告", crawled_at, 66485 + ) + assert len(results) == 0 + + +def test_parse_gxgp_missing_title(): + response = make_api_response([ + {"title": "", "publishDate": 1746720000000, "purchaseName": "x", "articleId": 1} + ]) + crawled_at = datetime(2026, 5, 9, 10, 0, 0) + results = parse_gxgp_api_response( + response, "ZcyAnnouncement1", "采购公告", crawled_at, 66485 + ) + assert len(results) == 0 + + +def test_extract_pagination(): + response = make_api_response([]) + pagination = extract_pagination(response) + assert pagination["total"] == 0 + assert pagination["page_no"] == 1 + assert pagination["has_next"] is False + + +def test_parse_dahuagov_html(): + html = """ +
+