""" 政策雷达 - 政府网站抓取工具集 每个函数返回标准化格式: {"title": str, "date": str, "url": str, "content": str} 列表函数返回: [{"title": str, "date": str, "url": str}, ...] """ import re import json import urllib.request import urllib.error import ssl from html import unescape USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" TIMEOUT = 15 # 忽略 SSL 证书验证(部分政府站点使用自签名证书) _SSL_CTX = ssl.create_default_context() _SSL_CTX.check_hostname = False _SSL_CTX.verify_mode = ssl.CERT_NONE def _fetch(url, headers=None): """通用 HTTP GET,返回解码后的 HTML 文本""" hdrs = {"User-Agent": USER_AGENT} if headers: hdrs.update(headers) req = urllib.request.Request(url, headers=hdrs) with urllib.request.urlopen(req, timeout=TIMEOUT, context=_SSL_CTX) as resp: raw = resp.read() for enc in ("utf-8", "gbk", "gb2312"): try: return raw.decode(enc) except UnicodeDecodeError: continue return raw.decode("utf-8", errors="replace") def _extract_paragraphs(html): """从 HTML 中提取正文段落(通用)""" paragraphs = re.findall(r"]*>(.*?)

", html, re.DOTALL) result = [] for p in paragraphs: text = re.sub(r"<[^>]+>", "", p).strip() text = unescape(text) if text and len(text) > 10: result.append(text) return result # ============================================================ # 优先级 1: 中央政府门户 # ============================================================ def gov_search(keyword, page=1): """ 中国政府网 - 搜索 API(可搜索新闻发布会等) URL: https://sousuo.www.gov.cn/search-gov/data?t=zhengcelibrary&q=... 返回 JSON,包含 title/url/summary 注意: 索引有 1-2 周延迟,最新内容可能尚未收录 """ import urllib.parse q = urllib.parse.quote(keyword) api_url = ( f"https://sousuo.www.gov.cn/search-gov/data?t=zhengcelibrary" f"&q={q}&p={page}&n=20&timetype=timeqb&mintime=&maxtime=" ) html = _fetch(api_url) try: data = json.loads(html) except json.JSONDecodeError: return [] results = [] search_result = data.get("searchVO", {}).get("catMap", {}).get("zhengcelibrary", {}) items = search_result.get("listVO", []) if isinstance(search_result, dict) else [] for item in items: title = re.sub(r"<[^>]+>", "", item.get("title", "")).strip() url = item.get("url", "") date = item.get("pubtime", "")[:10] if item.get("pubtime") else "" if title and url: results.append({"title": title, "date": date, "url": url}) return results def gov_zhengce_list(): """ 中国政府网 - 国务院政策文件库 URL: https://www.gov.cn/zhengce/ 结构: 服务端渲染,
  • 标题日期
  • """ html = _fetch("https://www.gov.cn/zhengce/") items = re.findall( r'
  • \s*]*>\s*([^<]+)\s*\s*\s*([\d-]+)\s*', html, ) results = [] for url, title, date in items: title = title.strip() if not title: continue if url.startswith("./"): url = "https://www.gov.cn/zhengce/" + url[2:] results.append({"title": title, "date": date, "url": url}) return results def gov_article(url): """ 中国政府网 - 政策文章正文 正文在
    内的

    标签中 """ html = _fetch(url) # 提取标题 title_match = re.search(r'([^<]+)', html) title = title_match.group(1).strip() if title_match else "" # 提取发布日期 date_match = re.search(r"发布时间[::]\s*([\d年\-./]+)", html) date = date_match.group(1).strip() if date_match else "" # 提取正文 content_paras = _extract_paragraphs(html) return { "title": title, "date": date, "url": url, "content": "\n\n".join(content_paras), } def scio_press_list(): """ 国新办 - 新闻发布会文字实录 注意: scio.gov.cn 有反爬保护,此处使用 PBC 站点作为镜像 PBC 的 "国新办举行新闻发布会" 系列完整保留了国新办原文 URL: https://www.pbc.gov.cn/goutongjiaoliu/113456/113469/index.html """ html = _fetch( "https://www.pbc.gov.cn/goutongjiaoliu/113456/113469/index.html" ) items = re.findall( r']*href="(/goutongjiaoliu/[^"]+)"[^>]*>([^<]{10,})', html, ) results = [] seen = set() for href, title in items: title = title.strip() if title and href not in seen and "国新办" in title: seen.add(href) results.append( { "title": title, "date": "", "url": "https://www.pbc.gov.cn" + href, } ) return results def xinhuanet_list(): """ 新华社 - 受权发布/时政 注意: 使用 www.news.cn(新域名),旧 xinhuanet.com 只有 2021 年旧数据 URL: https://www.news.cn/ 结构: 服务端渲染,标题在

    文章链接格式: /{channel}/{YYYYMMDD}/{32-hex}/c.html """ html = _fetch("https://www.news.cn/") pattern = r'
    ]*>]*href=[\'"](https?://www\.news\.cn/[^"\']+/20\d{6}/[a-f0-9]{32}/c\.html)[\'"][^>]*>([^<]+)' items = re.findall(pattern, html) results = [] seen = set() for url, title in items: title = title.strip() if title and url not in seen: seen.add(url) date_match = re.search(r"/(20\d{6})/", url) date = "" if date_match: d = date_match.group(1) date = f"{d[:4]}-{d[4:6]}-{d[6:8]}" results.append({"title": title, "date": date, "url": url}) return results def xinhuanet_article(url): """ 新华社 - 文章正文 标题:

    ...

    正文:
    内的

    标签 """ html = _fetch(url) title_match = re.search(r']*>.*?]*>([^<]+)', html, re.DOTALL) if not title_match: title_match = re.search(r'([^<]+)', html) title = title_match.group(1).strip() if title_match else "" # 提取正文 content_match = re.search(r'id="detailContent"[^>]*>(.*?)

    \s*(?:标题 """ html = _fetch( "https://www.pbc.gov.cn/goutongjiaoliu/113456/113469/index.html" ) items = re.findall( r']*href="(/goutongjiaoliu/[^"]+)"[^>]*>([^<]{10,})', html, ) results = [] seen = set() for href, title in items: title = title.strip() if title and href not in seen and len(title) > 5: seen.add(href) # 从 URL 提取日期 (格式: 2026071518015458230) date_match = re.search(r"/(20\d{6})\d{11}/", href) date = "" if date_match: d = date_match.group(1) date = f"{d[:4]}-{d[4:6]}-{d[6:8]}" results.append( { "title": title, "date": date, "url": "https://www.pbc.gov.cn" + href, } ) return results def pbc_article(url): """中国人民银行 - 文章正文(正文在

    标签中)""" html = _fetch(url) title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" content_paras = _extract_paragraphs(html) return { "title": title, "date": "", "url": url, "content": "\n\n".join(content_paras), } def mof_list(): """ 财政部 - 财政新闻/政策 URL: https://www.mof.gov.cn/zhengwuxinxi/caizhengxinwen/ 结构: 服务端渲染,标题 + 日期 """ html = _fetch("https://www.mof.gov.cn/zhengwuxinxi/caizhengxinwen/") items = re.findall( r'href="([^"]*t2026[^"]+\.htm)"[^>]*>\s*([^<]{10,80})\s*\s*(?:.*?(\d{4}[-/]\d{2}[-/]\d{2}))?', html, re.DOTALL, ) results = [] seen = set() for url, title, date in items: title = title.strip() if title and url not in seen: seen.add(url) if url.startswith("./") or url.startswith("t20"): url = "https://www.mof.gov.cn/zhengwuxinxi/caizhengxinwen/" + url.replace("./", "") results.append({"title": title, "date": date or "", "url": url}) return results def mof_article(url): """财政部 - 文章正文""" html = _fetch(url) title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" date_match = re.search(r"发布日期[::]\s*([\d年\-./]+)", html) date = date_match.group(1).strip() if date_match else "" content_paras = _extract_paragraphs(html) return { "title": title, "date": date, "url": url, "content": "\n\n".join(content_paras), } def ndrc_list(page=1): """ 发改委 - 产业政策/规划 URL: https://www.ndrc.gov.cn/xxgk/ 结构: 服务端渲染,

    分页: index_{page-1}.html """ if page == 1: list_url = "https://www.ndrc.gov.cn/xxgk/" else: list_url = f"https://www.ndrc.gov.cn/xxgk/index_{page - 1}.html" html = _fetch(list_url) items = re.findall( r']*href="([^"]+)"[^>]*>\s*([^<]{10,})\s*\s*\s*([\d/]+)\s*', html, ) results = [] for url, title, date in items: title = title.strip() if not title: continue if url.startswith("./"): url = "https://www.ndrc.gov.cn" + url[1:] results.append({"title": title, "date": date, "url": url}) return results def ndrc_article(url): """发改委 - 文章正文(正文在 div.TRS_Editor 内)""" html = _fetch(url) title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" editor_match = re.search(r'class="TRS_Editor"[^>]*>(.*?)
    \s*<', html, re.DOTALL) if editor_match: content_paras = _extract_paragraphs(editor_match.group(1)) else: content_paras = _extract_paragraphs(html) return { "title": title, "date": "", "url": url, "content": "\n\n".join(content_paras), } def csrc_list(page=1): """ 证监会 - 主席讲话/政策公告 URL: https://www.csrc.gov.cn/csrc/c100028/ 结构: HTML 页面(API 返回空结果,需用 common/searchList 获取 HTML) HTML URL: /common/searchList/{channelid}?_isAgg=true&_isJson=true&_pageSize=18&_template=index&page={N} """ channel_id = "c100028" url = f"https://www.csrc.gov.cn/common/searchList/{channel_id}?_isAgg=true&_isJson=true&_pageSize=18&_template=index&page={page}" html = _fetch(url) items = re.findall( r']*href="(/csrc/c100028/c\d+/content\.shtml)"[^>]*>([^<]{10,})', html, ) results = [] seen = set() for path, title in items: title = title.strip() if title and path not in seen: seen.add(path) results.append( { "title": title, "date": "", "url": "https://www.csrc.gov.cn" + path, } ) return results def csrc_article(url): """证监会 - 文章正文""" html = _fetch(url) title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" content_paras = _extract_paragraphs(html) return { "title": title, "date": "", "url": url, "content": "\n\n".join(content_paras), } def mofcom_list(): """ 商务部 - 外贸/消费政策 URL: https://www.mofcom.gov.cn/ 结构: 政策列表页使用 JSON API API: /api-gateway/jpaas-publish-server/front/page/build/unit """ html = _fetch("https://www.mofcom.gov.cn/") # 首页新闻是 SSR items = re.findall( r']*href="([^"]+)"[^>]*>\s*([^<]{10,80})\s*', html, ) results = [] seen = set() for url, title in items: title = title.strip() if title and url not in seen and len(title) > 8: seen.add(url) if url.startswith("/"): url = "https://www.mofcom.gov.cn" + url results.append({"title": title, "date": "", "url": url}) return results def mofcom_article(url): """商务部 - 文章正文(正文在 div.wms-con 内)""" html = _fetch(url) title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" content_match = re.search(r'class="wms-con"[^>]*>(.*?)
    \s*<', html, re.DOTALL) if content_match: content_paras = _extract_paragraphs(content_match.group(1)) else: content_paras = _extract_paragraphs(html) return { "title": title, "date": "", "url": url, "content": "\n\n".join(content_paras), } # ============================================================ # 优先级 3: 权威媒体转载 # ============================================================ def people_editorial_list(): """ 人民日报 - 社论/观点 URL: http://opinion.people.com.cn/GB/8213/49160/index.html 结构: 主页面有多个子栏目,每个子栏目页面有
  • 文章列表 文章URL格式: /n1/2026/0623/c223228-40745676.html (相对) 或完整URL """ base = "http://opinion.people.com.cn/GB/8213/49160/" html = _fetch(base + "index.html") # 获取所有子栏目链接(完整URL) sub_urls = re.findall(r'href="(http://opinion\.people\.com\.cn/GB/8213/49160/\d+/index\.html)"', html) sub_urls = list(dict.fromkeys(sub_urls)) results = [] seen = set() for sub in sub_urls: try: sub_html = _fetch(sub) # 解析
  • 项目(注意:人民网的链接使用单引号 href='...') lis = re.findall(r"]*>(.*?)
  • ", sub_html, re.DOTALL) for li in lis: # 匹配单引号或双引号的 href m = re.search(r"href=['\"](/n1/\d{4}/\d{4}/c[\d-]+\.html)['\"][^>]*>([^<]{10,})", li) if m: url_path, title = m.groups() title = title.strip() url = "http://opinion.people.com.cn" + url_path if title and url not in seen: seen.add(url) date_match = re.search(r"/n1/(20\d{2})/(\d{4})/", url) date = "" if date_match: year, md = date_match.groups() date = f"{year}-{md[:2]}-{md[2:]}" results.append({"title": title, "date": date, "url": url}) except Exception: continue return results def people_article(url): """人民日报 - 文章正文(正文在

    标签中,class=rm_txt_zw 的容器内)""" html = _fetch(url) title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" # 正文在 p 标签中 content_paras = _extract_paragraphs(html) return { "title": title, "date": "", "url": url, "content": "\n\n".join(content_paras), } def cctv_list(page=1): """ 央视新闻 - 新闻联播文字版 URL: https://news.cctv.com/lbj/ 结构: JSONP API API: https://news.cctv.com/2019/07/gaiban/cmsdatainterface/page/china_{page}.jsonp?callback=china 注意: 需要设置 Referer 头 """ api_url = f"https://news.cctv.com/2019/07/gaiban/cmsdatainterface/page/china_{page}.jsonp?callback=china" raw = _fetch(api_url, headers={"Referer": "https://news.cctv.com/"}) # 解析 JSONP: china({...}) json_match = re.search(r"[^(]+\((\{.*\})\)", raw, re.DOTALL) if not json_match: return [] try: data = json.loads(json_match.group(1)) except json.JSONDecodeError: return [] items = data.get("data", {}).get("list", []) if isinstance(data.get("data"), dict) else [] results = [] for item in items: title = item.get("title", "").strip() url = item.get("url", item.get("link", "")) date = item.get("focus_date", item.get("date", "")) if title and url: results.append( { "title": title, "date": str(date)[:10], "url": url, } ) return results def cctv_article(url): """央视新闻 - 文章正文(正文在 div.content_area 内)""" html = _fetch(url) title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" content_match = re.search(r'class="content_area"[^>]*>(.*?)\s*<', html, re.DOTALL) if content_match: content_paras = _extract_paragraphs(content_match.group(1)) else: content_paras = _extract_paragraphs(html) return { "title": title, "date": "", "url": url, "content": "\n\n".join(content_paras), } def ce_list(): """ 经济日报 - 部委政策解读 URL: http://www.ce.cn/ 结构: 服务端渲染,文章链接格式: /xwzx/gnsz/gdxw/202607/t20260721_xxxxxxx.shtml 政策相关文章在 /gnsz/ (国内时政) 和 /xwzx/ (新闻) 栏目下 """ html = _fetch("http://www.ce.cn/") items = re.findall( r']*href="([^"]*t2026[^"]+\.shtml)"[^>]*>([^<]{10,})', html, ) results = [] seen = set() for url, title in items: title = title.strip() # 过滤政策相关文章(国内时政/新闻栏目) if "/gnsz/" not in url and "/xwzx/" not in url and "/syj/" not in url: continue if title and url not in seen: seen.add(url) if url.startswith("/"): url = "http://www.ce.cn" + url # 从 URL 提取日期 date_match = re.search(r"/(20\d{6})/", url) date = "" if date_match: d = date_match.group(1) date = f"{d[:4]}-{d[4:6]}-{d[6:8]}" results.append({"title": title, "date": date, "url": url}) return results def ce_article(url): """经济日报 - 文章正文(正文在 div.content.clearfix 内)""" html = _fetch(url) title_match = re.search(r'id="NewsArticleTitle"[^>]*>([^<]+)', html) if not title_match: title_match = re.search(r"([^<]+)", html) title = title_match.group(1).strip() if title_match else "" content_match = re.search(r'class="content clearfix"[^>]*>(.*?)\s*<', html, re.DOTALL) if content_match: content_paras = _extract_paragraphs(content_match.group(1)) else: content_paras = _extract_paragraphs(html) return { "title": title, "date": "", "url": url, "content": "\n\n".join(content_paras), } # ============================================================ # 统一入口 # ============================================================ SCRAPERS = { "gov.cn": {"list": gov_zhengce_list, "article": gov_article, "name": "中国政府网"}, "xinhuanet": {"list": xinhuanet_list, "article": xinhuanet_article, "name": "新华社"}, "pbc": {"list": pbc_list, "article": pbc_article, "name": "中国人民银行"}, "mof": {"list": mof_list, "article": mof_article, "name": "财政部"}, "ndrc": {"list": ndrc_list, "article": ndrc_article, "name": "发改委"}, "csrc": {"list": csrc_list, "article": csrc_article, "name": "证监会"}, "mofcom": {"list": mofcom_list, "article": mofcom_article, "name": "商务部"}, "people": {"list": people_editorial_list, "article": people_article, "name": "人民日报"}, "cctv": {"list": cctv_list, "article": cctv_article, "name": "央视新闻"}, "ce": {"list": ce_list, "article": ce_article, "name": "经济日报"}, "scio": {"list": scio_press_list, "article": pbc_article, "name": "国新办(PBC镜像)"}, } def scrape_list(site_key, **kwargs): """统一列表抓取入口""" scraper = SCRAPERS.get(site_key) if not scraper: return [] return scraper["list"](**kwargs) def scrape_article(site_key, url): """统一文章抓取入口""" scraper = SCRAPERS.get(site_key) if not scraper: return {"title": "", "date": "", "url": url, "content": ""} return scraper["article"](url)