#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
涡阳县人民政府 - 环境影响评价 爬虫
站点: https://www.gy.gov.cn (涡阳县, 非广元!)
栏目: 政府信息公开 > 涡阳县生态环境分局 > 建设项目环评文件审批 (branchId=806, columnId=111001001)

两种数据源:
  A. XxgkSearch 搜索: /XxgkSearch/?branchId=885&keyword=环境影响评价&field=title&match=like&sort=relevant&page=N
     (搜索结果页, 分页为 ?page=N 形式; branchId=885 下仅 2 条, 全站 37 条)
  B. XxgkContent 列表: /XxgkContent/showList/806/111001001/page_{n}.html
     (列表栏目, 分页 page_N.html 清晰, 共 159 条 / 11 页)  ← 默认数据源

用法:
  python3 crawl_gyxxgk_eia.py                # 抓 B 列表, 默认 1 页
  python3 crawl_gyxxgk_eia.py --pages 5      # 抓 B 列表 5 页
  python3 crawl_gyxxgk_eia.py --search --pages 2   # 抓 A 搜索前 2 页
  python3 crawl_gyxxgk_eia.py --search --branch 806 --pages 2  # A 搜索指定分支
"""

import re
import sys
import time
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.gy.gov.cn"
LIST_TPL = BASE_URL + "/XxgkContent/showList/806/111001001/page_%d.html"
SEARCH_URL = BASE_URL + "/XxgkSearch/"
SITE_NAME = "涡阳县-环境影响评价"
CATEGORY = "环境影响评价"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Referer": BASE_URL + "/",
}
TIMEOUT = 30

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb  # noqa: E402


def fetch(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=TIMEOUT, verify=False)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if attempt == 2:
                print("  [WARN] 获取失败 (%d/3): %s - %s" % (attempt + 1, url, e), file=sys.stderr)
                return ""
        time.sleep(1.5)
    return ""


def clean_title(t):
    """标题清洗: 去空白/控制字符, 压缩多余空格"""
    if not t:
        return ""
    t = t.replace("\u3000", " ").replace("\xa0", " ").replace("\n", " ").replace("\r", " ").replace("\t", " ")
    t = re.sub(r"\s+", " ", t).strip()
    return t


def extract_list_items(html):
    """B 列表: section.m-tglist > ul > li > a(title) + span(date)"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    # 列表主体区域
    section = soup.find("section", class_=lambda c: c and "m-tglist" in (c if isinstance(c, str) else " ".join(c)))
    if not section:
        return items
    for li in section.find_all("li"):
        a_tag = li.find("a", href=True)
        if not a_tag:
            continue
        href = a_tag.get("href", "")
        if "XxgkContent/show" not in href:
            continue
        title = clean_title(a_tag.get("title") or a_tag.get_text(" ", strip=True))
        if not title:
            continue
        span_tag = li.find("span")
        date = span_tag.get_text(strip=True) if span_tag else ""
        items.append((urljoin(BASE_URL, href), title, date))
    return items


def extract_search_items(html):
    """A 搜索: 结构与列表相同 (li > a + span)"""
    return extract_list_items(html)


def absolutize_attachments(content_div, soup_html):
    """把正文 HTML 内的相对附件/图片链接绝对化, 返回 (新HTML, 附件列表)"""
    attachments = []
    for a in content_div.find_all("a", href=True):
        href = a["href"].strip()
        if not href or href.startswith("javascript"):
            continue
        low = href.lower()
        is_attach = ("/upload_bz/" in href or ".pdf" in low or ".doc" in low or ".xls" in low
                     or ".zip" in low or ".rar" in low or ".wps" in low or ".jpg" in low or ".png" in low)
        if is_attach:
            abs_url = urljoin(BASE_URL, href)
            a["href"] = abs_url
            name = clean_title(a.get_text(" ", strip=True)) or href.split("/")[-1].split("?")[0]
            attachments.append({"name": name, "url": abs_url})
    return soup_html, attachments


def extract_detail(html):
    """详情页: 标题 div.u-title; 日期 meta PubDate/发布时间; 正文 div#zoom(保留 <p> HTML)"""
    soup = BeautifulSoup(html, "html.parser")

    # 标题
    title = ""
    t = soup.find("div", class_="u-title")
    if t:
        title = clean_title(t.get_text(" ", strip=True))
    if not title:
        tt = soup.find("title")
        if tt and tt.string:
            title = clean_title(re.sub(r"[-—–].*$", "", tt.string.strip()))

    # 日期
    date = ""
    m = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2}(?:\s+\d{1,2}:\d{2})?)", html)
    if m:
        date = m.group(1)
    else:
        m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="([^"]+)"', html, re.I)
        if m:
            date = m.group(1)

    # 正文: div#zoom (g-detailbox) 保留 <p> HTML
    content = ""
    attachments = []
    content_div = soup.find("div", id="zoom")
    if not content_div:
        content_div = soup.find("div", class_=lambda c: c and "g-detailbox" in (c if isinstance(c, str) else " ".join(c)))
    if content_div:
        # 去掉“文本下载”按钮等非正文元素
        for node in content_div.select(".wzbot, .download_btn, .share-main, script, style"):
            node.decompose()
        # 附件链接绝对化
        _, attachments = absolutize_attachments(content_div, None)
        # 只保留 <p> 元素, 保留内嵌 <a>/<img>
        parts = []
        for p in content_div.find_all("p"):
            if p.find("a") or p.find("img"):
                ph = str(p)
            else:
                # 纯文本 <p>, 保留标签但清掉内部多余空白
                txt = p.get_text(" ", strip=True)
                if not txt:
                    continue
                inner = re.sub(r"\s+", " ", txt).strip()
                ph = "<p>" + inner + "</p>"
            parts.append(ph)
        content = "\n".join(parts)
        # 兜底: 正文无 <p> 但有直接文本 (如裸文本/&nbsp; 内容)
        if not parts:
            direct = []
            for node in content_div.children:
                if getattr(node, "name", None) is None:
                    txt = re.sub(r"[\s\u3000\u00a0]+", " ", str(node)).strip()
                    if txt:
                        direct.append("<p>" + txt + "</p>")
            if direct:
                content = "\n".join(direct)

    # 附件区: div.m-download (位于正文容器之外)
    dl_div = soup.find("div", class_=lambda c: c and "m-download" in (c if isinstance(c, str) else " ".join(c)))
    if dl_div:
        for a in dl_div.find_all("a", href=True):
            href = a["href"].strip()
            if not href or href.startswith("javascript"):
                continue
            abs_url = urljoin(BASE_URL, href)
            a["href"] = abs_url
            name = clean_title(a.get_text(" ", strip=True)) or href.split("/")[-1].split("?")[0]
            if not any(x["url"] == abs_url for x in attachments):
                attachments.append({"name": name, "url": abs_url})
        # 把附件追加到正文末尾 (保留 <p> HTML)
        for a in dl_div.find_all("a", href=True):
            if a.get("href", "").startswith("http") and "/upload_bz/" in a["href"]:
                content += '\n<p><a href="%s">%s</a></p>' % (a["href"], clean_title(a.get_text(" ", strip=True)))

    # 若正文无 p 但有附件链接也保留
    if not content and attachments:
        content = "\n".join('<p><a href="%s">%s</a></p>' % (a["url"], a["name"]) for a in attachments)

    return title, date, content, attachments


def build_item(item_url, title, date, content, attachments):
    import json as _json
    return {
        "site_name": SITE_NAME,
        "source_url": item_url,
        "url": item_url,
        "pub_date": date,
        "title": title,
        "content": content,
        "category": CATEGORY,
        "attachments": _json.dumps(attachments, ensure_ascii=False) if attachments else "",
    }


def crawl_list(pages):
    """B 列表: /XxgkContent/showList/806/111001001/page_{n}.html"""
    print("涡阳县-环境影响评价: 抓取 B 列表栏目 (806/111001001 建设项目环评文件审批) %d 页" % pages)
    all_items = []
    for idx in range(1, pages + 1):
        list_url = LIST_TPL % idx
        print("  [列表页 %d/%d] %s" % (idx, pages, list_url))
        html = fetch(list_url)
        if not html:
            continue
        items = extract_list_items(html)
        if not items:
            print("    -> 无数据, 停止")
            break
        print("    -> 找到 %d 条" % len(items))
        for item_url, item_title, item_date in items:
            time.sleep(0.3)
            detail_html = fetch(item_url)
            if not detail_html:
                print("    [FAIL] %s" % item_title[:40])
                continue
            title, date, content, attachments = extract_detail(detail_html)
            if not title:
                title = item_title
            if not date:
                date = item_date
            all_items.append(build_item(item_url, title, date, content, attachments))
            print("    [%d] %s (%s, %d字%s)" % (
                len(all_items), title[:40],
                "ok" if content else "empty", len(content),
                " +%d附" % len(attachments) if attachments else ""))
    return all_items


def crawl_search(pages, branch):
    """A 搜索: /XxgkSearch/?branchId=..&keyword=环境影响评价&field=title&match=like&sort=relevant&page=N"""
    print("涡阳县-环境影响评价: 抓取 A 搜索 (关键词=环境影响评价, branchId=%s) %d 页" % (branch, pages))
    all_items = []
    kw = requests.utils.quote("环境影响评价")
    for idx in range(1, pages + 1):
        params = {
            "keyword": "环境影响评价",
            "field": "title",
            "match": "like",
            "sort": "relevant",
            "page": idx,
        }
        if branch:
            params["branchId"] = branch
        search_url = SEARCH_URL + "?" + "&".join("%s=%s" % (k, requests.utils.quote(str(v))) for k, v in params.items())
        print("  [搜索页 %d/%d] %s" % (idx, pages, search_url))
        html = fetch(search_url)
        if not html:
            continue
        items = extract_search_items(html)
        if not items:
            print("    -> 无数据, 停止")
            break
        print("    -> 找到 %d 条" % len(items))
        for item_url, item_title, item_date in items:
            time.sleep(0.3)
            detail_html = fetch(item_url)
            if not detail_html:
                print("    [FAIL] %s" % item_title[:40])
                continue
            title, date, content, attachments = extract_detail(detail_html)
            if not title:
                title = item_title
            if not date:
                date = item_date
            all_items.append(build_item(item_url, title, date, content, attachments))
            print("    [%d] %s (%s, %d字%s)" % (
                len(all_items), title[:40],
                "ok" if content else "empty", len(content),
                " +%d附" % len(attachments) if attachments else ""))
    return all_items


def main():
    pages = 1
    mode = "list"   # list=B 列表栏目; search=A 搜索
    branch = ""     # A 搜索的 branchId, 默认全站
    i = 0
    argv = sys.argv[1:]
    while i < len(argv):
        if argv[i] == "--pages" and i + 1 < len(argv):
            pages = int(argv[i + 1]); i += 2
        elif argv[i] == "--search":
            mode = "search"; i += 1
        elif argv[i] == "--branch" and i + 1 < len(argv):
            branch = argv[i + 1]; i += 2
        else:
            i += 1

    if mode == "search":
        all_items = crawl_search(pages, branch)
        source = "A"
    else:
        all_items = crawl_list(pages)
        source = "B"

    if all_items:
        push_to_searchdb(all_items, batch_label=SITE_NAME)
        empty = sum(1 for it in all_items if not (it.get("content") or "").strip())
        with_p = sum(1 for it in all_items if "<p" in (it.get("content") or ""))
        n_att = sum(1 for it in all_items if it.get("attachments"))
        print("\n完成! 数据源=%s 总条数=%d 空正文=%d 含<p>=%d 附件条数=%d" % (
            source, len(all_items), empty, with_p, n_att))
    else:
        print("\n无任何数据入库")
    return 0


if __name__ == "__main__":
    sys.exit(main())
