#!/usr/bin/env python3
"""
安义县-生态环境局-主动回应（环评公示）
http://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/just_list.shtml
适用: 服务器端运行 (使用 push_to_searchdb)
运行: python3 crawl_anyi_v2.py --pages 5   (初始5页)
      python3 crawl_anyi_v2.py --pages 1   (日跑增量)
"""

import sys, re, json
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup, Tag

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

BASE_URL = "http://anyi.nc.gov.cn"
LIST_PATH = "/ayxzf/xsthjjgggs2021/just_list.shtml"
SITE_NAME = "安义县生态环境局-主动回应"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "http://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/just_list.shtml",
}
session = requests.Session()
session.headers.update(HEADERS)


def get_page_url(n):
    if n <= 1:
        return urljoin(BASE_URL, LIST_PATH)
    base = LIST_PATH.replace(".shtml", "")
    return urljoin(BASE_URL, f"{base}_{n}.shtml")


def parse_list(html, base_url):
    soup = BeautifulSoup(html, "html.parser")
    res = []
    for li in soup.select("ul.pageList > li"):
        a = li.find("a")
        if not a:
            continue
        t = a.get("title", "").strip() or a.get_text(strip=True)
        if not t:
            continue
        u = urljoin(base_url, a.get("href", "").strip())
        s = li.find("span", class_="time")
        d = s.get_text(strip=True) if s else ""
        res.append((t, u, d))
    return res


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    h1 = soup.find("h1", class_="article-title")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        mt = soup.find("meta", attrs={"name": "ArticleTitle"})
        if mt and mt.get("content"):
            title = mt["content"].strip()
    date_str = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        date_str = md["content"].strip()

    cp = []
    cd = soup.find("div", class_="article-content-body")
    if not cd:
        cd = soup.find(id="zoomcon")
    if cd:
        ucap = cd.find("ucapcontent")
        if ucap:
            cd = ucap

    if cd:
        for child in list(cd.children):
            if isinstance(child, str):
                t = child.strip()
                if t:
                    cp.append(t)
            elif isinstance(child, Tag):
                n = child.name.lower() if child.name else ""
                if n == "p":
                    txt = ""
                    for ch in child.children:
                        if isinstance(ch, Tag):
                            cn = ch.name.lower()
                            if cn == "span":
                                txt += ch.get_text(strip=True)
                            elif cn == "br":
                                txt += "\n"
                            elif cn == "a":
                                lt = ch.get_text(strip=True)
                                hr = ch.get("href", "")
                                txt += f"[{lt}]({hr})" if hr else lt
                            elif cn == "o:p":
                                continue
                            else:
                                txt += ch.get_text(strip=True)
                        elif isinstance(ch, str):
                            t = ch.strip()
                            if t:
                                txt += t
                    txt = re.sub(r"[ \t]+", " ", txt).strip()
                    if txt:
                        cp.append(txt)
                elif n == "table":
                    rows = child.find_all("tr")
                    if rows:
                        hc = rows[0].find_all(["td", "th"])
                        cc = len(hc)
                        if cc > 0:
                            md_lines = []
                            md_lines.append("| " + " | ".join(
                                x.get_text(strip=True) or " " for x in hc) + " |")
                            md_lines.append("| " + " | ".join(["---"] * cc) + " |")
                            for r in rows[1:]:
                                cells = r.find_all(["td", "th"])
                                rd = [x.get_text(strip=True) or " " for x in cells]
                                while len(rd) < cc:
                                    rd.append(" ")
                                md_lines.append("| " + " | ".join(rd[:cc]) + " |")
                            cp.append("\n".join(md_lines))
                elif n in ("ul", "ol", "div"):
                    t = child.get_text("\n", strip=True).strip()
                    if t:
                        cp.append(t)

    attachments = []
    if cd:
        for a_tag in cd.find_all("a"):
            href = a_tag.get("href", "").strip()
            if not href:
                continue
            ext = re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar|txt)(\?|$)', href, re.I)
            if ext or "/files/" in href:
                nm = a_tag.get_text(strip=True) or "附件"
                fu = urljoin(url, href)
                attachments.append({"name": nm, "url": fu})
                cp.append(f"[{nm}]({fu})")

    return title, date_str, "\n\n".join(cp), attachments


def crawl(max_pages=5):
    all_items = []
    for pn in range(1, max_pages + 1):
        pu = get_page_url(pn)
        print(f"\n{'='*50}")
        print(f"[PAGE {pn}] {pu}")
        try:
            r = session.get(pu, timeout=30)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  HTTP {r.status_code}")
                continue
        except Exception as e:
            print(f"  ERROR {e}")
            continue
        soup = BeautifulSoup(r.text, "html.parser")
        if not soup.select("ul.pageList > li"):
            print(f"  No more items, stop at page {pn}")
            break
        items = parse_list(r.text, pu)
        print(f"  {len(items)} items")

        for idx, (t, du, ds) in enumerate(items, 1):
            print(f"\n  [{idx}/{len(items)}] {t[:55]}...")
            try:
                dr = session.get(du, timeout=30)
                dr.encoding = "utf-8"
                if dr.status_code != 200:
                    print(f"    HTTP {dr.status_code}")
                    continue
                dt, dd, dc, da = extract_detail(dr.text, du)
                if not dt:
                    dt = t
                if not dd:
                    dd = ds
            except Exception as e:
                print(f"    ERROR {e}")
                continue
            if not dc:
                print(f"    WARNING: empty content (WAF?)")
                continue

            summary = re.sub(r"\s+", "", dc)[:200] if dc else ""
            print(f"    -> 标题: {dt[:35]}... 日期: {dd} 正文: {len(dc)} 附件: {len(da)}")
            all_items.append({
                "site_name": SITE_NAME, "source_url": du, "url": du,
                "title": dt, "pub_date": dd, "summary": summary,
                "content": dc,
                "attachments": json.dumps(da, ensure_ascii=False) if da else "",
            })

    if all_items:
        print(f"\n[*] 入库 {len(all_items)} 条...")
        push_to_searchdb(all_items, SITE_NAME)
        print(f"  DONE: {len(all_items)}")
    else:
        print("No items to push")


if __name__ == "__main__":
    mp = 5
    if len(sys.argv) > 2 and sys.argv[1] == "--pages":
        mp = int(sys.argv[2])
    crawl(max_pages=mp)
