#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
和县人民政府 — 建设项目环评文件审批 爬虫
site: www.hx.gov.cn (安徽马鞍山含山县)
栏目: /xxgk/opennessContent/?branch_id=57a3df762c262ea9a00aad3b&column_code=350101
      (和县经济开发区管委会-意见征集(环评公众参与公示))
CMS: 安徽政务公开平台 (OpennessContent/OpennessTarget, 新版 branch_id+column_code)
列表: AJAX /xxgk/opennessTarget/?branch_id=..&column_code=..&page=N
      返回 table tr>td.bt>a (标题) + td.cwrq (日期), <pagination pagecount total>
分页: &page=N (20条/页, 36页共703条, <pagination currentpage pagecount total>)
详情: /xxgk/openness/detail/content/{id}.html
详情结构: h1 标题 + meta PubDate + div#zoom 正文(文字/表格/内嵌PDF embed)
防护: JS cookie 挑战 → 需带 Cookie: token_verified=true
正文规则: 保留表格HTML、\n\n分段、embed PDF转<p><a>附件、去装饰
"""
import re
import sys
import os
import time
import sqlite3
import copy
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "和县经济开发区管委会-意见征集"
GROUP_NAME = "安徽"
DOMAIN = "www.hx.gov.cn"
BASE_URL = "https://www.hx.gov.cn"
BRANCH_ID = "57a3df762c262ea9a00aad3b"
COLUMN_CODE = "30100"
TOTAL_PAGES = 4                # <pagination pagecount="4" total="78">
SCRIPT_NAME = os.path.basename(__file__)

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Cookie": "token_verified=true",   # JS cookie 挑战
    "X-Requested-With": "XMLHttpRequest",  # 和县校验AJAX头, 缺则返回框架页
    "Referer": "https://www.hx.gov.cn/xxgk/opennessContent/?branch_id=57a3df762c262ea9a00aad3b&column_code=30100",
}
TIMEOUT = 30
session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    try:
        resp = session.get(url, timeout=TIMEOUT)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None
        if "document.cookie='token_verified" in resp.text[:500]:
            # 挑战未过, 重试一次(带cookie)
            resp = session.get(url, timeout=TIMEOUT)
            resp.encoding = "utf-8"
            if resp.status_code != 200:
                return None
        return resp.text
    except Exception:
        return None


def clean_title(title):
    """标题清洗: strip &middot;&nbsp; 实体前缀 + 省略号截断后缀"""
    title = re.sub(r'^[\s\xa0·\u00b7]+', '', title or '')
    title = re.sub(r'\s*\.{3,}\s*$', '', title)
    return title.strip()


def parse_list(html):
    """解析列表: table tr>td.bt>a + td.cwrq 日期"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for tr in soup.find_all("tr"):
        tds = tr.find_all("td")
        if len(tds) < 3:
            continue
        a = tds[1].find("a")
        if not a or not a.get("href"):
            continue
        href = a["href"].strip()
        if not href.startswith("http"):
            href = urljoin(BASE_URL, href)
        title = a.get_text(strip=True)
        date = tds[2].get_text(strip=True)[:10]
        items.append({"title": clean_title(title), "url": href, "publish_date": date})
    return items


def get_total_pages(html):
    """从 <pagination pagecount="N"> 提取总页数"""
    m = re.search(r'<pagination[^>]*pagecount="(\d+)"', html)
    if m:
        return int(m.group(1))
    return TOTAL_PAGES


def _normalize_text(el):
    """段落文本: <br>转\n, 压缩空白, 去空行"""
    brs = el.find_all("br")
    for br in brs:
        br.replace_with("\n")
    text = el.get_text()
    lines = [re.sub(r'[ \t\xa0\u3000]+', ' ', ln).strip() for ln in text.split("\n")]
    lines = [ln for ln in lines if ln]
    return "\n".join(lines)


def process_zoom_div(div, page_url):
    """正文div处理: 保留表格HTML, \n\n分段, embed PDF转附件, 去装饰图/重复段落"""
    parts = []
    seen_texts = set()
    attachments = []  # (name, abs_url)
    pdf_idx = 0

    for el in div.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
            continue
        if el.name in ("p", "div", "center"):
            # 内嵌PDF embed → 附件链接
            embeds = el.find_all("embed")
            for emb in embeds:
                src = emb.get("src")
                if src:
                    if not src.startswith("http"):
                        src = urljoin(page_url, src)
                    pdf_idx += 1
                    name = "环评文件全文.pdf" if pdf_idx == 1 else "环评文件全文(%d).pdf" % pdf_idx
                    if not any(u == src for _, u in attachments):
                        attachments.append((name, src))
                emb.decompose()
            # 附件a链接 (files/ 或 .pdf/.doc)
            file_links = []
            for a in el.find_all("a"):
                href = a.get("href")
                if not href:
                    continue
                href = str(href).strip()
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href, re.I):
                    name = a.get_text(strip=True) or href.split("/")[-1]
                    if not href.startswith("http"):
                        href = urljoin(page_url, href)
                    file_links.append((a, name, href))
            for a, name, href in file_links:
                if not any(u == href for _, u in attachments):
                    attachments.append((name, href))
                a.decompose()
            # 装饰图片移除
            imgs = el.find_all("img")
            for img in imgs:
                img.decompose()
            # 段落内嵌表格: 表格HTML + 去表格后的段落文本
            tbls = el.find_all("table")
            if tbls:
                p_copy = copy.deepcopy(el)
                for t in p_copy.find_all("table"):
                    t.decompose()
                text = _normalize_text(p_copy)
                if text and text not in seen_texts:
                    seen_texts.add(text)
                    parts.append(text)
                for t in tbls:
                    parts.append(str(t))
            else:
                text = _normalize_text(el)
                if text and text not in seen_texts:
                    seen_texts.add(text)
                    parts.append(text)
        elif el.name == "br":
            continue
        else:
            text = _normalize_text(el)
            if text and text not in seen_texts:
                seen_texts.add(text)
                parts.append(text)

    body = "\n\n".join(parts)
    # 附件统一转 <p><a>名称内嵌URL
    for name, url in attachments:
        if 'href="%s"' % url in body:
            continue
        if body:
            body += "\n\n"
        body += '<p><a href="%s" target="_blank">%s</a></p>' % (url, name)
    return body, attachments


def extract_detail(html, page_url=""):
    """详情解析: 返回 (title, date, content, attachments)"""
    soup = BeautifulSoup(html, "html.parser")
    # 标题: h1
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = clean_title(h1.get_text(strip=True))
    # 日期: meta PubDate
    date = ""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        date = m["content"].strip()[:10]
    # 正文: div#zoom
    content = ""
    attachments = []
    zoom = soup.find("div", attrs={"id": "zoom"})
    if zoom:
        # 移除意见收集表单噪声
        form = zoom.find("form", attrs={"id": "collect_form"})
        if form:
            form.decompose()
        content, attachments = process_zoom_div(zoom, page_url)
    return title, date, content, attachments


def main():
    import argparse
    parser = argparse.ArgumentParser(description=f"{SITE_NAME}爬虫")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数(默认1)")
    parser.add_argument("--script-name", default=SCRIPT_NAME, help="脚本名(写库)")
    args = parser.parse_args()

    pages_to_crawl = args.pages
    script_name = args.script_name
    print(f"🔍 {SITE_NAME} — 爬取 {pages_to_crawl} 页 (共{TOTAL_PAGES}页)")

    conn = sqlite3.connect(SEARCH_DB, timeout=30)
    cur = conn.cursor()

    new_total = 0
    skip_total = 0
    total_items = 0

    for page in range(1, pages_to_crawl + 1):
        page_url = "%s/xxgk/opennessTarget/?branch_id=%s&column_code=%s&page=%d" % (
            BASE_URL, BRANCH_ID, COLUMN_CODE, page)
        if page == 1:
            page_url = "%s/xxgk/opennessTarget/?branch_id=%s&column_code=%s" % (
                BASE_URL, BRANCH_ID, COLUMN_CODE)
        html = fetch(page_url)
        if not html:
            print(f"  [p{page:2d}/{pages_to_crawl}] ✗ HTTP失败")
            continue
        if page == 1:
            tp = get_total_pages(html)
            if tp and tp < pages_to_crawl:
                pages_to_crawl = tp
                print(f"  ℹ 站点实际总页数 {tp}，本次爬取上限调整")

        items = parse_list(html)
        if not items:
            print(f"  [p{page:2d}/{pages_to_crawl}] ✗ 未解析到条目")
            continue

        page_new = 0
        page_skip = 0
        for item in items:
            total_items += 1
            exists = cur.execute(
                "SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?",
                (item["url"], SITE_NAME)
            ).fetchone()
            if exists:
                page_skip += 1
                skip_total += 1
                continue

            detail_html = fetch(item["url"])
            if detail_html:
                d_title, d_date, content, attachments = extract_detail(detail_html, item["url"])
            else:
                d_title, d_date, content, attachments = "", "", "", []

            title = d_title or item["title"]
            date = d_date or item.get("publish_date", "")
            if not content:
                page_skip += 1
                skip_total += 1
                continue

            has_table = 1 if "<table" in content else 0
            att_urls = ",".join(u for _, u in attachments)
            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(title, content, publish_date, page_url, source_url, site_name, "
                    " group_name, script_name, has_table, attachments, status) "
                    "VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (title, content, date, item["url"], DOMAIN, SITE_NAME,
                     GROUP_NAME, script_name, has_table, att_urls, "published")
                )
                if cur.rowcount > 0:
                    row_id = cur.lastrowid
                    try:
                        plain = re.sub(r"<[^>]+>", " ", content)
                        plain = re.sub(r"\s+", " ", plain).strip()[:200] or title
                        cur.execute(
                            "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                            (row_id, title, SITE_NAME, plain)
                        )
                    except Exception:
                        pass
                    page_new += 1
                    new_total += 1
            except Exception as e:
                print(f"    ! 入库异常 {item['url']}: {e}")

        conn.commit()
        print(f"  [p{page:2d}/{pages_to_crawl}] ✓ {len(items)}条 (新增{page_new} 跳过{page_skip})")
        time.sleep(0.3)

    conn.close()
    print(f"\n📊 完成！列表 {total_items} 条 | 新增: {new_total} | 跳过: {skip_total}")


if __name__ == "__main__":
    main()
