#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
黄骅市人民政府 — 生态环境 爬虫
site: www.huanghua.gov.cn
栏目: /huanghua/c108928/list.shtml?xiangChannelCode=c108914 (渤海新区黄骅市生态环境局-生态环境)
CMS: UCAP (createPageHTML + ucap_printPageStr)
列表: table.zb_tab > tr > td.name_detail > a.detaillink (20条/页, 4页共73条)
分页: list.shtml / list_{N}.shtml?xiangChannelCode=c108914
详情: /huanghua/c108928/YYYYMM/{uuid}.shtml
详情结构: div.file_detail > h1 标题 + div 正文(Word p段落/表格/附件)
标题: meta ArticleTitle; 日期: meta PubDate
正文规则: 保留表格HTML、\\n\\n分段、附件转<p><a>名称内嵌URL、去装饰图/重复段落
"""
import re
import sys
import os
import time
import sqlite3
import copy
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "黄骅市-生态环境"
GROUP_NAME = "河北"
DOMAIN = "www.huanghua.gov.cn"
BASE_URL = "http://www.huanghua.gov.cn"
CHANNEL = "c108928"            # 生态环境栏目
XIANG_CHANNEL = "c108914"      # 乡镇级信息公开平台参数
TOTAL_PAGES = 4                # createPageHTML(...,4,1,'list','shtml',73)
SCRIPT_NAME = os.path.basename(__file__)

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 30
session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    try:
        resp = session.get(url, timeout=TIMEOUT)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None
        return resp.text
    except Exception:
        return None


def clean_title(title):
    """标题清洗: strip &middot;&nbsp; 实体前缀 + 省略号截断后缀"""
    title = re.sub(r'^[\s\xa0·\u00b7]+', '', title or '')
    title = re.sub(r'\s*\.{3,}\s*$', '', title)
    return title.strip()


def parse_list(html):
    """解析列表: table.zb_tab 行, td.name_detail>a.detaillink, 第4个td为日期"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    table = soup.find("table", class_="zb_tab")
    if not table:
        return items
    for tr in table.find_all("tr"):
        td_a = tr.find("td", class_="name_detail")
        if not td_a:
            continue
        a = td_a.find("a", class_="detaillink")
        if not a or not a.get("href"):
            continue
        href = a["href"].strip()
        if not href.startswith("http"):
            href = urljoin(BASE_URL, href)
        title = a.get_text(strip=True)
        date = ""
        tds = tr.find_all("td")
        if len(tds) >= 4:
            date = tds[3].get_text(strip=True)[:10]
        items.append({"title": clean_title(title), "url": href, "publish_date": date})
    return items


def _normalize_text(el):
    """段落文本: <br>转\\n, 压缩空白, 去空行"""
    brs = el.find_all("br")
    for br in brs:
        br.replace_with("\n")
    text = el.get_text()
    lines = [re.sub(r'[ \t\xa0\u3000]+', ' ', ln).strip() for ln in text.split("\n")]
    lines = [ln for ln in lines if ln]
    return "\n".join(lines)


def process_content_div(div, page_url):
    """正文div处理: 保留表格HTML, \\n\\n分段, 附件转<p><a>名称内嵌URL, 去装饰图/重复段落"""
    parts = []
    seen_texts = set()
    attachments = []  # (name, abs_url)

    for el in div.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
            continue
        if el.name in ("p", "div"):
            # 附件链接提取 (正文内 {id}/files/xxx) — 相对路径须按详情页URL解析
            # ⚠️ 不能在 find_all 迭代中 decompose (bs4会把后续元素置None), 先收集后统一移除
            file_links = []
            for a in el.find_all("a"):
                href = a.get("href")
                if not href:
                    continue
                href = str(href).strip()
                if "/files/" in href or re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href, re.I):
                    name = a.get_text(strip=True) or href.split("/")[-1]
                    if not href.startswith("http"):
                        href = urljoin(page_url, href)
                    file_links.append((a, name, href))
            for a, name, href in file_links:
                attachments.append((name, href))
                a.decompose()  # 从段落移除, 避免正文重复
            # 装饰图片 (bullet gif) 移除 — 同样不能迭代中decompose
            imgs = el.find_all("img")
            for img in imgs:
                img.decompose()
            # 段落内嵌表格: 表格HTML + 去表格后的段落文本 (防三重提取)
            tbls = el.find_all("table")
            if tbls:
                p_copy = copy.deepcopy(el)
                for t in p_copy.find_all("table"):
                    t.decompose()
                text = _normalize_text(p_copy)
                if text and text not in seen_texts:
                    seen_texts.add(text)
                    parts.append(text)
                for t in tbls:
                    parts.append(str(t))
            else:
                text = _normalize_text(el)
                if text and text not in seen_texts:
                    seen_texts.add(text)
                    parts.append(text)
        elif el.name == "br":
            continue
        else:
            text = _normalize_text(el)
            if text and text not in seen_texts:
                seen_texts.add(text)
                parts.append(text)

    body = "\n\n".join(parts)
    # 附件: 名称内嵌URL, 多附件独立成段
    for name, url in attachments:
        if body:
            body += "\n\n"
        body += '<p><a href="%s" target="_blank">%s</a></p>' % (url, name)
    return body, attachments


def extract_detail(html, page_url=""):
    """详情解析: 返回 (title, date, content, attachments)"""
    soup = BeautifulSoup(html, "html.parser")
    # 标题: meta ArticleTitle -> h1
    title = ""
    m = soup.find("meta", attrs={"name": "ArticleTitle"})
    if m and m.get("content"):
        title = clean_title(m["content"])
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = clean_title(h1.get_text(strip=True))
    # 日期: meta PubDate
    date = ""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        date = m["content"].strip()[:10]
    # 正文: div.file_detail 内 h1 之后的第一个 div (跳过 relate_append)
    content = ""
    attachments = []
    fd = soup.find("div", class_="file_detail")
    if fd:
        content_div = None
        for child in fd.find_all(recursive=False):
            if child.name != "div":
                continue
            cls = " ".join(child.get("class", []))
            if "relate_append" in cls:
                continue
            content_div = child
            break
        if content_div is not None:
            content, attachments = process_content_div(content_div, page_url)
    # 附件区 relate_append > ul#relate_list > li > a (UCAP标准附件位置)
    ra = soup.find("div", class_="relate_append")
    if ra:
        for a in ra.find_all("a"):
            href = a.get("href")
            if not href:
                continue
            href = str(href).strip()
            if "/files/" in href or re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href, re.I):
                name = a.get_text(strip=True) or href.split("/")[-1]
                if not href.startswith("http"):
                    href = urljoin(page_url, href)
                if not any(u == href for _, u in attachments):
                    attachments.append((name, href))
    # 附件统一转 <p><a>名称内嵌URL, 多附件独立成段 (正文内已拼的不重复)
    for name, url in attachments:
        if 'href="%s"' % url in content:
            continue
        if content:
            content += "\n\n"
        content += '<p><a href="%s" target="_blank">%s</a></p>' % (url, name)
    return title, date, content, attachments


def get_total_pages(html):
    """从 createPageHTML('page_tag',N,...) 提取总页数"""
    m = re.search(r"createPageHTML\('page_tag'\s*,\s*(\d+)\s*,", html)
    if m:
        return int(m.group(1))
    return TOTAL_PAGES


def main():
    import argparse
    parser = argparse.ArgumentParser(description=f"{SITE_NAME}爬虫")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数(默认1)")
    parser.add_argument("--script-name", default=SCRIPT_NAME, help="脚本名(写库)")
    args = parser.parse_args()

    pages_to_crawl = args.pages
    script_name = args.script_name
    print(f"🔍 {SITE_NAME} — 爬取 {pages_to_crawl} 页 (共{TOTAL_PAGES}页)")

    conn = sqlite3.connect(SEARCH_DB, timeout=30)
    cur = conn.cursor()

    new_total = 0
    skip_total = 0
    total_items = 0

    # 先抓第1页确定总页数
    for page in range(1, pages_to_crawl + 1):
        page_suffix = "" if page == 1 else "_{}".format(page)
        list_url = "{}/huanghua/{}/list{}.shtml?xiangChannelCode={}".format(
            BASE_URL, CHANNEL, page_suffix, XIANG_CHANNEL)
        html = fetch(list_url)
        if not html:
            print(f"  [p{page:2d}/{pages_to_crawl}] ✗ HTTP失败")
            continue
        if page == 1:
            tp = get_total_pages(html)
            if tp and tp < pages_to_crawl:
                pages_to_crawl = tp
                print(f"  ℹ 站点实际总页数 {tp}，本次爬取上限调整")

        items = parse_list(html)
        if not items:
            print(f"  [p{page:2d}/{pages_to_crawl}] ✗ 未解析到条目")
            continue

        page_new = 0
        page_skip = 0
        for item in items:
            total_items += 1
            exists = cur.execute(
                "SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?",
                (item["url"], SITE_NAME)
            ).fetchone()
            if exists:
                page_skip += 1
                skip_total += 1
                continue

            detail_html = fetch(item["url"])
            if detail_html:
                d_title, d_date, content, attachments = extract_detail(detail_html, item["url"])
            else:
                d_title, d_date, content, attachments = "", "", "", []

            title = d_title or item["title"]
            date = d_date or item.get("publish_date", "")
            if not content:
                page_skip += 1
                skip_total += 1
                continue

            has_table = 1 if "<table" in content else 0
            att_urls = ",".join(u for _, u in attachments)
            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(title, content, publish_date, page_url, source_url, site_name, "
                    " group_name, script_name, has_table, attachments, status) "
                    "VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (title, content, date, item["url"], DOMAIN, SITE_NAME,
                     GROUP_NAME, script_name, has_table, att_urls, "published")
                )
                if cur.rowcount > 0:
                    row_id = cur.lastrowid
                    try:
                        plain = re.sub(r"<[^>]+>", " ", content)
                        plain = re.sub(r"\s+", " ", plain).strip()[:200] or title
                        cur.execute(
                            "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                            (row_id, title, SITE_NAME, plain)
                        )
                    except Exception:
                        pass
                    page_new += 1
                    new_total += 1
            except Exception as e:
                print(f"    ! 入库异常 {item['url']}: {e}")

        conn.commit()
        print(f"  [p{page:2d}/{pages_to_crawl}] ✓ {len(items)}条 (新增{page_new} 跳过{page_skip})")
        time.sleep(0.3)

    conn.close()
    print(f"\n📊 完成！列表 {total_items} 条 | 新增: {new_total} | 跳过: {skip_total}")


if __name__ == "__main__":
    main()
