#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
咸宁高新区-通知公告 爬虫
站点: gxq.xianning.gov.cn
栏目: /zwgk/fdzdgknr/tzgg/
CMS: TRS (TRS_Documents + page_fgw.js 分页)
列表: 静态HTML, 每页20条, index_N.shtml 分页
详情: div.con.m15 正文容器
"""
import re, sys, time, requests, sqlite3, argparse
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://gxq.xianning.gov.cn"
LIST_PATH = "/zwgk/fdzdgknr/tzgg/"
DB_PATH = "/root/search.db"
SITE_NAME = "咸宁高新区-通知公告"
GROUP = "湖北"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
MAX_PAGES = 5


def init_db():
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [ERROR] fetch: {e}", file=sys.stderr)
        return None


def extract_list_items(soup):
    """提取文章列表"""
    items = []
    for ul in soup.find_all("ul"):
        lis = ul.find_all("li")
        # 找包含/tzgg/链接的ul（文章列表）
        tzgg_count = 0
        for li in lis:
            a = li.find("a", href=True)
            if a and "/tzgg/" in a["href"]:
                tzgg_count += 1
        if tzgg_count < 5:
            continue
        for li in lis:
            a = li.find("a", href=True)
            if not a:
                continue
            href = a["href"].strip()
            title = a.get("title") or a.get_text(strip=True) or ""
            title = title.strip()
            if not title or len(title) < 6:
                continue
            if "/tzgg/" not in href:
                continue
            full_url = href if href.startswith("http") else urljoin(BASE_URL + LIST_PATH, href)
            # 日期
            li_text = li.get_text(" ", strip=True)
            date = ""
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', li_text)
            if m:
                date = m.group(1)
            if not any(item["url"] == full_url for item in items):
                items.append({"title": title, "url": full_url, "date": date})
        break
    return items


def extract_date(soup):
    """从详情页提取日期"""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        return m["content"].strip()[:10]
    # xxgk-info可能包含日期
    info = soup.find("div", class_="xxgk-info")
    if info:
        m2 = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', info.get_text())
        if m2:
            return m2.group(1)
    return ""


def extract_content(soup):
    """提取正文HTML"""
    con = soup.find("div", class_="con m15")
    if not con:
        return ""
    
    parts = []
    for child in con.children:
        tag = getattr(child, "name", None)
        child_str = str(child)
        
        if tag == "p":
            text = child.get_text("\n", strip=True)
            if text:
                inner_html = re.sub(r'\s+', ' ', child_str.strip())
                parts.append(inner_html)
        elif tag == "table":
            parts.append(child_str.strip())
        elif tag in ("div", "section"):
            tbl = child.find("table")
            if tbl:
                parts.append(str(tbl).strip())
            else:
                text = child.get_text("\n", strip=True)
                if text:
                    parts.append(f"<p>{text}</p>")
        elif tag in ("ul", "ol"):
            parts.append(child_str.strip())
    
    return "\n\n".join(parts)


def save_to_db(conn, title, pub_date, content, url):
    conn.execute("""
        INSERT OR IGNORE INTO gov_raw
            (title, page_url, source_url, content, site_name, publish_date, group_name, script_name)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?)
    """, (title, url, SITE_NAME, content, SITE_NAME, pub_date, GROUP, "crawl_xianninggxq_tzgg.py"))
    conn.commit()
    rid = conn.execute("SELECT last_insert_rowid()").fetchone()[0]
    conn.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                 (rid, title, SITE_NAME, content[:200] if content else ""))
    conn.commit()
    return rid


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="爬取页数")
    args = parser.parse_args()

    print(f"[{SITE_NAME}] 开始爬取，pages={args.pages}")

    conn = init_db()
    new_count = 0
    skip_count = 0
    err_count = 0

    for page in range(1, args.pages + 1):
        if page == 1:
            url = BASE_URL + LIST_PATH
        else:
            url = BASE_URL + LIST_PATH + f"index_{page}.shtml"

        print(f"  --- 第{page}页: {url} ---")
        soup = get_soup(url)
        if not soup:
            print(f"  [ERROR] 无法获取第{page}页")
            err_count += 1
            continue

        items = extract_list_items(soup)
        if not items:
            print(f"  [WARN] 第{page}页无文章，可能已到尾页")
            break

        print(f"  获取到 {len(items)} 篇文章")

        for idx, item in enumerate(items):
            url = item["url"]
            exists = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone()
            if exists:
                skip_count += 1
                continue

            print(f"  [{page}-{idx+1}] {item['title'][:40]}...", end=" ", flush=True)

            detail_soup = get_soup(url)
            if not detail_soup:
                print("[ERROR 获取详情]")
                err_count += 1
                continue

            title = item["title"]
            pub_date = item.get("date") or extract_date(detail_soup)
            content = extract_content(detail_soup)

            if not title or not content:
                print("[跳过 无内容]")
                skip_count += 1
                continue

            rid = save_to_db(conn, title, pub_date, content, url)
            new_count += 1
            print(f"OK id={rid}")

            time.sleep(1)

    conn.close()
    print(f"\n[{SITE_NAME}] 完成: +{new_count} 跳过{skip_count} 错误{err_count}")


if __name__ == "__main__":
    main()
