#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
固镇县人民政府 - 通知公告爬虫
CMS: Lonsun（蓝汛）
列表: ul.doc_list > li > a[title] + span
分页: /content/column/10930360?pageIndex=N（90页，20条/页）
详情: div.newscontnet + meta ArticleTitle/PubDate
"""

import re, sys, json, time, requests, sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.guzhen.gov.cn"
LIST_URL = BASE_URL + "/content/column/10930360?pageIndex={}"
LIST_P1 = BASE_URL + "/gzdt/tzgg/index.html"
DB_PATH = "/root/search.db"
SITE_NAME = "固镇县人民政府-通知公告"
CATEGORY = GROUP = "安徽"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
MAX_PAGES = 5

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn

def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        return BeautifulSoup(r.text, "html.parser")
    except:
        return None

def extract_list_items(soup):
    items = []
    ul = soup.find("ul", class_=lambda x: x and "doc_list" in str(x))
    if not ul:
        return items
    for li in ul.find_all("li", recursive=False):
        a = li.find("a")
        if not a or not a.get("href"):
            continue
        title = (a.get("title") or a.get_text(strip=True) or "").strip()
        if not title:
            continue
        href = a["href"].strip()
        full = href if href.startswith("http") else urljoin(BASE_URL, href)
        span = li.find("span")
        date = span.get_text(strip=True) if span else ""
        items.append({"title": title, "url": full, "date": date})
    return items

def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None, None, None, None, None, None
    except Exception as e:
        print("  [ERROR] fetch: %s" % e, file=sys.stderr)
        return None, None, None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")

    title = ""
    m = soup.find("meta", attrs={"name": "ArticleTitle"})
    if m and m.get("content"):
        title = m["content"].strip()

    pub_date = ""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        pub_date = m["content"].strip()[:10]

    src_url = ""
    m = soup.find("meta", attrs={"name": "ContentSource"})
    if m and m.get("content"):
        src_url = m["content"].strip()

    div = soup.find("div", class_="newscontnet") or soup.find("div", class_="j-fontContent")
    parts = []
    attach = []

    if div:
        for noise in div.find_all(["script", "style"]):
            noise.decompose()

        def _has_table_ancestor(el):
            p = el.parent
            while p and p != div:
                if p.name == "div" and p.find("table"):
                    return True
                p = p.parent
            return False

        for child in div.find_all(["p", "table", "img"], recursive=True):
            if child.name == 'table':
                tbl_html = html_table_to_html(child, url)
                if tbl_html:
                    parts.append(tbl_html)
            elif child.name == "p":
                if _has_table_ancestor(child):
                    continue
                txt = child.get_text(" ", strip=True)
                if txt and len(txt) > 2:
                    parts.append(txt)
            elif child.name == "img":
                src = child.get("src", "")
                alt = child.get("alt", "")
                if src:
                    if not src.startswith("http"):
                        src = urljoin(url, src)
                    parts.append('<p><a href="%s">查看图片</a></p>' % (src,))

        # 附件
        for a in div.find_all("a", href=True):
            h = a["href"].strip()
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", h, re.I):
                fh = h if h.startswith("http") else urljoin(url, h)
                attach.append({"name": a.get_text(strip=True) or h.split("/")[-1], "url": fh})

    content = "\n\n".join(parts)
    if len(content.strip()) < 20:
        content = '<p><a href="%s">%s</a></p>' % (url, title or url.split("/")[-1])
    summary = content[:200] if len(content) > 200 else content
    return title, pub_date, src_url, content, summary, json.dumps(attach, ensure_ascii=False) if attach else ""

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=MAX_PAGES)
    parser.add_argument("--full", action="store_true")
    # Handle bare numeric arg for _MAX_PG
    import sys as _SYS2
    _MAX_PG = int(_SYS2.argv[-1]) if len(_SYS2.argv) > 1 and _SYS2.argv[-1].isdigit() else None
    if _MAX_PG is not None:
        print('[AutoPg] max_pages=' + str(_MAX_PG))
        _SYS2.argv.pop()
    args = parser.parse_args()
    max_pages = 90 if args.full else args.max_pages

    session = requests.Session()
    all_items, seen = [], set()

    for pn in range(1, max_pages + 1):
        url = LIST_P1 if pn == 1 else LIST_URL.format(pn)
        print("列表页 %d/%d: %s" % (pn, max_pages, url))
        soup = get_soup(url)
        if not soup:
            print("  FAILED")
            continue
        items = extract_list_items(soup)
        if not items:
            print("  空列表，停止")
            break
        n = 0
        for it in items:
            if it["url"] not in seen:
                seen.add(it["url"])
                all_items.append(it)
                n += 1
        print("  本页%d条，新增%d条，累计%d条" % (len(items), n, len(all_items)))
        if n == 0:
            break
        time.sleep(0.5)

    print("\n共%d篇文章，开始抓取详情..." % len(all_items))
    conn = init_db()
    ins, err = 0, 0
    SQL = "INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)"

    for idx, it in enumerate(all_items):
        try:
            print("  [%d/%d] %s..." % (idx + 1, len(all_items), it["title"][:40]))
            t, d, s, c, sm, a = extract_detail(it["url"])
            if t is None:
                err += 1
                continue
            conn.execute(SQL, (SITE_NAME, s or it["url"], it["url"], (t or it["title"]).strip(), d or it["date"],
                               sm or c[:200] if c else it["title"], c, CATEGORY, a, GROUP))
            conn.commit()
            ins += 1
            time.sleep(0.3)
        except Exception as e:
            print("  ERROR: %s" % e)
            err += 1
    conn.close()
    print("\n完成！新增%d条，错误%d条" % (ins, err))

if __name__ == "__main__":
    main()
