#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
襄城县人民政府 - 公告公示 爬虫
CMS: 自定义PHP
列表: /zwgk/010001/subpage.html (第1页), /zwgk/010001/{N}.html (第N页, N=2~5)
分页: 20条/页, 5页
详情: /zwgk/010001/YYYYMMDD/uuid.html
详情标题: <title>
详情正文: div.content

用法:
  python3 crawl_xiangcheng.py            # 默认5页
  python3 crawl_xiangcheng.py --pages 3  # 指定页数
"""

import re, sys, json, time, requests, sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.xiangchengxian.gov.cn"
LIST_FIRST = BASE_URL + "/zwgk/010001/subpage.html"
DB_PATH = "/root/search.db"
SITE_NAME = "襄城县-公告公示"
CATEGORY = GROUP = "公示公告"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
MAX_PAGES = 5
TIMEOUT = 30


def fetch(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if attempt == 2:
                print("  [WARN] 获取失败: %s - %s" % (url, e), file=sys.stderr)
                return ""
            time.sleep(2)


def extract_list_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.select("ul li a"):
        href = a.get("href", "")
        title = a.get_text(strip=True)
        if not href or not title or len(title) < 10:
            continue
        # 详情页URL含日期路径，如 /20260716/uuid.html
        if "subpage" in href or href.endswith(".html") and "/20" in href:
            full_url = urljoin(BASE_URL, href)
            # 从URL提取日期
            date = ""
            m = re.search(r"/(20\d{2})(\d{2})(\d{2})/", href)
            if m:
                date = "%s-%s-%s" % (m.group(1), m.group(2), m.group(3))
            items.append((full_url, title, date))
    return items


def extract_detail(html):
    soup = BeautifulSoup(html, "html.parser")
    
    # 标题：<title> 或 h3.ewb-article-tt
    title_tag = soup.find("title")
    title = title_tag.get_text(strip=True) if title_tag else ""
    if not title:
        h3 = soup.find("h3", class_=lambda c: c and "article-tt" in str(c))
        if h3:
            title = h3.get_text(strip=True)
    
    date = ""
    m = re.search(r"(\d{4}-\d{2}-\d{2})", html)
    if m:
        date = m.group(1)
    
    # 正文：div.ewb-article-content 内含 <p> 段落
    content = ""
    content_div = soup.find("div", class_=lambda c: c and "article-content" in str(c))
    if content_div:
        parts = []
        for p in content_div.find_all("p"):
            # 将 <br> 替换为换行
            for br in p.find_all("br"):
                br.replace_with("\n")
            # 手动拼接文本，保留 <br> 产生的换行
            texts = [s for s in p.stripped_strings if s]
            if not texts:
                continue
            if len(texts) == 1:
                text = texts[0]
            else:
                text = "\n".join(texts)
            parts.append(text)
        content = "\n\n".join(parts)
    
    return title, date, content


def push_to_db(conn, items):
    saved = 0
    skipped = 0
    for url, title, date, content in items:
        summary = content[:200].replace("\n", " ") if content else title
        try:
            cur = conn.execute(
                "INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, date_rank, summary, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_xiangcheng.py')",
                (url, title, SITE_NAME, date, content, date, summary, "")
            )
            if cur.rowcount > 0:
                saved += 1
            else:
                skipped += 1
        except Exception as e:
            print("  [ERR] DB: %s - %s" % (url, e), file=sys.stderr)
            skipped += 1
    return saved, skipped


def main():
    pages = MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a == "--pages" and i + 1 < len(sys.argv):
            pages = int(sys.argv[i + 1])
            break

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    all_items = []

    list_urls = [LIST_FIRST]
    list_urls.extend("%s/zwgk/010001/%d.html" % (BASE_URL, n) for n in range(2, pages + 1))

    print("襄城县-公告公示: 爬取 %d 页" % pages)

    for idx, list_url in enumerate(list_urls, 1):
        print("  [列表页 %d/%d] %s" % (idx, pages, list_url))
        html = fetch(list_url)
        if not html:
            continue

        items = extract_list_items(html)
        if not items:
            print("    -> 无数据，停止")
            break

        print("    -> 找到 %d 条" % len(items))

        for item_url, item_title, item_date in items:
            time.sleep(0.5)
            print("    [%d] %s..." % (len(all_items) + 1, item_title[:40]), end=" ", flush=True)
            detail_html = fetch(item_url)
            if not detail_html:
                print("fail")
                continue

            title, date, content = extract_detail(detail_html)
            if not title:
                title = item_title
            if not date:
                date = item_date

            all_items.append((item_url, title, date, content))
            status = "ok" if content else "empty"
            print("(%s, %d字)" % (status, len(content)))

    if all_items:
        saved, skipped = push_to_db(conn, all_items)
        conn.commit()
        print("\n完成! 新增: %d, 跳过: %d, 总共: %d" % (saved, skipped, len(all_items)))

    conn.close()
    return 0


if __name__ == "__main__":
    sys.exit(main())
