#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_ny_tzgg.py - 宁阳县-通知公告
CMS: TRS Hanweb xxgk search.jsp
列表: POST /module/xxgk/search.jsp?standardXxgk=1&infotypeId=TAE41&area=
      POST data: divid=div4&currpage=N&vc_title=&vc_number=&vc_all=&standardXxgk=1&infotypeId=TAE41&vc_xxgkarea=NY1225544454455A-1
详情: /art/YYYY/M/D/art_171773_{id}.html?xxgkhide=1
标题: div.main-fl-tit / meta[ArticleTitle]
日期: div.main-fl-riqi (含"发布日期：YYYY-MM-DD") / meta[PubDate]
内容: div.bt-left > p (跳过前导元数据div)
"""
import requests
import re
import sys
import os
import time
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE_URL = "http://www.ny.gov.cn"
LIST_URL = BASE_URL + "/col/col171773/index.html"
SITE_NAME = "宁阳县-通知公告"
GROUP = "山东"
INDUSTRY = "其他"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": LIST_URL + "?vc_xxgkarea=NY1225544454455A-1&number=TAE41",
}

SEARCH_URL = BASE_URL + "/module/xxgk/search.jsp?standardXxgk=1&infotypeId=TAE41&divid=div4"
POST_DATA_TPL = "currpage={page}&vc_title=&vc_number=&vc_all=&standardXxgk=1&infotypeId=TAE41&vc_xxgkarea=NY1225544454455A-1"


def get_session():
    """Create session"""
    session = requests.Session()
    session.headers.update(HEADERS)
    return session


def fetch_list_page(session, page_no):
    """Fetch one page of list data via search.jsp POST"""
    url = SEARCH_URL
    data = POST_DATA_TPL.format(page=page_no)
    try:
        r = session.post(url, data=data, timeout=60)
        r.encoding = "utf-8"
        html = r.text
        # Extract total count
        count_m = re.search(r"共(\d+)条", html)
        page_m = re.search(r"共\s*(\d+)\s*页", html)
        total = int(count_m.group(1)) if count_m else 0
        total_pages = int(page_m.group(1)) if page_m else 0
        return html, total, total_pages
    except Exception as e:
        print(f"  [API ERROR] page {page_no}: {e}")
        return None, 0, 0


def parse_list(html):
    """Parse search.jsp list HTML to extract (url, title, date) tuples"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.find_all("li"):
        a = li.find("a")
        b = li.find("b")
        if not a or not a.get("href"):
            continue
        href = a.get("href", "").strip()
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date_text = b.get_text(strip=True) if b else ""
        date_m = re.match(r"(\d{4}-\d{2}-\d{2})", date_text)
        date_str = date_m.group(1) if date_m else ""
        if href.startswith("/"):
            href = BASE_URL + href
        items.append((href, title, date_str))
    return items


def fetch_detail(session, url):
    """Fetch detail page HTML"""
    try:
        r = session.get(url, timeout=60)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"ERR:{e}")
        return None


def extract_detail(html, url):
    """Extract title, date, content from detail page"""
    soup = BeautifulSoup(html, "html.parser")

    # Title from meta[ArticleTitle] or div.main-fl-tit
    title = ""
    meta_title = soup.find("meta", attrs={"name": re.compile(r"ArticleTitle", re.I)})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        tit_div = soup.find("div", class_="main-fl-tit")
        if tit_div:
            title = tit_div.get_text(strip=True)

    # Date from meta[PubDate] or div.main-fl-riqi
    date_str = ""
    for name in ["PublishDate", "PubDate", "publishdate", "pubdate"]:
        meta_date = soup.find("meta", attrs={"name": re.compile(name, re.I)})
        if meta_date and meta_date.get("content"):
            raw = meta_date["content"].strip()
            m = re.match(r"(\d{4}-\d{2}-\d{2})", raw)
            if m:
                date_str = m.group(1)
                break
    if not date_str:
        riqi = soup.find("div", class_="main-fl-riqi")
        if riqi:
            m = re.search(r"(\d{4}-\d{2}-\d{2})", riqi.get_text())
            if m:
                date_str = m.group(1)

    # Content from div.bt-left or div.main-fl.bt-left
    content = ""
    content_div = soup.find("div", class_="bt-left")
    if not content_div:
        content_div = soup.find("div", class_="main-fl")
    if content_div:
        # Find all p tags in the content area
        parts = []
        # Collect p tags that have actual text content
        for p in content_div.find_all("p", recursive=True):
            text = p.get_text(strip=True)
            if text and text not in ("\xa0", ""):
                # Skip p tags inside tables
                if p.find_parent("table"):
                    continue
                parts.append(text)
        # Also capture tables
        for table in content_div.find_all("table", recursive=True):
            # Skip the metadata table (xxgk_table)
            if "xxgk_table" in table.get("class", []):
                continue
            parts.append(str(table))
        content = "\n\n".join(parts)

    # Summary
    summary = re.sub(r"<[^>]+>", "", content)[:200] if content else ""

    return title, date_str, content, summary


def save_to_db(items_data):
    """Batch insert into gov_raw and gov_search (FTS)"""
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=10)
    c = conn.cursor()
    inserted = 0
    fts_batch = []
    for title, date_str, content, summary, page_url in items_data:
        try:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                continue
            c.execute(
                """INSERT INTO gov_raw 
                   (page_url, title, site_name, publish_date, content, summary,
                    source_url, category, industry, group_name, script_name)
                   VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                (page_url, title, SITE_NAME, date_str, content, summary,
                 page_url, "政府公告", INDUSTRY, GROUP, "crawl_ny_tzgg.py"),
            )
            new_id = c.lastrowid
            if new_id:
                fts_batch.append((new_id, title, SITE_NAME, summary or ""))
            inserted += 1
        except Exception as e:
            print(f"  [DB ERROR] {title[:30]}: {e}")
    conn.commit()

    if fts_batch:
        for rid, title, site, summary in fts_batch:
            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                    (rid, title, site, summary),
                )
            except Exception as e:
                print(f"  [FTS ERROR] id={rid}: {e}")
        conn.commit()
        print(f"  FTS同步: {len(fts_batch)}条")

    conn.close()
    return inserted


def main():
    import argparse
    parser = argparse.ArgumentParser(description=f"{SITE_NAME}爬虫")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数")
    parser.add_argument("--full", action="store_true", help="全量爬取")
    args = parser.parse_args()

    max_pages = 999 if args.full else args.pages
    session = get_session()
    total_new = 0

    for page_no in range(1, max_pages + 1):
        print(f"[{time.strftime('%H:%M:%S')}] 第{page_no}页...", end=" ", flush=True)
        html, total, total_pages = fetch_list_page(session, page_no)
        if not html:
            print("失败")
            break

        items = parse_list(html)
        if not items:
            print("无数据，停止")
            break

        print(f"{len(items)}条 (共{total}条/{total_pages}页)", flush=True)

        batch = []
        for href, title, date_str in items:
            detail_html = fetch_detail(session, href)
            if not detail_html:
                continue
            detail_title, detail_date, content, summary = extract_detail(detail_html, href)
            final_title = detail_title or title
            final_date = detail_date or date_str
            batch.append((final_title, final_date, content, summary, href))
            time.sleep(0.3)

        saved = save_to_db(batch)
        total_new += saved
        print(f"  入库{saved}/{len(batch)}条 (累计{total_new})", flush=True)

        # Incremental mode
        if not args.full and saved == 0 and len(batch) > 0:
            print("  全部已存在，增量停止")
            break

    print(f"\n===== 完成 =====")
    print(f"新增入库: {total_new} 条")
    print(f"站点: {SITE_NAME}")
    print(f"分组: {GROUP}")


if __name__ == "__main__":
    main()
