#!/usr/bin/env python3
"""
临川区人民政府 - 多栏目爬虫
Columns:
  col1559: 政策文件 (infotypeId=D00002D00004)
  col4795: 环境保护 (infotypeId=D00004D00004)
  col1432: 政府信息公开 (infotypeId=D00004D00003)
CMS: Hanweb with xxgk/search.jsp API
"""
import os, re, sys, sqlite3, requests, json, time, argparse
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
BASE = "https://www.jxlc.gov.cn"
SEARCH_URL = BASE + "/module/xxgk/search.jsp"
PER_PAGE = 18

HEADERS_TEMPLATE = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "X-Requested-With": "XMLHttpRequest",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
}

COLUMNS = [
    {
        "name": "临川区人民政府-政策文件",
        "infotypeId": "D00002D00004",
        "referer": "https://www.jxlc.gov.cn/col/col1559/index.html?number=D00002D00004",
        "category": "政策文件",
        "divid": "div1432",
    },
    {
        "name": "临川区人民政府-环境保护",
        "infotypeId": "D00004D00004",
        "referer": "https://www.jxlc.gov.cn/col/col4795/index.html?number=D00004D00004D00007",
        "category": "环境保护",
        "divid": "div1432",
    },
    {
        "name": "临川区人民政府-政府信息公开",
        "infotypeId": "D00004D00003",
        "referer": "http://www.jxlc.gov.cn/col/col1432/",
        "category": "政府信息公开",
        "divid": "div1432",
    },
]

# DB helpers
def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    conn.row_factory = sqlite3.Row
    return conn

def is_dup(conn, page_url):
    return conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def insert_item(conn, site_name, title, page_url, publish_date, content, category):
    date_rank = int(publish_date.replace("-", "")) if publish_date and "-" in publish_date else 0
    summary = ""
    if content:
        text_soup = BeautifulSoup(content, "html.parser")
        plain = text_soup.get_text(strip=True)
        summary = plain[:200]
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (site_name, title, page_url, publish_date, content, date_rank, category, summary),
        )
        if cur.rowcount > 0:
            # Also write to trigram FTS table (gov_search_v3)
            try:
                conn.execute(
                    "INSERT INTO gov_search_v3 (title, content, source_url, publish_date, site_name) "
                    "VALUES (?, ?, ?, ?, ?)",
                    (title, (content or "")[:3000], page_url, publish_date, site_name),
                )
            except Exception as fts_err:
                print(f"    [FTS WARN] gov_search_v3 insert: {fts_err}")
            return True
        return False
    except Exception as e:
        print(f"    [ERR] insert: {e}")
        return False

# Parse search list page
def parse_page(html):
    items = []
    list_items = re.findall(r"<li>(.*?)</li>", html, re.DOTALL)
    for li in list_items:
        m = re.search(r"href='([^']+)'\s+title=\"([^\"]+)\"", li)
        d = re.search(r"<b>\s*(\d{4}-\d{2}-\d{2})\s*</b>", li)
        if m and d:
            href = m.group(1)
            title = m.group(2).strip()
            date_str = d.group(1).strip()
            url = href if href.startswith("http") else BASE + href
            items.append((title, url, date_str))
    return items

def fetch_page(infotypeId, divid, page_num):
    data = (
        f"infotypeId={infotypeId}&jdid=5&area=&divid={divid}"
        f"&vc_title=&vc_number=&currpage={page_num}"
        f"&vc_filenumber=&vc_all=&texttype=&fbtime="
    )
    resp = requests.post(SEARCH_URL, data=data, headers=HEADERS_TEMPLATE, timeout=30, verify=False)
    resp.encoding = "utf-8"
    return resp.text

# Parse detail page
def parse_detail(html):
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()

    content = ""
    zoom = soup.find("div", id="zoom")
    if zoom:
        content = str(zoom)
    if not content:
        for cid in ["content", "article", "text", "mainText"]:
            div = soup.find("div", id=cid)
            if div:
                content = str(div)
                break

    pub_date = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        pub_date = meta["content"].strip()
        m2 = re.match(r"(\d{4}-\d{1,2}-\d{1,2})", pub_date)
        if m2:
            pub_date = m2.group(1)

    if not title:
        t_tag = soup.find("title")
        if t_tag:
            title = t_tag.get_text(strip=True)

    return title, content, pub_date

# Process one column
def process_column(col, conn, max_pages=3):
    name = col["name"]
    infotypeId = col["infotypeId"]
    category = col["category"]
    divid = col["divid"]
    HEADERS_TEMPLATE["Referer"] = col["referer"]

    print(f"\n{'='*60}")
    print(f"[{name}] Starting...")

    html = fetch_page(infotypeId, divid, 1)
    items_page1 = parse_page(html)
    tr = re.search(r'nTotalCount"\s*value="(\d+)"', html)
    total_records = int(tr.group(1)) if tr else len(items_page1)
    total_pages = (total_records + PER_PAGE - 1) // PER_PAGE
    pages_to_scan = min(total_pages, max_pages)
    print(f"[{name}] Total: {total_records} records, {total_pages} pages, scanning {pages_to_scan}")

    all_active = []
    all_skipped = 0
    for page in range(1, pages_to_scan + 1):
        if page == 1:
            items = items_page1
        else:
            try:
                items = parse_page(fetch_page(infotypeId, divid, page))
            except Exception as e:
                print(f"  ERROR page {page}: {e}")
                time.sleep(5)
                continue

        for title, url, date_str in items:
            if date_str >= CUTOFF_DATE:
                all_active.append((title, url, date_str))
            else:
                all_skipped += 1

        if page % 50 == 0:
            print(f"  Scanned {page}/{pages_to_scan}, {len(all_active)} within cutoff")

        conn.commit()
        time.sleep(0.5)

    print(f"[{name}] Scan done: {len(all_active)} within cutoff, {all_skipped} older")

    new_count = dup_count = error_count = 0
    for idx, (title, url, date_str) in enumerate(all_active, 1):
        try:
            if is_dup(conn, url):
                dup_count += 1
                continue

            detail_resp = requests.get(url, headers={"User-Agent": "Mozilla/5.0"}, timeout=30, verify=False)
            detail_resp.encoding = "utf-8"
            detail_title, content, pub_date = parse_detail(detail_resp.text)

            if not detail_title:
                detail_title = title
            if not pub_date:
                pub_date = date_str

            if insert_item(conn, name, detail_title, url, pub_date, content, category):
                new_count += 1
            else:
                dup_count += 1
        except Exception as e:
            error_count += 1
            print(f"  ERROR {title[:40]}: {e}")

        conn.commit()
        time.sleep(0.3)
        if idx % 20 == 0:
            print(f"  Detail {idx}/{len(all_active)}: new={new_count}, dup={dup_count}, err={error_count}")

    print(f"[{name}] Done: +{new_count} new, {dup_count} dup, {error_count} err")
    return {"site": name, "new": new_count, "dup": dup_count, "err": error_count}

# Main
def main():
    parser = argparse.ArgumentParser(description="Crawl 临川区人民政府 columns")
    parser.add_argument(
        "--pages",
        type=int,
        default=3,
        help="Max pages to scan per column (default: 3 for daily, max 999 for full crawl)",
    )
    args = parser.parse_args()
    max_pages = min(args.pages, 999)

    conn = get_conn()
    results = []
    for col in COLUMNS:
        try:
            r = process_column(col, conn, max_pages=max_pages)
            results.append(r)
        except Exception as e:
            print(f"[{col['name']}] FATAL: {e}")
            results.append({"site": col["name"], "error": str(e)})

    conn.close()

    print(f"\n{'='*60}")
    print("Summary:")
    for r in results:
        status = f"+{r.get('new',0)} new, {r.get('dup',0)} dup" if "new" in r else f"ERROR: {r.get('error','')}"
        print(f"  {r['site']:30s} {status}")
    print(json.dumps(results, ensure_ascii=False))

if __name__ == "__main__":
    main()
