#!/usr/bin/env python3
"""


临川区人民政府 - 多栏目爬虫
Columns:
  col1559: 政策文件 (infotypeId=D00002D00004)
  col4795: 环境保护 (infotypeId=D00004D00004)
  col1432: 政府信息公开 (infotypeId=D00004D00003)
CMS: Hanweb with xxgk/search.jsp API
"""
import os, re, sys, sqlite3, requests, json, time
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
BASE = "https://www.jxlc.gov.cn"
SEARCH_URL = BASE + "/module/xxgk/search.jsp"
PER_PAGE = 18

HEADERS_TEMPLATE = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'X-Requested-With': 'XMLHttpRequest',
    'Content-Type': 'application/x-www-form-urlencoded; charset=UTF-8'
}

COLUMNS = [
    {
        "name": "临川区人民政府-政策文件",
        "infotypeId": "D00002D00004",
        "referer": "https://www.jxlc.gov.cn/col/col1559/index.html?number=D00002D00004",
        "category": "政策文件",
        "divid": "div1432",
    },
]

# ── DB helpers ─────────────────────────────────────────────
def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    conn.row_factory = sqlite3.Row
    return conn

def is_dup(conn, page_url):
    return conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def insert_item(conn, site_name, title, page_url, publish_date, content, category):
    date_rank = int(publish_date.replace("-", "")) if publish_date and "-" in publish_date else 0
    summary = ""
    if content:
        text_soup = BeautifulSoup(content, 'html.parser')
        plain = text_soup.get_text(strip=True)
        summary = plain[:200]
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (site_name, title, page_url, publish_date, content, date_rank, category, summary)
        )
        return cur.rowcount > 0
    except Exception as e:
        print(f"    [ERR] insert: {e}")
        return False

# ── Parse search list page ────────────────────────────────
def parse_page(html):
    items = []
    list_items = re.findall(r'<li>(.*?)</li>', html, re.DOTALL)
    for li in list_items:
        m = re.search(r"href='([^']+)'\s+title=\"([^\"]+)\"", li)
        d = re.search(r'<b>\s*(\d{4}-\d{2}-\d{2})\s*</b>', li)
        if m and d:
            href = m.group(1)
            title = m.group(2).strip()
            date_str = d.group(1).strip()
            url = href if href.startswith("http") else BASE + href
            items.append((title, url, date_str))
    return items

def fetch_page(infotypeId, divid, page_num):
    data = (
        f"infotypeId={infotypeId}&jdid=5&area=&divid={divid}"
        f"&vc_title=&vc_number=&currpage={page_num}"
        f"&vc_filenumber=&vc_all=&texttype=&fbtime="
    )
    resp = requests.post(SEARCH_URL, data=data, headers=HEADERS_TEMPLATE, timeout=30, verify=False)
    resp.encoding = "utf-8"
    return resp.text

# ── Parse detail page ──────────────────────────────────────
def parse_detail(html):
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()

    content = ""
    zoom = soup.find("div", id="zoom")
    if zoom:
        content = str(zoom)
    if not content:
        for cid in ["content", "article", "text", "mainText"]:
            div = soup.find("div", id=cid)
            if div:
                content = str(div)
                break

    pub_date = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        pub_date = meta["content"].strip()
        m2 = re.match(r"(\d{4}-\d{1,2}-\d{1,2})", pub_date)
        if m2:
            pub_date = m2.group(1)

    if not title:
        t_tag = soup.find("title")
        if t_tag:
            title = t_tag.get_text(strip=True)

    return title, content, pub_date

# ── Process one column ────────────────────────────────────
def process_column(col, conn):
    name = col["name"]
    infotypeId = col["infotypeId"]
    category = col["category"]
    divid = col["divid"]
    HEADERS_TEMPLATE["Referer"] = col["referer"]

    print(f"\n{'='*60}")
    print(f"[{name}] Starting...")

    html = fetch_page(infotypeId, divid, 1)
    items_page1 = parse_page(html)
    tr = re.search(r'nTotalCount"\s*value="(\d+)"', html)
    total_records = int(tr.group(1)) if tr else len(items_page1)
    total_pages = (total_records + PER_PAGE - 1) // PER_PAGE
    print(f"[{name}] Total: {total_records} records, {total_pages} pages")

    # Collect items within cutoff
    all_active = []
    all_skipped = 0
    for page in range(1, min(total_pages, _MAX_PG or total_pages)+1):
        if page == 1:
            items = items_page1
        else:
            try:
                items = parse_page(fetch_page(infotypeId, divid, page))
            except Exception as e:
                print(f"  ERROR page {page}: {e}")
                time.sleep(5)
                continue

        for title, url, date_str in items:
            if date_str >= CUTOFF_DATE:
                all_active.append((title, url, date_str))
            else:
                all_skipped += 1

        if page % 50 == 0:
            print(f"  Scanned {page}/{total_pages}, {len(all_active)} within cutoff")

        conn.commit()
        time.sleep(0.5)

    print(f"[{name}] Scan done: {len(all_active)} within cutoff, {all_skipped} older")

    # Fetch detail pages
    new_count = dup_count = error_count = 0
    for idx, (title, url, date_str) in enumerate(all_active, 1):
        try:
            if is_dup(conn, url):
                dup_count += 1
                continue

            detail_resp = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=30, verify=False)
            detail_resp.encoding = "utf-8"
            detail_title, content, pub_date = parse_detail(detail_resp.text)

            if not detail_title:
                detail_title = title
            if not pub_date:
                pub_date = date_str

            if insert_item(conn, name, detail_title, url, pub_date, content, category):
                new_count += 1
            else:
                dup_count += 1
        except Exception as e:
            error_count += 1
            print(f"  ERROR {title[:40]}: {e}")

        conn.commit()
        time.sleep(0.3)
        if idx % 20 == 0:
            print(f"  Detail {idx}/{len(all_active)}: new={new_count}, dup={dup_count}, err={error_count}")

    print(f"[{name}] Done: +{new_count} new, {dup_count} dup, {error_count} err")
    return {"site": name, "new": new_count, "dup": dup_count, "err": error_count}

# ── Main ───────────────────────────────────────────────────
def main():
    conn = get_conn()
    results = []
    for col in COLUMNS:
        try:
            r = process_column(col, conn)
            results.append(r)
        except Exception as e:
            print(f"[{col['name']}] FATAL: {e}")
            results.append({"site": col["name"], "error": str(e)})

    conn.close()

    print(f"\n{'='*60}")
    print("Summary:")
    for r in results:
        status = f"+{r.get('new',0)} new, {r.get('dup',0)} dup" if 'new' in r else f"ERROR: {r.get('error','')}"
        print(f"  {r['site']:30s} {status}")
    print(json.dumps(results, ensure_ascii=False))

if __name__ == "__main__":
    main()
