#!/usr/bin/env python3
"""乌海市人民政府-通知公告 (fixed pagination)
http://www.wuhai.gov.cn/wuhai/xxgk4/jbxxgk46/tzgg48/index.html
Pagination: ?page=N (totalpage=3)
"""
import requests, re, sys, os, subprocess, time
from bs4 import BeautifulSoup

SITE_NAME = "乌海市人民政府-通知公告"
BASE_URL = "http://www.wuhai.gov.cn"
LIST_URL = "http://www.wuhai.gov.cn/wuhai/xxgk4/jbxxgk46/tzgg48/index.html"
HEADERS = {"User-Agent": "Mozilla/5.0"}
DB_PATH = "/mnt/data/search.db"

def fetch(url):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    return r.text

def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.find("ul", class_="wh_thr_list")
    if not ul:
        return items, 0, 0
    for li in ul.find_all("li", class_="qs_clear"):
        date_span = li.find("span", class_="date")
        a_tag = li.find("a", title=True)
        if not a_tag or not date_span:
            continue
        href = a_tag.get("href", "")
        if not href:
            continue
        full_url = BASE_URL + href if href.startswith("/") else href
        title = a_tag.get("title", "")
        date_str = date_span.get_text(strip=True)
        items.append((title.strip(), full_url, date_str))
    total = 1
    hidden = soup.find("input", {"name": "article_paging_list_hidden"})
    if hidden and hidden.get("totalpage"):
        total = int(hidden["totalpage"])
    return items, total

def parse_detail(html):
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    title_div = soup.find("div", id="title")
    if title_div:
        title = title_div.get_text(strip=True)
    if not title:
        t_tag = soup.find("title")
        if t_tag:
            t = t_tag.get_text(strip=True)
            title = t.replace("-乌海市政府", "").replace("-通知公告", "").strip()
    content_div = soup.find("div", id="wh_x_c") or soup.find("div", class_="wh_xl_c")
    content = str(content_div) if content_div else ""
    date_str = ""
    sj = soup.find("span", id="shijian")
    if sj:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", sj.get_text())
        if m:
            date_str = m.group(1)
    if not date_str:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", html)
        if m:
            date_str = m.group(1)
    return title, content, date_str

def crawl(max_pages=5):
    all_items = []
    for page in range(1, max_pages + 1):
        page_url = LIST_URL + ("?page=" + str(page) if page > 1 else "")
        html = fetch(page_url)
        if not html:
            break
        items, total_pages = parse_list(html)
        all_items.extend(items)
        print("  Page {}: {} items (total: {} pages)".format(page, len(items), total_pages))
        if not items or page >= total_pages:
            break
        time.sleep(0.3)
    return all_items

def main():
    pages = 5
    if len(sys.argv) > 1:
        pages = int(sys.argv[1])
    items = crawl(pages)
    if not items:
        print("No items")
        sys.exit(1)
    new_count = 0
    dup_count = 0
    for title, page_url, pub_date in items:
        html = fetch(page_url)
        if not html:
            continue
        dt, content, dd = parse_detail(html)
        final_title = dt or title
        final_date = dd or pub_date
        summary = re.sub(r"<[^>]+>", " ", content)[:200] if content else ""
        summary = re.sub(r"\s+", " ", summary).strip()
        has_table = 1 if "<table" in content else 0
        safe = {
            "site": SITE_NAME.replace("'", "''"),
            "url": page_url.replace("'", "''"),
            "title": final_title.replace("'", "''"),
            "date": final_date,
            "summary": summary.replace("'", "''"),
            "content": content.replace("'", "''")
        }
        sql = "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,summary,content,industry,group_name,has_table) VALUES('{}','{}','{}','{}','{}','{}','{}','政府公告','内蒙古',{})".format(
            safe["site"], safe["url"], safe["url"], safe["title"], safe["date"], safe["summary"], safe["content"], has_table)
        subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, timeout=30)
        result = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, "SELECT id FROM gov_raw WHERE page_url='{}'".format(safe["url"])],
            capture_output=True, text=True, timeout=30)
        row_id = result.stdout.strip()
        if row_id and row_id.isdigit():
            fts_sql = "INSERT OR REPLACE INTO gov_search(rowid,title,site_name,summary) VALUES({},'{}','{}','{}')".format(
                row_id, safe["title"], safe["site"], safe["summary"])
            subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, fts_sql], capture_output=True, timeout=30)
            new_count += 1
            print("  + {} | {}".format(final_title[:30], final_date))
        else:
            dup_count += 1
    print("\nDone: {} new, {} dup".format(new_count, dup_count))

if __name__ == "__main__":
    main()
