#!/usr/bin/env python3
import os
"""


肇东市人民政府 - 公示公告 爬虫
https://www.hljzhaodong.gov.cn/zd/c31/dh_lby.shtml
API: /common/search/{channelId}?_isAgg=false&_isJson=true&_pageSize=15&_template=index&page={page}
"""
import re
import sys
import json
import time
import sqlite3
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
BASE_URL = "https://www.hljzhaodong.gov.cn/zd/c31/dh_lby.shtml"
DOMAIN = "www.hljzhaodong.gov.cn"
SITE_NAME = "肇东市公示公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CHANNEL_ID = "21d5554a160b4c54860635454084dafb"
MAX_PAGES = 80
INCREMENTAL = "--incremental" in sys.argv
YEAR_LIMIT = 3

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "application/json,text/html,*/*",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

conn = sqlite3.connect(DB_PATH, timeout=60)
c = conn.cursor()

cutoff_date = datetime.now(timezone(timedelta(hours=8))).replace(tzinfo=None) - timedelta(days=365 * YEAR_LIMIT)

def fetch_list_api(page):
    url = f"https://www.hljzhaodong.gov.cn/common/search/{CHANNEL_ID}?_isAgg=false&_isJson=true&_pageSize=15&_template=index&page={page}"
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        data = r.json()
        return data.get("data", {}).get("results", [])
    except Exception as e:
        print(f"  [API ERROR] page {page}: {e}")
        return None

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  [ERROR] fetch {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta_title.get("content", "") if meta_title else ""
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True)

    date_str = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date:
        content = meta_date.get("content", "")
        dm = re.match(r'(\d{4}-\d{2}-\d{2})', content)
        if dm:
            date_str = dm.group(1)

    content_html = ""
    # Strategy 1: find con div inside content div (avoids header search box)
    content_div = soup.find("div", class_="content")
    if content_div:
        con_div = content_div.find("div", class_="con")
        if con_div:
            content_html = str(con_div)
    # Strategy 2: fallback - take the LAST con div on the page
    if not content_html:
        all_con = soup.find_all("div", class_="con")
        if all_con and len(all_con) > 0:
            content_html = str(all_con[-1])
    # Strategy 3: take the actual content div itself
    if not content_html and content_div:
        content_html = str(content_div)

    if not content_html:
        content_html = "<p>内容加载失败</p>"

    return title, content_html, date_str

def article_exists(page_url):
    c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url = ?", (page_url,))
    return c.fetchone()[0] > 0

def sync_fts(row_id, title, site_name):
    try:
        c.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                  (row_id, title, site_name, ""))
    except:
        pass

def insert_article(title, page_url, content_html, date_str):
    display_url = page_url.replace("http://", "").replace("https://", "")
    summary = ""

    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
    row = c.fetchone()
    existing_id = row[0] if row else None

    try:
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_zhaodong.py')""",
            (page_url, title, content_html, date_str, SITE_NAME, display_url, summary, ""))

        row_id = existing_id or c.lastrowid
        if row_id:
            sync_fts(row_id, title, SITE_NAME)
        return True
    except Exception as e:
        print(f"  [DB ERROR] {e}")
        return False

def main():
    total_new = 0
    total_updated = 0
    total_skipped = 0

    print(f"  Pages to crawl: 1-{MAX_PAGES}")

    for page in range(1, min(MAX_PAGES, _MAX_PG or MAX_PAGES)+1):
        print(f"  List page {page}/{MAX_PAGES}")

        results = fetch_list_api(page)
        if results is None:
            continue
        if not results:
            print(f"    No more results, stopping at page {page}")
            break

        print(f"    Found {len(results)} items")

        for item in results:
            title = item.get("title", "")
            api_url = item.get("url", "")
            date_str_full = item.get("publishedTimeStr", "")
            date_str = date_str_full[:10] if len(date_str_full) >= 10 else ""

            page_url = api_url.replace("http://www.hljzhaodong.gov.cn/", "https://www.hljzhaodong.gov.cn/")
            page_url = page_url.replace("https://www.hljzhaodong.gov.cn//", "https://www.hljzhaodong.gov.cn/")

            if date_str:
                try:
                    item_date = datetime.strptime(date_str, "%Y-%m-%d")
                    if item_date < cutoff_date.replace(tzinfo=None):
                        total_skipped += 1
                        continue
                except:
                    pass

            exists = article_exists(page_url)
            if INCREMENTAL and exists:
                total_skipped += 1
                continue

            print(f"    Fetching: {title[:35]}...", end=" ")
            d_title, content_html, detail_date = fetch_detail(page_url)

            if not d_title:
                d_title = title
            if not detail_date and date_str:
                detail_date = date_str

            # Only insert if there's actual content
            if content_html and content_html != "<p>内容加载失败</p>":
                if insert_article(d_title, page_url, content_html, detail_date):
                    if exists:
                        total_updated += 1
                        print("updated")
                    else:
                        total_new += 1
                        print("inserted")
                else:
                    print("FAILED")
            else:
                print(f"no content, skipped")
                total_skipped += 1

            time.sleep(0.3)

        conn.commit()
        print(f"  [COMMIT] page {page} done")

    conn.close()
    print(f"\n  ✅ 完成: 新增 {total_new}, 更新 {total_updated}, 跳过 {total_skipped}")

if __name__ == "__main__":
    main()
