#!/usr/bin/env python3
"""
佛山市三水大塘镇 - 政府信息公开（其他栏目）
https://www.ss.gov.cn/fsssdt/gkmlpt/index#2913
CMS: 广东省统一政府信息公开平台 (gkmlpt)
API: /fsssdt/gkmlpt/api/all/2913?page=N&sid=757177
"""
import sys, re, json, time, requests, sqlite3
from bs4 import BeautifulSoup
from datetime import datetime
import urllib3
urllib3.disable_warnings()

DB_PATH = "/root/search.db"
SITE_NAME = "佛山市三水大塘镇-其他"
CATEGORY = "其他"
GROUP = "广东"
BASE = "https://www.ss.gov.cn"
SITE_PREFIX = "fsssdt"
SID = "757177"
COLUMN_ID = 2913
API_URL = f"{BASE}/{SITE_PREFIX}/gkmlpt/api/all/{COLUMN_ID}?page={{}}&sid={SID}"
MAX_PAGES = 22

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "application/json, text/plain, */*",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn


def save_article(conn, title, page_url, publish_date, content, publisher=""):
    summary = content[:200] if content else title
    summary = re.sub(r"\s+", " ", summary).strip()
    try:
        cur = conn.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date, summary, content, category, group_name)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
            (SITE_NAME, page_url, page_url, title.strip(), publish_date,
             summary, content, CATEGORY, GROUP),
        )
        return cur.rowcount > 0
    except Exception as e:
        print(f"  DB error: {e}", file=sys.stderr)
        return False


def fetch_detail(url):
    """Fetch detail page, return (title, content)"""
    try:
        r = requests.get(url, headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
            "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
            "Accept-Language": "zh-CN,zh;q=0.9",
        }, timeout=30, verify=False)
        r.encoding = 'utf-8'
    except Exception as e:
        return None, None
    if r.status_code != 200:
        return None, None

    soup = BeautifulSoup(r.text, 'html.parser')

    # Title from <title> tag
    title = None
    title_tag = soup.find('title')
    if title_tag:
        t = title_tag.get_text(strip=True)
        # Sometimes title has "| site name" suffix
        parts = t.split('|')
        if parts and parts[0].strip():
            title = parts[0].strip()

    # Content from div.article-content
    content_div = soup.find('div', class_=re.compile(r'article-content', re.I))
    if not content_div:
        return title, None

    # Extract text from paragraphs, preserve tables
    parts = []
    for elem in content_div.find_all(['p', 'table'], recursive=True):
        if elem.name == 'p':
            if elem.find_parent('table'):
                continue
            text = elem.get_text('', strip=True)
            if text:
                parts.append(text)
        elif elem.name == 'table':
            parts.append(str(elem))

    content = '\n\n'.join(parts) if parts else ''
    if len(content.strip()) < 20:
        content = None

    return title, content


def main():
    max_pages = MAX_PAGES
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            if sys.argv[1] in ('--incremental',):
                max_pages = 1

    print(f"[{SITE_NAME}] max pages: {max_pages}", file=sys.stderr)
    conn = init_db()
    total_new = 0

    # Clean old data
    conn.execute("DELETE FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
    conn.commit()
    print(f"  Cleaned old data", file=sys.stderr)

    session = requests.Session()
    session.headers.update(HEADERS)

    for page in range(1, max_pages + 1):
        api_url = API_URL.format(page)
        print(f"[LIST] Page {page}: {api_url}", file=sys.stderr)

        try:
            r = session.get(api_url, timeout=30)
            if r.status_code != 200:
                print(f"  HTTP {r.status_code}, stopping", file=sys.stderr)
                break
            data = r.json()
        except Exception as e:
            print(f"  API error: {e}", file=sys.stderr)
            break

        articles = data.get('articles', [])
        if not articles:
            print(f"  No articles, stopping", file=sys.stderr)
            break

        print(f"  Found {len(articles)} articles", file=sys.stderr)

        for a in articles:
            title = a.get('title', '')
            article_url = a.get('url', '')
            ts = a.get('create_time', a.get('date', 0))
            date_str = datetime.fromtimestamp(ts).strftime('%Y-%m-%d') if ts else ''

            if not title or not article_url:
                continue

            print(f"  [{date_str}] {title[:40]}...", file=sys.stderr)

            # Check if exists
            cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (article_url,))
            if cur.fetchone():
                continue

            det_title, det_content = fetch_detail(article_url)
            final_title = det_title or title
            final_content = det_content or f"[{final_title}]({article_url})"

            if save_article(conn, final_title, article_url, date_str, final_content):
                total_new += 1
                if total_new <= 5:
                    print(f"  + {date_str} {final_title[:50]}", file=sys.stderr)

            time.sleep(0.3)

        conn.commit()
        time.sleep(1)

    conn.commit()
    conn.close()
    print(f"\nDone: {total_new} new articles inserted", file=sys.stderr)
    print(total_new)


if __name__ == '__main__':
    main()
