#!/usr/bin/env python3
"""Crawl lnfc.gov.cn - 凤城市-部门动态 (TRS WCM, glistN.html pagination)"""
import sys, re, json, sqlite3, os, ssl, urllib.request
from urllib.parse import urljoin

SITE_NAME = "凤城市-部门动态"
BASE_URL = "https://www.lnfc.gov.cn"
OUTPUT_FILE = "/root/gov_crawler/lnfc_output.jsonl"
DB_PATH = "/mnt/data/search.db"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

def http_get(url, timeout=20):
    req = urllib.request.Request(url, headers={
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
        "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    })
    try:
        resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        return ""

def extract_articles(html):
    articles = []
    pat = re.compile(
        r'<div class="time">([^<]+)</div>\s*<div class="name">\s*<a[^>]+href="([^"]+)"[^>]*title="([^"]*)"',
        re.IGNORECASE
    )
    for m in pat.finditer(html):
        date = m.group(1).strip()
        href = m.group(2)
        title = m.group(3).strip()
        url = urljoin(BASE_URL, href)
        if title:
            articles.append((url, title, date))
    return articles

def extract_detail(html):
    tm = re.search(r'<div class="center-title">(.*?)</div>', html, re.DOTALL)
    title = re.sub(r'<[^>]+>', '', tm.group(1)).strip() if tm else ""
    
    dm = re.search(r'发布日期：(\d{4}-\d{2}-\d{2})', html)
    date = dm.group(1) if dm else ""
    
    cm = re.search(r'<div class="center-info">(.*?)</div>', html, re.DOTALL)
    content = cm.group(1) if cm else ""
    
    attachments = []
    if content:
        for a in re.finditer(r'<a\s+href="([^"]+)"[^>]*>([^<]+)</a>', content):
            href, text = a.group(1), a.group(2).strip()
            full = urljoin(BASE_URL, href)
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', href, re.I):
                attachments.append({"name": text, "url": full})
        content = re.sub(r'\s*<br\s*/?>\s*', '\n', content)
        content = re.sub(r'</p>\s*', '</p>\n', content)
        content = re.sub(r'\n{3,}', '\n\n', content).strip()
    
    return title, date, content, json.dumps(attachments, ensure_ascii=False)

def main():
    max_pages = int(sys.argv[1]) if len(sys.argv) > 1 else 5
    total = existing = 0
    
    for page in range(1, max_pages + 1):
        n = page - 1
        if n == 0:
            url = BASE_URL + "/fcszf/xwdt/bmdt/glist.html"
        else:
            url = BASE_URL + "/fcszf/xwdt/bmdt/glist" + str(n) + ".html"
        
        print(f"  Page {page}...", file=sys.stderr)
        html = http_get(url)
        if not html:
            print(f"  [WARN] Empty", file=sys.stderr)
            break
        
        articles = extract_articles(html)
        if not articles:
            print(f"  [INFO] No articles", file=sys.stderr)
            break
        
        print(f"   {len(articles)} articles", file=sys.stderr)
        
        for url, title, date in articles:
            try:
                dh = http_get(url)
                if dh:
                    dt, detail_date, content, att_j = extract_detail(dh)
                    dt = dt or title
                    date = date or detail_date
                else:
                    dt, content, att_j = title, "", "[]"
                
                rec = {"site_name": SITE_NAME, "source_url": url, "page_url": url,
                       "title": dt, "publish_date": date, "content": content,
                       "summary": title[:200], "attachments": att_j}
                
                if os.path.exists(DB_PATH):
                    try:
                        db = sqlite3.connect(DB_PATH, timeout=60)
                        c = db.cursor()
                        c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=?", (url,))
                        if c.fetchone()[0] > 0:
                            existing += 1; total += 1; db.close(); continue
                        db.close()
                    except: pass
                
                with open(OUTPUT_FILE, "a", encoding="utf-8") as f:
                    f.write(json.dumps(rec, ensure_ascii=False) + "\n")
                total += 1
                if total % 20 == 0:
                    print(f"    ... {total}", file=sys.stderr)
            except Exception as e:
                print(f"    [ERR] {e}", file=sys.stderr)
    
    print(total)

if __name__ == "__main__":
    main()
