#!/usr/bin/env python3
"""新余市发展和改革委员会-项目建设动态 (fgw.xinyu.gov.cn) 爬虫

CMS: UCAP
列表: /fgw/xmjsdt/zwgk_xxgklist.shtml (静态分页: zwgk_xxgklist_N.shtml, 16页)
详情: div.article-title标题 + UCAPCONTENT正文 + meta PubDate日期
注意: 只爬/fgw/xmjsdt/路径的本地文章，跳过微信/市政府/省发改委外链
"""

import os, sqlite3, re, time, requests, sys
from datetime import datetime, timezone, timedelta
from urllib3 import disable_warnings, exceptions
disable_warnings(exceptions.InsecureRequestWarning)

SITE_NAME = "新余市发改委-项目建设动态"
BASE = "https://fgw.xinyu.gov.cn"
LIST_TPL = "https://fgw.xinyu.gov.cn/fgw/xmjsdt/zwgk_xxgklist.shtml"
LIST_PAGE_TPL = "https://fgw.xinyu.gov.cn/fgw/xmjsdt/zwgk_xxgklist_%d.shtml"
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
CUTOFF = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
MAX_PAGES = 16

def fetch(url, max_retry=3):
    for att in range(max_retry):
        try:
            r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if att < max_retry - 1:
                time.sleep(2)
    return None

def parse_list_page(html):
    items = []
    for m in re.finditer(r'<li[^>]*>(.*?)</li>', html, re.S):
        li = m.group(1)
        link_m = re.search(r'<a[^>]*href="([^"]+)"[^>]*>([^<]+)</a>', li)
        if not link_m:
            continue
        href = link_m.group(1)
        title = link_m.group(2).strip()
        
        # Only crawl local /fgw/xmjsdt/ pages (project articles, not nav sections)
        if not href.startswith("/fgw/xmjsdt/"):
            continue
        
        # Full URL
        if href.startswith("http"):
            full_url = href
        else:
            full_url = BASE + href if href.startswith("/") else BASE + "/" + href
        
        # Extract date from li
        date_str = ""
        date_m = re.search(r'(\d{4})[-/](\d{1,2})[-/](\d{1,2})', li)
        if date_m:
            y, mo, d = date_m.group(1), date_m.group(2), date_m.group(3)
            date_str = f"{y}-{mo.zfill(2)}-{d.zfill(2)}"
        
        if date_str and date_str < CUTOFF:
            continue
        
        items.append({"title": title, "url": full_url, "pub_date": date_str})
    return items

def fetch_detail(url):
    html = fetch(url)
    if not html:
        return "", "", ""
    
    # Title from article-title div
    title = ""
    m = re.search(r'class="article-title"[^>]*>(.*?)</div>', html, re.S)
    if m:
        title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
        title = re.sub(r'\s+', ' ', title).strip()
        title = re.sub(r'\s*访问量[：:].*$', '', title).strip()
    if not title:
        m = re.search(r'<title>(.*?)</title>', html)
        if m:
            title = m.group(1).strip()
    
    # Content from UCAPCONTENT
    content = ""
    m = re.search(r'<UCAPCONTENT>(.*?)</UCAPCONTENT>', html, re.S)
    if m:
        content = m.group(1).strip()
    if not content:
        m = re.search(r'class="article-content[^"]*"[^>]*>(.*?)</div>', html, re.S)
        if m:
            content = m.group(1).strip()
    
    content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.S)
    content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.S)
    
    # Date
    date_str = ""
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{2}-\d{2})', html)
    if m:
        date_str = m.group(1)
    if not date_str:
        m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{2}-\d{2})', html)
        if m:
            date_str = m.group(1)
    if not date_str:
        for d in re.finditer(r'(\d{4}-\d{1,2}-\d{1,2})', html):
            ctx = html[max(0,d.start()-30):d.end()+10]
            if 'date' in ctx.lower() or '时间' in ctx or '发布' in ctx:
                date_str = d.group(1)
                break
    
    return title, content, date_str

def main():
    test = "--test" in sys.argv
    mode = "TEST" if test else "FULL"
    print(f"\n=== {SITE_NAME} ({mode}) ===", flush=True)
    
    all_items = []
    for pg in range(1, MAX_PAGES + 1):
        url = LIST_TPL if pg == 1 else LIST_PAGE_TPL % pg
        html = fetch(url)
        if not html:
            print(f"  Page {pg}: fetch failed", flush=True)
            break
        items = parse_list_page(html)
        print(f"  Page {pg}: {len(items)} local items", flush=True)
        all_items.extend(items)
        if len(items) == 0:
            break
    
    print(f"  Total: {len(all_items)} local items within 3yr", flush=True)
    
    if not all_items:
        print("  No items to crawl")
        return
    
    if test:
        all_items = all_items[:5]
    
    conn = None if test else sqlite3.connect(DB_PATH)
    cursor = conn.cursor() if conn else None
    total_new = 0
    done = 0
    
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}... ", end="", flush=True)
        title, content, pub_date = fetch_detail(item["url"])
        if not title:
            title = item["title"]
        if not pub_date:
            pub_date = item["pub_date"]
        
        if test:
            print(f"title={title[:50]} content={len(content)}B date={pub_date}")
            continue
        
        try:
            cursor.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, tags) VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, title[:500], item["url"], content, pub_date, "", "项目建设"),
            )
            if cursor.rowcount > 0:
                total_new += 1
                print(f"{len(content)}B")
            else:
                print("skip")
        except Exception as e:
            print(f"DB error: {e}")
        
        done += 1
        if done % 30 == 0 and conn:
            conn.commit()
    
    if conn:
        conn.commit()
        conn.execute(
            "INSERT INTO gov_search(rowid, title, site_name, summary, content) "
            "SELECT r.rowid, r.title, r.site_name, r.summary, r.content "
            "FROM gov_raw r WHERE r.rowid NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
            (SITE_NAME,)
        )
        conn.commit()
        c = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        cnt = c.fetchone()[0]
        conn.close()
        print(f"\n=== 完成 ===", flush=True)
        print(f"  新增: {total_new}条", flush=True)
        print(f"  累计: {cnt}条", flush=True)

if __name__ == "__main__":
    main()
