#!/usr/bin/env python3
"""
本溪市溪湖区人民政府 - 每日增量爬虫
栏目：pzjgxx (批准结果), pzfwxx (批准服务)
每日增量模式
"""
import urllib.request, re, os, sys, time, json
from datetime import datetime, timedelta

SITE_BASE = "http://www.xihu.gov.cn"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
TEMP_DB = os.environ.get("TEMP_DB", "/root/search.db")
DAYS_BACK = 3  # Look back 3 days for daily crawl

COLUMNS = {
    "xihu_pzjg": SITE_BASE + "/publicity/zdxxgz/zdlyxxgk/zdjsxm/pzjgxx",
    "xihu_pzfw": SITE_BASE + "/publicity/zdxxgz/zdlyxxgk/zdjsxm/pzfwxx",
}

def get_max_pages(list_url):
    """Get total pages from list page"""
    req = urllib.request.Request(list_url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=15)
    html = resp.read().decode("utf-8")
    m = re.search(r"共(\d+)页", html)
    if m:
        return int(m.group(1))
    return 1

def fetch_list(url):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    html = resp.read().decode("utf-8")
    articles = []
    col_name = "pzjgxx" if "pzjgxx" in url else "pzfwxx"
    for m in re.finditer(rf"(202\d-\d{{2}}-\d{{2}}).*?<a[^>]*href=\"(/publicity/[^\"]*{col_name}/\d+)\"[^>]*>\s*(.*?)\s*</a>", html, re.DOTALL):
        date = m.group(1)
        href = m.group(2)
        title = re.sub(r"<[^>]+>", "", m.group(3)).strip()
        if len(title) > 15:
            if not href.startswith("http"):
                href = SITE_BASE + href
            articles.append({"title": title, "url": href, "date": date})
    return articles

def main():
    import sqlite3
    
    conn = sqlite3.connect(TEMP_DB, timeout=60)
    c = conn.cursor()
    c.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT, content TEXT, source_url TEXT UNIQUE, site_name TEXT, publish_date TEXT)")
    
    cutoff = (datetime.now() - timedelta(days=DAYS_BACK)).strftime("%Y-%m-%d")
    
    for site_name, list_url in COLUMNS.items():
        print(f"\n=== {site_name} ===")
        try:
            total_pages = get_max_pages(list_url)
            print(f"  Pages: {total_pages}")
            
            all_articles = []
            base = list_url.rstrip("/")
            
            for page in range(1, total_pages + 1):
                url = base if page == 1 else f"{base}_{page}"
                try:
                    arts = fetch_list(url)
                    # Only keep recent articles
                    recent = [a for a in arts if a["date"] >= cutoff]
                    all_articles.extend(recent)
                    if not recent:
                        # If first page has no recent items, earlier pages won't either
                        if page == 1:
                            all_articles.extend(arts[:5])  # Keep first 5 as buffer
                        else:
                            break
                except Exception as e:
                    print(f"  Page {page} error: {e}")
                time.sleep(0.3)
            
            # Deduplicate by URL
            seen = set()
            unique_articles = []
            for a in all_articles:
                if a["url"] not in seen:
                    seen.add(a["url"])
                    unique_articles.append(a)
            
            print(f"  Articles to check: {len(unique_articles)}")
            
            new_count = 0
            for art in unique_articles:
                c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url=?", (art["url"],))
                if c.fetchone()[0] > 0:
                    continue
                
                # Fetch detail
                try:
                    req = urllib.request.Request(art["url"], headers=HEADERS)
                    resp = urllib.request.urlopen(req, timeout=30)
                    html = resp.read().decode("utf-8")
                    content = ""
                    m = re.search(r'class="(?:content|article|mainText|bt_content)[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
                    if m: content = m.group(1)
                    if not content:
                        m = re.search(r'class="mainContent"[^>]*>(.*?)</div>\s*</div>\s*<!-- content', html, re.DOTALL)
                        if m: content = m.group(1)
                    if content:
                        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
                        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
                        if len(re.sub(r'<[^>]+>', '', content).strip()) >= 50:
                            c.execute("INSERT OR IGNORE INTO gov_raw VALUES (?,?,?,?,?,?)",
                                      (None, art["title"], content, art["url"], site_name, art["date"]))
                            conn.commit()
                            new_count += 1
                except Exception as e:
                    print(f"  Detail error: {art['url'][-40:]}: {e}")
                time.sleep(0.5)
            
            print(f"  New: {new_count}")
        except Exception as e:
            print(f"  Error: {e}")
    
    conn.close()
    print(f"\nDone at {datetime.now().strftime('%H:%M:%S')}")

if __name__ == "__main__":
    main()
