#!/usr/bin/env python3
import os
"""常熟经济技术开发区 - 信息发布爬虫
静态HTML分页 (?page=N) + requests详情页
"""
import requests, re, sqlite3, sys, time, os
from datetime import datetime, timedelta

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "常熟经济技术开发区-信息发布"
BASE = "https://www.changshu-china.com"
LIST_URL = BASE + "/part-xinxifabu.html"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_PAGES = 5
H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url):
    for retry in range(2):
        try:
            r = requests.get(url, headers=H, timeout=30)
            if r.status_code == 200:
                r.encoding = "utf-8"
                return r.text
        except:
            time.sleep(2)
    return None

def parse_list(html):
    items = []
    for li in re.findall(r'<li class="work-activity-list-item[^>]*>(.*?)</li>', html, re.DOTALL):
        href_m = re.search(r'href="([^"]+)"', li)
        if not href_m:
            continue
        href = href_m.group(1)
        # Skip external links
        if href.startswith("http") and BASE not in href:
            continue
        
        title = (re.search(r'work-activity-list-item-text[^>]*>(.*?)<', li) or [None, ""])[1]
        date = (re.search(r'style="color:#cdcdcd;"[^>]*>(.*?)<', li) or [None, ""])[1]
        
        if title:
            items.append({
                "url": href if href.startswith("http") else BASE + href,
                "title": title.strip(),
                "date": date.strip()
            })
    return items

def extract_detail(html):
    title = ""
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html)
    if m:
        title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    
    publish_date = ""
    m = re.search(r'时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if m:
        publish_date = m.group(1)
    if not publish_date:
        m = re.findall(r'(\d{4}-\d{2}-\d{2})', html)
        if m:
            publish_date = m[0]
    
    content = ""
    m = re.search(r'<article[^>]*>(.*?)</article>', html, re.DOTALL)
    if m:
        content = m.group(1)
        content = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', content, flags=re.DOTALL|re.I)
        # Remove h1 (title is duplicated), keep section content
        content = re.sub(r'<h1[^>]*>.*?</h1>', '', content, flags=re.DOTALL)
        content = re.sub(r'\s*(style|class|align|lang|dir|width|height|border|cellpadding|cellspacing)="[^"]*"', '', content)
        content = content.strip()
    
    return title, publish_date, content

def run(max_pages=None):
    if max_pages is None:
        max_pages = MAX_PAGES
    print(f"\n{'='*50}\n🚀 {SITE_NAME}\n{'='*50}")
    
    all_items = []
    for pg in range(1, max_pages + 1):
        url = LIST_URL if pg == 1 else f"{LIST_URL}?page={pg}"
        html = fetch(url)
        if not html:
            print(f"  ⚠️ 第{pg}页取不到")
            break
        items = parse_list(html)
        # Remove duplicates that span pages
        existing_urls = {i["url"] for i in all_items}
        new_items = [i for i in items if i["url"] not in existing_urls]
        if not new_items:
            print(f"  → 第{pg}页无新条目")
            break
        print(f"📋 第{pg}页: {len(new_items)} 条")
        all_items.extend(new_items)
        time.sleep(0.5)
    
    if not all_items:
        print("  ⚠️ 无数据")
        return
    
    print(f"\n📊 共 {len(all_items)} 条 (去重后)")
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    new = skip = no_body = 0
    
    for i, item in enumerate(all_items, 1):
        if item["date"] and item["date"] < CUTOFF:
            skip += 1
            continue
        
        html = fetch(item["url"])
        if not html:
            print(f"  ⚠️ 详情取不到: {item['title'][:30]}")
            skip += 1
            continue
        
        title, date, content = extract_detail(html)
        if not title:
            title = item["title"]
        if not date:
            date = item["date"]
        
        text_len = len(re.sub(r'<[^>]+>', '', content).strip()) if content else 0
        if text_len < 10:
            no_body += 1
            continue
        
        try:
            dr = 0
            try:
                dr = int(datetime.strptime(date, "%Y-%m-%d").timestamp())
            except:
                dr = int(time.time())
            summary = re.sub(r'<[^>]+>', '', content).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,content,summary,date_rank,category) VALUES(?,?,?,?,?,?,?,?,?)",
                (SITE_NAME, item["url"], item["url"], title, date, content, summary, dr, "信息发布")
            )
            if conn.total_changes:
                new += 1
            else:
                skip += 1
        except Exception as e:
            print(f"  ❌ DB: {e}")
            skip += 1
        
        if i % 10 == 0:
            conn.commit()
            print(f"  ...{i}/{len(all_items)}")
        time.sleep(0.3)
    
    conn.commit()
    try:
        conn.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
        conn.execute("INSERT INTO gov_search(rowid,title,site_name,summary) SELECT rowid,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        conn.commit()
    except Exception as e:
        print(f"⚠️ FTS: {e}")
    conn.close()
    
    print(f"\n✅ {SITE_NAME}: 新增{new}, 空正文{no_body}, 跳过{skip}")

if __name__ == "__main__":
    mp = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(mp)
