#!/usr/bin/env python3
"""
Crawl 阜新市细河区 - 便民通知 (fixed)
https://www.fxxh.gov.cn/channel/list/11354.html
"""
import requests, re, sqlite3, sys, os, time
from datetime import datetime, timedelta
import os

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
NAME = "细河区-便民通知"
DOMAIN = "www.fxxh.gov.cn"

PAGES = 10
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0"}
NOW = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def extract_content(html):
    """Extract content from detail page - try multiple patterns"""
    # Pattern 1: Detailcontent div
    for pat in ['id="Detailcontent"', 'id=Detailcontent']:
        idx = html.find(pat)
        if idx >= 0:
            div_start = html.rfind("<div", 0, idx)
            if div_start >= 0:
                section = html[div_start:]
                depth = 0
                for i in range(len(section)):
                    if section[i:i+4] == "<div" and (i+4 >= len(section) or section[i+4] in " >\n\r\t"):
                        depth += 1
                    elif section[i:i+6] == "</div>":
                        depth -= 1
                        if depth == 0:
                            gt_pos = section.find(">", 0, i)
                            if gt_pos > 0:
                                content = section[gt_pos+1:i]
                            else:
                                content = section[7:i]
                            # Clean
                            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
                            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
                            content = re.sub(r'<!--.*?-->', '', content, flags=re.DOTALL)
                            return content.strip()
    
    # Pattern 2: news_inner
    for pat in ['class="news_inner"', 'class=news_inner']:
        idx = html.find(pat)
        if idx >= 0:
            div_start = html.rfind("<div", 0, idx)
            if div_start >= 0:
                section = html[div_start:]
                depth = 0
                for i in range(len(section)):
                    if section[i:i+4] == "<div" and (i+4 >= len(section) or section[i+4] in " >\n\r\t"):
                        depth += 1
                    elif section[i:i+6] == "</div>":
                        depth -= 1
                        if depth == 0:
                            gt_pos = section.find(">", 0, i)
                            if gt_pos > 0:
                                content = section[gt_pos+1:i]
                            else:
                                content = section[7:i]
                            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
                            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
                            content = re.sub(r'<!--.*?-->', '', content, flags=re.DOTALL)
                            return content.strip()
    
    return ""

def fetch_detail(session, url):
    """Fetch detail page"""
    try:
        r = session.get(url, timeout=30)
        if r.status_code != 200:
            return None, None, None, None
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        return None, None, None, None
    
    content = extract_content(html)
    
    # Title
    title_m = re.search(r"<title>(.*?)<", html)
    title = title_m.group(1).strip() if title_m else ""
    
    # Date from meta
    date_str = ""
    date_m = re.search(r'<meta name="PubDate"[^>]*content="([^"]+)"', html)
    if date_m:
        raw = date_m.group(1).strip().split(" ")[0]
        try:
            dt = datetime.strptime(raw, "%Y-%m-%d")
            date_str = dt.strftime("%Y-%m-%d")
        except:
            try:
                dt = datetime.strptime(raw, "%Y-%-m-%-d")
                date_str = dt.strftime("%Y-%m-%d")
            except:
                date_str = raw
    
    summary = re.sub(r'<[^>]+>', '', content)[:200] if content else ""
    summary = re.sub(r'\s+', ' ', summary).strip()
    
    return title, date_str, content, summary

def parse_list(html):
    """Extract article items from list page - handle both full URLs and relative"""
    items = []
    for m in re.finditer(
        r'<a class="list1-ul-link" href="([^"]+)"[^>]*>'
        r'.*?<div class="list1-ul-link-inner">(.*?)</div>'
        r'.*?<div class="tabs2Right-part2">\s*([\d-]+)\s*</div>',
        html, re.DOTALL
    ):
        url = m.group(1).strip()
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        date = m.group(3).strip()
        if title and url:
            items.append({"url": url, "title": title, "date": date})
    return items

def insert_item(conn, item):
    """Insert or replace one item"""
    title = item["title"].replace("'", "''")
    page_url = item["url"].replace("'", "''")
    content = item.get("content", "").replace("'", "''")
    summary = item.get("summary", "").replace("'", "''")
    date = item.get("date", "").replace("'", "''")
    
    date_rank = 0
    try:
        dt = datetime.strptime(date, "%Y-%m-%d")
        date_rank = int(dt.timestamp())
    except:
        date_rank = int(time.time())
    
    sql = f"""INSERT OR REPLACE INTO gov_raw 
        (site_name, page_url, title, publish_date, content, summary, date_rank, category, script_name)
        VALUES ('{NAME}', '{page_url}', '{title}', '{date}',
                '{content}', '{summary}', {date_rank}, '便民通知', 'crawl_fxxh.py')"""
    try:
        conn.execute(sql)
        conn.commit()
        return True
    except Exception as e:
        print(f"    ❌ DB: {e}")
        return False

def main():
    print(f"\n{'='*60}")
    print(f"🚀 {NAME} 爬虫")
    print(f"   页面: 1~{PAGES}")
    print(f"   日期截止: {CUTOFF_DATE}")
    print(f"{'='*60}\n")
    
    session = requests.Session()
    session.headers.update(HEADERS)
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    
    total_new = 0
    total_skipped = 0
    start_time = time.time()
    
    for page in range(1, PAGES + 1):
        if page == 1:
            list_url = "https://www.fxxh.gov.cn/channel/list/11354.html"
        else:
            list_url = f"https://www.fxxh.gov.cn/channel/list/11354_{page}.html"
        
        print(f"📄 第{page}/{PAGES}页")
        
        try:
            r = session.get(list_url, timeout=20)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  ❌ HTTP {r.status_code}")
                continue
            items = parse_list(r.text)
        except Exception as e:
            print(f"  ❌ 错误: {e}")
            continue
        
        if not items:
            print(f"  ⚠️ 无数据")
            continue
        
        print(f"  📋 {len(items)} 条")
        
        # Filter by date
        recent = [it for it in items if it["date"] >= CUTOFF_DATE]
        total_skipped += len(items) - len(recent)
        if not recent:
            print(f"  ✅ 已无3年内数据，停止")
            break
        
        for idx, item in enumerate(recent):
            # Check existing content
            has = conn.execute(
                "SELECT 1 FROM gov_raw WHERE page_url=? AND content IS NOT NULL AND content != ''",
                (item["url"],)
            ).fetchone()
            
            if has:
                print(f"  [{idx+1}/{len(recent)}] ⏭️ {item['title'][:35]}...")
                continue
            
            print(f"  [{idx+1}/{len(recent)}] {item['title'][:35]}...", end=" ", flush=True)
            
            title, date, content, summary = fetch_detail(session, item["url"])
            
            if content:
                item["content"] = content
                item["summary"] = summary or item["title"]
                item["title"] = title or item["title"]
                item["date"] = date or item["date"]
                if insert_item(conn, item):
                    total_new += 1
                    print(f"✅ {len(content):,}字")
                else:
                    print(f"❌ DB失败")
            else:
                # Still try insert with list page data
                item["content"] = ""
                item["summary"] = item["title"]
                if insert_item(conn, item):
                    print(f"⚠️ 无正文，已保存标题")
                else:
                    print(f"❌ DB失败")
            
            time.sleep(0.5)
        
        print()
    
    elapsed = time.time() - start_time
    print(f"{'='*60}")
    print(f"📊 完成！新增 {total_new} 条 | 跳过 {total_skipped} 条")
    print(f"⏱ 耗时: {elapsed:.0f}s ({elapsed/60:.1f}min)")
    print(f"{'='*60}")
    
    conn.close()
    return total_new

if __name__ == "__main__":
    main()
