#!/usr/bin/env python3
"""
Crawl 息烽县 - 项目环评
https://www.xifeng.gov.cn/zwgk/zdlygk/sthj/xmhp_5741524/
"""
import requests, re, sqlite3, os, time
from datetime import datetime, timedelta
import os

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
NAME = "息烽县-项目环评"
DOMAIN = "www.xifeng.gov.cn"
BASE = "https://www.xifeng.gov.cn"
LIST_PATH = "/zwgk/zdlygk/sthj/xmhp_5741524"

PAGES = 7
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0"}
NOW = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def extract_content(html):
    """Extract from TRS UEDITOR div"""
    idx = html.find('class="trs_editor_view TRS_UEDITOR')
    if idx < 0:
        idx = html.find('class="content_1"')
    if idx > 0:
        div_start = html.rfind("<div", 0, idx)
        if div_start >= 0:
            section = html[div_start:]
            depth = 0
            content = ""
            for i in range(len(section)):
                if section[i:i+4] == "<div" and (i+4 >= len(section) or section[i+4] in " >\n\r\t"):
                    depth += 1
                elif section[i:i+6] == "</div>":
                    depth -= 1
                    if depth == 0:
                        gt_pos = section.find(">", 0, i)
                        content = section[gt_pos+1:i] if gt_pos > 0 else section[7:i]
                        break
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
            content = re.sub(r'<!--.*?-->', '', content, flags=re.DOTALL)
            return content.strip()
    return ""

def fetch_detail(session, url):
    """Fetch detail page"""
    try:
        r = session.get(url, timeout=30)
        if r.status_code != 200:
            return None, None, None
        r.encoding = "utf-8"
        html = r.text
    except:
        return None, None, None
    
    content = extract_content(html)
    
    # Title
    title_m = re.search(r"<title>(.*?)<", html)
    title = title_m.group(1).strip() if title_m else ""
    title = re.sub(r'-\s*息烽县人民政府门户网站.*$', '', title).strip()
    
    # Date
    date_str = ""
    dm = re.search(r'<meta name="PubDate"[^>]*content="([^"]+)"', html, re.I)
    if dm:
        raw = dm.group(1).strip().split(" ")[0]
        try:
            date_str = datetime.strptime(raw, "%Y-%m-%d").strftime("%Y-%m-%d")
        except:
            date_str = raw
    
    summary = re.sub(r'<[^>]+>', '', content)[:200] if content else ""
    summary = re.sub(r'\s+', ' ', summary).strip()
    
    return title, date_str, content

def parse_list(html):
    """Extract items from list page - parse within <li> to avoid cross-element matches"""
    items = []
    for li in re.findall(r'<li>(.*?)</li>', html, re.DOTALL):
        m = re.search(
            r'<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})</span>',
            li, re.DOTALL
        )
        if m:
            url = m.group(1).strip()
            title = m.group(2).strip()
            if not title:
                title = re.sub(r'<[^>]+>', '', m.group(3)).strip()
            date = m.group(4).strip()
            if url and title:
                items.append({"url": url, "title": title, "date": date})
    return items

def insert_item(conn, item):
    """Insert or replace"""
    title = item["title"].replace("'", "''")
    page_url = item["url"].replace("'", "''")
    content = item.get("content", "").replace("'", "''")
    summary = item.get("summary", "").replace("'", "''")
    date = item.get("date", "").replace("'", "''")
    date_rank = 0
    try:
        date_rank = int(datetime.strptime(date, "%Y-%m-%d").timestamp())
    except:
        date_rank = int(time.time())
    
    sql = f"""INSERT OR REPLACE INTO gov_raw
        (site_name, page_url, title, publish_date, content, summary, date_rank, category)
        VALUES ('{NAME}', '{page_url}', '{title}', '{date}',
                '{content}', '{summary}', {date_rank}, '环评审批')"""
    try:
        conn.execute(sql)
        conn.commit()
        return True
    except Exception as e:
        print(f"    ❌ DB: {e}")
        return False

def main():
    print(f"\n{'='*60}")
    print(f"🚀 {NAME}")
    print(f"   页面: 1~{PAGES}")
    print(f"   日期截止: {CUTOFF_DATE}")
    print(f"{'='*60}\n")
    
    session = requests.Session()
    session.headers.update(HEADERS)
    conn = sqlite3.connect(SEARCH_DB)
    conn.execute("PRAGMA journal_mode=WAL")
    
    total_new = 0
    start_time = time.time()
    
    for page in range(0, PAGES):
        if page == 0:
            list_url = f"{BASE}{LIST_PATH}/index.html"
        else:
            list_url = f"{BASE}{LIST_PATH}/index_{page+1}.html"
        
        print(f"📄 第{page+1}/{PAGES}页")
        
        try:
            r = session.get(list_url, timeout=20)
            if r.status_code != 200:
                print(f"  ❌ HTTP {r.status_code}")
                continue
            r.encoding = "utf-8"
            items = parse_list(r.text)
        except Exception as e:
            print(f"  ❌ 错误: {e}")
            continue
        
        if not items:
            print(f"  ⚠️ 无数据")
            continue
        
        print(f"  📋 {len(items)} 条")
        
        for idx, item in enumerate(items):
            if item["date"] < CUTOFF_DATE:
                continue
            
            has = conn.execute(
                "SELECT 1 FROM gov_raw WHERE page_url=? AND content IS NOT NULL AND content != ''",
                (item["url"],)
            ).fetchone()
            
            if has:
                print(f"  [{idx+1}/{len(items)}] ⏭️ {item['title'][:35]}...")
                continue
            
            print(f"  [{idx+1}/{len(items)}] {item['title'][:35]}...", end=" ", flush=True)
            
            title, date, content = fetch_detail(session, item["url"])
            
            if content:
                item["content"] = content
                item["summary"] = re.sub(r'<[^>]+>', '', content)[:200]
                item["summary"] = re.sub(r'\s+', ' ', item["summary"]).strip()
                item["title"] = title or item["title"]
                item["date"] = date or item["date"]
                if insert_item(conn, item):
                    total_new += 1
                    print(f"✅ {len(content):,}字")
                else:
                    print(f"❌ DB")
            else:
                item["content"] = ""
                item["summary"] = item["title"]
                if insert_item(conn, item):
                    print(f"⚠️ 无正文")
                else:
                    print(f"❌ DB")
            
            time.sleep(0.5)
        
        print()
    
    elapsed = time.time() - start_time
    print(f"{'='*60}")
    print(f"📊 完成！新增 {total_new} 条")
    print(f"⏱ 耗时: {elapsed:.0f}s ({elapsed/60:.1f}min)")
    print(f"{'='*60}")
    conn.close()
    return total_new

if __name__ == "__main__":
    main()
