#!/usr/bin/env python3
"""
Crawl 沾化区-通知公告 - Hanweb JPage (fix content extraction)
http://www.zhanhua.gov.cn/col/col118071/index.html
"""
import requests, re, sqlite3, time
from datetime import datetime, timedelta
import os

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
NAME = "沾化区-通知公告"
BASE = "http://www.zhanhua.gov.cn"
NOW = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0"}

def extract_content(html):
    for cls in ['class="bg-article"', 'class="content"']:
        idx = html.find(cls)
        if idx < 0: continue
        div_start = html.rfind("<div", 0, idx)
        if div_start < 0: continue
        section = html[div_start:]
        depth = 0
        content = ""
        for i in range(len(section)):
            if section[i:i+4] == "<div" and (i+4 >= len(section) or section[i+4] in " >\n\r\t"):
                depth += 1
            elif section[i:i+6] == "</div>":
                depth -= 1
                if depth == 0:
                    gt_pos = section.find(">", 0, i)
                    content = section[gt_pos+1:i] if gt_pos > 0 else section[7:i]
                    break
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<!--.*?-->', '', content, flags=re.DOTALL)
        if content.strip():
            return content.strip()
    return ""

def fetch_detail(session, url):
    try:
        r = session.get(url, timeout=30)
        if r.status_code != 200: return None, None, None
        r.encoding = "utf-8"
        html = r.text
    except: return None, None, None
    content = extract_content(html)
    # Use meta ArticleTitle for clean title (no prefix, full chemical names)
    tm = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    if not tm:
        tm = re.search(r"<title>(.*?)<", html)
        if tm:
            title_text = tm.group(1).strip()
            # Remove common site name prefixes
            title_text = re.sub(r'^沾化区人民政府\s*通知公告\s*', '', title_text)
            title = title_text.strip()
        else:
            title = ""
    else:
        title = tm.group(1).strip()
    # Date: try multiple sources
    dm = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    if not dm:
        dm = re.search(r'<meta name="publishdate" content="([^"]*)"', html)
    if not dm:
        # Also try any date pattern
        dm = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    date_str = dm.group(1)[:10] if dm else ""
    summary = re.sub(r'<[^>]+>', '', content)[:200] if content else title
    summary = re.sub(r'\s+', ' ', summary).strip()
    return title, date_str, content

def parse_list_page(html):
    items = []
    for li in re.findall(r'<li>(.*?)</li>', html, re.DOTALL):
        m = re.search(r'<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})</span>', li, re.DOTALL)
        if m:
            url = m.group(1)
            if not url.startswith("http"): url = BASE + url
            items.append({"url": url, "title": m.group(2).strip(), "date": m.group(4).strip()})
    return items

def insert_item(conn, item):
    t = item["title"].replace("'", "''")
    u = item["url"].replace("'", "''")
    c = item.get("content", "").replace("'", "''")
    s = item.get("summary", "").replace("'", "''")
    d = item.get("date", "").replace("'", "''")
    dr = 0
    try: dr = int(datetime.strptime(d, "%Y-%m-%d").timestamp())
    except: dr = int(time.time())
    sql = f"INSERT OR REPLACE INTO gov_raw (site_name, page_url, title, publish_date, content, summary, date_rank, category) VALUES ('{NAME}','{u}','{t}','{d}','{c}','{s}',{dr},'政府公告')"
    try:
        conn.execute(sql)
        conn.commit()
        return True
    except Exception as e:
        print(f"    DB ERROR: {e}")
        return False

def main():
    print(f"\n{'='*50}")
    print(f"🚀 {NAME}")
    print(f"{'='*50}\n")
    session = requests.Session()
    session.headers.update(HEADERS)
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    
    r = session.get(f"{BASE}/col/col118071/index.html", timeout=20)
    r.encoding = "utf-8"
    items = parse_list_page(r.text)
    print(f"📋 首页: {len(items)} 条\n")
    
    total_new = 0
    for idx, item in enumerate(items):
        if item["date"] < CUTOFF_DATE:
            continue
        has = conn.execute("SELECT 1 FROM gov_raw WHERE page_url=? AND content IS NOT NULL AND content!=''", (item["url"],)).fetchone()
        if has:
            print(f"  [{idx+1}/{len(items)}] ⏭️ {item['title'][:35]}...")
            continue
        print(f"  [{idx+1}/{len(items)}] {item['title'][:35]}...", end=" ", flush=True)
        title, date, content = fetch_detail(session, item["url"])
        if title or content:
            item["content"] = content or ""
            item["summary"] = (re.sub(r'<[^>]+>', '', content)[:200] if content else title).strip()
            item["title"] = title or item["title"]
            item["date"] = date or item["date"]
            if insert_item(conn, item):
                total_new += 1
                print(f"✅ {len(content or ''):,}字")
            else:
                print(f"❌ DB")
        else:
            print(f"❌ 获取失败")
        time.sleep(0.5)
    
    print(f"\n{'='*50}")
    print(f"📊 新增 {total_new} 条")
    print(f"{'='*50}")
    conn.close()

if __name__ == "__main__":
    main()
