#!/usr/bin/env python3
"""
聊城市生态环境局-环保要闻 (sthjj.liaocheng.gov.cn)
==============================================
CMS: 政府门户 (layui 动态列表)
列表: /channel_t_189_13529/?page=N
详情: /channel_t_189_13529/doc_XXXXX.html
正文: div.content-text
JS 分页: ?page=N 参数可用

用法:
    python3 crawl_liaocheng.py          # 全量
    python3 crawl_liaocheng.py --test   # 测试 5 条
"""

import re, sys, os, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
import os
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

# ── 配置 ──
SITE_NAME  = "聊城市生态环境局-环保要闻"
BASE_URL   = "http://sthjj.liaocheng.gov.cn"
LIST_URL   = "http://sthjj.liaocheng.gov.cn/channel_t_189_13529/"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES  = 5
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  \u26a0 fetch fail: {e}")
        return None

def parse_list(html):
    """\u63d0\u53d6\u5217\u8868\u6761\u76ee (url, title, pub_date)"""
    items = []
    for m in re.finditer(r'<div class="item">(.*?)</div>', html, re.DOTALL):
        item = m.group(1)
        link = re.search(r'href="([^"]+)"[^>]*title="([^"]*)"', item)
        if not link:
            link = re.search(r'href="([^"]+)"[^>]*>([^<]+)', item)
        date = re.search(r'<span class="time">(\d{4}-\d{2}-\d{2})</span>', item)
        if link:
            href = link.group(1).strip()
            title = link.group(2).strip() if len(link.groups()) >= 2 else ""
            if title and len(title) > 5:
                items.append({
                    "title": title,
                    "url": href if href.startswith('http') else BASE_URL + href,
                    "pub_date": date.group(1) if date else ""
                })
    return items

def fetch_detail(url):
    """\u83b7\u53d6\u8be6\u60c5\u9875 title, content, pub_date"""
    html = fetch(url)
    if not html:
        return {"title": "", "content": "", "pub_date": ""}

    # Title
    title = ""
    m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]*)"', html, re.I)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>([^<]+)', html)
        if m:
            title = re.sub(r'[-_—|].*', '', m.group(1)).strip()

    # Date
    pub_date = ""
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="([^"]+)', html, re.I)
    if m:
        pub_date = m.group(1)[:10]
    if not pub_date:
        m = re.search(r'\u65f6\u95f4\uff1a\s*(\d{4}-\d{2}-\d{2})', html)
        if m:
            pub_date = m.group(1)

    # Content
    content = ""
    m = re.search(r'class="content-text"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1)
    if content:
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r' style="[^"]*"', '', content)
        content = content.strip()

    return {"title": title, "content": content, "pub_date": pub_date}

def to_db(items):
    """\u6279\u91cf\u5199\u5165 search.db"""
    if not items:
        print("  \u23ad \u65e0\u6570\u636e")
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, title, page_url, content, publish_date, summary, tags) "
                "VALUES (?,?,?,?,?,?,?)",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "\u73af\u4fdd\u8981\u95fb",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    # FTS sync
    db.execute(
        "INSERT INTO gov_search(rowid, title, site_name, summary, content) "
        "SELECT r.id, r.title, r.site_name, r.summary, r.content "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    db.commit()
    db.close()
    print(f"  \U0001f4be \u5165\u5e93: \u65b0\u589e{ok}, \u8df3\u8fc7{fail}, FTS\u5df2\u540c\u6b65")
    return ok, fail

def crawl(test=False):
    """\u5168\u91cf\u722c\u53d6"""
    all_items = []
    seen_urls = set()

    for page in range(1, MAX_PAGES + 1):
        url = f"{LIST_URL}?page={page}" if page > 1 else LIST_URL
        html = fetch(url)
        if not html:
            continue
        items = parse_list(html)
        if not items:
            print(f"  \u2139 \u7b2c{page}\u9875\u65e0\u6570\u636e\uff0c\u5230\u8fbe\u672b\u9875")
            break
        # Filter by 3 years and dedup
        new = 0
        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])
            if item["pub_date"] and item["pub_date"] < CUTOFF:
                continue
            all_items.append(item)
            new += 1
        first_date = items[0].get("pub_date", "?")
        last_date = items[-1].get("pub_date", "?")
        print(f"  \U0001f4c4 \u7b2c{page}\u9875: {len(items)}\u6761 ({first_date} ~ {last_date}), \u65b0\u589e{new}")

    if test:
        all_items = all_items[:5]
        print(f"  \U0001f9ea \u6d4b\u8bd5\u6a21\u5f0f: \u53ea\u5904\u7406\u524d {len(all_items)} \u6761")

    print(f"\n  \U0001f4ca \u5217\u8868\u6c47\u603b: {len(all_items)} \u6761\uff08\u8fd13\u5e74\uff09")
    if not all_items:
        return 0, 0

    # Crawl details
    results = []
    for i, item in enumerate(all_items, 1):
        detail = fetch_detail(item["url"])
        entry = {
            "title": detail["title"] or item["title"],
            "url": item["url"],
            "content": detail["content"],
            "pub_date": detail["pub_date"] or item.get("pub_date", ""),
        }
        results.append(entry)
        print(f"  [{i}/{len(all_items)}] {entry['pub_date']} {entry['title'][:40]}... ({len(detail['content'])}B)")
        time.sleep(0.3)

    ok, fail = to_db(results)
    return ok, fail

if __name__ == "__main__":
    test = "--test" in sys.argv
    print(f"\n\U0001f4e1 [{SITE_NAME}] {'\u6d4b\u8bd5\u6a21\u5f0f' if test else '\u5168\u91cf'} (\u6700\u591a{MAX_PAGES}\u9875, \u8fd13\u5e74)")
    t0 = time.time()
    ok, fail = crawl(test=test)
    print(f"  \u23f1 \u8017\u65f6: {time.time()-t0:.1f}s")
    print(f"  \u2705 \u65b0\u589e: {ok}  \u274c \u8df3\u8fc7: {fail}")
