#!/usr/bin/env python3
import os
"""
crawl_dmxhbj.py — 大名县生态环境分局 (daming.gov.cn)
栏目: 公告公示(2181) | 生态环境(2191)
"""

import re, time, os
from datetime import datetime, timedelta
from urllib.parse import urljoin
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.daming.gov.cn/dmxxxgk/bmxx/zfbm/xhbj"
GOVSEARCH = "http://www.hd.gov.cn/govsearch"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
}
THREE_YEARS_AGO = datetime.now() - timedelta(days=3*365)
SITE_NAME = "大名县生态环境分局"

CATEGORIES = {
    "2181": "公告公示",
    "2191": "生态环境",
}

def get_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

def insert_article(conn, art):
    conn.execute(
        """INSERT OR IGNORE INTO gov_raw
           (site_name, title, page_url, publish_date, content)
           VALUES (?, ?, ?, ?, ?)""",
        (SITE_NAME, art["title"], art["url"], art["publish_date"], art["content"])
    )

def parse_date(date_str):
    """try: 2026年04月27日"""
    try:
        return datetime.strptime(date_str, "%Y年%m月%d日")
    except:
        try:
            ds = date_str.replace("年","-").replace("月","-").replace("日","")
            return datetime.strptime(ds, "%Y-%m-%d")
        except:
            return None

def get_cat_path(cat_id):
    """Get URL path for category"""
    if cat_id == "2191":
        return f"{BASE_URL}/2163/2186/{cat_id}"
    return f"{BASE_URL}/2163/{cat_id}"

def parse_list_page(cat_id, page_num):
    """Parse static list page"""
    cat_path = get_cat_path(cat_id)
    if page_num == 1:
        url = f"{cat_path}/index_10369.html"
    else:
        url = f"{cat_path}/index_10369_{page_num}.html"

    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
    except:
        return []
    if resp.status_code != 200:
        return []

    html = resp.text
    # Capture each row from <div class="row"> to </div> followed by <div id="Tip
    rows = re.findall(
        r'<div class="row">(.*?)</div>\s*<div id="Tip',
        html, re.DOTALL
    )

    items = []
    for row in rows:
        # Skip table header row (has OrderBy)
        if 'orderBy=' in row:
            continue

        m = re.search(r'<a href="([^"]+)"[^>]*>([^<]+)</a>', row)
        if not m:
            continue
        rel_url = m.group(1).strip()
        title = re.sub(r'\s+', ' ', m.group(2)).strip()
        if not title:
            continue

        d = re.search(r'class="fbrq"[^>]*title="([^"]+)"', row)
        if not d:
            continue

        dt = parse_date(d.group(1))
        if dt is None or dt < THREE_YEARS_AGO:
            continue

        full_url = urljoin(url, rel_url).split("?")[0]
        items.append({"url": full_url, "title": title, "date": dt.strftime("%Y-%m-%d")})

    return items

def parse_list_jsp(cat_id, page_num):
    """Parse JSP list page (11+)"""
    cat_path = get_cat_path(cat_id)
    puburl = f"{cat_path}/?id=2163/{cat_id}"
    url = f"{GOVSEARCH}/simp_govdmx_list.jsp?page={page_num}&pubURL={puburl}"

    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
    except:
        return []
    if resp.status_code != 200:
        return []

    html = resp.text
    if "没有找到记录" in html or "no_object_found" in html:
        return []

    rows = re.findall(r'<div class="row">(.*?)</div>', html, re.DOTALL)

    items = []
    for row in rows:
        if 'orderBy=' in row:
            continue
        if not re.search(r'fbrq', row):
            continue

        m = re.search(r'<a href="([^"]+)"[^>]*>([^<]+)</a>', row)
        if not m:
            continue
        full_url = m.group(1).strip().split("?")[0]
        title = re.sub(r'\s+', ' ', m.group(2)).strip()
        if not title:
            continue

        d = re.search(r'fbrq[^>]*>([^<]+)<', row)
        if not d:
            continue

        dt = parse_date(d.group(1).strip())
        if dt is None or dt < THREE_YEARS_AGO:
            continue

        items.append({"url": full_url, "title": title, "date": dt.strftime("%Y-%m-%d")})

    return items

def crawl_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=20)
        resp.encoding = "utf-8"
    except:
        return "", "", ""
    if resp.status_code != 200:
        return "", "", ""

    html = resp.text
    content = ""
    m = re.search(r'<div class="content">(.*?)</div>\s*</div>\s*<!--', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    if not content:
        m = re.search(r'<div class="content">(.*?)</div>', html, re.DOTALL)
        if m:
            content = m.group(1).strip()

    source = ""
    m = re.search(r'发布机构[：:]\s*</b>\s*([^<]+)', html)
    if m:
        source = m.group(1).strip()

    pub_date = ""
    m = re.search(r'发布日期[：:]\s*</b>\s*([^<]+)', html)
    if m:
        pub_date = m.group(1).strip()

    return content, source, pub_date

def main():
    conn = get_db()
    total_inserted = 0

    for cat_id, cat_label in CATEGORIES.items():
        print(f"\n{'='*40}")
        print(f"[{cat_label}] 开始爬取列表...")

        all_items = []
        # Static pages 1-20
        for page in range(1, 21):
            items = parse_list_page(cat_id, page)
            if not items:
                break
            all_items.extend(items)
            print(f"  page {page}: {len(items)} 条 (静态)")
            time.sleep(0.3)

        # JSP pages 11+
        for page in range(11, 100):
            items = parse_list_jsp(cat_id, page)
            if not items:
                break
            all_items.extend(items)
            print(f"  page {page}: {len(items)} 条 (JSP)")
            time.sleep(0.3)

        print(f"[{cat_label}] 列表合计: {len(all_items)} 条 (近3年)")
        if not all_items:
            continue

        cat_inserted = 0
        print(f"爬取详情 ({len(all_items)} 条)...")

        for i, item in enumerate(all_items):
            content, source, pub_date = crawl_detail(item["url"])
            if not content:
                continue

            art = {
                "title": item["title"],
                "url": item["url"],
                "publish_date": pub_date or item["date"],
                "source": source or SITE_NAME,
                "content": content,
            }
            insert_article(conn, art)
            cat_inserted += 1

            if i % 20 == 0:
                conn.commit()
            if i % 10 == 0:
                print(f"  [{i+1}/{len(all_items)}] {cat_inserted} 条 -- {item['title'][:30]}")
            time.sleep(0.3)

        conn.commit()
        total_inserted += cat_inserted
        print(f"[{cat_label}] 完成: {cat_inserted}/{len(all_items)} 条入库")

    conn.close()
    print(f"\n{'='*50}")
    print(f"大名县生态环境分局 全部完成!")
    print(f"总入库: {total_inserted} 条")
    print(f"{'='*50}")

if __name__ == "__main__":
    main()
