#!/usr/bin/env python3
"""望江县—建设项目环境影响评价审批爬虫
URL: https://www.wangjiang.gov.cn/grassroots/column/19635638?catId=1018541
CMS: 安庆政府网站群 (Ls.pagination)
"""

import requests, re, sqlite3, sys, os
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "望江县-环评审批"
BASE_URL = "https://www.wangjiang.gov.cn"
LIST_URL = "/grassroots/column/19635638?catId=1018541"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
}


def fetch_list():
    """Fetch the list page."""
    url = BASE_URL + LIST_URL
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    return r.text


def parse_list(html):
    """Parse list page to extract items.
    List: div.lm_main_lists > div.xxgk_nav_con > ul.doc_list > li.clearfix
    """
    soup = BeautifulSoup(html, "html.parser")
    items = []
    con = soup.find("div", class_="xxgk_nav_con")
    if not con:
        return items
    for li in con.find_all("li", class_="clearfix"):
        a = li.find("a", class_="title")
        date_span = li.find("span", class_="date")
        if a and date_span:
            href = a.get("href", "").strip()
            title = a.get("title", "") or a.get_text(strip=True)
            if not href or not title:
                continue
            date_str = date_span.get_text(strip=True)
            items.append({
                "title": title,
                "date": date_str,
                "url": href if href.startswith("http") else urljoin(BASE_URL, href),
            })
    return items


def fetch_detail(url):
    """Fetch detail page and extract content.
    Content: div.j-fontContent.newscontnet.minh500
    """
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
    except Exception:
        return {"content": "", "full_title": ""}

    # Title from meta
    full_title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"}) or soup.find("meta", attrs={"Name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        full_title = meta_title["content"].strip()

    # Content
    content = ""
    for cls in ["j-fontContent newscontnet minh500", "j-fontContent", "newscontnet"]:
        content_div = soup.find("div", class_=cls)
        if content_div:
            content = str(content_div)
            break
    if not content:
        content_div = soup.find("div", class_=lambda c: c and "j-fontContent" in c if c else False)
        if content_div:
            content = str(content_div)

    return {"content": content, "full_title": full_title}


def save_to_db(items):
    """Insert items into DB."""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for item in items:
        content_clean = re.sub(r'<[^>]+>', '', item["content"]).strip()
        c.execute("""INSERT OR IGNORE INTO gov_raw
            (title, page_url, site_name, summary, content, publish_date)
            VALUES (?, ?, ?, ?, ?, ?)""",
            (item["title"], item["url"], SITE_NAME,
             content_clean[:500], item["content"], item["date"]))
        if c.rowcount > 0:
            inserted += 1
    conn.commit()
    conn.close()
    return inserted


def main():
    is_incremental = len(sys.argv) >= 2
    cutoff_days = int(sys.argv[1]) if is_incremental else 9999
    print(f"=== {SITE_NAME} === (增量: {is_incremental}, cutoff={cutoff_days}天)")

    # Step 1-2: Fetch list
    html = fetch_list()
    all_items = parse_list(html)
    print(f"  列表: {len(all_items)} 条")
    if not all_items:
        print("  无数据")
        return

    # Step 3: Date filter
    now = datetime.now()
    cutoff = (now - timedelta(days=cutoff_days)).strftime("%Y-%m-%d") if is_incremental else CUTOFF_DATE
    filtered = [it for it in all_items if it["date"] >= cutoff]
    print(f"  日期过滤后: {len(filtered)} 条 (>= {cutoff})")

    if not filtered:
        print("  无新数据")
        return

    # Step 4: De-duplicate
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    existing = set()
    for item in filtered:
        c.execute("SELECT page_url FROM gov_raw WHERE page_url=?", (item["url"],))
        if c.fetchone():
            existing.add(item["url"])
    conn.close()

    to_fetch = [it for it in filtered if it["url"] not in existing]
    print(f"  去重后: {len(to_fetch)} 条")

    # Step 5: Fetch details
    success = []
    for item in to_fetch:
        detail = fetch_detail(item["url"])
        if detail["content"]:
            success.append({
                "title": detail["full_title"] or item["title"],
                "url": item["url"],
                "date": item["date"],
                "content": detail["content"],
            })
        else:
            print(f"  跳过无内容: {item['title'][:40]}...")

    # Step 6: Save
    inserted = save_to_db(success)
    print(f"  入库: {inserted} 条 (成功/总数: {len(success)}/{len(to_fetch)})")

    print(f"=== {SITE_NAME} 完成 ===")


if __name__ == "__main__":
    main()
