#!/usr/bin/env python3
"""
定远县-建设项目环境影响评价 (www.dingyuan.gov.cn)
============================================
CMS: 滁州市统一信息公开平台 (AJAX API)
列表API: /chuzhou/site/label/8888?labelName=publicInfoList&pageIndex=N
详情: /public/161054677/{id}.html, <div class="xxgk_contnet"> 含正文

用法:
    python3 crawl_dingyuan_hjbh.py          # 全量(最多4页)
    python3 crawl_dingyuan_hjbh.py --test   # 测试 5 条
"""

import re, sys, os, time, json
from datetime import datetime, timedelta, timezone
import requests, urllib3
import os
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
from bs4 import BeautifulSoup

SITE_NAME  = "定远县-建设项目环境影响评价"
API_URL    = "https://www.dingyuan.gov.cn/chuzhou/site/label/8888"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES  = 4  # 3yr cutoff within page 4
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "X-Requested-With": "XMLHttpRequest",
    "Referer": "https://www.dingyuan.gov.cn/public/column/161054677?type=4&catId=170070671&action=list",
}

API_PARAMS = {
    "labelName": "publicInfoList",
    "siteId": "2653861",
    "pageSize": "20",
    "pageIndex": "1",
    "action": "list",
    "isDate": "true",
    "dateFormat": "yyyy-MM-dd",
    "length": "80",
    "organId": "161054677",
    "catId": "170070671",
    "type": "4",
}

def fetch_list(page):
    params = dict(API_PARAMS)
    params["pageIndex"] = str(page)
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ! list page {page} fail: {str(e)[:60]}")
        return None

def parse_list(html):
    items = []
    for m in re.finditer(
        r'<li class="mc">\s*<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)</a>.*?<li class="rq">([^<]+)',
        html, re.DOTALL
    ):
        url = m.group(1)
        title = re.sub(r'\s+', ' ', m.group(2)).strip()
        pub_date = m.group(3).strip()
        if not url.startswith("http"):
            url = "https://www.dingyuan.gov.cn" + url if url.startswith("/") else "https://www.dingyuan.gov.cn/" + url
        items.append({"title": title, "url": url, "pub_date": pub_date})
    return items

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        html = r.text
        soup = BeautifulSoup(html, "html.parser")
        # 正文容器: gkwz_contnet (真正正文) > xxgkcontent > contentbox
        d = soup.find("div", class_=lambda c: c and isinstance(c, str) and "gkwz_contnet" in c)
        if not d:
            d = soup.find("div", class_=lambda c: c and isinstance(c, str) and "xxgkcontent" in c)
        if not d:
            d = soup.find("div", id="contentbox") or soup.find("div", class_="contentbox")
        if not d:
            # 兜底: xxgk_contnet 内去掉工具块
            d = soup.find("div", class_="xxgk_contnet")
            if not d:
                return ""
            for el in d.select(".xxgk_newtit, .wzewm, .newsinfo, .clear, .ls-article-menu, [class*=share], [class*=print], [class*=ewm], [class*=qrcode]"):
                el.decompose()
        # 删除脚本/样式/隐藏块
        for el in d.select("script, style, .xxgk_zclist, .wzewm, [class*=qrcode], [class*=share], .ls-article-menu"):
            el.decompose()
        # 附件/图片链接绝对化
        base = url[: url.rfind("/") + 1]
        for a in d.find_all("a", href=True):
            h = a["href"].strip()
            if h.startswith("/"):
                a["href"] = "https://www.dingyuan.gov.cn" + h
            elif h.startswith(("./", "../")) or (not h.startswith(("http", "javascript", "mailto", "#"))):
                from urllib.parse import urljoin
                a["href"] = urljoin(base, h)
        for img in d.find_all("img", src=True):
            s = img["src"].strip()
            if s.startswith("/"):
                img["src"] = "https://www.dingyuan.gov.cn" + s
        out = str(d)
        out = re.sub(r"<!--.*?-->", "", out, flags=re.S)
        out = re.sub(r"\s{2,}", " ", out)
        return out.strip()
    except Exception as e:
        print(f"  ! detail fail: {str(e)[:60]}")
        return ""

def to_db(items):
    if not items:
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR REPLACE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, tags, script_name) VALUES (?,?,?,?,?,?,?, 'crawl_dingyuan_hjbh.py')",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "建设项目环境影响评价",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
    #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
    db.commit()
    db.execute(
        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    db.commit()
    db.close()
    return ok, fail

def crawl(test=False):
    all_items = []
    seen_urls = set()
    for page in range(1, MAX_PAGES + 1):
        html = fetch_list(page)
        if not html:
            break
        items = parse_list(html)
        if not items:
            print(f"  [Page {page}] empty, reached end")
            break
        new = 0
        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])
            if item["pub_date"] and item["pub_date"] < CUTOFF:
                continue
            all_items.append(item)
            new += 1
        f = items[0]["pub_date"] if items else "?"
        l = items[-1]["pub_date"] if items else "?"
        print(f"  [Page {page}] {len(items)} items ({f} ~ {l}), new: {new}")
        if new == 0:
            break
    print(f"  Total: {len(all_items)} items within 3yr")
    if test:
        all_items = all_items[:5]
        print(f"  TEST mode: {len(all_items)} items")
    if not all_items:
        return 0, 0
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}... ", end="", flush=True)
        content = fetch_detail(item["url"])
        item["content"] = content
        print(f"{len(content)}B")
    ok, fail = to_db(all_items)
    return ok, fail

if __name__ == "__main__":
    test = "--test" in sys.argv
    mode = "TEST" if test else "FULL"
    print()
    print(f"[{SITE_NAME}] {mode}")
    t0 = time.time()
    ok, fail = crawl(test=test)
    print(f"  Time: {round(time.time()-t0, 1)}s")
    print(f"  New: {ok} Skip: {fail}")
