#!/usr/bin/env python3
"""Crawl gzfyht.com - 贵州逢源恒通 项目公示"""
import re, sys, os, json, time, ssl, gzip
import urllib.request, urllib.error
from bs4 import BeautifulSoup
import sqlite3

ssl._create_default_https_context = ssl._create_unverified_context
ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

BASE_URL = "https://www.gzfyht.com"
SITE_NAME = "贵州逢源恒通-项目公示"
GROUP_NAME = "企业环评"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break


def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30, context=ctx)
    raw = resp.read()
    if raw[:2] == b'\x1f\x8b':
        raw = gzip.decompress(raw)
    return raw.decode("utf-8", errors="replace")


def extract_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    link_map = {}
    for a in soup.find_all("a", href=True):
        h = a["href"]
        if "/nd.jsp" not in h or "fromColId=105" not in h:
            continue
        t = a.get_text(strip=True)
        if not t or len(t) < 5:
            continue
        if re.match(r"^\d{4}-\d{2}-\d{2}$", t):
            continue
        if not h.startswith("http"):
            h = BASE_URL + h if h.startswith("/") else BASE_URL + "/" + h
        if h not in link_map or len(t) > len(link_map[h]["title"]):
            link_map[h] = {"title": t, "url": h}
    items = list(link_map.values())
    return items


def parse_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    h1 = soup.select_one("h1")
    if h1:
        title = h1.get_text(strip=True)

    date_text = ""
    for m in re.finditer(r"(\d{4}-\d{2}-\d{2})", html):
        date_text = m.group(1)
        break

    content = ""
    attachments = []
    content_div = soup.select_one(".richContent")
    if not content_div:
        for sel in ["[class*=content]", "[class*=article]", ".text", "#content"]:
            el = soup.select_one(sel)
            if el and len(el.get_text(" ", strip=True)) > 50:
                content_div = el
                break
    if not content_div:
        return {"title": title, "date": date_text, "content": "", "attachments": []}

    parts = []
    # ⚠️ 页面布局用外层<table>，find_parent("table") 会误匹配
    # 只跳过 content_div 内部的表格
    content_tables = content_div.find_all("table")

    for el in content_div.find_all(["p", "table", "img"], recursive=True):
        # 跳过在 content_div 内部表格中的 <p>
        if el.name == "p":
            in_content_table = False
            for ct in content_tables:
                if el in ct.find_all(recursive=False) or el in ct.descendants:
                    in_content_table = True
                    break
            if in_content_table:
                continue

        # 跳过内部表格中的子表格
        if el.name == "table":
            in_content_table = False
            for ct in content_tables:
                if ct != el and el in ct.descendants:
                    in_content_table = True
                    break
            if in_content_table:
                continue

        if el.name == "img":
            src = el.get("src", "")
            if "icon_" in src.lower():
                continue
            if src.startswith("//"):
                src = "https:" + src
            elif src.startswith("/"):
                src = BASE_URL + src
            elif not src.startswith("http"):
                src = url.rsplit("/", 1)[0] + "/" + src
            alt = el.get("alt", "") or ""
            parts.append(f"![{alt}]({src})")
            continue

        text = el.get_text(" ", strip=True)
        if text:
            parts.append(text)

    content = "\n\n".join(p for p in parts if p.strip())
    content = content.replace("\u00a0", " ").replace("&nbsp;", " ")
    return {"title": title, "date": date_text, "content": content, "attachments": []}


def main():
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    c.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
    c.execute("DELETE FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    conn.commit()
    print("[清理] 已清理该站旧数据", flush=True)

    all_items = []
    html = fetch(f"{BASE_URL}/col.jsp?id=105")
    if not html:
        print("[ERROR] Cannot fetch list page", file=sys.stderr)
        sys.exit(1)
    items = extract_items(html)
    print(f"[INFO] List page: {len(items)} items", flush=True)
    all_items.extend(items)

    try:
        html2 = fetch(f"{BASE_URL}/col.jsp?id=105&pageNo=2")
        items2 = extract_items(html2)
        if items2:
            existing_urls = {i["url"] for i in all_items}
            new = [i for i in items2 if i["url"] not in existing_urls]
            if new:
                print(f"[INFO] Page 2: {len(new)} new items", flush=True)
                all_items.extend(new)
    except:
        pass

    print(f"[INFO] Total items: {len(all_items)}", flush=True)
    new_count = skip_count = empty_count = 0

    for idx, item in enumerate(all_items):
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
        if c.fetchone():
            skip_count += 1
            continue
        if idx % 5 == 0:
            print(f"[INFO] [{idx}/{len(all_items)}] {item['title'][:40]}...", file=sys.stderr, end=" ", flush=True)
        html = fetch(item["url"])
        if not html:
            print("FETCH FAILED", file=sys.stderr)
            continue
        detail = parse_detail(html, item["url"])
        title = detail["title"] or item["title"]
        date_text = detail["date"]
        content = detail["content"]
        if not content.strip():
            empty_count += 1
            if idx % 5 == 0:
                print("EMPTY", file=sys.stderr)
            continue
        plain = re.sub(r'\s+', ' ', content).strip()
        summary = plain[:200] if plain else title[:200]
        date_rank = 0
        if date_text:
            m2 = re.search(r"(\d{4})-(\d{2})-(\d{2})", date_text)
            if m2:
                date_rank = int(m2.group(1) + m2.group(2) + m2.group(3))
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank) VALUES (?,?,?,?,?,?,?,?,?)",
                (title, SITE_NAME, GROUP_NAME, item["url"], date_text, content, summary, "[]", date_rank)
            )
            if c.rowcount > 0:
                new_count += 1
        except Exception as e:
            if idx % 5 == 0:
                print(f"DB ERROR: {e}", file=sys.stderr)
        time.sleep(0.3)

    conn.commit()
    conn.close()
    print(f"\n[RESULT] {SITE_NAME}: {new_count} new, {skip_count} skip, {empty_count} empty", flush=True)


if __name__ == "__main__":
    main()
