#!/usr/bin/env python3
"""crawl_zybz.py - 播州区生态环境信息公开爬虫 (TRS CMS)"""

import sys, re, time, hashlib, urllib.parse, sqlite3, os, argparse

BASE_URL = "https://www.zybz.gov.cn/zfbm/sthjj/zfxxgk_5668393/fdzdgknr_5668396/hjbh_5668401/"
TOTAL_PAGES = 21
DB_PATH = "/root/search.db"
SITE_NAME = "zybz"
SITE_TITLE = "播州区生态环境"
GROUP_NAME = "生态环境"
DELAY = 0.3

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

def fetch(url, retries=3):
    import requests
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, verify=False, timeout=20)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERR] {e}", flush=True)
                return None

def extract_list(html):
    items = []
    for m in re.finditer(r'<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<b>([^<]+)</b>', html, re.DOTALL):
        href, title, date = m.groups()
        if href.startswith("http"):
            url = href
        elif href.startswith("/"):
            url = "https://www.zybz.gov.cn" + href
        else:
            url = urllib.parse.urljoin(BASE_URL, href)
        items.append((title.strip(), url, date.strip()))
    return items

def get_detail(html):
    for pat in [r'class="trs_editor_view[^"]*"\s*>', r'class="TRS_Editor[^"]*"\s*>']:
        m = re.search(pat, html)
        if m:
            start = m.end()
            depth = 1
            pos = start
            while depth > 0 and pos < len(html):
                if html[pos:pos+5] in ("<div ", "<div\t"):
                    depth += 1
                    pos += 1
                elif html[pos:pos+6] == "</div>":
                    depth -= 1
                    pos += 1
                else:
                    pos += 1
            content_html = html[start:pos]
            text = clean(content_html)
            if text and len(text) > 50:
                return text
    return None

def clean(html):
    for t in ("script", "style"):
        html = re.sub(f"<{t}[^>]*>.*?</{t}>", "", html, flags=re.DOTALL)
    html = re.sub(r"<br\s*/?>", "\n", html)
    html = re.sub(r"</p>", "\n", html)
    html = re.sub(r"</div>", "\n", html)
    html = re.sub(r"</tr>", "\n", html)
    html = re.sub(r"</li>", "\n", html)
    text = re.sub(r"<[^>]+>", "", html)
    for o, n in (("&nbsp;", " "), ("&amp;", "&"), ("&lt;", "<"), ("&gt;", ">"), ("&quot;", '"'), ("\r\n", "\n")):
        text = text.replace(o, n)
    return "\n".join(l.strip() for l in text.split("\n") if l.strip())

def page_url(n):
    return BASE_URL if n == 1 else BASE_URL + f"index_{n-1}.html"

def main():
    p = argparse.ArgumentParser()
    p.add_argument("--pages", type=int, default=None)
    args = p.parse_args()
    n = min(args.pages or TOTAL_PAGES, TOTAL_PAGES)
    import urllib3
    urllib3.disable_warnings()
    conn = sqlite3.connect(DB_PATH)
    new = skip = err = 0
    print(f"Starting: {SITE_TITLE} ({n}/{TOTAL_PAGES} pages)", flush=True)
    for pn in range(1, n + 1):
        u = page_url(pn)
        print(f"\nPage {pn}/{n}: {u}", flush=True)
        html = fetch(u)
        if not html:
            err += 1
            continue
        items = extract_list(html)
        if not items:
            print("  No items found", flush=True)
            continue
        print(f"  {len(items)} items", flush=True)
        for title, url, date in items:
            existing = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone()
            if existing:
                skip += 1
                continue
            dh = fetch(url)
            c = get_detail(dh) if dh else None
            if not c or len(c) < 50:
                skip += 1
                continue
            summary = title[:200]
            conn.execute("INSERT INTO gov_raw (title, summary, content, page_url, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                         (title, summary, c[:3000], url, date, SITE_NAME))
            conn.execute("INSERT INTO gov_search_v3 (title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                         (title, c[:3000], url, date, SITE_NAME))
            new += 1
            conn.commit()
            time.sleep(DELAY)
        time.sleep(DELAY * 2)
    conn.close()
    print(f"\n=== Complete: new={new}, skip={skip}, err={err} ===", flush=True)

if __name__ == "__main__":
    main()
