#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_ysia_tzgg.py - 榆神工业区(榆林经济技术开发区)-通知公告
URL: https://ysia.yl.gov.cn/xwzx/tzgg/
CMS: 重构后的WCM, index_N.html分页, 15条/页
列表: /xwzx/tzgg/index.html (page1), index_N.html (page2+)
详情: /xwzx/tzgg/YYYYMM/tYYYYMMDD_id.html
标题: <h1 class="tit">
日期: xy-msg span (时间：YYYY-MM-DD HH:MM:SS)
正文: div.details#article > div.TRS_UEDITOR > p
"""
import requests, re, sys, os, time, json
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "https://ysia.yl.gov.cn"
LIST_BASE = "/xwzx/tzgg"
SITE_NAME = "榆神工业区(榆林经济技术开发区)-通知公告"
GROUP = "陕西"
INDUSTRY = "环境公示"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
SESSION = requests.Session()


def fetch(url):
    try:
        r = SESSION.get(url, headers=HEADERS, timeout=60)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("ERR:%s" % e)
        return None


def parse_list(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.find_all("a", href=True):
        h = a["href"]
        t = a.get_text(strip=True)
        if not t or len(t) < 6:
            continue
        if h.startswith("./") and h.endswith(".html") and "/t2" in h:
            full = BASE_URL + LIST_BASE + "/" + h[2:]
            items.append((full, t))
    return items


def parse_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    # Title from h1.tit
    title = ""
    h1 = soup.find("h1", class_=lambda c: c and isinstance(c, str) and "tit" in c)
    if h1:
        title = h1.get_text(strip=True)

    # Date from xy-msg
    date_str = ""
    msg = soup.find("div", class_=lambda c: c and isinstance(c, str) and "xy-msg" in c)
    if msg:
        txt = msg.get_text(strip=True)
        m = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
        if m:
            date_str = m.group(1)

    # Content from div.details#article > TRS_UEDITOR
    content = ""
    detail_div = soup.find("div", class_=lambda c: c and isinstance(c, str) and "details" in c)
    if not detail_div:
        detail_div = soup.find(id="article")

    if detail_div:
        ue = detail_div.find(class_=lambda c: c and isinstance(c, str) and "TRS_UEDITOR" in c)
        if ue:
            parts = []
            for p in ue.find_all("p"):
                txt = p.get_text(strip=True)
                if txt:
                    parts.append(txt)
            content = "\n".join(parts)

    return title, date_str, content


def push_to_db(items):
    """Insert into search.db via subprocess sqlite3"""
    import subprocess
    total = 0
    for url, title, pub_date, content, site_name, script_name, industry in items:
        sql = "INSERT OR IGNORE INTO gov_raw (page_url, title, publish_date, content, site_name, script_name, industry) VALUES ('%s', '%s', '%s', '%s', '%s', '%s', '%s');" % (
            url.replace("'", "''"),
            (title or "").replace("'", "''"),
            (pub_date or "").replace("'", "''"),
            (content or "").replace("'", "''"),
            (site_name or "").replace("'", "''"),
            (script_name or "").replace("'", "''"),
            (industry or "").replace("'", "''"),
        )
        r = subprocess.run(
            ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql],
            capture_output=True, text=True, timeout=10
        )
        if r.returncode == 0:
            total += 1
    return total


def main():
    pages = MAX_PAGES
    if len(sys.argv) > 1:
        try:
            pages = int(sys.argv[1].replace("--pages=", ""))
        except:
            pass

    script_name = os.path.basename(__file__)
    all_items = []

    for page in range(pages):
        if page == 0:
            url = BASE_URL + LIST_BASE + "/index.html"
        else:
            url = BASE_URL + LIST_BASE + "/index_%d.html" % page

        print("[%s] %s..." % (str(page + 1), url), end=" ")
        html = fetch(url)
        if not html:
            print("FAIL")
            continue

        links = parse_list(html)
        print("%d links" % len(links))

        for link_url, link_title in links:
            print("  %s" % link_title[:40], end="... ")
            detail_html = fetch(link_url)
            if not detail_html:
                print("ERR")
                continue
            title, pub_date, content = parse_detail(detail_html, link_url)
            if not title:
                title = link_title
            print("OK (%d字)" % len(content))
            all_items.append((link_url, title, pub_date, content, SITE_NAME, script_name, INDUSTRY))
            time.sleep(0.5)

    if not all_items:
        print("未获取到任何数据")
        return

    # Insert
    total = 0
    for item in all_items:
        total += push_to_db([item])

    print("\n入库: %d/%d 条" % (total, len(all_items)))

    # FTS sync
    if total > 0:
        import subprocess
        ids = []
        for url, _, _, _, _, _, _ in all_items[:total]:
            r = subprocess.run(
                ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, "SELECT rowid FROM gov_raw WHERE page_url='%s';" % url.replace("'", "''")],
                capture_output=True, text=True, timeout=10
            )
            if r.stdout.strip():
                ids.append(r.stdout.strip())
        if ids:
            for rid in ids:
                sql = "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) SELECT %s, title, site_name, substr(content, 1, 300) FROM gov_raw WHERE rowid=%s;" % (rid, rid)
                subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, timeout=10)
            print("FTS同步: %d条" % len(ids))

    print("\n===== 完成 =====")
    print("新增入库: %d 条" % total)
    print("站点: %s" % SITE_NAME)


if __name__ == "__main__":
    main()
