#!/usr/bin/env python3
import os
# -*- coding: utf-8 -*-
"""
工程建设验收公示网 - 通用多栏目 (水保验收 id=2, 环评公示 id=4, 环保验收 id=3 等)
Usage:
  python3 crawl_yanshougs_sbys.py                     # 水保验收(id=2), 日跑5页
  python3 crawl_yanshougs_sbys.py --id 4 --name yanshougs_hpgs   # 环评公示
  python3 crawl_yanshougs_sbys.py --full              # 全量爬全部页
"""
import sys, os, re, time, json, requests
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DOMAIN = "https://www.yanshougs.com"
DEFAULT_ID = "2"
DEFAULT_NAME = "yanshougs_sbys"
DEFAULT_PAGES = 5
MAX_TOTAL_PAGES = 1473

def extract_table_tags(html_str):
    tbl_soup = BeautifulSoup(html_str, "html.parser")
    tables = tbl_soup.find_all("table")
    replacements = []
    for i, tb in enumerate(tables):
        key = "___TBL_%d___" % i
        replacements.append((key, str(tb)))
        tb.replace_with(key)
    return str(tbl_soup), replacements

def is_attachment_link(href, text=""):
    if re.search(r"\.(docx?|pdf|pptx?|xlsx?|jpg|png|gif)(\$|\?)", href, re.IGNORECASE):
        return True
    if "/download/" in href:
        return True
    if re.search(r"\.(pdf|doc|docx|xlsx?|pptx?)", text, re.IGNORECASE):
        return True
    return False

def fetch(url, encoding="utf-8"):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = encoding
        return r.text
    except Exception as e:
        print("  FETCH ERROR %s: %s" % (url, e), file=sys.stderr)
        return None

def parse_list_items(html):
    items = []
    soup = BeautifulSoup(html, "lxml")
    rows = soup.select("table tr")
    for row in rows[1:]:
        cells = row.find_all("td")
        if len(cells) >= 3:
            a2 = cells[1].find("a")
            if not a2:
                continue
            href = a2["href"].strip() if a2.get("href") else ""
            if not href:
                continue
            title = a2.get_text(strip=True)
            if not title or len(title) < 5:
                title = cells[1].get_text(strip=True)
            date = cells[3].get_text(strip=True) if len(cells) >= 4 else cells[2].get_text(strip=True)
            full_url = href if href.startswith("http") else DOMAIN + href
            items.append({"url": full_url, "title": title, "date": date})
    return items

def get_content(url, title_from_list):
    html = fetch(url)
    if not html:
        return "", "", [], title_from_list
    soup = BeautifulSoup(html, "lxml")
    full_title = title_from_list
    dt = soup.select_one("div.dir_c_title")
    if dt:
        t = dt.get_text(strip=True)
        if t:
            full_title = t
    pub_date = ""
    di = soup.select_one("div.dir_c_time")
    if di:
        m = re.search(r"(\d{4}[-/]\d{2}[-/]\d{2})", di.get_text())
        if m:
            pub_date = m.group(1).replace("/", "-")
    if not pub_date:
        m = re.search(r"(\d{4}[-/]\d{2}[-/]\d{2})", html)
        if m:
            pub_date = m.group(1).replace("/", "-")
    body = soup.select_one("div.dir_c_content")
    content_html = ""
    attachments = []
    if body:
        for a_tag in body.find_all("a", href=True):
            h = a_tag["href"].strip()
            txt = a_tag.get_text(strip=True)
            if is_attachment_link(h, txt):
                fname = txt or h.split("/")[-1]
                attach_url = h if h.startswith("http") else DOMAIN + h
                attachments.append({"url": attach_url, "name": fname})
                a_tag.replace_with("[%s](%s)" % (fname, attach_url))
        body_html = str(body)
        body_html, table_markers = extract_table_tags(body_html)
        body_soup = BeautifulSoup(body_html, "html.parser")
        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")
        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()
        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, "\n\n" + tbl_html + "\n\n")
    if not content_html or len(content_html.strip()) < 20:
        content_html = '<p><a href="%s">%s</a></p>' % (url, full_title)
    return content_html, pub_date, attachments, full_title

def main():
    import argparse
    parser = argparse.ArgumentParser(description="艺工爇中语注版爱渐")
    parser.add_argument("--id", default=DEFAULT_ID, help="栏目ID (默认2=水保验收)")
    parser.add_argument("--name", default=DEFAULT_NAME, help="系织名称")
    parser.add_argument("--pages", type=int, default=None, help="爱变页東 (默认5)")
    parser.add_argument("--full", action="store_true", help="全量爱取全部页")
    # Handle bare numeric arg for _MAX_PG
    import sys as _SYS2
    _MAX_PG = int(_SYS2.argv[-1]) if len(_SYS2.argv) > 1 and _SYS2.argv[-1].isdigit() else None
    if _MAX_PG is not None:
        print('[AutoPg] max_pages=' + str(_MAX_PG))
        _SYS2.argv.pop()
    args = parser.parse_args()

    col_id = args.id
    site_name = args.name
    api_base = "https://www.yanshougs.com/publicity_list?id=%s" % col_id

    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (site_name,))
    existing = c.fetchone()[0]

    if args.full:
        max_pages = MAX_TOTAL_PAGES
    elif args.pages is not None:
        max_pages = args.pages
    else:
        max_pages = DEFAULT_PAGES

    print("栏目ID=%s 等皹=%s: 已有 %d 条, 爱取 %d 页" % (col_id, site_name, existing, max_pages), file=sys.stderr)

    all_new = 0
    for pg in range(1, max_pages + 1):
        page_url = "%s&page=%d" % (api_base, pg)
        html = fetch(page_url)
        if not html:
            print("  第%d页 获取失败" % pg, file=sys.stderr)
            continue
        items = parse_list_items(html)
        if not items:
            print("  第%d页 无条集存" % pg, file=sys.stderr)
            break
        print("  第%d页: %d 条" % (pg, len(items)), file=sys.stderr)
        for it in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (it["url"], site_name))
            if c.fetchone():
                continue
            content, pub_date, attachments, full_title = get_content(it["url"], it["title"])
            date = it["date"] or pub_date
            try:
                c.execute(
                    """INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, source_url, status, attachments)
                       VALUES (?,?,?,?,?,?,?,?)""",
                    (site_name, it["url"], full_title[:500], content, date[:20],
                     it["url"], "synced", json.dumps(attachments, ensure_ascii=False))
                )
                all_new += 1
                print("    + %s (%s) body:%dB attach:%d" % (full_title[:50], date, len(content), len(attachments)), file=sys.stderr)
            except Exception as e:
                print("    ! 入库失败: %s" % e, file=sys.stderr)
        time.sleep(0.3)

    conn.commit()
    print("\n完戔! 新增 %d 条" % all_new, file=sys.stderr)



    conn.close()

if __name__ == "__main__":
    main()
