#!/usr/bin/env python3
"""
yanshougs.com (工程建设验收公示网) - 环评公示爬虫
==================================================
API: /publicity_list?id=4&page=N 获取列表
详情页获取正文内容

用法:
  python3 crawl_yanshougs.py                    # 增量爬（第1页，含正文）
  python3 crawl_yanshougs.py --pages=5          # 爬5页
  python3 crawl_yanshougs.py 1                  # 兼容旧配置(数字参数=页数)
  python3 crawl_yanshougs.py --full             # 全量爬所有页
  python3 crawl_yanshougs.py --backfill-body    # 补爬已有记录的缺失正文
  python3 crawl_yanshougs.py --pages=1 --no-body  # 仅标题/URL/日期，不爬正文
"""

import os, re, sys, time, json, sqlite3
import requests
from bs4 import BeautifulSoup

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")

SITE_NAME = "工程建设验收公示网"
DOMAIN = "https://www.yanshougs.com"
CATEGORY_ID = 4
CATEGORY_NAME = "环评公示"
GROUP = "全国"
INDUSTRY = "环评公示"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

API_BASE = "https://www.yanshougs.com/publicity_list?id=%d" % CATEGORY_ID
MAX_PAGES = 105
DEFAULT_PAGES = 1
stats = {"new": 0, "skip": 0, "errors": 0, "body_errors": 0}


def get_total_pages():
    try:
        r = requests.get("https://www.yanshougs.com/list/4.html", headers=HEADERS, timeout=10)
        m = re.search(r"list/4_(\d+)\.html", r.text)
        if m:
            return max(int(m.group(1)), 105)
    except:
        pass
    return MAX_PAGES


def fetch_list_page(page_num):
    url = "%s&page=%d" % (API_BASE, page_num)
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    return r.text


def parse_items(html):
    items = []
    soup = BeautifulSoup(html, "lxml")
    rows = soup.select("table tr")
    for row in rows[1:]:
        cells = row.find_all("td")
        if len(cells) < 4:
            continue
        a_tag = cells[1].find("a")
        if not a_tag:
            continue
        href = a_tag.get("href", "").strip()
        if not href:
            continue
        title = a_tag.get_text(strip=True)
        builder = cells[2].get_text(strip=True)
        date = cells[3].get_text(strip=True)
        full_url = href if href.startswith("http") else DOMAIN + href
        items.append((title, full_url, date, builder))
    return items


def fetch_body(page_url):
    """获取详情页正文HTML和摘要"""
    try:
        r = requests.get(page_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")
        content_div = soup.select_one("div.dir_c_content")
        body_html = ""
        summary = ""
        if content_div:
            for tag in content_div.find_all(["script", "style"]):
                tag.decompose()
            body_html = str(content_div)
            summary = content_div.get_text(strip=True)[:200]
        return body_html, summary
    except Exception as e:
        print("    body error: %s" % e, file=sys.stderr)
        return "", ""


def store_record(title, page_url, publish_date, body_html="", summary=""):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(title, page_url, source_url, site_name, publish_date, category, industry, group_name, content, summary) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (title.strip(), page_url, DOMAIN, SITE_NAME, publish_date,
             CATEGORY_NAME, INDUSTRY, GROUP, body_html, summary),
        )
        conn.commit()
        is_new = conn.total_changes > 0

        # Also update if existing but body empty
        if not is_new and body_html:
            row = conn.execute("SELECT id, content FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row and (not row[1] or row[1].strip() == ""):
                conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (body_html, summary, row[0]))
                conn.commit()
                is_new = True

        if is_new:
            row = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row:
                _c2 = sqlite3.connect(SEARCH_DB, timeout=60)
                try:
                    _c2.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name) VALUES (?,?,?)",
                                (row[0], title.strip(), SITE_NAME))
                    _c2.commit()
                except:
                    pass
                finally:
                    _c2.close()
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except Exception as e:
        stats["errors"] += 1
        print("  DB error: %s" % e, file=sys.stderr)
    finally:
        conn.close()


def backfill_body():
    """补爬已有记录的缺失正文"""
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    rows = conn.execute(
        "SELECT id, page_url, title FROM gov_raw WHERE site_name=? AND (content IS NULL OR content='')",
        (SITE_NAME,)
    ).fetchall()
    conn.close()
    print("Found %d records missing body" % len(rows))
    for i, (rid, url, title) in enumerate(rows):
        body, summary = fetch_body(url)
        if body:
            conn = sqlite3.connect(SEARCH_DB, timeout=60)
            conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (body, summary, rid))
            conn.commit()
            conn.close()
            print("  [%d/%d] OK: %s" % (i + 1, len(rows), title[:40]))
        else:
            print("  [%d/%d] FAIL: %s" % (i + 1, len(rows), title[:40]))
        time.sleep(0.3)
    print("\nBackfill done!")


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description="yanshougs ringping gongshi crawler")
    parser.add_argument("--full", action="store_true", help="crawl all pages")
    parser.add_argument("--pages", type=int, default=None, help="pages to crawl")
    parser.add_argument("--no-body", action="store_true", help="skip body (title/URL/date only)")
    parser.add_argument("--backfill-body", action="store_true", help="backfill missing body for existing records")
    parser.add_argument("n", nargs="?", type=int, default=None, help="legacy arg (page count)")
    args = parser.parse_args()

    os.chdir(BASE_DIR)

    if args.backfill_body:
        backfill_body()
        sys.exit(0)

    total_pages = get_total_pages()
    print("Total pages: %d" % total_pages)

    if args.full:
        pages_to_crawl = total_pages
    elif args.pages is not None:
        pages_to_crawl = args.pages
    elif args.n is not None:
        pages_to_crawl = args.n
    else:
        pages_to_crawl = DEFAULT_PAGES

    pages_to_crawl = min(pages_to_crawl, total_pages)
    print("Will crawl %d pages..." % pages_to_crawl)

    for p in range(1, pages_to_crawl + 1):
        try:
            html = fetch_list_page(p)
            items = parse_items(html)
            if not items:
                print("  Page %d: empty" % p)
                break
            print("  Page %d: %d items" % (p, len(items)))
            for title, url, date, builder in items:
                if args.no_body:
                    store_record(title, url, date)
                else:
                    body_html, summary = fetch_body(url)
                    store_record(title, url, date, body_html, summary)
                time.sleep(0.3)
        except Exception as e:
            print("  Error page %d: %s" % (p, e))

    print("\nDone! New:%d Skip:%d Errors:%d BodyErrors:%d" % (
        stats["new"], stats["skip"], stats["errors"], stats["body_errors"]))
