#!/usr/bin/env python3
"""太仓环保公众网-环评公示爬虫
站点: http://www.tchbgz.com/
栏目: 环评公示 (?hpgs/)
分页: /?hpgs_N/ (N=2..11), 15条/页, 共11页约165条
CMS: 自定义企业CMS
"""

import os, re, sys, time, json, subprocess
from bs4 import BeautifulSoup
import requests

DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
BASE_URL = "http://www.tchbgz.com"
MAX_PAGES = 11
SITE_NAME = "太仓环保公众网-环评公示"
INDUSTRY = "环评公示"
GROUP = "企业"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def fetch(url, retries=3):
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url}: {e}", file=sys.stderr)
                return None


def parse_list(html):
    """解析列表页，返回 [ {page_url, title, date}, ... ]"""
    items = []
    soup = BeautifulSoup(html, "html.parser")

    for li in soup.select("li.pagearticlelist"):
        left = li.select_one("div.pagelistleft a")
        right = li.select_one("div.pagelistright")
        if not left:
            continue
        href = left.get("href", "")
        if not href or "hpgs" not in href:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href

        title = left.get("title", "") or left.get_text(strip=True)
        if not title:
            continue

        date_str = right.get_text(strip=True) if right else ""
        # Normalize date
        dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", date_str)
        publish_date = dm.group(1) if dm else ""

        items.append({
            "page_url": href,
            "title": title.strip(),
            "date": publish_date,
        })
    return items


def parse_detail(html, url):
    """解析详情页，返回 (title, publish_date, content_text, content_html)"""
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    publish_date = ""
    content = ""
    content_html = ""

    # Title from <title> tag
    title_tag = soup.find("title")
    if title_tag:
      t = title_tag.get_text(strip=True)
      t = t.split("-环评公示")[0].split("-")[0].strip()
      if t:
        title = t

    # Publish date from detail page
    body_text = soup.get_text()
    dm = re.search(r"发布时间[：:]?\s*(\d{4}-\d{1,2}-\d{1,2})", body_text)
    if dm:
      publish_date = dm.group(1)

    # Content from articleshow div
    content_div = soup.select_one("div.articleshow")
    if not content_div:
      content_div = soup.select_one("div.a-content-pagearticle")

    if content_div:
      # Extract tables as HTML
      tables = content_div.find_all("table")
      table_htmls = []
      for t in tables:
        t.extract()  # remove from tree
        table_htmls.append(str(t))

      # Extract <p> text from remaining content
      paragraphs = []
      for p in content_div.find_all("p"):
        # Skip empty paragraphs
        text = p.get_text(strip=True)
        if text:
          paragraphs.append(text)

      # Attach embed when content is short
      if not paragraphs and not table_htmls:
        # Fallback: whole text
        content = content_div.get_text(strip=True)
      else:
        content = "\n\n".join(paragraphs)
        if table_htmls:
          content += "\n\n" + "\n".join(table_htmls)

      content_html = str(content_div)

    return title, publish_date, content, content_html


def crawl():
    max_pages_str = sys.argv[1] if len(sys.argv) > 1 else str(MAX_PAGES)
    try:
        max_pages = int(max_pages_str)
    except ValueError:
        max_pages = MAX_PAGES

    print(f"[{SITE_NAME}] Starting crawl, max_pages={max_pages}", flush=True)
    all_items = []

    # Page 1
    url = f"{BASE_URL}/?hpgs/"
    html = fetch(url)
    if not html:
        print("[ERROR] Cannot fetch page 1", file=sys.stderr)
        return
    items = parse_list(html)
    print(f"  Page 1: {len(items)} items", flush=True)
    all_items.extend(items)

    # Pages 2+
    for page in range(2, max_pages + 1):
        url = f"{BASE_URL}/?hpgs_{page}/"
        html = fetch(url)
        if not html:
            break
        items = parse_list(html)
        if not items:
            break
        print(f"  Page {page}: {len(items)} items", flush=True)
        all_items.extend(items)
        if len(items) < 15:
            break  # last page

    print(f"Total list items: {len(all_items)}", flush=True)

    # Fetch details
    new_count = 0
    dup_count = 0
    error_count = 0

    for item in all_items:
        page_url = item["page_url"]
        print(f"  Fetching: {item['title'][:40]}...", flush=True)

        detail_html = fetch(page_url)
        if not detail_html:
            error_count += 1
            continue

        title, publish_date, content, content_html = parse_detail(detail_html, page_url)

        # Use list date as fallback
        if not publish_date:
            publish_date = item.get("date", "")
        if not title:
            title = item.get("title", "")
        if not title:
            error_count += 1
            continue

        # Save to DB
        try:
            result = save_to_db(
                page_url=page_url,
                title=title,
                publish_date=publish_date,
                content=content,
                content_html=content_html,
            )
            if result == "new":
                new_count += 1
            elif result == "dup":
                dup_count += 1
            else:
                error_count += 1
        except Exception as e:
            print(f"  [ERROR] DB insert: {e}", file=sys.stderr)
            error_count += 1

    print(f"\n=== {SITE_NAME} Done ===", flush=True)
    print(f"New: {new_count}, Dup: {dup_count}, Error: {error_count}", flush=True)


def save_to_db(page_url, title, publish_date, content, content_html):
    """插入到search.db"""
    # Check if URL already exists
    check_sql = f"SELECT rowid FROM gov_raw WHERE page_url = '{page_url.replace(chr(39), chr(39)+chr(39))}'"
    result = subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, check_sql],
        capture_output=True, text=True, timeout=10,
    )
    if result.stdout.strip():
        return "dup"

    # Escape single quotes
    def esc(s):
        return s.replace("'", "''") if s else ""

    # Clean content - remove excessive whitespace
    if content:
        content = re.sub(r"\s{3,}", "\n\n", content.strip())
    if not content:
        content = ""

    summary = content[:200] if content else ""

    sql = f"""INSERT INTO gov_raw (page_url, title, publish_date, content, site_name, industry, summary)
VALUES (
  '{esc(page_url)}',
  '{esc(title)}',
  '{esc(publish_date)}',
  '{esc(content)}',
  '{esc(SITE_NAME)}',
  '{INDUSTRY}',
  '{esc(summary)}'
)"""

    result = subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql],
        capture_output=True, text=True, timeout=10,
    )
    if result.returncode != 0 and "UNIQUE" not in result.stderr:
        print(f"  [DB ERROR] {result.stderr}", file=sys.stderr)
        return "error"

    # Sync FTS
    sync_sql = f"""INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
SELECT rowid, title, site_name, summary
FROM gov_raw WHERE page_url = '{esc(page_url)}' AND rowid NOT IN (SELECT rowid FROM gov_search)"""

    subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sync_sql],
        capture_output=True, text=True, timeout=10,
    )

    return "new"


if __name__ == "__main__":
    crawl()
