#!/usr/bin/env python3
"""
crawl_soochowchem.py - 宁夏东吴农化股份有限公司 新闻/环评公示
URL: http://www.soochowchem.com/news.html
结构: 单页46条列表，/news_detail/id/N.html 详情，正文在 td.zw
"""
import sys, re, requests, subprocess
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
SITE_NAME = "宁夏东吴农化-新闻公示"
INDUSTRY = "环境公示"
BASE_URL = "http://www.soochowchem.com"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}


def esc(s):
    """Escape single quotes for SQLite"""
    return (s or "").replace("'", "''")


def parse_date(text):
    m = re.search(r"(\d{4}[-/]\d{1,2}[-/]\d{1,2})", text)
    return m.group(1).replace("/", "-") if m else ""


def fetch_detail(url):
    """Fetch detail page, return (title, content, publish_date)"""
    try:
        r = requests.get(url, timeout=15, headers=HEADERS)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None, None, None
    except Exception as e:
        print(f"  [ERR] {url}: {e}", flush=True)
        return None, None, None

    soup = BeautifulSoup(r.text, "html.parser")
    h1 = soup.find("h1")
    title = h1.get_text(strip=True) if h1 else ""

    pub_date = parse_date(r.text)

    zws = soup.find_all("td", class_="zw")
    if not zws:
        zws = soup.find_all(class_="zw")
    if zws:
        # Take the longest .zw td (first one is footer with contact info)
        zw = max(zws, key=lambda x: len(x.get_text(strip=True)))
    else:
        # Fallback: td with largest text
        tds = soup.find_all("td")
        zw = max(tds, key=lambda x: len(x.get_text(strip=True))) if tds else None

    content = ""
    if zw:
        # Extract content preserving paragraph breaks
        # First try <p> tags
        ps = zw.find_all("p")
        if ps:
            paragraphs = []
            for p in ps:
                txt = p.get_text(strip=True)
                if txt and len(txt) > 5:
                    paragraphs.append(txt)
            if paragraphs:
                content = "\n\n".join(paragraphs)
        if not content:
            for br in zw.find_all("br"):
                br.replace_with("\n")
            content = zw.get_text(strip=True)
        # Clean title prefix from content
        if title and content.startswith(title):
            content = content[len(title):].strip()
        date_prefix = f"发布时间：{pub_date}"
        if date_prefix in content:
            content = content.split(date_prefix, 1)[-1].strip()

    if not title:
        title_tag = soup.find("title")
        title = title_tag.get_text(strip=True) if title_tag else ""

    return title, content, pub_date


def insert_one(page_url, title, content, pub_date):
    """Insert into search.db via sqlite3 CLI"""
    summary = content[:200] if content else ""
    sql = f"""INSERT INTO gov_raw (page_url, title, publish_date, content, site_name, industry, summary)
VALUES (
  '{esc(page_url)}',
  '{esc(title)}',
  '{esc(pub_date)}',
  '{esc(content)}',
  '{esc(SITE_NAME)}',
  '{INDUSTRY}',
  '{esc(summary)}'
)"""
    result = subprocess.run(
        ["sqlite3", DB_PATH, sql],
        capture_output=True, text=True, timeout=10,
    )
    if result.returncode != 0 and "UNIQUE" not in result.stderr:
        print(f"  [DB ERROR] {result.stderr}", file=sys.stderr)
        return False

    # FTS sync
    fts_sql = f"""INSERT INTO gov_search (rowid, title, site_name, summary)
SELECT rowid, title, site_name, summary
FROM gov_raw WHERE page_url = '{esc(page_url)}' AND rowid NOT IN (SELECT rowid FROM gov_search)"""
    subprocess.run(
        ["sqlite3", DB_PATH, fts_sql],
        capture_output=True, text=True, timeout=10,
    )
    return True


def crawl():
    max_pages = 5
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            pass

    print(f"[{SITE_NAME}] Starting crawl...", flush=True)

    # Single list page
    url = f"{BASE_URL}/news.html"
    r = requests.get(url, timeout=15, headers=HEADERS)
    r.encoding = "utf-8"
    if r.status_code != 200:
        print(f"[ERR] List page HTTP {r.status_code}", file=sys.stderr)
        return

    soup = BeautifulSoup(r.text, "html.parser")
    links = soup.find_all("a", href=True)
    news_items = []
    for a in links:
        href = a["href"]
        if "news_detail" in href:
            title = a.get_text(strip=True)
            full_url = href if href.startswith("http") else f"{BASE_URL}{href}"
            news_items.append((full_url, title))

    print(f"Found {len(news_items)} items on list page", flush=True)

    new_count = 0
    dup_count = 0
    err_count = 0

    for i, (detail_url, list_title) in enumerate(news_items):
        print(f"  [{i+1}/{len(news_items)}] {list_title[:50]}...", flush=True)
        title, content, pub_date = fetch_detail(detail_url)
        final_title = title or list_title
        if insert_one(detail_url, final_title, content, pub_date):
            new_count += 1
        else:
            dup_count += 1

    print(f"\nDone! New: {new_count}, Duplicates: {dup_count}, Errors: {err_count}", flush=True)


if __name__ == "__main__":
    crawl()
