#!/usr/bin/env python3
"""Capchem - 公告信息（环评公示）"""
import requests, re, sqlite3, os, sys
from bs4 import BeautifulSoup
from datetime import datetime

BASE = "https://www.capchem.com"
LIST_URL = "/news/3/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://www.capchem.com/",
}
DB = "/root/search.db"
site_name = "新宙邦-公告公示"

seen_urls = set()
count = 0

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        content_div = soup.select_one("div.e_richText-11") or soup.select_one("div.e_container-6")
        content = ""
        if content_div:
            content = str(content_div)
        else:
            body = soup.find("body")
            if body:
                content = str(body)
        # Date from detail
        pub_date = ""
        page_text = soup.get_text()
        for p in re.findall(r'发布时间[：:]\s*(\d{4}[-年]\d{1,2}[-月]\d{1,2})', page_text):
            pub_date = p.replace("年","-").replace("月","-")
            break
        return content.strip(), pub_date
    except Exception as e:
        print(f"  [WARN] detail error: {e}")
        return "", ""

conn = sqlite3.connect(DB, timeout=60)
c = conn.cursor()
try:
    r = requests.get(BASE + LIST_URL, headers=HEADERS, timeout=15)
    r.encoding = "utf-8"
except Exception as e:
    print(f"[ERR] {e}")
    sys.exit(1)

soup = BeautifulSoup(r.text, "html.parser")
items = soup.select("div.cbox-23.p_loopitem")
print(f"Found {len(items)} items")

for item in items:
    a = item.find("a")
    if not a:
        continue
    href = a.get("href", "")
    title = a.get_text(strip=True)
    if not href or "News_detail" not in href:
        continue

    # Extract date from beginning of text (YYYYMM-DD format)
    pub_date = ""
    m = re.match(r'(\d{6})[-\u4e00-\u9fff]', title)
    if m:
        y = m.group(1)[:4]
        m_d = m.group(1)[4:]
        pub_date = f"{y}-{m_d[:2]}-{m_d[2:]}"
        title = re.sub(r'^\d{6}[-年]?\s*', '', title).strip()
    # Also try full date pattern
    if not pub_date:
        m = re.search(r'(\d{4})[-\u4e00-\u9fff](\d{1,2})[-\u6708](\d{1,2})', title)
        if m:
            pub_date = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"
            title = re.sub(r'\d{4}[-\u5e74]\d{1,2}[-\u6708]\d{1,2}', '', title).strip()

    url = BASE + href if href.startswith("/") else href
    if url in seen_urls:
        continue
    seen_urls.add(url)

    c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
    if c.fetchone():
        print(f"  [SKIP] {title[:40]}... (exists)")
        continue

    content, detail_date = get_detail(url)
    if not content or len(content) < 50:
        print(f"  [SKIP] {title[:40]}... (empty content)")
        continue

    if detail_date:
        pub_date = detail_date

    summary = re.sub(r"<[^>]+>", "", content)
    summary = re.sub(r"\s+", " ", summary).strip()[:200]

    c.execute(
        "INSERT INTO gov_raw (title, summary, content, page_url, publish_date, category, site_name) VALUES (?,?,?,?,?,?,?)",
        (title.strip(), summary, content, url, pub_date, "环保公示", site_name),
    )
    conn.commit()
    count += 1
    print(f"  [{count}] {title[:50]} ({pub_date})")

conn.close()
print(f"\n===== DONE: {count} new records =====")
