#!/usr/bin/env python3
"""
安徽思科环境系统工程技术有限公司 - 项目公示 (ciscoep.com)
Yun300 CMS / 云千站群
列表API: /comp/portalResNews/list.do?compId=portalResNews_list-15890012838618906&cid=1&size=6&currentPage=N
详情: /news/N.html
正文: article.summary > div.lead > p
"""
import re, sys, os, json, time
import urllib.request, urllib.error
from bs4 import BeautifulSoup
import sqlite3

BASE_URL = "http://www.ciscoep.com"
SITE_NAME = "安徽思科环境-项目公示"
GROUP_NAME = "企业环评"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
LIST_API = "/comp/portalResNews/list.do?compId=portalResNews_list-15890012838618906&cid=1&size=6&currentPage="
PAGE_SIZE = 6

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break


def fetch(url, api=False):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=30).read().decode("utf-8", errors="replace")
    except Exception as e:
        print(f"[WARN] fetch failed: {url} - {e}", file=sys.stderr)
        return ""


def extract_items_from_api(html):
    """Extract items from API response HTML"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.select("a.newLinkBox"):
        href = a.get("href", "")
        if not href:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href

        h2 = a.select_one("h2")
        title = h2.get_text(strip=True) if h2 else ""
        if not title or len(title) < 5:
            continue

        # Date from leftTimeBox
        date = ""
        lb = a.select_one(".leftTimeBox")
        if lb:
            d = lb.select_one(".newData")
            ym = lb.select_one(".newYearMon")
            if ym and d:
                day = d.get_text(strip=True)
                ymt = ym.get_text(strip=True)
                date = f"{ymt}-{day}" if day else ymt

        # Also check newToolBox
        if not date:
            tb = a.select_one(".newToolBox")
            if tb:
                sp = tb.select_one(".data1")
                if sp:
                    txt = sp.get_text(strip=True)
                    m = re.search(r"(\d{4})\s*[-_]\s*(\d{1,2})\s*[-_]\s*(\d{1,2})", txt)
                    if m:
                        date = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"

        items.append({"title": title, "url": href, "date": date})
    return items


def parse_detail(html, url):
    """Parse detail page"""
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    h1 = soup.select_one("h1.title-box")
    if h1:
        title = h1.get_text(strip=True)

    date_text = ""
    time_el = soup.select_one(".p_time")
    if time_el:
        txt = time_el.get_text(strip=True)
        m = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
        if m:
            date_text = m.group(1)

    content = ""
    attachments = []

    lead = soup.select_one(".lead")
    if not lead:
        return {"title": title, "date": date_text, "content": "", "attachments": []}

    parts = []
    for el in lead.find_all(["p", "table", "img"], recursive=True):
        if el.name == "p" and el.find_parent("table"):
            continue
        if el.name == "table" and el.find_parent("table"):
            continue

        if el.name == "img":
            src = el.get("src", "")
            if "icon_" in src.lower():
                continue
            if src.startswith("//"):
                src = "http:" + src
            elif src.startswith("/"):
                src = BASE_URL + src
            elif not src.startswith("http"):
                src = url.rsplit("/", 1)[0] + "/" + src
            alt = el.get("alt", "") or ""
            parts.append(f"![{alt}]({src})")
            continue

        if el.name == "table":
            rows = []
            for tr in el.find_all("tr"):
                cells = [c.get_text(" ", strip=True) for c in tr.find_all(["td", "th"])]
                if cells:
                    # Check for attachments in cells
                    for cell in tr.find_all(["td", "th"]):
                        for a in cell.find_all("a", href=True):
                            ah = a["href"]
                            if re.search(r'\.(docx?|pdf|xlsx?|rar|zip)$', ah, re.I):
                                atitle = a.get_text(strip=True) or "附件"
                                if ah.startswith("/"):
                                    ah = BASE_URL + ah
                                elif not ah.startswith("http"):
                                    ah = url.rsplit("/", 1)[0] + "/" + ah
                                attachments.append({"title": atitle, "url": ah})
                    rows.append("| " + " | ".join(cells) + " |")
            if rows:
                col_count = len(rows[0].split("|")) - 2
                sep = "| " + " | ".join(["---"] * col_count) + " |"
                rows.insert(1, sep)
                parts.append("\n".join(rows))
            continue

        # <p> text
        text = el.get_text(" ", strip=True)
        if text:
            for a in el.find_all("a", href=True):
                ah = a["href"]
                if re.search(r'\.(docx?|pdf|xlsx?|rar|zip)$', ah, re.I):
                    atitle = a.get_text(strip=True) or "附件"
                    if ah.startswith("/"):
                        ah = BASE_URL + ah
                    elif not ah.startswith("http"):
                        ah = url.rsplit("/", 1)[0] + "/" + ah
                    attachments.append({"title": atitle, "url": ah})
            parts.append(text)

    content = "\n\n".join(p for p in parts if p.strip())
    content = content.replace("\u00a0", " ").replace("&nbsp;", " ")

    seen = set()
    unique_att = []
    for att in attachments:
        if att["url"] not in seen:
            seen.add(att["url"])
            unique_att.append(att)

    return {"title": title, "date": date_text, "content": content, "attachments": unique_att}


def main():
    conn = sqlite3.connect(DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()

    # Clean old data
# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)     c.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)     c.execute("DELETE FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    conn.commit()
    print("[清理] 已清理该站旧数据", flush=True)

    all_items = []
    total_pages = 4  # From JS config

    if _MAX_PG:
        total_pages = min(total_pages, _MAX_PG)

    for pg in range(1, total_pages + 1):
        api_url = f"{BASE_URL}{LIST_API}{pg}"
        html = fetch(api_url, api=True)
        if not html:
            print(f"[WARN] Page {pg}: empty response", file=sys.stderr)
            continue
        items = extract_items_from_api(html)
        print(f"[INFO] Page {pg}: {len(items)} items", flush=True)
        if not items:
            break
        all_items.extend(items)

    print(f"[INFO] Total items from list: {len(all_items)}", flush=True)

    new_count = 0
    skip_count = 0
    empty_count = 0
    total_att = 0

    for idx, item in enumerate(all_items):
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
        if c.fetchone():
            skip_count += 1
            continue

        if idx % 5 == 0:
            print(f"[INFO] [{idx}/{len(all_items)}] {item['title'][:40]}...", file=sys.stderr, end=" ", flush=True)

        html = fetch(item["url"])
        if not html:
            print("FETCH FAILED", file=sys.stderr)
            continue

        detail = parse_detail(html, item["url"])
        title = detail["title"] or item["title"]
        date_text = detail["date"] or item["date"]
        content = detail["content"]
        attachments_list = detail["attachments"]

        if not content.strip():
            empty_count += 1
            if idx % 5 == 0:
                print("EMPTY", file=sys.stderr)
            continue

        plain = re.sub(r'\s+', ' ', content).strip()
        summary = plain[:200] if plain else title[:200]
        att_json = json.dumps(attachments_list, ensure_ascii=False) if attachments_list else "[]"

        date_rank = 0
        if date_text:
            m2 = re.search(r"(\d{4})-(\d{2})-(\d{2})", date_text)
            if m2:
                date_rank = int(m2.group(1) + m2.group(2) + m2.group(3))

        try:
            c.execute(
                """INSERT OR IGNORE INTO gov_raw 
                (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank)
                VALUES (?,?,?,?,?,?,?,?,?)""",
                (title, SITE_NAME, GROUP_NAME, item["url"], date_text, content, summary, att_json, date_rank)
            )
            if c.rowcount > 0:
                new_count += 1
                total_att += len(attachments_list)
        except Exception as e:
            if idx % 5 == 0:
                print(f"DB ERROR: {e}", file=sys.stderr)

        time.sleep(0.3)

    conn.commit()
    conn.close()

    print(f"\n[RESULT] {SITE_NAME}: {new_count} new, {skip_count} skip, {empty_count} empty, {total_att} attachments", flush=True)


if __name__ == "__main__":
    main()
