#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""crawl_hbak.py - 河北安科工程技术有限公司-评价项目业绩信息 (www.hbak.cn)
站点: 静态 ASP (GBK), 无WAF, 企业站 (2014年停更, 历史数据)
列表: news.asp?id=4&cID=3 → ?id=4&page={N}&cID=3 (共150页)
详情: newsny.asp?id={id}&cID=3
运行: python3 crawl_hbak.py [--pages=N] [--dryrun]
"""
import sys, os, re, json, time, requests, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

SITE_NAME = "河北安科-评价项目业绩"
BASE = "http://www.hbak.cn"
MAX_PAGES_DEFAULT = 5
CUTOFF = (datetime.now() - timedelta(days=10*365)).strftime("%Y-%m-%d")  # 历史站, 放宽10年
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
OUT = "/tmp/hbak.jsonl"
DB_PATH = "/root/search.db"
SCRIPT_NAME = "crawl_hbak.py"


def _drop_container_parts(parts):
    """剔除「容器段」：find_all(['p','div']) 会同时收下容器 <div> 与它内部的 <p>，
    导致同一内容重复（容器那份常还被 get_text(strip=True) 拍平）。
    判据：先按值去重，再剔除被其它段完全包含的段。保护：剔除后为空则返回去重结果。
    """
    if not parts:
        return parts
    ps = [p for p in parts if isinstance(p, str)]
    if len(ps) != len(parts):
        return parts
    seen, uniq = set(), []
    for p in parts:
        if p not in seen:
            seen.add(p); uniq.append(p)
    keep = [a for a in uniq
            if not (len(a) >= 40 and any(b is not a and b and b in a for b in uniq))]
    return keep if keep else uniq


def parse_pages_arg():
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            return int(a.split("=")[1])
    return MAX_PAGES_DEFAULT


def fetch(url):
    try:
        r = requests.get(url, headers={"User-Agent": UA}, timeout=20, verify=False)
        r.encoding = "gbk"
        return r.text
    except Exception:
        return ""


def list_items(page):
    if page == 1:
        url = f"{BASE}/news.asp?id=4&cID=3"
    else:
        url = f"{BASE}/news.asp?id=4&page={page}&cID=3"
    html = fetch(url)
    if not html:
        return []
    items = []
    for m in re.finditer(r'href="(newsny\.asp\?id=(\d+)&cID=3)"[^>]*>([^<]{8,120})', html):
        items.append({"title": m.group(3).strip(), "url": f"{BASE}/{m.group(1)}"})
    return items


def parse_detail(url):
    html = fetch(url)
    if not html:
        return "", "", ""
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    m = re.search(r"<title>([^<]*)</title>", html, re.S)
    if m:
        title = re.sub(r"\s*[-_].*$", "", m.group(1)).strip()
    content = ""
    soup = BeautifulSoup(html, "html.parser")
    # 正文容器: page_about (标题+表格) > sub_right (面包屑+正文)
    zoom = soup.find("div", class_=re.compile(r"page_about", re.I))
    if not zoom:
        zoom = soup.find("div", class_=re.compile(r"sub_right", re.I))
    if not zoom:
        zoom = soup.find("div", id=re.compile(r"zoom|content|article", re.I)) or soup.find("div", class_=re.compile(r"content|article|text|zoom", re.I))
    if zoom:
        for el in zoom.find_all(class_=re.compile(r"share|qrcode|print", re.I)):
            el.decompose()
        parts = []
        for el in zoom.find_all(["p", "pre", "table", "div"], recursive=True):
            if el.name in ("p", "pre") and el.find_parent("table"):
                continue
            if el.name == "table" and el.find_parent("table"):
                continue
            if el.name == "table":
                parts.append(str(el))
                continue
            if el.name == "div":
                if el.find(["p", "pre", "table"], recursive=False):
                    continue
                txt = el.get_text(" ", strip=True)
                if txt:
                    parts.append(f"<p>{txt}</p>")
                continue
            txt = el.get_text(" ", strip=True)
            if txt:
                # 保留段落内部 HTML (a/img 不拍平), 压缩多余空白
                parts.append(re.sub(r"\s+", " ", str(el)).strip())
        dedup = []
        for p in parts:
            if p and (not dedup or dedup[-1] != p):
                dedup.append(p)
        dedup = _drop_container_parts(dedup)
        content = "\n\n".join(dedup)
    if len(content.strip()) < 50:
        # fallback: 取 "作者：" 之后到 "版权所有" 之前的正文文本
        body_text = re.sub(r"<script.*?</script>", "", html, flags=re.S)
        body_text = re.sub(r"<[^>]+>", "\n", body_text)
        lines = [l.strip() for l in body_text.split("\n") if l.strip()]
        try:
            start = next(i for i, l in enumerate(lines) if "作者" in l)
        except StopIteration:
            start = 0
        end = len(lines)
        for i, l in enumerate(lines):
            if "版权所有" in l or "Copyright" in l:
                end = i
                break
        body_lines = [l for l in lines[start+1:end] if l not in ("首页", "&gt;", "信息公示", "评价项目业绩信息", "联系我们") and not l.startswith("&")]
        if body_lines and body_lines[0].startswith("-->"):
            body_lines[0] = body_lines[0].lstrip("-->").strip()
        content = "\n\n".join([f"<p>{l}</p>" for l in body_lines if l])
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}" target="_blank">{title}</a></p>'
    # 详情页无日期 → 从正文找年份
    date_text = ""
    m = re.search(r"(20\d{2})年", content[:500])
    if m:
        date_text = m.group(1) + "-01-01"
    return title, date_text, content


def main():
    max_pages = parse_pages_arg()
    dryrun = "--dryrun" in sys.argv
    print(f"[hbak] pages={max_pages} cutoff={CUTOFF}", flush=True)
    results = []
    for pn in range(1, max_pages + 1):
        items = list_items(pn)
        if not items:
            print(f"  第{pn}页 0 条, 停止", flush=True)
            break
        print(f"  第{pn}页 {len(items)} 条", flush=True)
        for it in items:
            try:
                title, date_text, content = parse_detail(it["url"])
                if date_text and date_text < CUTOFF:
                    continue
                results.append({"url": it["url"], "title": title or it["title"], "date": date_text, "content": content})
            except Exception as e:
                print(f"    ✗ {str(e)[:60]}", flush=True)
            time.sleep(0.3)
    print(f"[完成] {len(results)} 条", flush=True)
    if dryrun:
        for r in results[:3]:
            print(json.dumps(r, ensure_ascii=False)[:200])
        return
    with open(OUT, "w", encoding="utf-8") as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + "\n")
    if os.path.exists(DB_PATH):
        conn = sqlite3.connect(DB_PATH, timeout=120)
        conn.execute("PRAGMA busy_timeout=120000")
        c = conn.cursor()
        added = 0
        for r in results:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (r["url"],))
            if c.fetchone():
                continue
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank, script_name) VALUES (?,?,?,?,?,?,?,?,?,?)",
                (r["title"], SITE_NAME, "河北", r["url"], r["date"], r["content"], "", "",
                 int(r["date"].replace("-", "")) if re.match(r"\d{4}-\d{2}-\d{2}", r["date"]) else 0, SCRIPT_NAME))
            if c.rowcount:
                added += 1
        conn.commit()
        conn.close()
        print(f"[入库] 新增 {added}", flush=True)


if __name__ == "__main__":
    main()
