#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
海科化工 - 公司新闻
站点: haike.haikegroup.com/hk_news.html
CMS: 静态 HTML (hk_detailed/hk_news/ID.html)
列表: a[href=/hk_detailed/hk_news/3767.html] (标题在 a 内, 日期 2025年11月14)
详情: div#article 正文
分页: /hk_news/p/N.html
用法: python3 crawl_haike_gsgg.py [--pages=N | N]
"""
import re, sys, os, json, html as html_mod
from urllib.parse import urljoin
import requests, sqlite3

SITE_NAME = "海科化工-公司新闻"
BASE_URL = "http://haike.haikegroup.com"
LIST_URL = "http://haike.haikegroup.com/hk_news.html"
DOMAIN = "haike.haikegroup.com"
GROUP = "企业"
CUTOFF = "2023-01-01"
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Referer": LIST_URL,
}

def clean_title(t):
    t = re.sub(r"&middot;|&nbsp;|&#160;|\u200b|\ufeff", "", t or "")
    t = re.sub(r"^[•·]\s*", "", t)
    return re.sub(r"\s+", " ", t).strip()

def normalize_date(s):
    m = re.search(r"(\d{4})[年\-/. ](\d{1,2})[月\-/. ](\d{1,2})", s or "")
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        return f"{y:04d}-{mo:02d}-{d:02d}"
    return ""

def fetch(url, retries=3):
    for i in range(retries):
        try:
            resp = requests.get(url, timeout=30, headers=HEADERS, verify=False)
            resp.encoding = "utf-8"
            if resp.status_code == 200:
                return resp.text
        except Exception:
            if i < retries - 1:
                import time; time.sleep(2)
    return None

def extract_list_items(html_text, base_url):
    items = []
    if not html_text:
        return items
    seen = set()
    # li 块: <li>日期 <a href="/hk_detailed/hk_news/ID.html"><h1 title="标题">...</h1></a></li>
    MONTHS = {"Jan":1,"Feb":2,"Mar":3,"Apr":4,"May":5,"Jun":6,"Jul":7,"Aug":8,"Sep":9,"Oct":10,"Nov":11,"Dec":12}
    for m in re.finditer(
        r'<li[^>]*>(.*?)</li>',
        html_text, re.DOTALL):
        block = m.group(1)
        am = re.search(r'<a[^>]+href="([^"]*hk_detailed/hk_news/\d+\.html)"[^>]*><h1[^>]*title="([^"]*)"', block, re.DOTALL)
        if not am:
            am = re.search(r'<a[^>]+href="([^"]*hk_detailed/hk_news/\d+\.html)"[^>]*>(.*?)</a>', block, re.DOTALL)
        if not am:
            continue
        href = am.group(1).strip()
        if am.re.groups == 2 and 'title="' in block:
            title = clean_title(am.group(2))
        else:
            title = clean_title(re.sub(r"<[^>]+>", "", am.group(2)))
        if len(title) < 4:
            continue
        if href in seen:
            continue
        seen.add(href)
        # 日期: <h1>10</h1><span>Jun 2026</span>
        pub_date = ""
        dm = re.search(r'<h1>\s*(\d{1,2})\s*</h1>\s*<span>\s*([A-Za-z]{3})\s+(\d{4})', block)
        if dm:
            day, mon, yr = int(dm.group(1)), MONTHS.get(dm.group(2)), int(dm.group(3))
            if mon:
                pub_date = f"{yr:04d}-{mon:02d}-{day:02d}"
        detail_url = urljoin(base_url, href) if not href.startswith("http") else href
        items.append({"title": title, "url": detail_url, "pub_date": pub_date, "href_raw": href})
    return items

def parse_detail(url):
    html_text = fetch(url)
    if not html_text:
        return None
    title = ""
    m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html_text)
    if m:
        title = clean_title(m.group(1))
    if not title:
        m = re.search(r"<title>(.*?)</title>", html_text, re.DOTALL)
        if m:
            title = clean_title(re.sub(r"\s*[-_—]\s*.*$", "", m.group(1).strip()))
    pub_date = ""
    m = re.search(r'<meta\s+name="(?:PubDate|publishdate|publish_date)"\s+content="([^"]*)"', html_text, re.I)
    if m:
        pub_date = normalize_date(m.group(1))
    body_html = ""
    m = re.search(r'<div[^>]+id="article"[^>]*>(.*)', html_text, re.DOTALL)
    if not m:
        m = re.search(r'<div[^>]+id="content"[^>]*>(.*)', html_text, re.DOTALL)
    if m:
        raw = m.group(1)
        # 优先找正文结束: id="footer" / id="share" / class="footer" 或 script
        cut = re.search(
            r'<div[^>]+id="(?:footer|share|print)"'
            r'|<div[^>]+class="[^"]*(?:footer|share|print)[^"]*"'
            r'|<script[^>]*>|<div[^>]+id="rightsidebar"',
            raw)
        if cut:
            body_html = raw[:cut.start()]
        else:
            end = raw.rfind("</div>")
            body_html = raw[:end + 6] if end != -1 else raw
    body_html = re.sub(r"<script[^>]*>.*?</script>", "", body_html, flags=re.S | re.I)
    body_html = re.sub(r"<style[^>]*>.*?</style>", "", body_html, flags=re.S | re.I)
    atts = []
    for m in re.finditer(r'<a[^>]*href="([^"]+)"[^>]*>([^<]{0,120})</a>', body_html):
        href, name = m.group(1).strip(), clean_title(m.group(2))
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|7z|et|dps)$", href, re.I):
            abs_url = href if href.startswith("http") else (BASE_URL + href if href.startswith("/") else urljoin(url, href))
            if not any(a["url"] == abs_url for a in atts):
                atts.append({"name": name or href.split("/")[-1], "url": abs_url})
    def abs_link(m):
        href = m.group(1).strip()
        if href.startswith("http") or href.startswith("javascript") or href.startswith("#"):
            return m.group(0)
        if href.startswith("//"):
            return m.group(0).replace(href, "http:" + href)
        return m.group(0).replace(href, BASE_URL + href if href.startswith("/") else urljoin(url, href))
    body_html = re.sub(r'href="([^"]+)"', abs_link, body_html)
    body_html = re.sub(r'src="([^"]+)"', abs_link, body_html)
    body_html = re.sub(r'<p>\s*</p>', "", body_html)
    body_html = re.sub(r'<p[^>]*>', "<p>", body_html)
    if atts:
        has_att = any(a["url"] in body_html for a in atts)
        if not has_att:
            att_html = "".join(f'<p><a href="{a["url"]}">{html_mod.escape(a["name"])}</a></p>' for a in atts)
            body_html = body_html.rstrip() + "\n" + att_html
    text_len = len(re.sub(r"<[^>]+>", "", body_html).strip())
    has_img = "<img" in body_html
    has_video = "<iframe" in body_html or "video" in body_html
    if text_len < 5 and not has_img and not has_video and not atts:
        return {"title": title, "content": "", "pub_date": pub_date, "attachments": []}
    return {"title": title, "content": body_html, "pub_date": pub_date, "attachments": atts}

def _store(detail, it):
    row = {
        "site_name": SITE_NAME,
        "source_url": it["url"],
        "page_url": it["url"],
        "title": detail["title"] or it["title"],
        "publish_date": detail["pub_date"] or it["pub_date"],
        "content": detail["content"],
        "summary": (detail["content"] or "")[:500],
        "category": "环境保护",
        "script_name": os.path.basename(__file__),
    }
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()
    try:
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name) VALUES (?,?,?,?,?,?,?,?,?)",
            (row["site_name"], row["source_url"], row["page_url"], row["title"], row["publish_date"], row["content"], row["summary"], row["category"], row["script_name"]))
        conn.commit()
    except Exception:
        pass
    conn.close()

def crawl(pages=1):
    stored = skipped = 0
    seen_urls = set()
    for page in range(1, pages + 1):
        if page == 1:
            html_text = fetch(LIST_URL)
        else:
            html_text = fetch(f"{BASE_URL}/hk_news/p/{page}.html")
        if not html_text:
            break
        items = extract_list_items(html_text, LIST_URL)
        if not items:
            break
        for it in items:
            if it["url"] in seen_urls:
                continue
            seen_urls.add(it["url"])
            if not it["pub_date"] or it["pub_date"] < CUTOFF:
                continue
            detail = parse_detail(it["url"])
            if not detail:
                continue
            if not detail.get("content"):
                detail = {"title": it["title"], "publish_date": it["pub_date"], "content": "", "pub_date": it["pub_date"]}
            detail["title"] = detail["title"] or it["title"]
            detail["publish_date"] = detail.get("publish_date") or it["pub_date"]
            _store(detail, it)
            if detail["content"]:
                stored += 1
            else:
                skipped += 1
        print(f"[{SITE_NAME}] P{page}: {len(items)} items", flush=True)
        time.sleep(0.3)
    print(f"[{SITE_NAME}] Result: {stored} stored, {skipped} skipped")

if __name__ == "__main__":
    import time
    _MAX_PG = 1
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            try:
                _MAX_PG = int(a.split("=", 1)[1])
            except ValueError:
                pass
        elif a.isdigit():
            _MAX_PG = int(a)
    crawl(pages=_MAX_PG)
