#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
G批 山东人力环保-环评公告 (renlihuanbao.com)
http://renlihuanbao.com/zxns/zxns1/
CMS: 自研 PHP, GBK 编码
列表: <a href="/zxns/zxns1/YYYY-MM-DD/NNN.html">· 标题</a>
分页: 待确认 (可能首页即全量)
详情: /zxns/zxns1/YYYY-MM-DD/NNN.html -> div.lbyx (标题+公示时间+附件zip)
"""
import sys, os, re, time, html as html_lib, urllib.parse, sqlite3, json

BASE_URL = "http://renlihuanbao.com"
LIST_URL = "http://renlihuanbao.com/zxns/zxns1/"
SITE_NAME = "山东人力环保-环评公告"
GROUP_NAME = "企业"
SCRIPT_NAME = "crawl_renli_zxns.py"
CATEGORY = "环境保护"

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

DB_PATH = "/root/search.db"
ENC = "gbk"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

_DAYS_CUTOFF = 365 * 3
_CUTOFF_DATE = time.strftime("%Y-%m-%d", time.localtime(time.time() - _DAYS_CUTOFF * 86400))


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"^\s*(?:·|&middot;)?\s*", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch(url, timeout=30):
    for _ in range(2):
        try:
            import requests, urllib3
            urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
            r = requests.get(url, headers=HEADERS, timeout=timeout, verify=False)
            if r.status_code == 200:
                r.encoding = ENC
                return r.text
        except Exception:
            pass
    return None


def fetch_list(page_idx):
    if page_idx == 1:
        url = LIST_URL
    else:
        url = f"{LIST_URL}zxns1_{page_idx}.html"
    return fetch(url)


def parse_list(html):
    items = []
    if BeautifulSoup is None:
        return items
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.find_all("a", href=True):
        href = a.get("href", "").strip()
        if not re.search(r"/zxns/zxns1/\d{4}-\d{2}-\d{2}/\d+\.html$", href):
            continue
        title = clean_title(a.get_text(strip=True))
        if not title or len(title) < 4:
            continue
        # 日期从 URL 提取
        m = re.search(r"(\d{4}-\d{2}-\d{2})", href)
        pd = m.group(1) if m else ""
        url = urllib.parse.urljoin(BASE_URL, href)
        items.append({"title": title, "url": url, "pub_date": pd})
    seen = set()
    uniq = []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            uniq.append(it)
    return uniq


def extract_clean_text(content_html):
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")
    for tag in soup.find_all(["script", "style"]):
        tag.decompose()
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "").strip()
        text = a_tag.get_text(strip=True) or os.path.basename(href.split("?")[0]) or "附件"
        if not href or href.startswith(("javascript:", "#")):
            a_tag.unwrap()
            continue
        full_url = urllib.parse.urljoin(BASE_URL, href)
        new_a = soup.new_tag("a", href=full_url, target="_blank")
        new_a.string = text
        a_tag.replace_with(new_a)
    for tag in soup.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
        tag.unwrap()
    for tag in soup.find_all(["div", "p"]):
        if not tag.get_text(strip=True) and not tag.find(["img", "table", "a"]):
            tag.decompose()
    from html import escape
    seen_a = set()
    parts = []
    for el in soup.find_all(["table", "p"]):
        if el.name == "table":
            parts.append(str(el))
        elif el.name == "p" and not el.find_parent("table"):
            inner = []
            for c in el.contents:
                name = getattr(c, "name", None)
                if name == "img":
                    inner.append(str(c))
                elif name in ("a", "table", "br"):
                    inner.append(str(c))
                elif name:
                    inner.append(str(c))
                else:
                    inner.append(escape(str(c)))
            t = "".join(inner).strip()
            if t:
                parts.append(f"<p>{t}</p>")
    for div in soup.find_all(["div", "td"]):
        if div.find_parent("table") and div.name == "div":
            continue
        if div.find(["p", "table", "ul"]):
            continue
        direct = "".join(str(c) for c in div.contents if isinstance(c, str)).strip()
        if direct and len(direct) >= 2:
            parts.append(f"<p>{escape(direct)}</p>")
        for a_in in div.find_all("a", href=True):
            href = a_in.get("href", "").strip()
            text = a_in.get_text(strip=True) or os.path.basename(href.split("?")[0]) or "附件"
            if href and not href.startswith(("javascript:", "#")):
                key = (text, urllib.parse.urljoin(BASE_URL, href))
                if key in seen_a:
                    continue
                seen_a.add(key)
                parts.append(f'<p><a href="{key[1]}" target="_blank">{key[0]}</a></p>')
    return "\n\n".join(parts) if parts else escape(content_html).strip()


def parse_detail(html):
    result = {"title": "", "publish_date": "", "content": ""}
    soup = BeautifulSoup(html, "html.parser")
    lbyx = soup.find("div", class_="lbyx")
    if lbyx:
        # 标题: 最长 span
        best = ""
        for span in lbyx.find_all("span"):
            txt = span.get_text(strip=True)
            if len(txt) > len(best):
                best = txt
        result["title"] = clean_title(best)
        m = re.search(r"公示时间[：:]\s*(\d{4})年(\d{1,2})月(\d{1,2})日", lbyx.get_text())
        if m:
            result["publish_date"] = f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
        # 正文: 只删标题 div (含 font-size:26px 的居中标题), 保留附件 div
        clone = BeautifulSoup(str(lbyx), "html.parser")
        for div in clone.find_all("div"):
            style = div.get("style", "")
            if "text-align: center" in style and ("font-size:26px" in style or "font-size: 26px" in style):
                div.decompose()
            elif "text-align: center" in style and "padding: 6px" not in style and not div.find("a"):
                div.decompose()
        result["content"] = extract_clean_text(str(clone))
    if not result["title"]:
        title_tag = soup.find("title")
        if title_tag:
            result["title"] = clean_title(title_tag.get_text(strip=True))
    if not result["publish_date"]:
        m2 = re.search(r"20\d{2}[-/年]\d{1,2}[-/月]\d{1,2}", html[:8000])
        if m2:
            result["publish_date"] = m2.group(0).replace("年", "-").replace("月", "-").replace("/", "-")[:10]
    return result


def crawl(pages=1, json_out=None):
    stored = skipped = 0
    seen_urls = set()
    for page in range(1, pages + 1):
        html = fetch_list(page)
        if not html:
            break
        items = parse_list(html)
        if not items:
            break
        for it in items:
            if it["url"] in seen_urls:
                continue
            seen_urls.add(it["url"])
            if it["pub_date"] and it["pub_date"] < _CUTOFF_DATE:
                continue
            detail = parse_detail(fetch(it["url"]) or "")
            detail["title"] = detail["title"] or it["title"]
            detail["publish_date"] = detail["publish_date"] or it["pub_date"]
            _store(detail, it, json_out)
            if detail["content"]:
                stored += 1
            else:
                skipped += 1
        time.sleep(0.3)
    log(f"Result: {stored} stored, {skipped} skipped")


def _store(detail, it, json_out):
    row = {
        "site_name": SITE_NAME,
        "source_url": it["url"],
        "page_url": it["url"],
        "title": detail["title"],
        "publish_date": detail["publish_date"],
        "content": detail["content"],
        "summary": (detail["content"] or "")[:500],
        "category": CATEGORY,
        "script_name": SCRIPT_NAME,
    }
    if json_out:
        with open(json_out, "a", encoding="utf-8") as f:
            f.write(json.dumps(row, ensure_ascii=False) + "\n")
        return
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()
    try:
        c.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, source_url, page_url, title, publish_date, content, summary, category, script_name)"
            " VALUES (?,?,?,?,?,?,?,?,?)",
            (row["site_name"], row["source_url"], row["page_url"], row["title"],
             row["publish_date"], row["content"], row["summary"], row["category"], row["script_name"]))
        conn.commit()
    except Exception:
        pass
    conn.close()


def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)


if __name__ == "__main__":
    json_out = None
    if "--json" in sys.argv:
        json_out = "/tmp/crawl_test.jsonl"
    crawl(_PAGES, json_out)
