#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
G批 TRS 广西模板 - {SITE_NAME}
http://sthjj.shiyan.gov.cn/hjgl/hjyxpj/hpgs/
CMS: TRS 广西集约化
列表: <li><span>日期</span><a href="./tXXX.shtml" title="标题">标题</a></li>
分页: createPageHTML(N,0,"index","shtml","104") -> index_1.shtml
详情: div.TRS_UEDITOR / div.view.TRS_UEDITOR + <meta PubDate>
"""
import sys, os, re, time, html as html_lib, urllib.parse, sqlite3, json

BASE_URL = "http://sthjj.shiyan.gov.cn"
LIST_URL = "http://sthjj.shiyan.gov.cn/hjgl/hjyxpj/hpgs/"
SITE_NAME = "十堰市生态环境局-环评公示"
GROUP_NAME = "湖北"
SCRIPT_NAME = "crawl_shiyan_hpgs.py"
CATEGORY = "环境保护"

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

DB_PATH = "/root/search.db"
ENC = "utf-8"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

_DAYS_CUTOFF = 365 * 3
_CUTOFF_DATE = time.strftime("%Y-%m-%d", time.localtime(time.time() - _DAYS_CUTOFF * 86400))


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"^\s*(?:&middot;|·|\u00b7)?\s*(?:&nbsp;|\u00a0)?\s*", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch(url, referer=None, timeout=30):
    h = dict(HEADERS)
    if referer:
        h["Referer"] = referer
    for _ in range(2):
        try:
            import requests, urllib3
            urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
            r = requests.get(url, headers=h, timeout=timeout, verify=False)
            if r.status_code == 200:
                r.encoding = ENC
                return r.text
        except Exception:
            pass
    return None


def fetch_list(page_idx):
    if page_idx == 1:
        url = LIST_URL
    else:
        url = LIST_URL.rstrip("/") + f"/index_{page_idx-1}.shtml"
    return fetch(url)


def parse_list(html):
    items = []
    if BeautifulSoup is None:
        return items
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.find_all("a", href=True):
        href = a.get("href", "").strip()
        if not href or href.startswith(("javascript:", "#")):
            continue
        if not re.search(r"t\d+(_\d+)?\.shtml$", href):
            continue
        title = clean_title(a.get("title", "") or a.get_text(strip=True))
        if not title or len(title) < 4:
            continue
        # 父容器找日期
        pd = ""
        parent = a.find_parent(["li", "div"])
        if parent:
            m = re.search(r"20\d{2}-\d{1,2}-\d{1,2}", parent.get_text())
            if m:
                pd = m.group(0)
        url = urllib.parse.urljoin(LIST_URL, href)
        items.append({"title": title, "url": url, "pub_date": pd})
    # 去重保序
    seen = set()
    uniq = []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            uniq.append(it)
    return uniq


def extract_clean_text(content_html):
    """输出纯 HTML：段落 <p>、链接 <a href target=_blank>、表格保留、图片 <img>、附件独立成段"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")
    for tag in soup.find_all(["script", "style"]):
        tag.decompose()
    for img in soup.find_all("img"):
        src0 = img.get("src", "")
        if any(k in src0 for k in ("un-collect", "collect", "share", "favicon", "qrcode", "ewm")):
            img.decompose()
    # 链接/附件 → HTML 内嵌绝对 URL
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "").strip()
        text = a_tag.get_text(strip=True) or os.path.basename(href.split("?")[0]) or "附件"
        if not href or href.startswith(("javascript:", "#")):
            a_tag.unwrap()
            continue
        full_url = urllib.parse.urljoin(BASE_URL, href)
        new_a = soup.new_tag("a", href=full_url, target="_blank")
        new_a.string = text
        a_tag.replace_with(new_a)
    for tag in soup.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
        tag.unwrap()
    attach_links = []
    for li in soup.find_all("li"):
        a_in = li.find("a", href=True)
        if a_in and not li.find_parent("table"):
            href = a_in.get("href", "").strip()
            text = a_in.get_text(strip=True) or os.path.basename(href.split("?")[0]) or "附件"
            if href and not href.startswith(("javascript:", "#")):
                attach_links.append((text, urllib.parse.urljoin(BASE_URL, href)))
    for li in soup.find_all("li"):
        if li.find("a", href=True) and not li.find_parent("table"):
            li.decompose()
    for tag in soup.find_all(["div", "p"]):
        if not tag.get_text(strip=True) and not tag.find(["img", "table", "a"]):
            tag.decompose()
    from html import escape
    parts = []
    for el in soup.find_all(["table", "p"]):
        if el.name == "table":
            parts.append(str(el))
        elif el.name == "p" and not el.find_parent("table"):
            inner = []
            for c in el.contents:
                name = getattr(c, "name", None)
                if name == "img":
                    src0 = c.get("src", "")
                    if any(k in src0.lower() for k in (".gif", "icon16", "file", "attach")):
                        continue
                    inner.append(str(c))
                elif name in ("a", "table", "br"):
                    inner.append(str(c))
                elif name:
                    inner.append(str(c))
                else:
                    inner.append(escape(str(c)))
            t = "".join(inner).strip()
            if t:
                parts.append(f"<p>{t}</p>")
    seen_a = set()
    for text, url in attach_links:
        key = (text, url)
        if key in seen_a:
            continue
        seen_a.add(key)
        parts.append(f'<p><a href="{url}" target="_blank">{text}</a></p>')
    # div 直文本转 <p> 段（防整页实体化）
    for div in soup.find_all(["div", "td"]):
        if div.find_parent("table") and div.name == "div":
            continue
        if div.find(["p", "table", "ul"]):
            continue
        direct = "".join(str(c) for c in div.contents if isinstance(c, str)).strip()
        if direct and len(direct) >= 2:
            parts.append(f"<p>{escape(direct)}</p>")
        for a_in in div.find_all("a", href=True):
            href = a_in.get("href", "").strip()
            text = a_in.get_text(strip=True) or os.path.basename(href.split("?")[0]) or "附件"
            if href and not href.startswith(("javascript:", "#")):
                key = (text, urllib.parse.urljoin(BASE_URL, href))
                if key in seen_a:
                    continue
                seen_a.add(key)
                parts.append(f'<p><a href="{key[1]}" target="_blank">{key[0]}</a></p>')
    return "\n\n".join(parts) if parts else escape(content_html).strip()


def parse_detail(html):
    result = {"title": "", "publish_date": "", "content": ""}
    soup = BeautifulSoup(html, "html.parser")
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        result["title"] = clean_title(meta_title["content"])
    if not result["title"]:
        title_tag = soup.find("title")
        if title_tag:
            t = title_tag.get_text(strip=True)
            t = re.split(r"\s*[-_—|]\s*(?:公告公示|环境影响评价信息|环境核查审批|项目受理|公示公告|新闻|信息公示|网站|人民政府|生态环境局)", t)[0].strip()
            result["title"] = t
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        result["publish_date"] = meta_date["content"][:10]
    if not result["publish_date"]:
        m = re.search(r"20\d{2}[-/年]\d{1,2}[-/月]\d{1,2}", html[:8000])
        if m:
            result["publish_date"] = m.group(0).replace("年", "-").replace("月", "-").replace("/", "-")[:10]
    content_div = soup.find("div", class_="doc-content")
    if not content_div:
        content_div = soup.find("div", class_=re.compile("TRS_UEDITOR|trs_editor_view")) or soup.find("div", class_="view")
    if content_div:
        result["content"] = extract_clean_text(str(content_div))
    return result


def crawl(pages=1, json_out=None):
    stored = skipped = 0
    seen_urls = set()
    for page in range(1, pages + 1):
        html = fetch_list(page)
        if not html:
            break
        items = parse_list(html)
        if not items:
            break
        for it in items:
            if it["url"] in seen_urls:
                continue
            seen_urls.add(it["url"])
            if it["pub_date"] and it["pub_date"] < _CUTOFF_DATE:
                continue
            detail = parse_detail(fetch(it["url"]) or "")
            if not detail["content"]:
                detail = {"title": it["title"], "publish_date": it["pub_date"], "content": ""}
            detail["title"] = detail["title"] or it["title"]
            detail["publish_date"] = detail["publish_date"] or it["pub_date"]
            _store(detail, it, json_out)
            if detail["content"]:
                stored += 1
            else:
                skipped += 1
        time.sleep(0.3)
    log(f"Result: {stored} stored, {skipped} skipped")


def _store(detail, it, json_out):
    row = {
        "site_name": SITE_NAME,
        "source_url": it["url"],
        "page_url": it["url"],
        "title": detail["title"],
        "publish_date": detail["publish_date"],
        "content": detail["content"],
        "summary": (detail["content"] or "")[:500],
        "category": CATEGORY,
        "script_name": SCRIPT_NAME,
    }
    if json_out:
        with open(json_out, "a", encoding="utf-8") as f:
            f.write(json.dumps(row, ensure_ascii=False) + "\n")
        return
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()
    try:
        c.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, source_url, page_url, title, publish_date, content, summary, category, script_name)"
            " VALUES (?,?,?,?,?,?,?,?,?)",
            (row["site_name"], row["source_url"], row["page_url"], row["title"],
             row["publish_date"], row["content"], row["summary"], row["category"], row["script_name"]))
        conn.commit()
    except Exception:
        pass
    conn.close()


def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)


if __name__ == "__main__":
    json_out = None
    if "--json" in sys.argv:
        json_out = "/tmp/crawl_test.jsonl"
    crawl(_PAGES, json_out)
