#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_jskydhb_xxgk.py — 江苏科易达环保科技股份有限公司「信息公开」栏目
http://www.jskydhb.com/index.php/List/index/cid/237.html
PHP ThinkPHP 风格:
- 列表: div.childcc > p > a[href=/Show/index/cid/237/id/{N}.html] + div.time(发布日期：[2026/06/15])
- 分页: /List/index/cid/237/p/{N}.html (5条/页, 共38页=189条)
- 详情: /Show/index/cid/237/id/{N}.html
    标题 div.news_content > h1, 日期 div.datetime, 正文 div.news_show
- 内容: 土壤污染状况调查报告/环评公示, 纯文本正文(无附件)
"""
import argparse
import html as html_lib
import os
import re
import sqlite3
import time
import urllib.parse

import requests

BASE = "http://www.jskydhb.com"
SITE_NAME = "江苏科易达环保"
SCRIPT_NAME = "crawl_jskydhb_xxgk.py"
GROUP_NAME = "江苏"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
TIMEOUT = 30


def clean_title(t):
    """标题清洗: strip &middot;&nbsp;&#32; 实体前缀与空白"""
    if not t:
        return ""
    t = t.strip()
    t = re.sub(r"&middot;|&nbsp;|&#160;|&#32;", "", t, flags=re.I)
    t = t.strip()
    t = re.sub(r"\s+", " ", t)
    return t


def html_to_text(html_seg):
    """HTML -> 文本, 保留 a 链接(内嵌URL)与表格, 去 script/style, \n\n 分段"""
    seg = html_seg
    links = []

    def _link_repl(m):
        href, txt = m.group(1), m.group(2)
        txt = re.sub(r"<[^>]+>", "", txt)
        txt = html_lib.unescape(txt).strip()
        if not txt or href.startswith("#") or href.startswith("javascript"):
            return txt or ""
        abs_url = urllib.parse.urljoin(BASE, href)
        links.append((href, abs_url, txt))
        return f"__LINK__{len(links)}__"

    seg = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, seg, flags=re.S | re.I)
    # 去 script/style
    seg = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", seg, flags=re.S | re.I)
    # 段落边界: 块级标签 -> \n\n
    seg = re.sub(r"</(p|div|li|tr|h[1-6]|br)>", "\n\n", seg, flags=re.I)
    seg = re.sub(r"<(p|div|li|tr|h[1-6]|br)[^>]*>", "\n", seg, flags=re.I)
    seg = re.sub(r"<[^>]+>", "", seg)
    seg = html_lib.unescape(seg)
    # 还原链接(内嵌URL)
    for i, (href, abs_url, txt) in enumerate(links, 1):
        seg = seg.replace(f"__LINK__{i}__", f'<a href="{abs_url}" target="_blank">{txt}</a>')
    seg = re.sub(r"\r", "", seg)
    seg = re.sub(r"\n{3,}", "\n\n", seg)
    seg = re.sub(r"[ \t\u00a0]+", " ", seg)
    return seg.strip(), links


def parse_list(html_src):
    """列表页 -> [(title, art_url, pub_date)]"""
    items = re.findall(
        r'<p><a href="(/index\.php/Show/index/cid/237/id/\d+\.html)"[^>]*>(.*?)</a></p>\s*<div class="time">发布日期[：:]\s*\[?(\d{4})/(\d{1,2})/(\d{1,2})\]?</div>',
        html_src, re.S)
    out = []
    for href, t, y, mo, d in items:
        art_url = urllib.parse.urljoin(BASE, href)
        title = clean_title(t)
        pub = "%s-%02d-%02d" % (y, int(mo), int(d))
        out.append((title, art_url, pub))
    return out


def parse_detail(html_src, art_url):
    """详情页 -> (title, publish_date, body_html, links)"""
    # 标题
    m = re.search(r'<div class="news_content">\s*<h1>(.*?)</h1>', html_src, re.S)
    title = clean_title(m.group(1)) if m else ""
    if not title:
        m = re.search(r"<title>(.*?)</title>", html_src, re.S)
        if m:
            title = clean_title(m.group(1).split("-")[0])
    # 日期 发布日期：2026/06/15
    m = re.search(r"发布日期[：:]\s*(\d{4})/(\d{1,2})/(\d{1,2})", html_src)
    publish = ""
    if m:
        publish = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
    # 正文: div.news_show 深度计数闭合(内部无嵌套div, 找第一个闭合)
    m = re.search(r'<div class="news_show">(.*?)</div>', html_src, re.S)
    body_html = ""
    if m:
        body_html = m.group(1).strip()
    return title or None, publish or None, body_html, []


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1, help="抓取页数")
    ap.add_argument("--db", default="/mnt/data/search.db", help="SQLite 数据库路径")
    ap.add_argument("--script-name", default=SCRIPT_NAME)
    args = ap.parse_args()

    db_path = args.db
    script_name = args.script_name
    if not db_path.startswith("/mnt/"):
        db_path = os.path.expanduser(db_path)

    conn = sqlite3.connect(db_path, timeout=60)
    cur = conn.cursor()
    new_count = 0
    dup_count = 0
    err_count = 0

    session = requests.Session()

    for page in range(1, args.pages + 1):
        if page == 1:
            url = f"{BASE}/index.php/List/index/cid/237.html"
        else:
            url = f"{BASE}/index.php/List/index/cid/237/p/{page}.html"
        try:
            r = session.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"[{page}] 请求失败 {url}: {e}")
            err_count += 1
            continue
        items = parse_list(r.text)
        print(f"[{page}] 列表 {len(items)} 条 {url}")
        if not items:
            continue
        for title, art_url, pub_date in items:
            try:
                cur.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=? AND script_name=?", (art_url, script_name))
                if cur.fetchone()[0] > 0:
                    dup_count += 1
                    continue
                dr = session.get(art_url, headers=HEADERS, timeout=TIMEOUT)
                dr.encoding = "utf-8"
                d_title, d_date, content_html, links = parse_detail(dr.text, art_url)
                if not content_html.strip():
                    print(f"  跳过空正文: {title[:30]}")
                    dup_count += 1
                    continue
                if d_title:
                    title = d_title
                if d_date:
                    pub_date = d_date
                content, links_out = html_to_text(content_html)
                att_txt = "\n\n".join(a[2] for a in links_out) if links_out else ""
                summary = content[:500] if content else ""
                cur.execute(
                    "INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, status, category, visits, content, tags, industry, attachments, group_name, has_table, script_name) "
                    "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
                    (SITE_NAME, art_url, art_url, title, pub_date, 0, summary, "", "", 0, content, "",
                     "other", att_txt, GROUP_NAME, 1 if "<table" in content else 0, script_name),
                )
                cur.execute(
                    "INSERT INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                    (cur.lastrowid, title, SITE_NAME, summary),
                )
                conn.commit()
                new_count += 1
                print(f"  + {pub_date} {title[:40]}")
            except Exception as e:
                err_count += 1
                print(f"  ERR {title[:30]}: {e}")
        time.sleep(0.3)

    conn.close()
    print(f"完成: 新增 {new_count} / 重复 {dup_count} / 错误 {err_count}")


if __name__ == "__main__":
    main()
