#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
江苏润环环境科技有限公司(www.jsrainfine.com) - 环评项目公示栏目
URL: http://www.jsrainfine.com/inside/4/15.html
CMS: ASP.NET (AspNetPager 分页)
列表: ul.newsnav > li > span.day2[日期] + a[href=/detail/{N}.html][title]
分页: /inside/4/15/page{N}.html, 5页, 15条/页 ≈75条
详情: /detail/{N}.html -> 302 -> /detail.aspx?id={N}
  标题 div.detailhead > span#lblhead, 日期 发布时间：
  正文 div.align > span#ContentPlaceHolder1_lblcontent (含表格/附件)
生产库 schema: gov_raw(id, site_name, source_url, page_url, title, publish_date, date_rank,
  summary, status, category, visits, content, tags, industry, attachments, group_name, has_table, script_name)
gov_search: fts5(title, site_name, summary, tokenize=trigram) rowid 对齐 gov_raw.id
"""
import argparse
import hashlib
import os
import re
import sqlite3
import time
from html import unescape
from urllib.parse import urljoin

import requests

BASE_URL = "http://www.jsrainfine.com"
LIST_URL = BASE_URL + "/inside/4/15.html"
SITE_NAME = "江苏润环环境-环评项目公示"
SCRIPT_NAME = "crawl_jsrainfine_xmgs.py"
GROUP_NAME = "江苏"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def clean_title(t):
    """清洗标题: strip 实体前缀"""
    t = unescape(t or "")
    t = re.sub(r"^[\s\xa0·\u00b7]+", "", t)
    t = re.sub(r"\s*\.{3,}\s*$", "", t)
    t = re.sub(r"\s+", " ", t).strip()
    return t


def parse_list(html_text, base_url):
    """解析列表页, 返回 [(title, url, date), ...]"""
    items = []
    pat = re.compile(
        r'<span class="day2[^"]*">\[?(\d{4}-\d{2}-\d{2})\]?</span>\s*<a\s+href="([^"]+)"[^>]*title="([^"]*)"',
        re.S,
    )
    for m in pat.finditer(html_text):
        date, href, title_attr = m.group(1), m.group(2), m.group(3)
        href = urljoin(base_url, href)
        if "/detail/" not in href:
            continue
        title = clean_title(title_attr)
        if not title:
            continue
        items.append((title, href, date.strip()))
    seen = set()
    out = []
    for it in items:
        if it[1] not in seen:
            seen.add(it[1])
            out.append(it)
    return out


def parse_detail(html_text, url):
    """解析详情页, 返回 (title, date, content_html, attachments)"""
    # 标题
    title = ""
    m = re.search(r'<div class="detailhead"[^>]*>\s*<span[^>]*>(.*?)</span>', html_text, re.S)
    if m:
        title = clean_title(re.sub(r"<[^>]+>", "", m.group(1)))
    # 日期
    date = ""
    dm = re.search(r"发布时间：(\d{4}-\d{2}-\d{2})", html_text)
    if dm:
        date = dm.group(1)
    # 正文: span#ContentPlaceHolder1_lblcontent 内含 UEditor 内容(p+span+img+附件a)
    # 结构: <span id="ContentPlaceHolder1_lblcontent"><p><span>...</span></p>...<p><img/><a href="pdf">附件</a></p><p><br/></p></span>
    # 不能按 span 深度计数(内容内嵌 span), 用 "最后一个 </p></span>" 或 "<p><br/></p></span>" 作为结束
    cm = re.search(r'<span id="ContentPlaceHolder1_lblcontent"[^>]*>(.*?)</p></span>', html_text, re.S)
    if not cm:
        cm = re.search(r'<span id="ContentPlaceHolder1_lblcontent"[^>]*>(.*?)</span>', html_text, re.S)
    if not cm:
        return title, date, "", []
    body = cm.group(1)
    # 剥离 script/style
    body = re.sub(r"<script[\s\S]*?</script>", "", body)
    body = re.sub(r"<style[\s\S]*?</style>", "", body)
    # 附件
    atts = []
    for a in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', body, re.S):
        href, txt = a.group(1), re.sub(r"<[^>]+>", "", a.group(2)).strip()
        if re.search(r"\.(?:pdf|docx?|xlsx?|zip|rar|wps|et|dps)(?:\?|$)", href, re.I) or txt.startswith("附件"):
            atts.append((href, urljoin(url, href), txt or href.split("/")[-1]))
    return title, date, body, atts


def html_to_text(html_text, attachments):
    """HTML -> 文本, 保留表格、\\n\\n 分段、附件嵌入"""
    tables = []

    def _save_tbl(m):
        tables.append(m.group(0))
        return f"@@TABLE{len(tables) - 1}@@"

    html_text = re.sub(r"<table[\s\S]*?</table>", _save_tbl, html_text)
    # 附件链接先替换为占位, 避免被拍平 (匹配原始相对路径或绝对URL两种形式)
    att_placeholders = []
    for orig_href, abs_url, txt in attachments:
        ph = f"@@ATT{len(att_placeholders)}@@"
        att_placeholders.append((ph, abs_url, txt))
        pat = r'<a[^>]+href="' + re.escape(orig_href) + r'"[^>]*>.*?</a>'
        if abs_url != orig_href:
            pat += r'|<a[^>]+href="' + re.escape(abs_url) + r'"[^>]*>.*?</a>'
        html_text = re.sub(pat, ph, html_text, flags=re.S)
    html_text = re.sub(r"</?(?:p|div|tr|br|h[1-6]|li)[^>]*>", "\n", html_text)
    html_text = re.sub(r"</?(?:span|font|b|strong|i|em|u|o:p|o|a)[^>]*>", "", html_text)
    html_text = html_text.replace("&nbsp;", " ").replace("\xa0", " ")
    html_text = unescape(html_text)
    # 兜底: 剥离残留 script/style/尾标签
    html_text = re.sub(r"<script[\s\S]*?</script>", "", html_text)
    html_text = re.sub(r"<style[\s\S]*?</style>", "", html_text)
    html_text = re.sub(r"</?(?:html|body)[^>]*>", "", html_text)
    html_text = re.sub(r"\n{3,}", "\n\n", html_text)
    lines = [ln.strip() for ln in html_text.split("\n")]
    out = [ln for ln in lines if ln]
    text = "\n\n".join(out)
    for i, t in enumerate(tables):
        text = text.replace(f"@@TABLE{i}@@", t)
    text = re.sub(r"\n{3,}", "\n\n", text)
    # 附件占位恢复为名称内嵌URL
    for ph, abs_url, txt in att_placeholders:
        if ph in text:
            text = text.replace(ph, f'<a href="{abs_url}" target="_blank">{txt}</a>')
        else:
            text += f'\n\n<p><a href="{abs_url}" target="_blank">{txt}</a></p>'
    return text.strip()


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1, help="抓取页数")
    ap.add_argument("--db", default="/mnt/data/search.db", help="SQLite 数据库路径")
    ap.add_argument("--script-name", default=SCRIPT_NAME)
    args = ap.parse_args()

    db_path = args.db
    script_name = args.script_name
    if not db_path.startswith("/mnt/"):
        db_path = os.path.expanduser(db_path)

    conn = sqlite3.connect(db_path, timeout=60)
    cur = conn.cursor()
    new_count = 0
    dup_count = 0
    err_count = 0

    session = requests.Session()

    for page in range(1, args.pages + 1):
        url = LIST_URL if page == 1 else f"{BASE_URL}/inside/4/15/page{page}.html"
        try:
            r = session.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"[{page}] 请求失败 {url}: {e}")
            err_count += 1
            continue
        items = parse_list(r.text, url)
        print(f"[{page}] 列表 {len(items)} 条 {url}")
        if not items:
            continue
        for title, art_url, pub_date in items:
            try:
                cur.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=? AND script_name=?", (art_url, script_name))
                if cur.fetchone()[0] > 0:
                    dup_count += 1
                    continue
                dr = session.get(art_url, headers=HEADERS, timeout=30, allow_redirects=True)
                dr.encoding = "utf-8"
                d_title, d_date, content_html, atts = parse_detail(dr.text, art_url)
                if not content_html.strip():
                    print(f"  跳过空正文: {title[:30]}")
                    dup_count += 1
                    continue
                if d_title:
                    title = d_title
                if d_date:
                    pub_date = d_date
                content = html_to_text(content_html, atts)
                att_txt = "\n\n".join(a[1] for a in atts) if atts else ""
                summary = content[:500] if content else ""
                cur.execute(
                    "INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, status, category, visits, content, tags, industry, attachments, group_name, has_table, script_name) "
                    "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
                    (SITE_NAME, art_url, art_url, title, pub_date, 0, summary, "", "", 0, content, "",
                     "other", att_txt, GROUP_NAME, 1 if "<table" in content else 0, script_name),
                )
                # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                conn.commit()
                cur.execute(
                    "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                    (cur.lastrowid, title, SITE_NAME, summary),
                )
                conn.commit()
                new_count += 1
                print(f"  + {pub_date} {title[:40]}")
            except Exception as e:
                err_count += 1
                print(f"  ERR {title[:30]}: {e}")
        time.sleep(0.3)

    conn.close()
    print(f"完成: 新增 {new_count} / 重复 {dup_count} / 错误 {err_count}")


if __name__ == "__main__":
    main()
