#!/usr/bin/env python3
"""
重庆市合川区双槐镇 - 其他文件
TRS CMS, 单页无分页
https://www.hc.gov.cn/bmjd/jz/shz_100615/zwgk_100619/fdzdgknr_100621/zcwj_100623/qtgw_101432/
"""
import requests
import re
import sys
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
BASE_URL = "https://www.hc.gov.cn"
LIST_URL = "https://www.hc.gov.cn/bmjd/jz/shz_100615/zwgk_100619/fdzdgknr_100621/zcwj_100623/qtgw_101432/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}
SITE_NAME = "合川区双槐镇其他文件"
SITE_URL = LIST_URL


def get_list(session):
    """获取列表页全部条目"""
    r = session.get(LIST_URL, headers=HEADERS, timeout=30, verify=False)
    r.encoding = "utf-8"
    html = r.text

    items = re.findall(
        r'<li>\s*<a href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span>([^<]*)</span>',
        html, re.DOTALL
    )
    results = []
    for href, title, date_str in items:
        full_url = urljoin(LIST_URL, href)
        results.append({
            "title": title.strip(),
            "url": full_url,
            "date": date_str.strip(),
        })
    return results


def get_detail(session, url):
    """获取详情页内容"""
    try:
        r = session.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        return {"content": "", "error": str(e)}

    # TRS Editor 内容
    content_m = re.search(
        r'<div[^>]*class="trs_editor_view[^"]*"[^>]*>(.*?)</div>\s*</div>',
        html, re.DOTALL
    )
    content = content_m.group(1).strip() if content_m else ""

    if content:
        content = re.sub(r'\s+style="[^"]*"', '', content)
        content = re.sub(r'\s+class="[^"]*"', '', content)
        content = content.strip()

    # attachments
    atts = re.findall(
        r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip)[^"]*)"[^>]*>([^<]+)',
        html
    )
    att_html = ""
    if atts:
        att_html = '<div class="attachments"><p><strong>附件：</strong></p><ul>'
        for att_href, att_name in atts:
            att_url = urljoin(url, att_href)
            att_html += f'<li><a href="{att_url}" target="_blank">{att_name.strip()}</a></li>'
        att_html += '</ul></div>'

    if att_html:
        content += att_html

    return {"content": content, "error": ""}


def save_to_db(conn, items):
    cur = conn.cursor()
    saved = 0
    for item in items:
        title = item["title"]
        url = item["url"]
        content = item["content"]
        pub_date = item["date"]

        if not content:
            continue

        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if cur.fetchone():
            continue

        summary = re.sub(r'<[^>]+>', '', content)[:200]
        content_plain = re.sub(r'<[^>]+>', '', content).strip()

        cur.execute(
            "INSERT OR IGNORE INTO gov_raw (title, page_url, content, summary, site_name, publish_date, source_url, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (title, url, content, summary, SITE_NAME, pub_date, url, "环评公示"),
        )
        cur.execute(
            "INSERT OR IGNORE INTO gov_search_v3 (title, content, source_url, publish_date, site_name) VALUES (?, ?, ?, ?, ?)",
            (title, content_plain, url, pub_date, SITE_NAME),
        )
        saved += 1
    conn.commit()
    return saved


def main():
    import sqlite3
    full_mode = "--full" in sys.argv
    conn = sqlite3.connect(DB_PATH)
    session = requests.Session()

    items = get_list(session)
    print(f"List items: {len(items)}")

    if not full_mode:
        cur = conn.cursor()
        for item in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
            if cur.fetchone():
                print("Existing record found, skipping")
                conn.close()
                return

    for item in items:
        detail = get_detail(session, item["url"])
        if detail.get("error"):
            print(f"  ERROR {item['title'][:30]}: {detail['error']}")
            continue
        item["content"] = detail["content"]

    db_saved = save_to_db(conn, items)
    conn.close()
    print(f"Done. Crawled: {len(items)}, New: {db_saved}")


if __name__ == "__main__":
    main()
