#!/usr/bin/env python3
"""Crawler for lucheng.gov.cn - 建设项目环境影响评价信息公示
CMS: Hanweb JPaaS (浙江网站集群)
List: API /api-gateway/jpaas-publish-server/front/page/build/unit
Detail: td.bt_content (使用深度匹配确保取出完整TD内容)
"""

import requests, re, sqlite3, sys, time
from datetime import datetime

API_URL = "https://www.lucheng.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE_URL = "https://www.lucheng.gov.cn"
DOMAIN = "www.lucheng.gov.cn"
SITE_NAME = "温州市鹿城区-建设项目环境影响评价信息公示"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36", "Referer": f"{SITE_URL}/col/col1229866057/index.html"}
DB_PATH = "/root/search.db"
INCREMENTAL = "--incremental" in sys.argv

API_PARAMS = {
    "parseType": "bulidstatic", "webId": "2634",
    "tplSetId": "diy36yBhJaoIH8wRG28xo", "pageType": "column",
    "tagId": "2021栏目列表", "editType": "null", "pageId": "1229866057"
}

def get_text(html):
    return re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', '', html)).strip()

def extract_td_content(html):
    """深度匹配: 找到第一个 td.bt_content 后, 数 <td>/</td> 深度找出正确闭合位置."""
    m = re.search(r'<td[^>]*class="bt_content"[^>]*>', html)
    if not m:
        return ""
    start = m.end()
    depth = 1
    pos = start
    while depth > 0 and pos < len(html):
        next_open = html.find('<td', pos)
        next_close = html.find('</td>', pos)
        if next_close == -1:
            break
        if next_open != -1 and next_open < next_close:
            depth += 1
            pos = next_open + 4
        else:
            depth -= 1
            if depth == 0:
                pos = next_close
            else:
                pos = next_close + 5
    return html[start:pos]

def fetch_list():
    r = requests.get(API_URL, params=API_PARAMS, headers=HEADERS, timeout=30, verify=False)
    dd = r.json()
    h = dd.get("data", {}).get("html", "")
    items = []
    for m in re.finditer(r'<div[^>]*class="news-item"[^>]*>.*?<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"[^>]*>.*?</a><span>([^<]+)</span>', h, re.DOTALL):
        path = m.group(1).strip()
        url = SITE_URL + path if path.startswith('/') else SITE_URL + '/' + path
        items.append((m.group(2).strip(), url, m.group(3).strip()))
    return items

def parse_detail(html):
    content = extract_td_content(html)
    pub_date = ""
    m = re.search(r'发布日期：(\d{4}[-/]\d{2}[-/]\d{2})', html)
    if m: pub_date = m.group(1).strip()
    atts = []
    for am in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>(.*?)</a>', html, re.DOTALL | re.IGNORECASE):
        au = am.group(1).strip()
        if au.startswith('http'): att_url = au
        elif au.startswith('/'): att_url = SITE_URL + au
        else: att_url = SITE_URL + '/' + au
        atts.append((get_text(am.group(2)), att_url))
    return content, pub_date, atts

def save(items):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    ins = 0
    for title, url, ds, content, pd, atts in items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone(): continue
        ah = '<div class="attachments">' + '<br>'.join(f'<a href="{u}" target="_blank">{t}</a>' for t, u in atts) + '</div>' if atts else ''
        fc = content + '\n' + ah if ah else content
        summary = get_text(content)[:200] if content else title
        c.execute("INSERT INTO gov_raw(title,page_url,content,summary,publish_date,site_name) VALUES(?,?,?,?,?,?)",
                  (title, url, fc, summary, ds or pd, SITE_NAME))
        ins += 1
    conn.commit()
    conn.close()
    return ins

def main():
    print(f"=== {SITE_NAME} ===")
    items = fetch_list()
    print(f"List: {len(items)} items")
    all_items = []
    for i, (title, url, ds) in enumerate(items, 1):
        if INCREMENTAL:
            conn = sqlite3.connect(DB_PATH, timeout=60)
            cu = conn.cursor()
            if cu.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone():
                conn.close()
                continue
            conn.close()
        print(f"  [{i}/{len(items)}] {title[:50]}...")
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        content, pd, atts = parse_detail(r.text)
        if not content: content = title
        all_items.append((title, url, ds, content, pd, atts))
        time.sleep(0.3)
    if all_items:
        n = save(all_items)
        print(f"Saved: {n} new items")
    else:
        print("No new items")

if __name__ == "__main__":
    main()
