#!/usr/bin/env python3
import os
"""上海市建设项目环评信息公开平台 - e2.sthj.sh.gov.cn:8081"""
import re, time, json, urllib.request, urllib.error, urllib.parse, sqlite3, ssl
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE = "https://e2.sthj.sh.gov.cn:8081/qygkweb"
LIST_URL = BASE + "/jsp/view/hjxxgk/jsxmbpq_list.jsp"
DETAIL_URL = BASE + "/jsxmxxgk/eiareport/action/jsxm_eiaReportDetail.do"
SITE_NAME = "上海市建设项目环评公示"
CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
DELAY = 1.0
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url, data=None, retries=3):
    for i in range(retries):
        try:
            if data:
                req = urllib.request.Request(url, data=data.encode(), headers=HEADERS)
            else:
                req = urllib.request.Request(url, headers=HEADERS)
            with urllib.request.urlopen(req, context=ctx, timeout=20) as r:
                return r.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(DELAY * (i+1))
                continue
            print("  [WARN] " + str(e)[:80], flush=True)
            return None

def parse_list(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.select("a[onclick*=openInfo]"):
        onclick = a.get("onclick", "")
        m_id = re.search(r"openInfo\('([^']+)'", onclick)
        m_type = re.search(r"openInfo\('[^']+','([^']+)'", onclick)
        if not m_id:
            continue
        uid = m_id.group(1)
        ptype = m_type.group(1) if m_type else "报告表"

        item_div = a.find("div", class_="item")
        if not item_div:
            continue

        name_el = item_div.find("div", class_="name")
        title = name_el.get_text(strip=True) if name_el else ""

        date_el = item_div.find("div", class_="id-item")
        date = ""
        if date_el:
            m_d = re.search(r"发布日期[：:](\d{4}-\d{2}-\d{2})", date_el.get_text())
            if m_d:
                date = m_d.group(1)

        district_el = item_div.find("span", class_="i-qy")
        district = district_el.get_text(strip=True) if district_el else ""

        js_el = item_div.find("div", class_="js")
        location = ""
        company = ""
        if js_el:
            txt = js_el.get_text(strip=True)
            m_c = re.search(r"建设单位[：:]\s*(.*?)(?:\s*$|\s*建设地点)", txt)
            if m_c:
                company = m_c.group(1).strip()
            m_l = re.search(r"建设地点[：:]\s*(.*?)(?:\s*建设单位)", txt)
            if m_l:
                location = m_l.group(1).strip()

        detail_url = DETAIL_URL + "?stEiaId=" + uid + "&type=" + urllib.parse.quote(ptype)
        full_title = title + "（" + district + "）" if district else title

        if date and date >= CUTOFF:
            items.append((detail_url, uid, full_title, date, company, location, district))
    return items

def parse_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    title_el = soup.select_one("div.bTitle")
    title = title_el.get_text(strip=True) if title_el else ""

    content_parts = []
    for step in soup.select("div.bstep"):
        content_parts.append(str(step))
    content = "\n".join(content_parts)

    # Find download buttons
    attach_links = []
    for btn in soup.select("input.btnDown"):
        onclick = btn.get("onclick", "")
        m = re.search(r"filedown\('([^']+)'", onclick)
        if m:
            attach_type = m.group(1)
            parent = btn.find_parent("li")
            label_text = ""
            if parent:
                label_text = parent.get_text(strip=True).split("：")[-1].replace("下载", "").strip()
            else:
                label_text = attach_type
            attach_links.append('<a href="' + url + '&filedown=' + attach_type + '">' + label_text + '</a>')

    if attach_links:
        content += '<div class="attachments">' + "<br/>".join(attach_links) + "</div>"

    return title, content

def save(items):
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    n = 0
    for url, uid, title, date, company, loc, dist in items:
        try:
            summary = "建设单位：" + company + " | 建设地点：" + loc if company else ""
            c.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (page_url, title, publish_date, site_name, source_url, summary, content)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (url, title, date, SITE_NAME, url, summary, ""))
            if c.rowcount > 0:
                n += 1
        except Exception as e:
            print("  [DB] " + str(e)[:60], flush=True)
    conn.commit()
    conn.close()
    return n

def update_content(items):
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    n = 0
    total = len(items)
    for i, (url, uid, title, date, company, loc, dist) in enumerate(items, 1):
        html = fetch(url)
        if not html:
            continue
        dt, dc = parse_detail(html, url)
        if dc:
            final_title = dt or title
            summary = "建设单位：" + company + " | 建设地点：" + loc if company else ""
            c.execute("UPDATE gov_raw SET title=?, content=?, summary=? WHERE page_url=?",
                     (final_title, dc, summary, url))
            if c.rowcount > 0:
                n += 1
        print("  [sh] 详情 " + str(i) + "/" + str(total) + ": " + title[:30] + "... " + str(len(dc)) + "字", flush=True)
        time.sleep(DELAY)
    conn.commit()
    conn.close()
    return n

def main():
    print("[sh] 上海市建设项目环评 - 截止: " + CUTOFF, flush=True)

    all_items = []
    for page in range(1, 7):
        data = "currentPage=" + str(page)
        print("[sh] 列表 " + str(page) + ": POST page=" + str(page), flush=True)
        html = fetch(LIST_URL, data)
        if not html:
            continue
        items = parse_list(html)
        if not items:
            print("  [END] 空", flush=True)
            break
        dates = sorted(set(d for _,_,_,d,_,_,_ in items))
        print("  -> " + str(len(items)) + "条, " + dates[0] + " ~ " + dates[-1], flush=True)
        all_items.extend(items)
        time.sleep(DELAY)

    print("\n[sh] 共 " + str(len(all_items)) + " 条", flush=True)

    saved = save(all_items)
    print("[sh] 基础信息入库 " + str(saved) + " 条", flush=True)

    updated = update_content(all_items)
    print("[sh] 正文更新 " + str(updated) + " 条", flush=True)

    conn = sqlite3.connect(DB_PATH)
    try:
        conn.execute("INSERT INTO gov_search(gov_search,rowid,title,site_name,summary) VALUES('rebuild',0,'','','')")
    except:
        pass
    conn.commit()
    conn.close()

    print("[sh] ✅ 完成，共 " + str(len(all_items)) + " 条", flush=True)

if __name__ == "__main__":
    main()
