#!/usr/bin/env python3
"""
上海市生态环境局 - 建设项目环境影响评价受理信息公示
仅爬列表页元数据 + PDF URL，正文由 PDF 内嵌链接承载
"""
import os, sys, re, logging, urllib.request, urllib.parse, sqlite3
from datetime import date

logging.basicConfig(level=logging.INFO, format='%(asctime)s [%(levelname)s] %(message)s')
log = logging.getLogger(__name__)

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "上海市生态环境局-环评受理"
LIST_URL = "https://link.sthj.sh.gov.cn/shhj/fa/cms/shhj/hpsl_list_login.jsp"
PDF_BASE = "https://link.sthj.sh.gov.cn/file/exchange_file/"
CUTOFF = "2023-06-18"
MAX_PAGE = 156

def fetch(page_no):
    data = urllib.parse.urlencode({'pageNo': page_no}).encode()
    req = urllib.request.Request(LIST_URL, data=data, method='POST')
    req.add_header('User-Agent', 'Mozilla/5.0')
    try:
        r = urllib.request.urlopen(req, timeout=20)
        return r.read().decode('utf-8', errors='replace')
    except Exception as e:
        log.warning(f"Page {page_no}: {e}")
        return None

def parse(html):
    recs = []
    for row in re.findall(r'<tr[^>]*height="79"[^>]*>(.*?)</tr>', html, re.DOTALL):
        tds = re.findall(r'<td[^>]*title="([^"]*)"', row)
        if len(tds) < 1:
            continue
        title = tds[0].strip()
        company = tds[1].strip() if len(tds) >= 2 else ""
        dr = tds[2].strip() if len(tds) >= 3 else ""
        pd_ = dr.split('~')[0].strip() if '~' in dr else ""
        pdf_m = re.search(r"openPdf\('([^']*)'\)", row)
        pdf_file = pdf_m.group(1) if pdf_m else ""
        pdf_url = PDF_BASE + urllib.parse.quote(pdf_file) if pdf_file else ""
        recs.append({'title': title, 'company': company, 'pub_date': pd_,
                     'date_range': dr, 'pdf_file': pdf_file, 'pdf_url': pdf_url})
    return recs

def main():
    log.info("上海生态环境局 - 环评受理")
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    
    new = 0
    skip = 0
    
    for pn in range(1, MAX_PAGE + 1):
        html = fetch(pn)
        if not html:
            continue
        recs = parse(html)
        if (pn - 1) % 20 == 0:
            log.info(f"Page {pn}/{MAX_PAGE} - new: {new}")
        
        for r in recs:
            if r['pub_date'] and r['pub_date'] < CUTOFF:
                continue
            
            # 用 page_url (PDF) 去重，无PDF时用title确保唯一
            pu = r['pdf_url'] or (LIST_URL + '?title=' + urllib.parse.quote(r['title']))
            if db.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (pu,)).fetchone():
                skip += 1
                continue
            
            summary = f"项目名称: {r['title']}\n建设单位: {r['company']}\n公示时间: {r['date_range']}"
            from html import escape as esc
            content = f'<a href="{esc(r["pdf_url"])}" target="_blank">{esc(r["title"])}</a>' if r['pdf_url'] else summary
            
            db.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date, summary, content, status, category)
                VALUES (?,?,?,?,?,?,?,'normal','环评受理')""",
                (SITE_NAME, LIST_URL, pu, r['title'],
                 r['pub_date'] or date.today().isoformat(), summary, content))
            if db.total_changes > 0:
                new += 1
            db.commit()
    
    db.close()
    log.info(f"Done! New: {new}, DupSkip: {skip}")

if __name__ == '__main__':
    main()
