#!/usr/bin/env python3
"""扬州市生态环境局 - 环评审批"""
import re, sys, os, json, urllib.parse
import urllib.request, urllib.error
from bs4 import BeautifulSoup
import sqlite3

BASE_URL = "https://sthj.yangzhou.gov.cn"
LIST_URL = BASE_URL + "/zfxxgk/fdzdgk/ywgz/hpsp/index.html"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE_NAME = "扬州市-环评审批"
SITE_DISPLAY = "扬州市生态环境局"
GROUP_NAME = "环评平台"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=30).read().decode("utf-8", errors="replace")
    except Exception as e:
        print(f"[WARN] fetch failed: {url} - {e}", file=sys.stderr)
        return ""

def fetch_api(page_no, page_size=14):
    """Fetch list page via API"""
    params = {
        "parseType": "bulidstatic",
        "webId": "5biGb8ANMvmWd52WUVRQu",
        "tplSetId": "Y3ymqFLmqouIDOIdPh2n8",
        "pageType": "column",
        "tagId": "ajax分页",
        "editType": "null",
        "pageId": "EpQ6v8pVFSqK88eqbxaXA",
        "paramJson": json.dumps({"pageNo": page_no, "pageSize": page_size}, ensure_ascii=False)
    }
    url = API_URL + "?" + urllib.parse.urlencode(params)
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30).read().decode("utf-8", errors="replace")
        data = json.loads(resp)
        return data.get("data", {}).get("html", "")
    except Exception as e:
        print(f"[WARN] API fetch failed page {page_no}: {e}", file=sys.stderr)
        return ""

def extract_items(html):
    """Extract items from API HTML"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for a in soup.select("ul.ul-list > li > a[href]"):
        href = a.get("href", "").strip()
        title = a.get("title", "") or ""
        ps = a.find_all("p")
        date_text = ""
        if len(ps) >= 2:
            if not title:
                title = ps[0].get_text(strip=True)
            date_text = ps[1].get_text(strip=True)
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append({"title": title.strip(), "url": href, "date": date_text})
    return items

def get_total_info(html):
    """Get total count and page size from pagination attributes"""
    soup = BeautifulSoup(html, "html.parser")
    pagination = soup.select_one("div.pagination")
    if not pagination:
        return 0, 14
    count_str = pagination.get("count", "0")
    rows_str = pagination.get("rows", "14")
    try:
        return int(count_str), int(rows_str)
    except:
        return 0, 14

def parse_detail(html, url):
    """Parse detail page content"""
    soup = BeautifulSoup(html, "html.parser")
    
    # Title
    title = ""
    title_div = soup.select_one("div.artTitle")
    if title_div:
        title = title_div.get_text(strip=True)
    if not title:
        meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta_title and meta_title.get("content"):
            title = meta_title["content"].strip()
    
    # Date from 生成日期
    date_text = ""
    for li in soup.select("ul.xxgk_head li"):
        bt = li.find("span", class_="xxgkBt")
        con = li.find("span", class_="xxgkCon")
        if bt and con and "生成日期" in bt.get_text():
            m = re.search(r"(\d{4}-\d{2}-\d{2})", con.get_text())
            if m:
                date_text = m.group(1)
                break
    if not date_text:
        meta_date = soup.find("meta", attrs={"name": "PubDate"})
        if meta_date and meta_date.get("content"):
            m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_date["content"])
            if m:
                date_text = m.group(1)
    
    # Content
    content_div = soup.select_one("div.artTxt")
    if not content_div:
        content_div = soup.select_one("div.bt-content")
    
    if not content_div:
        print(f"[WARN] No content found: {url}", file=sys.stderr)
        return {"title": title, "date": date_text, "content": "", "attachments": []}
    
    attachments = []
    parts = []
    
    for el in content_div.find_all(["p", "table", "img"], recursive=True):
        if el.name == "p" and el.find_parent("table"):
            continue
        if el.name == "table" and el.find_parent("table"):
            continue
        
        if el.name == "p":
            text = el.get_text(" ", strip=True)
            # Check for attachment links
            for a in el.find_all("a", href=True):
                href = a["href"].strip()
                if "download" in href or "fileUrl" in href or re.search(r'\.(zip|pdf|docx?|xlsx?|rar)$', href, re.I):
                    atitle = a.get_text(strip=True) or a.get("download", "")
                    if not href.startswith("http"):
                        href = BASE_URL + href
                    if not any(att["url"] == href for att in attachments):
                        attachments.append({"title": atitle, "url": href})
            if text:
                parts.append(text)
        
        elif el.name == 'table':
            tbl_html = html_table_to_html(el, url)
            if tbl_html:
                parts.append(tbl_html)
        elif el.name == "img":
            src = el.get("src", "")
            alt = el.get("alt", "")
            if src:
                if src.startswith("//"):
                    src = "https:" + src
                elif src.startswith("/"):
                    src = BASE_URL + src
                elif not src.startswith("http"):
                    src = url.rsplit("/", 1)[0] + "/" + src
                parts.append(f"![{alt}]({src})" if alt else f"![]({src})")
    
    # Deduplicate
    seen = set()
    unique_attachments = []
    for att in attachments:
        if not isinstance(att, dict):
            unique_attachments.append(att)
            continue
        if att["url"] not in seen:
            seen.add(att["url"])
            unique_attachments.append(att)
    
    content = "\n\n".join(parts)
    return {"title": title, "date": date_text, "content": content, "attachments": unique_attachments}


def main():
    # First API call to get page 1 data and total page info
    first_html = fetch_api(1, 14)
    if not first_html:
        print("[ERROR] Cannot fetch first page via API", file=sys.stderr)
        sys.exit(1)
    
    total_count, page_size = get_total_info(first_html)
    total_pages = (total_count + page_size - 1) // page_size if total_count > 0 else 1
    print(f"[INFO] Total count: {total_count}, page size: {page_size}, total pages: {total_pages}", file=sys.stderr)
    
    if _MAX_PG and total_pages > _MAX_PG:
        print(f"[INFO] Limiting to {_MAX_PG} pages (--pages={_MAX_PG})", file=sys.stderr)
        total_pages = _MAX_PG
    
    all_items = extract_items(first_html)
    print(f"[INFO] Found {len(all_items)} items on page 1/{total_pages}", file=sys.stderr)
    
    for pg in range(2, total_pages + 1):
        print(f"[INFO] Fetching page {pg}/{total_pages}", file=sys.stderr)
        html = fetch_api(pg, page_size)
        if not html:
            print(f"[WARN] Empty response for page {pg}", file=sys.stderr)
            continue
        items = extract_items(html)
        print(f"[INFO] Found {len(items)} items on page {pg}", file=sys.stderr)
        all_items.extend(items)
    
    print(f"[INFO] Total items from list: {len(all_items)}", file=sys.stderr)
    
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    
    new_count = 0
    empty_content = 0
    total_attachments = 0
    
    for item in all_items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
        if c.fetchone():
            continue
        
        print(f"[INFO] Fetching detail: {item['title'][:50]}...", file=sys.stderr)
        html = fetch(item["url"])
        if not html:
            print(f"[WARN] Cannot fetch detail, skipping: {item['url']}", file=sys.stderr)
            continue
        
        detail = parse_detail(html, item["url"])
        
        title = detail["title"] or item["title"]
        date_text = detail["date"] or item["date"]
        content = detail["content"]
        attachments = detail["attachments"]
        
        plain = re.sub(r'<[^>]+>', '', content)
        plain = re.sub(r'\s+', ' ', plain).strip()
        summary = plain[:200] if plain else ""
        
        if not content:
            empty_content += 1
            print(f"[WARN] Empty content: {title}", file=sys.stderr)
        
        date_rank = 0
        if date_text:
            m = re.search(r"(\d{4})-(\d{2})-(\d{2})", date_text)
            if m:
                date_rank = int(m.group(1) + m.group(2) + m.group(3))
        
        att_json = json.dumps(attachments, ensure_ascii=False) if attachments else "[]"
        
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank) VALUES (?,?,?,?,?,?,?,?,?)",
                (title, SITE_NAME, GROUP_NAME, item["url"], date_text, content, summary, att_json, date_rank)
            )
            if c.rowcount > 0:
                new_count += 1
                total_attachments += len(attachments)
        except Exception as e:
            print(f"[ERROR] DB insert failed: {title[:30]} - {e}", file=sys.stderr)
    
    conn.commit()
    conn.close()
    
    print(f"\n[RESULT] {SITE_NAME}: {new_count} new, {empty_content} empty content, {total_attachments} attachments")


if __name__ == "__main__":
    main()
