#!/usr/bin/env python3
"""
江苏省发展和改革委员会 - 行政许可公示
URL: https://fzggw.jiangsu.gov.cn/col/col59371/index.html
CMS: HanWeb/TRS with jpage dataproxy
Detail: /art/YYYY/M/D/art_59371_ARTID.html
"""

import requests, re, sys, os, json, time
from bs4 import BeautifulSoup
from datetime import datetime

DB_PATH = "/root/search.db"
SITE_NAME = "江苏省发改委-行政许可公示"
BASE_URL = "https://fzggw.jiangsu.gov.cn"
API_URL = BASE_URL + "/module/web/jpage/dataproxy.jsp"

API_PARAMS = {
    "col": "1",
    "appid": "1",
    "webid": "3",
    "path": "/",
    "columnid": "59371",
    "sourceContentType": "1",
    "unitid": "394197",
    "webname": "江苏省发展和改革委员会",
    "permissiontype": "0",
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

# CLI
import argparse
parser = argparse.ArgumentParser()
parser.add_argument("--pages", type=int, default=0, help="Pages to crawl (0 = all)")
parser.add_argument("--verbose", "-v", action="store_true")
args = parser.parse_args()

MAX_PAGES = args.pages if args.pages > 0 else 99

def log(msg):
    if args.verbose:
        print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}")

def fetch_list_page(page_num):
    params = dict(API_PARAMS)
    params["page"] = str(page_num)
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            log(f"API page {page_num} returned {r.status_code}")
            return None, 0
        data = r.text
        total_m = re.search(r"<totalrecord>(\d+)", data)
        total = int(total_m.group(1)) if total_m else 0
        records = re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", data, re.DOTALL)
        return records, total
    except Exception as e:
        log(f"Error fetching page {page_num}: {e}")
        return None, 0

def parse_record(html_frag):
    soup = BeautifulSoup(html_frag, "html.parser")
    a_tag = soup.find("a")
    if not a_tag:
        return None
    href = a_tag.get("href", "")
    if not href or "/art/" not in href:
        return None
    title = a_tag.get("title", "") or a_tag.get_text(strip=True)
    if not title:
        return None
    title = re.sub(r"\s+", " ", title).strip()
    url = BASE_URL + href if href.startswith("/") else href
    tds = soup.find_all("td")
    date = ""
    if len(tds) >= 3:
        date = tds[2].get_text(strip=True).replace("/", "-")
    return {"title": title, "url": url, "date": date}

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        if r.status_code != 200:
            log(f"  Detail {url[-50:]} HTTP {r.status_code}")
            return None, "", ""
    except Exception as e:
        log(f"  Detail error: {e}")
        return None, "", ""

    soup = BeautifulSoup(r.text, "html.parser")

    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta["content"].strip() if meta else ""

    if not title and soup.title:
        t = soup.title.get_text(strip=True)
        t = re.sub(r"^江苏省发展和改革委员会\s*", "", t).strip()
        if "行政许可公示" in t:
            t = t.replace("行政许可公示", "").strip()
        title = t

    # Content extraction
    content = ""
    shbt = soup.find("div", class_="shbt_content")
    if shbt:
        parts = []
        for child in shbt.children:
            if not hasattr(child, "name"):
                continue
            if child.name == "div" and "atable" in (child.get("class", "") or []):
                table = child.find("table")
                if table:
                    parts.append(str(table))
                else:
                    txt = child.get_text(separator="\n", strip=True)
                    if txt and len(txt) > 3:
                        parts.append(txt)
            elif child.name in ("p", "div"):
                txt = child.get_text(strip=True)
                if txt and len(txt) > 3:
                    parts.append(txt)
        content = "\n\n".join(parts)
    else:
        for candidate in soup.find_all("div", class_=lambda c: c and "content" in c if c else False):
            txt = candidate.get_text(strip=True)
            if len(txt) > 50:
                content = txt
                break

    content = re.sub(r"[ \t]+", " ", content).strip()

    attachments = []
    for a_tag in soup.find_all("a", href=True):
        href = a_tag["href"]
        ext = os.path.splitext(href)[1].lower()
        if ext in (".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"):
            if href.startswith("/"):
                href = BASE_URL + href
            elif not href.startswith("http"):
                href = BASE_URL + "/" + href.lstrip("/")
            name = a_tag.get_text(strip=True) or os.path.basename(href)
            attachments.append({"name": name, "url": href})
            content += f"\n[附件] {name}: {href}"

    return title, content, json.dumps(attachments, ensure_ascii=False)

# Main
all_items = []
seen_urls = set()

log("Fetching list pages...")
for page_num in range(1, MAX_PAGES + 1):
    records, total = fetch_list_page(page_num)
    if records is None:
        break
    new_count = 0
    for frag in records:
        item = parse_record(frag)
        if not item or item["url"] in seen_urls:
            continue
        seen_urls.add(item["url"])
        all_items.append(item)
        new_count += 1
    log(f"  Page {page_num}: +{new_count} (total: {len(all_items)}/{total})")
    if new_count == 0:
        break

print(f"\n列表合计: {len(all_items)} 条")

if not all_items:
    print("No items to process.")
    sys.exit(0)

# Import DB lib
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
try:
    from crawler_lib import push_to_searchdb, get_connection
except ImportError:
    import sqlite3
    def push_to_searchdb(items, site_name):
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA journal_mode=WAL")
        conn.execute("PRAGMA busy_timeout=30000")
        c = conn.cursor()
        new = skip = 0
        for item in items:
            try:
                c.execute(
                    """INSERT OR IGNORE INTO gov_raw 
                    (site_name, source_url, page_url, title, publish_date, date_rank, summary, content, attachments)
                    VALUES (?,?,?,?,?,?,?,?,?)""",
                    (site_name, item.get("url",""), item.get("url",""),
                     item.get("title",""), item.get("pub_date",""),
                     item.get("date_rank",0), item.get("summary",""),
                     item.get("content",""), item.get("attachments","[]"))
                )
                if c.lastrowid and c.lastrowid > 0:
                    new += 1
                else:
                    skip += 1
            except:
                skip += 1
        conn.commit()
        conn.close()
        return new, skip
    def get_connection():
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA journal_mode=WAL")
        conn.execute("PRAGMA busy_timeout=30000")
        return conn

current_year = datetime.now().year
items_for_db = []

for idx, item in enumerate(all_items):
    log(f"  [{idx+1}/{len(all_items)}] {item['title'][:40]}...")
    title, content, attachments = fetch_detail(item["url"])
    if not title:
        title = item["title"]

    pub_date = item["date"]
    if not pub_date:
        date_m = re.search(r"/art/(\d{4})/(\d{1,2})/(\d{1,2})/", item["url"])
        if date_m:
            pub_date = f"{date_m.group(1)}-{int(date_m.group(2)):02d}-{int(date_m.group(3)):02d}"

    date_rank = 0
    if pub_date and len(pub_date) >= 10:
        try:
            dt = datetime.strptime(pub_date[:10], "%Y-%m-%d")
            date_rank = int(dt.strftime("%Y%m%d"))
        except:
            pass

    summary = (BeautifulSoup(content, "html.parser").get_text(strip=True) if "<" in content else content)[:500]

    items_for_db.append({
        "title": title, "url": item["url"], "content": content,
        "pub_date": pub_date, "date_rank": date_rank,
        "summary": summary, "source_url": item["url"],
        "attachments": attachments, "page_url": item["url"],
    })

# Pre-reserve content for summary filter
# (dummy, summary already extracted above)

print(f"\n入库中 ({len(items_for_db)} 条)...")
new, skip = push_to_searchdb(items_for_db, SITE_NAME)
print(f"入库完成: new={new}, skip={skip}")

conn = get_connection()
count = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
conn.close()
print(f"DB中 {SITE_NAME} 共 {count} 条")
