#!/usr/bin/env python3
import os
"""Crawl 潍坊滨海经济技术开发区 - 公告公示"""
import sys, re, json, time, sqlite3
from datetime import datetime, timedelta
from urllib.request import Request, urlopen
from urllib.error import HTTPError, URLError

BASE = "http://www.wfbinhai.gov.cn"
API_URL = BASE + "/els-service/article/{page}/15"
CATAID = "1839936166166663168"  # 公告公示
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "潍坊滨海-公告公示"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch_api(page):
    data = json.dumps({"catas": [CATAID], "order": "up"}).encode()
    req = Request(API_URL.format(page=page), data=data,
                  headers={"Content-Type": "application/json;charset=utf-8"})
    try:
        with urlopen(req, timeout=15) as resp:
            return json.loads(resp.read())
    except Exception as e:
        print(f"  API error page {page}: {e}")
        return None

def fetch(url, retries=3):
    for i in range(retries):
        try:
            with urlopen(url, timeout=15) as resp:
                return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
    return ""

def extract_detail(html, base_url=BASE):
    title = date = ""
    m = re.search(r'<meta name="ArticleTitle"[^>]*content="([^"]*)"', html)
    if m: title = m.group(1).strip()
    m = re.search(r'<meta name="PubDate"[^>]*content="([^"]*)"', html)
    if m: date = m.group(1).strip()[:10]
    content = ""
    m = re.search(r'<div[^>]*id="ozoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m: content = m.group(1).strip()
    if content:
        def _fix_src(m):
            tag = m.group(0)
            s = re.search(r'src="([^"]*)"', tag)
            if s:
                src = s.group(1)
                if src.startswith("http://") or src.startswith("https://") or src.startswith("data:"):
                    return tag
                ab = (base_url.rstrip("/") + "/" + src.lstrip("/")) if not src.startswith("/") else (base_url.rstrip("/") + src)
                tag = tag.replace('src="' + src + '"', 'src="' + ab + '"')
            return tag
        content = re.sub(r'<img[^>]*>', _fix_src, content)
    return title, date, content

def main():
    result = fetch_api(1)
    if not result or not result.get("success"):
        print("Failed to fetch API")
        return
    total = result["data"]["elementsTotal"]
    total_pages = result["data"]["totalPages"]
    print(f"Total articles: {total}, pages: {total_pages}")

    db = sqlite3.connect(DB_PATH)
    total_inserted = 0
    processed = 0

    for page in range(1, total_pages + 1):
        result = fetch_api(page)
        if not result:
            continue
        articles = result["data"]["contents"]
        if not articles:
            break

        all_old = True
        for a in articles:
            if a["fwdate"][:10] >= CUTOFF:
                all_old = False
                break
        if all_old:
            print(f"  Page {page}: all older than {CUTOFF}, stopping")
            break

        for a in articles:
            fwdate = a["fwdate"][:10]
            if fwdate < CUTOFF:
                continue

            # Parse xxid (some have "xx" prefix)
            raw_id = a["xxid"]
            try:
                record_id = int(raw_id)
            except ValueError:
                # Strip non-numeric prefix
                clean = re.sub(r'^[a-zA-Z]+', '', raw_id)
                if not clean:
                    print(f"  Skip {raw_id}: non-numeric id")
                    continue
                record_id = int(clean)

            detail_url = f"{BASE}/{a['dq']}/{a['dwid']}/{a['xxid']}.html"
            exists = db.execute("SELECT 1 FROM gov_raw WHERE id=?", (record_id,)).fetchone()
            if exists:
                processed += 1
                continue

            html = fetch(detail_url)
            if not html:
                print(f"  Skip {a['subject'][:30]}... (fetch failed)")
                processed += 1
                continue

            title, pub_date, content = extract_detail(html)
            if not title and not content:
                print(f"  Skip {a['subject'][:30]}... (404/content empty)")
                processed += 1
                continue
            if not title:
                title = a["subject"]
            if not pub_date:
                pub_date = fwdate
            if not content:
                content = f"<p>{title}</p>"

            summary = re.sub(r'<[^>]+>', "", content)[:200].strip()
            try:
                db.execute(
                    "INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (record_id, title, content, pub_date, detail_url, detail_url, SITE_NAME, summary)
                )
                total_inserted += 1
            except Exception as e:
                print(f"  DB Error {a['xxid']}: {e}")

            time.sleep(0.2)

        db.commit()
        processed += len(articles)
        print(f"  Page {page}/{total_pages}: {total_inserted} inserted, {processed} processed")

    db.close()
    print("\nDone! {total_inserted} new records inserted")
if __name__ == "__main__":
    main()
