#!/usr/bin/env python3
import os
import sys, re, json, time, sqlite3, subprocess
from datetime import datetime, timedelta
BASE = "http://www.wschina.gov.cn"
LIST_URL = "/kfswsxwz/c00511/pc/list.html"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
SITE_NAME = "尉氏县-环境保护"
def fetch(url, retries=3):
    for i in range(retries):
        try:
            r = subprocess.run(["curl", "-sL", "--max-time", "15", url], capture_output=True, timeout=20)
            if r.returncode == 0 and r.stdout:
                return r.stdout.decode("utf-8", errors="replace")
        except: pass
        time.sleep(2)
    return ""
def extract_detail(html, base_url=BASE):
    title = pub_date = ""
    m = re.search(r'<meta name="ArticleTitle"[^>]*content="([^"]*)"', html)
    if m: title = m.group(1).strip()
    m = re.search(r'<meta name="PubDate"[^>]*content="([^"]*)"', html)
    if m: pub_date = m.group(1).strip()
    content = ""
    m = re.search(r'<div class="article-content[^"]*"[^>]*id="zoomcon"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m: content = m.group(1).strip()
    # Fix relative image URLs -> absolute
    if content:
        def _fix_img(m):
            tag = m.group(0)
            s = re.search(r'src="([^"]*)"', tag)
            if s:
                src = s.group(1)
                if src.startswith('http://') or src.startswith('https://'):
                    return tag
                ab = (base_url.rstrip('/') + '/' + src.lstrip('/')) if not src.startswith('/') else (base_url.rstrip('/') + src)
                tag = tag.replace('src="' + src + '"', 'src="' + ab + '"')
            return tag
        content = re.sub(r'<img[^>]*>', _fix_img, content)
    return title, pub_date, content
def main():
    html = fetch(BASE + LIST_URL)
    if not html: print("Failed"); return
    m = re.search(r'var articleList = (\[.*?\]);', html, re.DOTALL)
    if not m: print("No articleList"); return
    articles = json.loads(m.group(1))
    print("Total: {}".format(len(articles)))
    db = sqlite3.connect(DB_PATH)
    total = 0
    for a in articles:
        if a["pubDate"][:10] < CUTOFF: continue
        detail_url = BASE + json.loads(a["urls"])["pc"]
        detail_html = fetch(detail_url)
        if not detail_html: continue
        _, _, content = extract_detail(detail_html)
        if not content: content = "<p>" + a.get("summary", "") + "</p>"
        summary = re.sub(r'<[^>]+>', "", content)[:200].strip()
        sql = "INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary) VALUES (?, ?, ?, ?, ?, ?, ?, ?)"
        try:
            db.execute(sql, (int(a["id"]), a["title"], content, a["pubDate"][:10], detail_url, detail_url, SITE_NAME, summary))
            total += 1
        except Exception as e:
            print("  Error {}: {}".format(a["id"], e))
    db.commit(); db.close()
    print("Done! {} records".format(total))
if __name__ == "__main__":
    main()
