#!/usr/bin/env python3
"""Crawl mingguang.gov.cn - 明光市-建设项目环境影响评价审批"""
import sys, re, json, sqlite3, os, ssl, urllib.request
from urllib.parse import urljoin

SITE_NAME = "明光市-建设项目环境影响评价审批"
BASE_URL = "https://www.mingguang.gov.cn"
ORGAN_ID = "161056936"
CAT_ID = "170077272"
PAGE_URL = lambda n: BASE_URL + "/public/column/" + ORGAN_ID + "?type=4&catId=" + CAT_ID + "&action=list&pageIndex=" + str(n)
OUTPUT_FILE = "/root/gov_crawler/mingguang_output.jsonl"
DB_PATH = "/mnt/data/search.db"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

def http_get(url, timeout=20):
    req = urllib.request.Request(url, headers={
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
        "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    })
    try:
        resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        return ""

def extract_articles(html):
    articles = []
    pat = re.compile(
        r'<a\s+href="(https?://[^"]*/public/' + re.escape(ORGAN_ID) + r'/(\d+)\.html)"'
        r'[^>]*?class="title[^"]*"[^>]*?title="([^"]*)"', re.IGNORECASE)
    for m in pat.finditer(html):
        u, t = m.group(1), m.group(3).strip()
        if t: articles.append((u, t))
    return articles

def extract_detail(html):
    tm = re.search(r'<h1[^>]*class="newstitle"[^>]*>(.*?)</h1>', html, re.DOTALL)
    title = re.sub(r'<[^>]+>', '', tm.group(1)).strip() if tm else ""
    dm = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
    date = dm.group(1) if dm else ""
    attachments = []
    cm = re.search(r'class="gkwz_contnet\s*j-fontContent"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if not cm:
        cm = re.search(r'class="gkwz_contnet[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    content = cm.group(1) if cm else ""
    if content:
        for a in re.finditer(r'<a\s+href="([^"]+)"[^>]*>([^<]+)</a>', content):
            href, text = a.group(1), a.group(2).strip()
            full = urljoin(BASE_URL, href)
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', href, re.I) or 'downfile' in href.lower():
                attachments.append({"name": text, "url": full})
    return title, date, content, json.dumps(attachments, ensure_ascii=False)

def main():
    max_pages = int(sys.argv[1]) if len(sys.argv) > 1 else 5
    total = existing = 0
    for page in range(1, max_pages + 1):
        url = PAGE_URL(page)
        html = http_get(url)
        if not html: break
        articles = extract_articles(html)
        if not articles: break
        for url, title in articles:
            try:
                dh = http_get(url)
                if not dh: continue
                dt, date, content, att_j = extract_detail(dh)
                dt = dt or title
                rec = {"site_name": SITE_NAME, "source_url": url, "page_url": url,
                       "title": dt, "publish_date": date, "content": content,
                       "summary": title[:200], "attachments": att_j}
                if os.path.exists(DB_PATH):
                    try:
                        db = sqlite3.connect(DB_PATH, timeout=60)
                        c = db.cursor()
                        c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=?", (url,))
                        if c.fetchone()[0] > 0:
                            existing += 1; total += 1; db.close(); continue
                        db.close()
                    except: pass
                with open(OUTPUT_FILE, "a", encoding="utf-8") as f:
                    f.write(json.dumps(rec, ensure_ascii=False) + "\n")
                total += 1
            except: pass
    print(total)

if __name__ == "__main__":
    main()
