#!/usr/bin/env python3
"""Crawler for macheng.gov.cn - 双随机一公开
CMS: 黄冈市信息公开平台
List: ul.xxgk_nav_list > li > a.title + span.date
Detail: div.gkwz_contnet
"""

import requests, re, sqlite3, sys, time
from datetime import datetime

SITE_URL = "http://www.macheng.gov.cn"
LIST_URL = "http://www.macheng.gov.cn/zwgk/public/column/6635527?type=4&catId=7030669&action=list"
DOMAIN = "www.macheng.gov.cn"
SITE_NAME = "麻城市-双随机一公开"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = "/root/search.db"
INCREMENTAL = "--incremental" in sys.argv

def get_text(html):
    return re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', '', html)).strip()

def fetch_list():
    r = requests.get(LIST_URL, headers=HEADERS, timeout=30, verify=False)
    r.encoding = "utf-8"
    items = []
    # Find the xxgk_nav_list
    m = re.search(r'xxgk_nav_list[^>]*>(.*?)</ul>', r.text, re.DOTALL)
    if not m: return items
    ul = m.group(1)
    for li in re.findall(r'<li[^>]*>(.*?)</li>', ul, re.DOTALL):
        link = re.search(r'href="(http://www.macheng.gov.cn/zwgk/public/\d+/\d+\.html)"', li)
        title_m = re.search(r'title="([^"]*)"', li)
        date_m = re.search(r'<span[^>]*>(\d{4}[-/]\d{2}[-/]\d{2})</span>', li)
        if link and title_m:
            items.append((title_m.group(1).strip(), link.group(1).strip(), date_m.group(1).strip() if date_m else ""))
    return items

def parse_detail(html):
    content = ""
    for cls in ["gkwz_contnet", "wzcon j-fontContent", "wzcon"]:
        m = re.search(r'<div[^>]*class="' + cls + r'"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
        if m and len(m.group(1).strip()) > 200:
            content = m.group(1).strip()
            break
    pub_date = ""
    m = re.search(r'(\d{4}[-/]\d{2}[-/]\d{2})', html)
    if m: pub_date = m.group(1).strip()
    atts = []
    for am in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd))"[^>]*>(.*?)</a>', html, re.DOTALL | re.IGNORECASE):
        au = am.group(1).strip()
        att_url = au if au.startswith('http') else (SITE_URL + au if au.startswith('/') else SITE_URL + '/' + au)
        atts.append((get_text(am.group(2)), att_url))
    return content, pub_date, atts

def save(items):
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    ins = 0
    for title, url, ds, content, pd, atts in items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone(): continue
        ah = '<div class="attachments">' + '<br>'.join(f'<a href="{u}" target="_blank">{t}</a>' for t, u in atts) + '</div>' if atts else ''
        fc = content + '\n' + ah if ah else content
        summary = get_text(content)[:200] if content else title
        c.execute("INSERT INTO gov_raw(title,page_url,content,summary,publish_date,site_name) VALUES(?,?,?,?,?,?)",
                  (title, url, fc, summary, ds or pd, SITE_NAME))
        ins += 1
    conn.commit()
    conn.close()
    return ins

def main():
    print(f"=== {SITE_NAME} ===")
    items = fetch_list()
    print(f"List: {len(items)} items")
    all_items = []
    for i, (title, url, ds) in enumerate(items, 1):
        if INCREMENTAL:
            conn = sqlite3.connect(DB_PATH)
            cu = conn.cursor()
            if cu.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone():
                conn.close()
                continue
            conn.close()
        print(f"  [{i}/{len(items)}] {title[:50]}...")
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        content, pd, atts = parse_detail(r.text)
        if not content: content = title
        all_items.append((title, url, ds, content, pd, atts))
        time.sleep(0.3)
    if all_items:
        n = save(all_items)
        print(f"Saved: {n} new items")
    else:
        print("No new items")

if __name__ == "__main__":
    main()
