#!/usr/bin/env python3
"""金昌市政务服务网 - 企业入驻园区专区/政策资讯/公告"""
import requests, sys, os, sqlite3, re, time, json
from datetime import datetime
from bs4 import BeautifulSoup

API_URL = "https://zwfw.gansu.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
BASE = "https://zwfw.gansu.gov.cn"
DB = "/root/search.db"
SITE_NAME = "金昌经开区"
TABLE = "site_jinchang"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json, text/plain, */*",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://zwfw.gansu.gov.cn/jinchang/ztfw/qyrzyqzq/zczx/gg/index.html",
}

API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "d560d670ebd349669177bda54cb1bbb7",
    "tplSetId": "2e48cb3b1d7e4d41aeab793605d882b4",
    "pageType": "column",
    "tagId": "当前栏目ajax分页",
    "pageId": "0c925712c2364beb834fec0622949224",
}

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_conn():
    conn = sqlite3.connect(DB, timeout=60)
    conn.execute(f"CREATE TABLE IF NOT EXISTS {TABLE} (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT UNIQUE, url TEXT, date TEXT, content TEXT, summary TEXT, created_at TEXT)")
    return conn

def fetch_list(page=1):
    params = dict(API_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page, "pageSize": 15}, ensure_ascii=False)
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        data = r.json()
        if data.get("success"):
            return data["data"]["html"]
        return None
    except Exception as e:
        print(f"  [ERROR] fetch list page {page}: {e}")
        return None

def parse_list(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for li in soup.find_all('li', class_='clearfix'):
        a = li.find('a')
        span = li.find('span')
        if not a or not a.get('href'):
            continue
        title = a.get('title', '').strip()
        if not title:
            title = a.get_text(strip=True)
        href = a['href']
        if href.startswith('/'):
            href = BASE + href
        date = span.get_text(strip=True) if span else ""
        items.append({'title': title, 'url': href, 'date': date})
    return items

def get_page_info(html):
    """Extract total pages from the pagination div."""
    soup = BeautifulSoup(html, 'html.parser')
    pagination = soup.find('div', class_='pagination')
    if pagination:
        count = int(pagination.get('count', 0))
        rows = int(pagination.get('rows', 15))
        total_pages = (count + rows - 1) // rows  # ceiling division
        return total_pages
    return 1

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERROR] fetch detail {url}: {e}")
        return None

def extract_text(html):
    """Extract clean text from detail page."""
    soup = BeautifulSoup(html, 'html.parser')
    article = soup.find('div', class_='article')
    if not article:
        return "", ""
    content = article.find('div', class_='content')
    if not content:
        return "", ""
    text = body_text(content)
    text = re.sub(r'\n{3,}', '\n\n', text)
    summary = text[:300] if len(text) > 300 else text
    return text, summary

def crawl(incremental=False):
    conn = get_conn()
    cur = conn.cursor()
    
    existing = set()
    if incremental:
        for row in cur.execute(f"SELECT title FROM {TABLE}"):
            existing.add(row[0])
    
    # Fetch page 1 to get total pages
    html = fetch_list(1)
    if not html:
        print("[ERROR] Could not fetch list")
        sys.exit(1)
    
    total_pages = get_page_info(html)
    total_count = 0
    all_items = parse_list(html)
    total_count += len(all_items)
    
    for p in range(2, total_pages + 1):
        html = fetch_list(p)
        if html:
            items = parse_list(html)
            all_items.extend(items)
            total_count += len(items)
            print(f"  Page {p}/{total_pages}: {len(items)} items")
        else:
            print(f"  Page {p}/{total_pages}: FAILED")
        time.sleep(0.5)
    
    print(f"Total items: {len(all_items)}")
    
    total_inserted = 0
    total_skipped = 0
    
    for item in all_items:
        title = item['title']
        if title in existing:
            total_skipped += 1
            continue
        
        detail_html = fetch_detail(item['url'])
        if not detail_html:
            total_skipped += 1
            continue
        
        text, summary = extract_text(detail_html)
        if not text or len(text) < 10:
            total_skipped += 1
            continue
        
        now = datetime.now().strftime('%Y-%m-%d %H:%M:%S')
        try:
            cur.execute(
                f"INSERT OR IGNORE INTO {TABLE} (title, url, date, content, summary, created_at) VALUES (?,?,?,?,?,?)",
                (title, item['url'], item['date'], text, summary, now)
            )
            if cur.rowcount > 0:
                total_inserted += 1
                if total_inserted <= 5 or total_inserted % 15 == 0:
                    print(f"  + {title[:50]}... ({item['date']})")
            else:
                total_skipped += 1
        except Exception as e:
            print(f"  [ERROR] DB insert: {e}")
    
    conn.commit()
    conn.close()
    print(f"\nDone! Inserted: {total_inserted}, Skipped: {total_skipped}")

if __name__ == '__main__':
    incremental = '--incremental' in sys.argv
    requests.packages.urllib3.disable_warnings()
    crawl(incremental=incremental)
