#!/usr/bin/env python3
"""
长武县人民政府 — 重大建设项目
https://www.changwu.gov.cn/zfxxgk/fdzdgk/zdjsxm2/
CMS: TRS/WCM
分页: index.html (0), index_1.html... (虚标31页)
详情: ./YYYYMM/tYYYYMMDD_XXXXXX.html
正文: div.article-content.article-content-body
"""
import requests
import sqlite3
import re
import os
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
SITE_NAME = "长武县-重大建设项目"
LIST_URL = "https://www.changwu.gov.cn/zfxxgk/fdzdgk/zdjsxm2/index.html"
BASE_URL = "https://www.changwu.gov.cn/zfxxgk/fdzdgk/zdjsxm2/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def fetch_page(page):
    if page == 0:
        url = LIST_URL
    else:
        url = f"{BASE_URL}index_{page}.html"
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    return r.text

def extract_items(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        title = a.get_text(strip=True)
        if not title or len(title) < 5:
            continue
        if "t202" not in href:
            continue

        if href.startswith("http"):
            full_url = href
        elif href.startswith("./"):
            full_url = BASE_URL + href[2:]
        elif href.startswith("/"):
            full_url = "https://www.changwu.gov.cn" + href
        else:
            full_url = BASE_URL + href

        date_str = ""
        span = li.find("span")
        if span:
            date_str = span.get_text(strip=True)

        items.append({"title": title, "url": full_url, "date": date_str})
    return items

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        for tag in soup(["script", "style", "nav", "footer", "header", "aside"]):
            tag.decompose()

        detail_date = ""
        attr = soup.find("div", class_=lambda c: c and "attr" in c.lower())
        if attr:
            m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", attr.get_text())
            if m:
                detail_date = m.group(1)

        content_div = soup.find("div", class_=lambda c: c and "article-content" in c.lower() and "body" in c.lower())
        if not content_div:
            content_div = soup.find("div", class_=lambda c: c and "article-content" in c.lower())
        if content_div:
            # Unwrap inline span tags so text stays inline
            for span in content_div.find_all("span"):
                span.unwrap()
            # Extract text paragraph by paragraph (get_text() inserts extra newlines)
            para_parts = []
            seen_texts = set()
            for el in content_div.find_all(["p", "li"]):
                text = el.get_text(strip=True)
                if text and text not in seen_texts:
                    para_parts.append(text)
                    seen_texts.add(text)
            if para_parts:
                content = "\n\n".join(para_parts)
            else:
                # Fallback for unusual structures
                content = body_text(content_div)
        else:
            content = body_text(soup)

        content = re.sub(r"\n{3,}", "\n\n", content)
        content = re.sub(r" {2,}", " ", content)
        return content.strip(), detail_date
    except Exception as e:
        print(f"  [ERROR] detail: {e}", flush=True)
        return "", ""

def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    existing = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        existing.add(row[0])
    latest_db_date = "1900-01-01"
    row = c.execute("SELECT MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()
    if row and row[0]:
        latest_db_date = row[0]
    conn.close()
    print(f"Latest DB date: {latest_db_date}", flush=True)
    print(f"Existing URLs: {len(existing)}", flush=True)

    all_items = []
    empty_pages = 0
    for page in range(31):
        html = fetch_page(page)
        items = extract_items(html)
        print(f"  Page {page+1}: {len(items)} items", flush=True)
        if len(items) == 0:
            empty_pages += 1
            if empty_pages >= 2:
                break
        else:
            empty_pages = 0
        all_items.extend(items)

    print(f"Total list items: {len(all_items)}", flush=True)

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    new_count = skip_count = error_count = 0

    for item in all_items:
        url = item["url"]
        title = item["title"]
        date_str = item["date"]
        if url in existing:
            skip_count += 1
            continue

        content, detail_date = fetch_detail(url)
        if not detail_date:
            detail_date = date_str
        if not content or len(content) < 50:
            print(f"  [SHORT] {title[:30]}... ({len(content)})", flush=True)
            if not content:
                error_count += 1
                continue

        if not detail_date:
            detail_date = datetime.now().strftime("%Y-%m-%d")
        detail_date = re.sub(r"[^\d-]", "", detail_date)[:10]
        dr = int(detail_date.replace("-", "")) if detail_date.count("-") == 2 else 0

        c.execute("""
            INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content, summary, date_rank)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?)
        """, (SITE_NAME, LIST_URL, url, title, detail_date, content, content[:500], dr))
        new_count += 1
        if new_count % 20 == 0:
            conn.commit()
            print(f"  Progress: {new_count} new / {skip_count} skip", flush=True)

    conn.commit()
    conn.close()
    print(f"\n{'='*50}", flush=True)
    print(f"Site: {SITE_NAME}", flush=True)
    print(f"Total list items: {len(all_items)}", flush=True)
    print(f"New: {new_count}", flush=True)
    print(f"Skipped (existing): {skip_count}", flush=True)
    print(f"Errors: {error_count}", flush=True)
    print(f"{'='*50}", flush=True)

if __name__ == "__main__":
    main()
