#!/usr/bin/env python3
"""鱼台县人民政府 - 建设项目环境影响评价"""
import requests, sys, re, os, sqlite3, urllib.parse, time
from datetime import datetime
from bs4 import BeautifulSoup

BASE = "http://www.yutai.gov.cn"
DB = "/root/search.db"
SITE_NAME = "鱼台县生态环境"
TABLE = "gov_raw"
SITE_SOURCE = "http://www.yutai.gov.cn/col/col61241/index.html"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": "http://www.yutai.gov.cn/col/col61241/index.html",
}

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_conn():
    conn = sqlite3.connect(DB, timeout=60)
    conn.execute(f"CREATE TABLE IF NOT EXISTS {TABLE} (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT UNIQUE, url TEXT, date TEXT, content TEXT, summary TEXT, created_at TEXT)")
    return conn

def extract_text(html):
    """Extract clean text from Hanweb detail HTML with proper paragraph breaks, tables, and attachments."""
    soup = BeautifulSoup(html, 'html.parser')
    zoom = soup.find('div', id='zoom')
    if not zoom:
        for cls in ['bt_content', 'article-content', 'content']:
            el = soup.find('div', class_=cls)
            if el:
                zoom = el
                break
    if not zoom:
        return "", ""
    
    parts = []
    
    # 1. Capture attachment links (doc/pdf) first
    attach_links = []
    for a in zoom.find_all("a", href=True):
        href = a["href"]
        fname = a.get_text(strip=True)
        if not fname:
            fname = href.split("/")[-1] if "/" in href else href
        # Check if it's a document download link
        if any(x in href.lower() for x in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip"]):
            if href.startswith("/"):
                href = "http://www.yutai.gov.cn" + href
            attach_links.append(f"[附件：{fname}]({href})")
        elif "downfile" in href.lower() or "download" in href.lower():
            if href.startswith("/"):
                href = "http://www.yutai.gov.cn" + href
            attach_links.append(f"[附件：{fname}]({href})")
    
    # 2. Capture table HTML (before they get destroyed by span unwrap)
    for table in zoom.find_all("table"):
        table_html = str(table)
        # Clean whitespace in HTML
        table_html = re.sub(r">\s+<", "><", table_html)
        table_html = re.sub(r"\s{2,}", " ", table_html)
        parts.append(table_html)
        table.decompose()
    
    # 3. Unwrap inline spans
    for span in zoom.find_all("span"):
        span.unwrap()
    
    # 4. Extract p/li text in document order
    seen = set()
    for el in zoom.find_all(["p", "li"]):
        text = el.get_text(strip=True)
        if text and text not in seen:
            parts.append(text)
            seen.add(text)
    
    # 5. Append attachment links at the end
    if attach_links:
        parts.append("\n---\n附件：")
        parts.extend(attach_links)
    
    text = "\n\n".join(parts) if parts else body_text(zoom)
    text = re.sub(r'\n{3,}', '\n\n', text)
    text = re.sub(r' {2,}', ' ', text)
    summary = text[:500] if len(text) > 500 else text
    return text.strip(), summary.strip()

def fetch_list(page=1):
    url = f"{BASE}/module/xxgk/search.jsp"
    data = {
        "divid": "div4",
        "infotypeId": "YTA551904",
        "jdid": "111",
        "area": "",
        "sortfield": "compaltedate:0",
        "standardXxgk": "1",
        "currpage": str(page),
    }
    try:
        r = requests.post(url, data=data, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERROR] fetch list page {page}: {e}")
        return None

def parse_list(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for li in soup.find_all('li'):
        a = li.find('a')
        b = li.find('b')
        if not a or not a.get('href'):
            continue
        title = a.get('title', '').strip()
        if not title:
            title = a.get_text(strip=True)
        href = a['href']
        if href.startswith('/'):
            href = BASE + href
        date = b.get_text(strip=True) if b else ""
        items.append({'title': title, 'url': href, 'date': date})
    return items

def get_page_info(html):
    """Extract total pages from list HTML"""
    # "共277条记录" and "共&nbsp;14&nbsp;页"
    m = re.search(r'共(\d+)条记录', html)
    total_records = int(m.group(1)) if m else 0
    # Find max page number from any funGoPage call (e.g. funGoPage('/...',14))
    pages = re.findall(r"funGoPage\('[^']+',(\d+)\)", html)
    total_pages = max(int(p) for p in pages) if pages else 1
    return total_records, total_pages

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERROR] fetch detail {url}: {e}")
        return None

def crawl(incremental=False):
    conn = get_conn()
    cur = conn.cursor()
    
    # Enable WAL mode to reduce locking
    cur.execute("PRAGMA journal_mode=WAL")
    
    # Get existing URLs for incremental
    existing = set()
    if incremental:
        for row in cur.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
            existing.add(row[0])
    
    total_inserted = 0
    total_skipped = 0
    
    # Fetch first page to get total pages
    html = fetch_list(1)
    if not html:
        print("[ERROR] Could not fetch list page 1")
        sys.exit(1)
    
    total_records, total_pages = get_page_info(html)
    print(f"Total: {total_records} records, {total_pages} pages")
    
    # Parse first page
    all_items = parse_list(html)
    
    # Fetch remaining pages
    for p in range(2, total_pages + 1):
        html = fetch_list(p)
        if html:
            all_items.extend(parse_list(html))
            print(f"  Page {p}/{total_pages}: fetched")
        else:
            print(f"  Page {p}/{total_pages}: FAILED")
        time.sleep(0.5)
    
    print(f"Total items in list: {len(all_items)}")
    
    for item in all_items:
        title = item['title']
        url = item['url']
        if url in existing:
            total_skipped += 1
            continue
        
        detail_html = fetch_detail(url)
        if not detail_html:
            total_skipped += 1
            continue
        
        text, summary = extract_text(detail_html)
        if not text or len(text) < 10:
            total_skipped += 1
            continue
        
        date_str = item.get('date', '')
        if not date_str:
            date_str = "2020-01-01"
        date_str = re.sub(r"[^\d-]", "", date_str)[:10]
        dr = int(date_str.replace("-", "")) if date_str.count("-") == 2 else 0
        
        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content, summary, date_rank) VALUES (?,?,?,?,?,?,?,?)",
                (SITE_NAME, SITE_SOURCE, url, title, date_str, text, summary, dr)
            )
            if cur.rowcount > 0:
                total_inserted += 1
                print(f"  + {title[:50]}... ({date_str})")
                # Sync to FTS immediately
                try:
                    # 2026-09-22: 先提交 gov_raw —— 下面手动写 FTS 会因触发器已写过同一
                    #   rowid 而 IntegrityError，若不先 commit，这条记录会被一并回滚（静默丢数据）
                    conn.commit()
                    cur.execute("INSERT OR REPLACE INTO gov_search(rowid, site_name, title, summary) VALUES (?,?,?,?)",
                        (cur.lastrowid, SITE_NAME, title, summary))
                except Exception:
                    pass
                if total_inserted % 10 == 0:
                    conn.commit()
            else:
                total_skipped += 1
        except Exception as e:
            print(f"  [ERROR] DB insert: {e}")
    
    conn.commit()
    conn.close()
    print(f"\nDone! Inserted: {total_inserted}, Skipped: {total_skipped}")

if __name__ == '__main__':
    incremental = '--incremental' in sys.argv
    requests.packages.urllib3.disable_warnings()
    crawl(incremental=incremental)
