#!/usr/bin/env python3
"""盐边县人民政府 — 服务公开（公共服务）
http://www.scyanbian.gov.cn/zwgk/wgktj/fwgk/index.shtml
TRS CMS，单页列表
"""
import os, sys, re, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.scyanbian.gov.cn/zwgk/wgktj/fwgk/"
SITE_NAME = "盐边县政府-服务公开"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
session = requests.Session()
session.headers.update(HEADERS)
session.verify = False
import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
import sqlite3

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def store_item(title, date_str, content, page_url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        c = conn.cursor()
        summary = (content.strip()[:200] if content.strip() else '')
        date_rank = 0
        if date_str:
            try:
                date_rank = int(datetime.strptime(date_str[:10], "%Y-%m-%d").timestamp())
            except:
                pass
        c.execute("""INSERT OR IGNORE INTO gov_raw (title, publish_date, content, page_url, source_url, site_name, summary, date_rank)
                     VALUES (?,?,?,?,?,?,?,?)""",
                  (title.strip(), date_str, content.strip(), page_url.strip(), page_url.strip(), SITE_NAME, summary, date_rank))
        a = c.rowcount; conn.commit(); conn.close(); return a
    except Exception as e:
        print(f"[ERROR] 写入失败: {e}"); return 0

def fetch(url, max_retries=3):
    for i in range(max_retries):
        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            print(f"[WARN] {url} 请求失败: {e}")
        if i < max_retries - 1:
            time.sleep(2)
    return None

def parse_list_page(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    nb = soup.find("div", class_="new-box")
    if not nb:
        return items
    for a in nb.find_all("a", href=True):
        title = a.get_text(strip=True)
        href = a["href"]
        if not title or not href:
            continue
        full_url = urljoin(BASE_URL, href)
        date_str = ""
        span = a.find_next_sibling("span")
        if span:
            date_str = span.get_text(strip=True)
        items.append((title, date_str, full_url))
    return items

def parse_detail_page(url):
    html = fetch(url)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")
    
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        t = soup.find("title")
        if t:
            title = t.get_text(strip=True).split("·")[0].strip()
    
    date_str = ""
    m = re.search(r'发布时间[：:]?\s*(\d{4}-\d{1,2}-\d{1,2})', soup.get_text())
    if m:
        date_str = m.group(1)
    
    content_parts = []
    body = soup.find("div", class_=lambda c: c and "main" in str(c))
    if body:
        table_texts = set()
        table_row_texts = set()
        table_full_texts = set()
        for table in body.find_all("table"):
            all_rows_txt = ""
            for td in table.find_all(["td", "th"]):
                t = td.get_text(strip=True)
                if t:
                    table_texts.add(t)
                    all_rows_txt += t
            for tr in table.find_all("tr"):
                row_txt = "".join(td.get_text(strip=True) for td in tr.find_all(["td", "th"]))
                if row_txt:
                    table_row_texts.add(row_txt)
            if all_rows_txt:
                table_full_texts.add(all_rows_txt)
        
        for el in body.find_all(["p", "table"]):
            if el.name == "p":
                txt = el.get_text(strip=True)
                if txt and len(txt) > 3:
                    skip_words = ["扫一扫", "分享到", "上一篇", "下一篇", "责任编辑", "编辑:", "发布时间", "阅读次数"]
                    if any(w in txt for w in skip_words):
                        continue
                    if txt in table_texts:
                        continue
                    if txt in table_row_texts:
                        continue
                    if txt in table_full_texts:
                        continue
                    content_parts.append(txt)
            elif el.name == 'table':
                tbl_html = html_table_to_html(el, url)
                if tbl_html:
                    parts.append(tbl_html)
    content = "\n\n".join(content_parts)
    
    attachments = []
    if body:
        for a in body.find_all("a", href=True):
            h = a["href"].lower()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ppt|pptx)$', h):
                fname = a.get_text(strip=True) or os.path.basename(a["href"])
                full_url = urljoin(url, a["href"])
                attachments.append(f"[{fname}]({full_url})")
    if attachments:
        if content:
            content += "\n\n---\n**附件：**\n" + "\n".join(attachments)
        else:
            content = "\n".join(attachments)
    
    return title, date_str, content

def main():
    print(f"[START] {SITE_NAME} 阈值: {THRESHOLD}")
    total_new = 0
    total_skip = 0
    total_err = 0
    
    page_url = "http://www.scyanbian.gov.cn/zwgk/wgktj/fwgk/index.shtml"
    html = fetch(page_url)
    if not html:
        print("[ERR] 获取列表页失败")
        return
    
    items = parse_list_page(html)
    print(f"\n[LIST] 共 {len(items)} 条")
    
    for title, date_str, detail_url in items:
        if date_str and date_str < THRESHOLD:
            print(f"  [SKIP] {date_str} {title[:40]}")
            total_skip += 1
            continue
        
        if detail_url.lower().endswith('.pdf'):
            st = store_item(title, date_str or "", f"[查看PDF]({detail_url})", detail_url)
            if st:
                total_new += 1
                print(f"  [PDF] {date_str} {title[:40]}")
            else:
                total_err += 1
            continue
        
        t, d, c = parse_detail_page(detail_url)
        if t and c:
            final_date = d or date_str or ""
            st = store_item(t, final_date, c, detail_url)
            if st:
                total_new += 1
                print(f"  [OK] {final_date} {t[:40]}")
            else:
                total_err += 1
                print(f"  [DUP] {final_date} {t[:40]}")
        else:
            total_err += 1
            print(f"  [ERR] {d or date_str or ''} {title[:40]} 解析失败 {detail_url}")
        time.sleep(0.3)
    
    print(f"\n[DONE] 新增: {total_new}, 跳过: {total_skip}, 失败: {total_err}")

if __name__ == "__main__":
    main()
