#!/usr/bin/env python3
"""沈丘县人民政府 — 公告公示 - 补跑失败项
"""
import os, sys, re, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "沈丘县人民政府-公告公示"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
session = requests.Session()
session.headers.update(HEADERS)
session.verify = False
import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
import sqlite3

def store_item(title, date_str, content, page_url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        c = conn.cursor()
        summary = (content.strip()[:200] if content.strip() else '')
        date_rank = 0
        if date_str:
            try:
                date_rank = int(datetime.strptime(date_str[:10], "%Y-%m-%d").timestamp())
            except:
                pass
        c.execute("""INSERT OR IGNORE INTO gov_raw (title, publish_date, content, page_url, source_url, site_name, summary, date_rank)
                     VALUES (?,?,?,?,?,?,?,?)""",
                  (title.strip(), date_str, content.strip(), page_url.strip(), page_url.strip(), SITE_NAME, summary, date_rank))
        a = c.rowcount; conn.commit(); conn.close(); return a
    except Exception as e:
        print(f"[ERROR] 写入失败: {e}"); return 0

def fetch(url, max_retries=3):
    for i in range(max_retries):
        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            print(f"[WARN] {url} 请求失败: {e}")
        if i < max_retries - 1:
            time.sleep(2)
    return None

def parse_detail_page(url):
    html = fetch(url)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")
    
    title = ""
    h1 = soup.find("h1", class_="font28")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        t = soup.find("title")
        if t:
            title = t.get_text(strip=True)
            for suffix in ["-沈丘县人民政府", "—沈丘县人民政府", "_沈丘县人民政府"]:
                if suffix in title:
                    title = title[:title.index(suffix)].strip()
                    break
    
    date_str = ""
    info_time = soup.find("div", class_="info-time")
    if info_time:
        m = re.search(r'(\d{4}-\d{2}-\d{2})', info_time.get_text())
        if m:
            date_str = m.group(1)
    
    content_parts = []
    content_div = soup.find("div", class_="news_content")
    if content_div:
        # 查找 p, div, table, img 标签
        for el in content_div.find_all(["p", "div", "table", "img"]):
            if el.name == "img":
                alt = el.get("alt", "")
                src = el.get("src", "")
                if src:
                    full_src = urljoin(url, src)
                    txt = f"![{alt}]({full_src})"
                    content_parts.append(txt)
                continue
            txt = el.get_text(strip=True)
            if txt and len(txt) > 3:
                if txt.startswith("字体大小") or "打印" in txt or "分享到" in txt:
                    continue
                if "上一篇：" in txt or "下一篇：" in txt:
                    continue
                if "点击数" in txt and "时间" in txt:
                    continue
                content_parts.append(txt)
            elif el.name == "table":
                # table with little text content but structure
                rows = []
                for tr in el.find_all("tr"):
                    cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                    if any(cells):
                        rows.append(" | ".join(cells))
                if rows:
                    content_parts.append("\n".join(rows))
    
    content = "\n\n".join(content_parts)
    
    # 附件
    attachments = []
    for a in soup.find_all("a", href=True):
        h = a["href"].lower()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ppt|pptx)$', h):
            fname = a.get_text(strip=True) or os.path.basename(a["href"])
            full_url = urljoin(url, a["href"])
            attachments.append(f"[{fname}]({full_url})")
    if attachments:
        if content:
            content += "\n\n---\n**附件：**\n" + "\n".join(attachments)
        else:
            content = "\n".join(attachments)
    
    return title, date_str, content

# 补跑失败的6条
urls = [
    "http://www.shenqiu.gov.cn/newslast_41310.html",
    "http://www.shenqiu.gov.cn/newslast_41381.html",
    "http://www.shenqiu.gov.cn/newslast_41380.html",
    "http://www.shenqiu.gov.cn/newslast_41379.html",
    "http://www.shenqiu.gov.cn/newslast_41378.html",
    "http://www.shenqiu.gov.cn/newslast_41037.html",
]

for url in urls:
    t, d, c = parse_detail_page(url)
    if t and c:
        st = store_item(t, d, c, url)
        if st:
            print(f"[OK] {d} {t[:50]}")
        else:
            print(f"[DUP] {d} {t[:50]}")
    else:
        print(f"[ERR] {url}")
    time.sleep(0.3)
