#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
岑巩县人民政府 - 生态环境栏目爬虫
https://www.qdncg.gov.cn/zwgk/zdlyxx/sthj/
CMS: TRS（拓尔思）
列表: ul.NewsList > li > a title + span date
分页: index.html -> index_1.html ... (GET)
正文: div.trs_editor_view
"""

import requests
import re
import time
import sqlite3
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

BASE_URL = "https://www.qdncg.gov.cn"
LIST_URL = "https://www.qdncg.gov.cn/zwgk/zdlyxx/sthj"
SITE_NAME = "岑巩县生态环境"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

DB_PATH = "/root/search.db"
DAYS_BACK = 365

KEYWORDS = ["环评", "环境影响评价", "环境影响报告书", "环境影响报告表",
            "环境影响评价报告", "环保验收", "竣工环境保护", "环境保护验收"]


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.text_factory = str
    return conn


def url_exists(conn, url):
    cur = conn.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (url,))
    return cur.fetchone() is not None


def insert_record(conn, record):
    conn.execute(
        """INSERT OR IGNORE INTO gov_raw
           (title, page_url, summary, content, site_name, publish_date)
           VALUES (?, ?, ?, ?, ?, ?)""",
        (record["title"], record["page_url"], record["summary"],
         record["content"], record["site_name"],
         record["publish_date"])
    )


def extract_clean_text(html):
    """提取格式良好的正文，保留段落结构"""
    text = re.sub(r'</?(b|span|font|strong|u|i|em|o:p)[^>]*>', '', html, flags=re.IGNORECASE)
    text = re.sub(r'<br\s*/?>', '\n', text)
    soup = BeautifulSoup(text, 'html.parser')
    return soup.get_text(separator='\n\n', strip=True)


def fetch_page(page_num):
    if page_num == 0:
        url = f"{LIST_URL}/index.html"
    else:
        url = f"{LIST_URL}/index_{page_num}.html"
    
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code == 200:
            return resp.text
        return None
    except Exception as e:
        print(f"  [ERROR] 列表页 {url} 请求失败: {e}")
        return None


def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="NewsList")
    if not ul:
        return []
    
    items = []
    for li in ul.find_all("li"):
        a_tag = li.find("a")
        span = li.find("span")
        if not a_tag:
            continue
        
        href = a_tag.get("href", "").strip()
        title = a_tag.get_text(strip=True)
        date_str = span.get_text(strip=True) if span else ""
        
        if not href or not title:
            continue
        
        if href.startswith("./"):
            href = href[1:]
        if not href.startswith("http"):
            href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href
        
        items.append((title, href, date_str))
    
    return items


def fetch_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None, []
        
        soup = BeautifulSoup(resp.text, "html.parser")
        
        content_div = soup.find("div", class_="trs_editor_view")
        if not content_div:
            content_div = soup.find("div", class_="Article_Con")
        if not content_div:
            content_div = soup.find("div", class_="ContentPageBox")
        
        content_text = ""
        if content_div:
            content_text = extract_clean_text(str(content_div))
        
        attachments = []
        if content_div:
            for a in content_div.find_all("a"):
                href = a.get("href", "").strip()
                text = a.get_text(strip=True)
                if not href or href.startswith("#") or "javascript" in href.lower():
                    continue
                if href.startswith("./"):
                    ahref = LIST_URL + href[1:]
                elif href.startswith("/"):
                    ahref = BASE_URL + href
                elif not href.startswith("http"):
                    ahref = LIST_URL + "/" + href
                else:
                    ahref = href
                
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|jpg|png)$', href.lower()):
                    attachments.append({"text": text, "url": ahref})
        
        return content_text, attachments
    
    except Exception as e:
        print(f"  [ERROR] 详情页 {url} 请求失败: {e}")
        return None, []


def should_crawl(title, date_str):
    if date_str:
        try:
            pub_date = datetime.strptime(date_str, "%Y-%m-%d")
            if pub_date < datetime.now() - timedelta(days=DAYS_BACK):
                return False
        except:
            pass
    
    for kw in KEYWORDS:
        if kw in title:
            return True
    return False


def main():
    print(f"=== 岑巩县生态环境爬虫 ===")
    print(f"目标: {LIST_URL}")
    print(f"时间范围: 最近{DAYS_BACK}天")
    
    conn = get_conn()
    
    total_found = 0
    total_inserted = 0
    page = 0
    
    while True:
        print(f"\n--- 第{page+1}页 ---")
        html = fetch_page(page)
        if not html:
            print(f"  [INFO] 第{page+1}页无内容，结束")
            break
        
        items = parse_list(html)
        if not items:
            print(f"  [INFO] 第{page+1}页无列表项，结束")
            break
        
        print(f"  本页 {len(items)} 条")
        
        for title, url, date_str in items:
            total_found += 1
            if not should_crawl(title, date_str):
                continue
            
            if url_exists(conn, url):
                print(f"  [-] {title[:50]}... 已存在")
                continue
            
            content, attachments = fetch_detail(url)
            if not content or len(content) < 50:
                print(f"  [!] {title[:50]}... 内容过短或无内容")
                continue
            
            summary = content[:200] if len(content) > 200 else content
            
            record = {
                "title": title,
                "page_url": url,
                "summary": summary,
                "content": content,
                "site_name": SITE_NAME,
                "publish_date": date_str,
            }
            
            insert_record(conn, record)
            total_inserted += 1
            conn.commit()
            
            print(f"  [+] {title[:50]}...")
            if attachments:
                print(f"     附件: {len(attachments)}个")
            
            time.sleep(0.5)
        
        # 检查下一页
        soup = BeautifulSoup(html, "html.parser")
        page_div = soup.find("div", class_="page")
        if page_div:
            script = page_div.find("script")
            if script and script.string:
                match = re.search(r'createPageHTML\(\s*(\d+)', script.string)
                if match:
                    last_page = int(match.group(1))
                    if page >= last_page:
                        print(f"  [DONE] 已到最后一页")
                        break
        
        page += 1
        time.sleep(1)
    
    conn.close()
    
    print(f"\n=== 完成 ===")
    print(f"总扫描: {total_found} 条")
    print(f"新增入库: {total_inserted} 条")


if __name__ == "__main__":
    main()
