#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
企业环保信息公示网 - 环评信息公示爬虫
https://www.ouryq.com/eia
CMS: WordPress
列表: table tr > td (全部在一页，无分页)
详情: div.content 正文
"""

import requests
import re
import sqlite3
import time
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

BASE_URL = "https://www.ouryq.com"
LIST_URL = "https://www.ouryq.com/eia"
SITE_NAME = "企业环保信息公示网-环评信息公示"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

DB_PATH = "/root/search.db"
DAYS_BACK = 365


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.text_factory = str
    return conn


def url_exists(conn, url):
    cur = conn.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (url,))
    return cur.fetchone() is not None


def insert_record(conn, record):
    conn.execute(
        """INSERT OR IGNORE INTO gov_raw
           (title, page_url, summary, content, site_name, publish_date)
           VALUES (?, ?, ?, ?, ?, ?)""",
        (record["title"], record["page_url"], record["summary"],
         record["content"], record["site_name"],
         record["publish_date"])
    )


def parse_date(date_str):
    """解析中文日期格式 '2026年3月24日' -> '2026-03-24'"""
    if not date_str:
        return ""
    date_str = date_str.strip()
    m = re.match(r'(\d{4})年(\d{1,2})月(\d{1,2})日', date_str)
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    # Also handle YYYYMMDD format
    m = re.match(r'(\d{4})(\d{2})(\d{2})', date_str)
    if m:
        return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
    return date_str


def fetch_page():
    """获取列表页"""
    try:
        resp = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code == 200:
            return resp.text
        return None
    except Exception as e:
        print(f"  [ERROR] 列表页请求失败: {e}")
        return None


def parse_list(html):
    """解析列表页表格，返回 (company, project_name, pub_date, detail_url, detail_text_as_title)"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    
    for tr in soup.find_all("tr"):
        tds = tr.find_all("td")
        if len(tds) != 8:
            continue
        
        # 第4列 = 项目名称 (作为标题)
        project_td = tds[3]
        project_name = project_td.get_text(strip=True)
        if not project_name or project_name == '项目名称':
            continue
        
        # 第1列 = 企业名称
        company = tds[0].get_text(strip=True)
        # 第2列 = 行业类别
        industry = tds[1].get_text(strip=True)
        # 第3列 = 县区
        county = tds[2].get_text(strip=True)
        # 第5列 = 地址
        address = tds[4].get_text(strip=True)
        # 第6列 = 起止日期 (不存)
        # 第7列 = 发布日期
        pub_date_raw = tds[6].get_text(strip=True)
        pub_date = parse_date(pub_date_raw)
        
        # 第8列 = 详情链接
        detail_link = tds[7].find("a")
        detail_url = ""
        if detail_link:
            href = detail_link.get("href", "").strip()
            if href:
                detail_url = href
        
        # 标题 = 企业名称 + 项目名称
        title = f"{company} - {project_name}" if company else project_name
        
        items.append({
            "title": title,
            "company": company,
            "project_name": project_name,
            "industry": industry,
            "county": county,
            "address": address,
            "pub_date": pub_date,
            "detail_url": detail_url,
        })
    
    return items


def fetch_detail(url):
    """获取详情页正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None, []
        
        soup = BeautifulSoup(resp.text, "html.parser")
        
        # 正文在 div.content 中
        content_div = soup.find("div", class_="content")
        if not content_div:
            return None, []
        
        # 表格前的导航文字去掉
        content_text = content_div.get_text(strip=True)
        
        # 附件
        attachments = []
        for a in content_div.find_all("a"):
            href = a.get("href", "").strip()
            text = a.get_text(strip=True)
            if not href or href.startswith("#") or "javascript" in href.lower():
                continue
            # 补全URL
            if href.startswith("/"):
                ahref = BASE_URL + href
            elif not href.startswith("http"):
                ahref = BASE_URL + "/" + href
            else:
                ahref = href
            
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|jpg|png)$', href.lower()):
                attachments.append({"text": text, "url": ahref})
        
        return content_text, attachments
    
    except Exception as e:
        print(f"  [ERROR] 详情页 {url} 请求失败: {e}")
        return None, []


def should_crawl(item):
    """判断是否应爬取：365天内"""
    if item["pub_date"]:
        try:
            pub_dt = datetime.strptime(item["pub_date"], "%Y-%m-%d")
            if pub_dt < datetime.now() - timedelta(days=DAYS_BACK):
                return False
        except:
            pass
    return True


def main():
    print(f"=== 企业环保信息公示网-环评爬虫 ===")
    print(f"目标: {LIST_URL}")
    print(f"时间范围: 最近{DAYS_BACK}天")
    
    conn = get_conn()
    
    print(f"\n--- 获取列表页 ---")
    html = fetch_page()
    if not html:
        print("[ERROR] 无法获取列表页")
        conn.close()
        return
    
    items = parse_list(html)
    print(f"列表页共 {len(items)} 条")
    
    total_found = len(items)
    total_inserted = 0
    total_skipped = 0
    total_outdated = 0
    
    for item in items:
        if not should_crawl(item):
            total_outdated += 1
            continue
        
        if url_exists(conn, item["detail_url"]):
            total_skipped += 1
            continue
        
        # 获取详情页正文
        content = ""
        attachments = []
        if item["detail_url"]:
            content, attachments = fetch_detail(item["detail_url"])
        
        if not content or len(content) < 20:
            short_title = item["title"][:50]
            print(f"  [!] {short_title}... 内容为空或无详情页")
            total_skipped += 1
            continue
        
        summary = content[:200] if len(content) > 200 else content
        
        record = {
            "title": item["title"],
            "page_url": item["detail_url"],
            "summary": summary,
            "content": content,
            "site_name": SITE_NAME,
            "publish_date": item["pub_date"],
        }
        
        insert_record(conn, record)
        conn.commit()
        total_inserted += 1
        
        short_title = item["title"][:50]
        print(f"  [+] {short_title}... ({item['pub_date']})")
        if attachments:
            print(f"     附件: {len(attachments)}个")
        
        time.sleep(0.5)
    
    conn.close()
    
    print(f"\n=== 完成 ===")
    print(f"总扫描: {total_found} 条")
    print(f"新增入库: {total_inserted} 条")
    print(f"已存在跳过: {total_skipped} 条")
    print(f"超出时间范围: {total_outdated} 条")


if __name__ == "__main__":
    main()
