#!/usr/bin/env python3
import os
"""陕西延长中煤榆林 - 通知公告(环评公示)
https://ylnh.sxycpc.com/web2016/xwzx/tzgg.htm
"""
import sys, re, sqlite3, requests, warnings
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
warnings.filterwarnings("ignore")

BASE_URL = "https://ylnh.sxycpc.com"
LIST_DIR = "/web2016/xwzx/"
SITE_NAME = "延长榆能化-通知公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUT_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0"}

def fetch(url, label=""):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("[ERROR] %s: %s" % (label, e))
        return ""

def parse_list(html):
    soup = BeautifulSoup(html, "lxml")
    items = []
    for a in soup.find_all("a", class_="c23609", href=True):
        title = a.get("title", "") or a.get_text(strip=True)
        href = a["href"].strip()
        # Find date in same row
        tr = a.find_parent("tr")
        date = ""
        if tr:
            sp = tr.find("span", class_="timestyle23609")
            if sp:
                date = sp.get_text(strip=True)
        if href.startswith("../../"):
            href = href.replace("../../", "/")
        items.append((title, href, date))
    return items

def parse_detail(html):
    soup = BeautifulSoup(html, "lxml")
    title = ""
    t = soup.find("title")
    if t:
        title = t.get_text(strip=True).split("-")[0].strip()
    pub_date = ""
    m = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if not m:
        m = re.search(r"发布日期[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if not m:
        # Fallback: from page meta or text
        m = re.search(r"(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}", html)
    if m:
        pub_date = m.group(1)
    content_div = soup.find("div", class_="c23430_content")
    if not content_div:
        content_div = soup.find("div", class_="v_news_content")
    if not content_div:
        content_div = soup.find(id="vsb_newscontent")
    if content_div:
        content_html = str(content_div)
    else:
        content_html = ""
    # Extract attachments outside content div (TRS CMS puts them in separate <tr>)
    attach_links = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if "virtual_attach_file" not in href:
            continue
        ext = href.split("e=")[-1] if "e=" in href else ""
        if ext not in (".pdf", ".doc", ".docx", ".zip", ".rar", ".xls", ".xlsx"):
            continue
        text = a.get_text(strip=True)
        if not text or len(text) < 2:
            span = a.find("span")
            if span:
                text = span.get_text(strip=True)
        if not text or len(text) < 2:
            text = "附件下载"
        full_url = href
        if full_url.startswith("../../"):
            full_url = href.replace("../../", BASE_URL + "/")
        attach_links.append((text, full_url))
    if attach_links:
        content_html += '<br><br><strong>附件：</strong><br>\n'
        for t, u in attach_links:
            content_html += f'<a href=\"{u}\" target=\"_blank\">{t}</a><br>\n'
    return title, pub_date, content_html

def main(mode="full"):
    start = datetime.now()
    print("[延长] mode=%s, cut_date=%s" % (mode, CUT_DATE))
    
    # Pages: tzgg.htm (page1), tzgg/2.htm (page2), tzgg/1.htm (page3 - oldest)
    page_urls = [LIST_DIR + "tzgg.htm", LIST_DIR + "tzgg/2.htm", LIST_DIR + "tzgg/1.htm"]
    
    all_items = []
    for url in page_urls:
        html = fetch(BASE_URL + url, url.split("/")[-1])
        if not html or len(html) < 500:
            continue
        items = parse_list(html)
        all_items.extend(items)
        if items:
            print("[延长] %s: %d条 (%s ~ %s)" % (url.split("/")[-1], len(items), items[0][2], items[-1][2]))
        if items and items[-1][2] < CUT_DATE:
            print("[延长] %s 超过3年线" % items[-1][2])
            break
    
    print("[延长] 列表共 %d 条" % len(all_items))
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    new, exists, skipped = 0, 0, 0
    
    for idx, (title, href, date) in enumerate(all_items):
        if date and date < CUT_DATE:
            skipped += 1
            continue
        
        detail_url = BASE_URL + href
        html = fetch(detail_url, "详情%d" % (idx+1))
        if not html:
            continue
        
        dt, pd, ch = parse_detail(html)
        title = dt or title
        pd = pd or date
        summary = BeautifulSoup(ch, "lxml").get_text(strip=True)[:200] if ch else title
        
        c.execute("INSERT OR IGNORE INTO gov_raw (title, page_url, site_name, summary, content, publish_date) VALUES (?,?,?,?,?,?)",
                  (title, detail_url, SITE_NAME, summary, ch, pd))
        if c.rowcount:
            new += 1
        else:
            exists += 1
        
        if (idx+1) % 10 == 0:
            conn.commit()
            el = (datetime.now()-start).total_seconds()
            print("[延长] %d/%d, 新增%d, 已存在%d, %ds" % (idx+1, len(all_items), new, exists, el))
    
    conn.commit()
    conn.close()
    el = (datetime.now()-start).total_seconds()
    print("[延长] 完成! 新增%d, 已存在%d, 跳过%d, 耗时%ds" % (new, exists, skipped, el))

if __name__ == "__main__":
    main(sys.argv[1] if len(sys.argv) > 1 else "full")
