#!/usr/bin/env python3
"""
临沂临港经济开发区-环境处罚信息 爬虫
CMS: VSB9
列表: http://www.lylgkfq.gov.cn/gk/fdgknr/zdly/ggjg/hjbhxx1/hjcfxx.htm
分页: hjcfxx/N.htm (N>=2, 共31页612条~20条/页)
详情: ../../../../../info/2372/NNNNN.htm (首页) / ../../../../../../info/2372/NNNNN.htm (子页)
内容: VSB内嵌PDF(virtual_attach_file.vsb?e=.pdf)
"""
import os, re, sys, json
from urllib import request
from urllib.parse import urljoin
import ssl

SITE_NAME = "临沂临港经济开发区-环境处罚信息"
LIST_BASE = "http://www.lylgkfq.gov.cn/gk/fdgknr/zdly/ggjg/hjbhxx1/"
BASE_URL = "http://www.lylgkfq.gov.cn"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

ssl._create_default_https_context = ssl._create_unverified_context

def fetch(url, timeout=20):
    req = request.Request(url, headers=HEADERS)
    try:
        resp = request.urlopen(req, timeout=timeout)
        return resp.read().decode('utf-8', errors='ignore')
    except Exception as e:
        print(f"  [ERROR] {url}: {e}", file=sys.stderr)
        return None

def parse_list(html, base_url):
    """Extract article links and dates from a list page."""
    items = []
    # Match li items with govnewslist
    for m in re.finditer(r'<li[^>]*id="line\d+_\d+"[^>]*>(.*?)</li>', html, re.DOTALL):
        li = m.group(1)
        if '本栏目暂无内容' in li:
            continue
        
        a = re.search(r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', li, re.DOTALL)
        if not a:
            continue
        
        href = a.group(1)
        title = re.sub(r'<[^>]+>', '', a.group(2)).strip()
        if not title or len(title) < 3:
            continue
        
        full_url = urljoin(base_url, href)
        # Clean up: remove leading ./ and normalize
        if full_url.startswith(base_url.rstrip('/') + '/info/'):
            pass  # already correct
        
        date = ''
        dm = re.search(r'<span[^>]*class="fdlidate"[^>]*>(.*?)</span>', li)
        if dm:
            date = dm.group(1).strip()
        
        items.append((title, full_url, date))
    
    return items

def parse_detail(html, url):
    """Extract PDF attachment URL as content."""
    content = ""
    attachments = []
    
    # Find PDF attachment links (virtual_attach_file.vsb?e=.pdf)
    for a_href, a_text in re.findall(r'<a[^>]*href="([^"]+\.vsb[^"]*e=\.pdf)"[^>]*>([^<]+)</a>', html):
        if '福昕' in a_text or 'Adobe' in a_text or '在线浏览' in a_text:
            continue
        if len(a_text.strip()) < 3:
            continue
        full_url = urljoin(url, a_href)
        attachments.append({"title": a_text.strip(), "url": full_url})
    
    # Also capture iframe PDF sources
    for m in re.finditer(r'<iframe[^>]*src="([^"]+\.vsb[^"]*e=\.pdf)"', html):
        src = m.group(1)
        full_url = urljoin(url, src)
        # Only add if not already captured as attachment
        if not any(a['url'] == full_url for a in attachments):
            attachments.append({"title": "PDF文档", "url": full_url})
    
    # Content: describe what the page contains
    if attachments:
        content = f"[内嵌PDF附件] {attachments[0]['title']}"
    
    # Try to get any text from .zw div
    m = re.search(r'<div[^>]*id="vsb_content"[^>]*>(.*?)</div>\s*</td>', html, re.DOTALL)
    if m:
        inner = m.group(1)
        # Extract p text (exclude iframe/script)
        parts = []
        for p in re.findall(r'<p[^>]*>(.*?)</p>', inner, re.DOTALL):
            text = re.sub(r'<[^>]+>', '', p).strip()
            if text and len(text) > 5 and '在线浏览' not in text and '福昕' not in text and 'Adobe' not in text:
                parts.append(text)
        if parts:
            content = '\n\n'.join(parts)
    
    return content, attachments

def crawl():
    """Main crawl function."""
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    
    new_count = 0
    skip_count = 0
    total_pages = 31  # Try up to 31, break on 404
    
    for page in range(1, total_pages + 1):
        if page == 1:
            list_url = LIST_BASE + 'hjcfxx.htm'
        else:
            list_url = LIST_BASE + f'hjcfxx/{page}.htm'
        
        print(f"[分页] 第{page}页: {list_url}", flush=True)
        html = fetch(list_url)
        if not html:
            print(f"  [完成] 第{page}页无法获取(404/超时)，停止", flush=True)
            break
        
        items = parse_list(html, list_url)
        if not items:
            print(f"  [完成] 第{page}页无数据", flush=True)
            break
        
        print(f"  找到 {len(items)} 条", flush=True)
        
        for title, url, date in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (url, SITE_NAME))
            if cur.fetchone():
                skip_count += 1
                continue
            
            detail_html = fetch(url)
            if not detail_html:
                print(f"  [跳过] 详情页失败: {title[:30]}", flush=True)
                skip_count += 1
                continue
            
            content, attachments = parse_detail(detail_html, url)
            attachments_json = json.dumps(attachments, ensure_ascii=False)
            
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw 
                   (site_name, page_url, title, content, publish_date, attachments, summary)
                   VALUES (?, ?, ?, ?, ?, ?, ?)""",
                (SITE_NAME, url, title, content, date, attachments_json, content[:200] if content else "")
            )
            if cur.rowcount > 0:
                new_count += 1
                print(f"  [新增] {title[:40]} [{date}]", flush=True)
            else:
                skip_count += 1
        
        conn.commit()
    
    conn.close()
    print(f"\n[DONE] {SITE_NAME}: 新增={new_count}, 跳过={skip_count}", flush=True)

if __name__ == "__main__":
    crawl()
