#!/usr/bin/env python3
import os
"""阳高县人民政府 — 生态环境保护（sthj2）"""

import sys, os, re, time, sqlite3, subprocess
from datetime import datetime, timedelta



import sys as _SYS
# 支持 --pages 参数
import argparse as _AP
_AP_PARSER = _AP.ArgumentParser()
_AP_PARSER.add_argument("--pages", type=int, default=0, help="限制页数")
_AP_ARGS, _ = _AP_PARSER.parse_known_args()
_MAX_PG = _AP_ARGS.pages if _AP_ARGS.pages > 0 else None
if _MAX_PG:
    print(f'[AutoPg] max_pages={_MAX_PG}')
# END AUTO PAGES
BASE = 'http://www.dtyg.gov.cn'
LIST_URL = BASE + '/ygxrmzfy/sthj2/list.shtml'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_CODE = 'dtyg'
SITE_NAME = '阳高县-生态环境保护'

HEADERS = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'

def fetch(url):
    r = subprocess.run(['curl', '-sL', '--max-time', '15', '-A', HEADERS, url],
        capture_output=True, timeout=20)
    if r.returncode == 0 and r.stdout:
        return r.stdout.decode('utf-8', errors='replace')
    return ''

def extract_list(html):
    """List items are all on the first page with data attributes"""
    items = []
    pattern = re.compile(
        r'<li[^>]*data-time="([^"]+)"[^>]*data-url="([^"]+)"[^>]*data-title="([^"]+)"',
        re.DOTALL
    )
    for m in pattern.finditer(html):
        pub_date = m.group(1).strip()[:10]
        url_path = m.group(2).lstrip('./')
        title = m.group(3).strip()
        # Fix URL
        full_url = BASE + '/' + url_path
        full_url = full_url.replace('//ygxrmzfy', '/ygxrmzfy')
        if 'ygxrmzfy/' not in full_url and '/ygxrmzfy' not in full_url:
            full_url = full_url.replace(BASE + '/', BASE + '/ygxrmzfy/')
        items.append((full_url, pub_date, title))
    return items

def extract_detail(html):
    title = ''
    pub_date = ''
    content = ''
    
    # Title from <ucaptitle>
    m = re.search(r'<ucaptitle>(.*?)</ucaptitle>', html, re.DOTALL)
    if m:
        title = m.group(1).strip()
    
    # Date from <publishtime>
    m = re.search(r'<publishtime>(.*?)</publishtime>', html, re.DOTALL)
    if m:
        pub_date = m.group(1).strip()[:10]
    
    # Fallback: meta PubDate
    if not pub_date:
        m = re.search(r'<meta\s+name="PubDate"[^>]*content="([^"]+)"', html)
        if m:
            pub_date = m.group(1).strip()[:10]
    
    # Content from TRS_Editor > ucapcontent
    m = re.search(r"<div\s+class=['\"]TRS_Editor['\"][^>]*>.*?<ucapcontent>(.*?)</ucapcontent>", html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        # Remove empty paragraphs
        content = re.sub(r'<p[^>]*>\s*(?:<br\s*/?>\s*)*</p>', '', content)
    else:
        # Fallback: look for content in xl_con3
        m = re.search(r'<div\s+class="xl_con3\s+detailCont"[^>]*>.*?<div\s+class=\'TRS_Editor\'[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
    
    # Append file attachments
    for fm in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|docx?|xlsx?|zip|rar))"[^>]*>([^<]+)</a>', html, re.DOTALL):
        furl = fm.group(1)
        fname = re.sub(r'<[^>]+>', '', fm.group(2)).strip()
        if not furl.startswith('http'):
            furl = BASE + ('/' if not furl.startswith('/') else '') + furl
        content += '\n<p><a href="{}" target="_blank">[附件] {}</a></p>'.format(furl, fname)
    
    return title, pub_date, content

def main():
    print('=' * 60)
    print('Crawling:', SITE_NAME)
    print('URL:', LIST_URL)
    print('Cutoff:', CUTOFF)
    print('=' * 60)
    
    list_html = fetch(LIST_URL)
    if not list_html:
        print('ERROR: Failed to fetch list page')
        return
    
    items = extract_list(list_html)
    print('Found {} articles on page 1'.format(len(items)))
    
    if not items:
        print('No items found. Check list page structure.')
        return
    
    total = 0
    skip_dates = 0
    skip_existing = 0
    empty = 0
    
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT, content TEXT, publish_date TEXT, source_url TEXT UNIQUE, page_url TEXT, site_name TEXT, summary TEXT)")
    
    for url, date_str, title in items:
        # Filter by date
        date_part = date_str[:10] if date_str else ''
        if date_part and date_part < CUTOFF:
            skip_dates += 1
            continue
        
        # Check if exists by source_url
        cur = db.execute("SELECT id FROM gov_raw WHERE source_url = ?", (url,))
        if cur.fetchone():
            skip_existing += 1
            continue
        
        # Fetch detail
        detail_html = fetch(url)
        if not detail_html:
            print('  Empty response: {}'.format(title[:40]))
            empty += 1
            continue
        
        detail_title, detail_date, content = extract_detail(detail_html)
        if not detail_title:
            detail_title = title
        if not detail_date:
            detail_date = date_part
        
        if not content or len(content.strip()) < 10:
            empty += 1
            continue
        
        summary = re.sub(r'<[^>]+>', '', content)[:200].strip()
        
        try:
            sql = "INSERT OR REPLACE INTO gov_raw (title, content, publish_date, source_url, page_url, site_name, summary, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'crawl_dtyg.py')"
            db.execute(sql, (detail_title, content, detail_date, url, url, SITE_NAME, summary))
            total += 1
        except Exception as e:
            print('  DB error: {} - {}'.format(e, title[:40]))
            continue
        
        if total % 20 == 0 and total > 0:
            print('  Progress: {} new, {} existing, {} date-skipped, {} empty'.format(total, skip_existing, skip_dates, empty))
            db.commit()
        
        time.sleep(0.3)
    
    db.commit()
    db.close()
    
    print('=' * 60)
    print('Summary for {}:'.format(SITE_NAME))
    print('  Total found:      {}'.format(len(items)))
    print('  Newly inserted:   {}'.format(total))
    print('  Date-skipped:     {}'.format(skip_dates))
    print('  Already exists:   {}'.format(skip_existing))
    print('  Empty body:       {}'.format(empty))
    print('=' * 60)

if __name__ == '__main__':
    main()
