#!/usr/bin/env python3
"""乌拉特前旗人民政府-生态环境"""
import re, sys, os, json, time, requests
from bs4 import BeautifulSoup

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
SITE_NAME = "乌拉特前旗人民政府-生态环境"
GROUP = "内蒙古巴彦淖尔"
MAX_PAGES = 5  # ~75 items
BASE_URL = "http://www.wltqq.gov.cn/zfxxgk/fdzdgknrwltqq/zdlyxxwltqq/hjbh"
LIST_URL = "http://www.wltqq.gov.cn/zfxxgk/fdzdgknrwltqq/zdlyxxwltqq/hjbh/index.html"
TOTAL_PAGES = 20  # pages 1-19 full, page 20 has 9 items

def fetch(url, encoding="utf-8"):
    r = requests.get(url, headers=HEADERS, timeout=30, allow_redirects=True)
    r.encoding = encoding
    return r.text

def resolve_url(rel_url):
    if rel_url.startswith("http"):
        return rel_url
    rel = rel_url.lstrip("./")
    return BASE_URL + "/" + rel

def extract_list(html):
    """Extract (title, url, date) from list page"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    tables = soup.find_all('table')
    if not tables:
        return items
    table = tables[0]
    for tr in table.find_all('tr')[1:]:  # skip header
        tds = tr.find_all(['td', 'th'])
        if len(tds) < 3:
            continue
        a = tds[1].find('a')
        if not a:
            continue
        title = a.get_text(strip=True)
        href = a.get('href', '').strip()
        date = tds[2].get_text(strip=True) if len(tds) > 2 else ""
        date = re.sub(r'\s+', '', date)
        if not re.match(r'\d{4}-\d{1,2}-\d{1,2}', date):
            date = ""
        items.append((title, href, date))
    return items

def extract_detail(html, page_url=""):
    """Extract body content from detail page"""
    soup = BeautifulSoup(html, 'html.parser')

    # Title
    title = ""
    t = soup.find('title')
    if t:
        title = t.get_text(strip=True)
        title = re.sub(r'__\s*生态环境__\s*乌拉特前旗人民政府.*$', '', title).strip()
        title = re.sub(r'__$', '', title).strip()

    # Date from page
    date = ""
    dm = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', html[:2000])
    if dm:
        date = dm.group(1)

    # Body
    body = ""
    for cls_name in ['TRS_UEDITOR', 'zw_content', 'm-con', 'article-content', 'content']:
        el = soup.find(class_=cls_name)
        if not el:
            continue
        parts = []
        for child in el.find_all(['p', 'table']):
            if child.name == 'table':
                parts.append(str(child))
            else:
                txt = child.get_text(strip=True)
                if txt:
                    parts.append(txt)
        if parts:
            body = '\n\n'.join(parts)
            break

    if not body:
        for div in soup.find_all('div'):
            txt = div.get_text(strip=True)
            if len(txt) > 200:
                paras = [p.get_text(strip=True) for p in div.find_all('p') if p.get_text(strip=True)]
                if paras:
                    body = '\n\n'.join(paras)
                    break

    # Check if body is essentially empty (< 20 chars of actual paragraph text)
    if body:
        # Extract only paragraph text for emptiness check (ignore table/scaffolding)
        para_text = ""
        for cls_name in ['TRS_UEDITOR', 'zw_content', 'm-con']:
            el = soup.find(class_=cls_name)
            if el:
                p_texts = [p.get_text(strip=True) for p in el.find_all('p') if len(p.get_text(strip=True)) > 5]
                if p_texts:
                    para_text = "\n\n".join(p_texts)
                break
        if len(para_text.strip()) < 20:
            body = ""
    if not body or len(body.strip()) < 20:
        # Embed images and attachments
        lines = ["[{}]({})".format(title, page_url)]
        # Look for images in the content area
        for cls_name in ['TRS_UEDITOR', 'zw_content', 'm-con']:
            el = soup.find(class_=cls_name)
            if el:
                imgs = el.find_all('img')
                for img in imgs:
                    src = img.get('src', '')
                    if src:
                        alt = img.get('alt', '图片')
                        lines.append("![{}]({})".format(alt, resolve_url(src)))
                # Look for attachment links
                for a in el.find_all('a', href=re.compile(r'\.(doc|docx|pdf|xls|xlsx|png|jpg|jpeg|gif|zip|rar)$', re.I)):
                    href = a.get('href', '')
                    atxt = a.get_text(strip=True) or "附件"
                    if href:
                        lines.append("[{}]({})".format(atxt, resolve_url(href)))
                break
        body = "\n\n".join(lines)

    return title, body, date

def save_to_db(items, site_name, group):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for title, url, date, body in items:
        try:
            c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, site_name, group_name, title, publish_date, summary, content, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'crawl_wltqq_hjbh.py')""",
                (url, site_name, group, title, date, body, body))
            inserted += 1
        except Exception as e:
            print("DB error: {} - {}".format(url, e))
    conn.commit()
    conn.close()
    return inserted

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES)
    args = parser.parse_args()

    max_pages = min(args.max_pages, TOTAL_PAGES)

    all_items = []
    for page in range(max_pages):
        if page == 0:
            url = LIST_URL
        else:
            url = BASE_URL + "/index_{}.html".format(page)
        try:
            html = fetch(url)
            items = extract_list(html)
            all_items.extend(items)
            print("Page {}: {} items".format(page + 1, len(items)))
        except Exception as e:
            print("Page {} FAILED: {}".format(page + 1, e))
        time.sleep(0.5)

    print("\nTotal list items: {}".format(len(all_items)))

    # Fetch details
    detail_items = []
    failed = 0
    for i, (title, rel_url, date) in enumerate(all_items):
        # Skip PDF links
        if rel_url.lower().endswith('.pdf'):
            detail_items.append((title, resolve_url(rel_url), date, "[PDF附件]"))
            continue
        url = resolve_url(rel_url)
        try:
            html = fetch(url)
            d_title, body, d_date = extract_detail(html, page_url=url)
            if not d_date:
                d_date = date
            if not d_title:
                d_title = title
            detail_items.append((d_title, url, d_date, body))
            if (i + 1) % 10 == 0:
                print("  progress: {}/{}".format(i + 1, len(all_items)))
        except Exception as e:
            print("  FAILED: {} - {}".format(url, e))
            failed += 1
        time.sleep(0.3)

    print("\nDetails fetched: {}, failed: {}".format(len(detail_items), failed))

    # Save to DB
    inserted = save_to_db(detail_items, SITE_NAME, GROUP)
    print("Inserted: {}".format(inserted))

    # Summary
    print("\n=== Summary ===")
    print("{} | {} | {} | {} | {}".format(SITE_NAME, inserted, 0, failed, 0))

if __name__ == "__main__":
    main()
