#!/usr/bin/env python3
"""
Crawl 宜昌高新区 - 通知公告 (gxq.yichang.gov.cn/list-23-1.html)
Same structure as list-103 but column ID 23.
"""
import json
import sys, re, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
import os

BASE_URL = "http://gxq.yichang.gov.cn"
COLUMN = "23"  # 通知公告
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "宜昌高新区 - 通知公告"
CATEGORY = "宜昌"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
})

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
def insert_to_db(records, conn):
    cursor = conn.cursor()
    inserted = 0
    skipped = 0
    for rec in records:
        if rec["date"] and rec["date"] < THREE_YEARS_AGO:
            skipped += 1
            continue
        content = rec.get("content", "")
        summary = content[:200] if content else ""
        try:
            cursor.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, page_url, title, publish_date, summary, content, category, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", (
                SITE_NAME, rec["url"], rec["title"], rec.get("date", ""),
                summary, content, CATEGORY, rec.get("attachments", "")
            ))
            if cursor.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB error: {e}")
    conn.commit()
    return inserted, skipped


def parse_list_page(html_text):
    records = []
    soup = BeautifulSoup(html_text, 'html.parser')
    for item in soup.find_all('div', class_='list-views'):
        bt_div = item.find('div', class_='listview-bt')
        date_div = item.find('div', class_='listview-date')
        if not bt_div:
            continue
        a = bt_div.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get_text(strip=True)
        if not href or not title:
            continue
        title = re.sub(r'<!--.*?-->', '', title).strip()
        if href.startswith('//'):
            href = 'http:' + href
        elif href.startswith('/'):
            href = BASE_URL + href
        elif not href.startswith('http'):
            href = BASE_URL + '/' + href.lstrip('/')
        date_str = date_div.get_text(strip=True) if date_div else ''
        dm = re.match(r'(\d{4}-\d{2}-\d{2})', date_str)
        if dm:
            date_str = dm.group(1)
        records.append({"title": title, "url": href, "date": date_str})
    return records


def extract_text_content(soup):
    """Extract clean text from detail page: paragraphs, tables, attachments."""
    content_div = soup.find('div', class_='txtcontent-div')
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'content' in str(c).lower() if c else False)

    if not content_div:
        return "", ""

    content_parts = []
    attachments = []

    # Iterate over direct children to preserve order
    for child in content_div.children:
        if child.name == 'p':
            txt = child.get_text(strip=True)
            if txt:
                content_parts.append(txt)
        elif child.name == 'table':
            rows = []
            for tr in child.find_all('tr'):
                cells = [td.get_text(strip=True) for td in tr.find_all(['td', 'th'])]
                if cells:
                    rows.append("| " + " | ".join(cells) + " |")
            if rows:
                content_parts.append("\n".join(rows))
        elif child.name and child.get_text(strip=True):
            # Other block elements (div, etc.) — check for tables inside
            for table in child.find_all('table'):
                rows = []
                for tr in table.find_all('tr'):
                    cells = [td.get_text(strip=True) for td in tr.find_all(['td', 'th'])]
                    if cells:
                        rows.append("| " + " | ".join(cells) + " |")
                if rows:
                    content_parts.append("\n".join(rows))

    # Also get any <p> that might be inside nested divs
    for p in content_div.find_all('p'):
        txt = p.get_text(strip=True)
        if txt and txt not in content_parts:
            content_parts.append(txt)

    # Attachments
    for a in content_div.find_all('a'):
        href = a.get('href', '')
        if any(href.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
            fname = a.get_text(strip=True) or os.path.basename(href)
            if not href.startswith('http'):
                href = BASE_URL + href if href.startswith('/') else BASE_URL + '/' + href.lstrip('/')
            attachments.append({"name": fname, "url": href})

    content = "\n\n".join(dict.fromkeys(content_parts))  # deduplicate preserving order

    # Empty content fallback
    if len(content.strip()) < 20:
        content = '<p><a href="{}">{}</a></p>'.format(
            "NONE",
            soup.find('h1').get_text(strip=True) if soup.find('h1') else "文件"
        )
        if attachments:
            for att in attachments:
                content += '\n📎 <p><a href="{}">{}</a></p>'.format(att['url'], att['name'])

    return content, json.dumps(attachments, ensure_ascii=False) if attachments else ""


def fetch_detail(url):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = 'utf-8'
    except Exception:
        return "", "", "", ""
    soup = BeautifulSoup(resp.text, 'html.parser')

    content, attachments = extract_text_content(soup)

    date_str = ""
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        date_str = meta['content'].strip()[:10]

    full_title = ""
    h1 = soup.find('h1')
    if h1:
        full_title = h1.get_text(strip=True)

    return content, date_str, full_title, attachments


def main():
    incremental = 'incremental' in sys.argv or sys.argv[-1] == '1'
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=5000")
    print(f"=== {SITE_NAME} ===")
    if incremental:
        print("Mode: incremental (page 1 only)")
        pages_to_crawl = 1
    else:
        pages_to_crawl = 5

    url1 = f"{BASE_URL}/list-{COLUMN}-1.html"
    resp = session.get(url1, timeout=30)
    resp.encoding = 'utf-8'
    all_records = parse_list_page(resp.text)
    print(f"  Page 1: {len(all_records)} records")

    if not incremental:
        soup = BeautifulSoup(resp.text, 'html.parser')
        pages_div = soup.find('div', id='pages')
        if pages_div:
            links = pages_div.find_all('a')
            total_pages = 1
            for a in links:
                m = re.search(rf'list-{COLUMN}-(\d+)\.html', a.get('href', ''))
                if m:
                    p = int(m.group(1))
                    if p > total_pages:
                        total_pages = p
            if total_pages > 0 and total_pages < pages_to_crawl:
                pages_to_crawl = total_pages
            total_match = re.search(r'(\d+)条', pages_div.get_text())
            total_rec = total_match.group(1) if total_match else '?'
            print(f"  Total: {total_rec} records, {total_pages} pages, crawling {pages_to_crawl}")

    for page in range(2, pages_to_crawl + 1):
        page_url = f"{BASE_URL}/list-{COLUMN}-{page}.html"
        try:
            resp = session.get(page_url, timeout=30)
            resp.encoding = 'utf-8'
            if resp.status_code != 200:
                break
            records = parse_list_page(resp.text)
            if not records:
                break
            print(f"  Page {page}: {len(records)} records")
            all_records.extend(records)
            time.sleep(0.3)
        except Exception as e:
            print(f"  Page {page} error: {e}")
            break

    print(f"\nTotal collected: {len(all_records)}")
    print("Fetching details...")
    for i, rec in enumerate(all_records):
        if i % 10 == 0:
            print(f"  {i}/{len(all_records)}", flush=True)
        content, date_from_detail, full_title, attachments = fetch_detail(rec["url"])
        if content:
            rec["content"] = content
        if date_from_detail:
            rec["date"] = date_from_detail
        if full_title:
            rec["title"] = full_title
        if attachments:
            rec["attachments"] = attachments

    inserted, skipped = insert_to_db(all_records, conn)
    conn.close()
    print(f"\n=== SUMMARY ===")
    print(f"Inserted: {inserted}, Skipped (old): {skipped}")


if __name__ == "__main__":
    main()
