#!/usr/bin/env python3
"""
Crawler: 重庆市南川区人民政府 - 公示公告
URL: http://www.cqnc.gov.cn/zwxx_197/gsgg/
Pagination: index.html, index_2.html, index_3.html (3 pages, ~32 articles)
Detail content: div.newsContent > div.trs_editor_view
Publish date: <meta name="PubDate" content="...">
"""
import re, sys, os, sqlite3, requests
from datetime import datetime, timedelta
from urllib.parse import urljoin
import os

BASE_URL = 'http://www.cqnc.gov.cn/zwxx_197/gsgg/'
SITE_NAME = '重庆市南川区人民政府-公示公告'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
HEADERS = {'User-Agent': 'Mozilla/5.0 (compatible; HermesCrawler/1.0)'}

CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')

def extract_content(html, detail_url):
    """Extract article content from detail page."""
    m = re.search(r'<div\s+class="newsContent"[^>]*id="zhengjicontent"[^>]*>(.*?)</div>\s*<!--\s*结束\s*文章内容', html, re.DOTALL)
    if m:
        content_div = m.group(1)
        m2 = re.search(r'<div\s+class="trs_editor_view[^"]*"[^>]*>(.*?)</div>\s*</div>', content_div, re.DOTALL)
        if m2:
            c = m2.group(1)
        else:
            c = content_div
    else:
        m = re.search(r'<div\s+class="trs_editor_view[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
        if m:
            c = m.group(1)
        else:
            m = re.search(r'<div\s+class="newsContent[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
            if m:
                c = m.group(1)
            else:
                return ''

    c = re.sub(r'<script[^>]*>.*?</script>', '', c, flags=re.DOTALL)
    c = re.sub(r'\sstyle="[^"]*width\s*:\s*\d+px[^"]*"', '', c, flags=re.I)
    # Convert relative image URLs to absolute
    c = re.sub(
        r'(<img[^>]*src\s*=\s*["\'])(/[^"\']+)(["\'])',
        lambda m: m.group(1) + urljoin('http://www.cqnc.gov.cn/', m.group(2).lstrip('/')) + m.group(3),
        c
    )
    c = re.sub(
        r'(<img[^>]*SRC\s*=\s*["\'])(/[^"\']+)(["\'])',
        lambda m: m.group(1) + urljoin('http://www.cqnc.gov.cn/', m.group(2).lstrip('/')) + m.group(3),
        c
    )
    return c.strip()

def get_pages():
    pages = [BASE_URL]
    for i in range(2, 50):
        url = urljoin(BASE_URL, f'index_{i}.html')
        try:
            r = requests.get(url, headers=HEADERS, timeout=10)
            if r.status_code == 200:
                pages.append(url)
            else:
                break
        except:
            break
    return pages

def parse_list(html, base_url):
    items = re.findall(
        r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>\s*<p>(.*?)</p>\s*<span>(.*?)</span>\s*</a>',
        html, re.DOTALL
    )
    results = []
    for href, title, p, date_str in items:
        detail_url = urljoin(base_url, href)
        title = title.strip() or re.sub(r'<[^>]+>', '', p).strip()
        results.append({'title': title, 'url': detail_url, 'date': date_str.strip()})
    return results

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print(f"  Error: {e}")
        return '', '', ''

    pub_date = ''
    m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
    if m:
        pub_date = m.group(1).split()[0] if ' ' in m.group(1) else m.group(1)

    title_m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    title = re.sub(r'<[^>]+>', '', title_m.group(1)).strip() if title_m else ''

    content = extract_content(html, url)
    return title, pub_date, content

def main():
    incremental = 'incremental' in sys.argv

    print(f"Crawler: {SITE_NAME}")
    print(f"Mode: {'INCREMENTAL' if incremental else 'FULL'}")
    print(f"Cutoff: {CUTOFF_DATE}")

    pages = get_pages()
    print(f"\nPages: {len(pages)}")
    if incremental:
        pages = pages[:1]

    all_articles = []
    for page_url in pages:
        try:
            r = requests.get(page_url, headers=HEADERS, timeout=15)
            r.encoding = 'utf-8'
            html = r.text
        except Exception as e:
            print(f"  Error: {e}")
            continue
        arts = parse_list(html, page_url)
        print(f"  {page_url.split('/')[-1] or 'index.html'}: {len(arts)} articles")
        all_articles.extend(arts)

    print(f"\nTotal: {len(all_articles)}")

    if not incremental:
        before = len(all_articles)
        all_articles = [a for a in all_articles if a['date'] >= CUTOFF_DATE]
        print(f"After 3yr filter: {len(all_articles)} (removed {before - len(all_articles)})")

    if not all_articles:
        print("No articles.")
        return

    # Insert
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    inserted = 0
    for art in all_articles:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, summary, status, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                (SITE_NAME, art['url'], art['url'], art['title'], art['date'],
                 '', 'active', int(art['date'].replace('-', '')) if art['date'] else 0)
            )
            if c.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB error: {e}")
    conn.commit()
    print(f"Inserted: {inserted}")

    if inserted == 0 and not incremental:
        print("All exist.")
        conn.close()
        return

    # Fetch details
    updated = 0
    errors = 0
    for i, art in enumerate(all_articles):
        c.execute("SELECT id, content FROM gov_raw WHERE source_url = ?", (art['url'],))
        row = c.fetchone()
        if not row:
            continue
        pid, existing = row
        if existing and len(existing) > 50:
            continue

        print(f"  [{i+1}/{len(all_articles)}] {art['title'][:40]}...")
        title, pub_date, content = get_detail(art['url'])
        if content:
            c.execute("UPDATE gov_raw SET content = ? WHERE id = ?", (content, pid))
            updated += 1
            print(f"    Content: {len(content)} chars")
        else:
            errors += 1
            print(f"    No content")
        if title and title != art['title']:
            c.execute("UPDATE gov_raw SET title = ? WHERE id = ?", (title, pid))
        if pub_date and pub_date != art['date']:
            c.execute("UPDATE gov_raw SET publish_date = ?, date_rank = ? WHERE id = ?",
                     (pub_date, int(pub_date.replace('-', '')), pid))
        conn.commit()

    conn.close()
    print(f"\nDone: {updated} updated, {errors} errors")

if __name__ == '__main__':
    main()
