#!/usr/bin/env python3
"""大武口-通知公告 爬虫"""
import os, re, sqlite3, time
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin



import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
BASE = 'https://www.dwk.gov.cn'
SITE_NAME = '大武口通知公告'
DB = os.getenv("SEARCH_DB", "/root/search.db")
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36'

tz = timezone(timedelta(hours=8))
cutoff = datetime.now(tz) - timedelta(days=365*3)
cutoff_str = cutoff.strftime('%Y-%m-%d')
print(f'Cutoff: {cutoff_str}')


def extract_zoomcon(html):
    """Extract content from mainText div (id=zoomcon)."""
    marker = 'id="zoomcon"'
    idx = html.find(marker)
    if idx < 0:
        return ''
    div_start = html.rfind('<div', 0, idx)
    if div_start < 0:
        return ''
    open_gt = html.find('>', div_start)
    if open_gt < 0:
        return ''
    start = open_gt + 1
    depth = 1
    i = start
    while depth > 0 and i < len(html):
        if html[i:i+4] == '<div':
            skip = html.find('>', i)
            if skip > i and skip - i < 100:
                tag = html[i+1:skip]
                if not tag.startswith('/') and not tag.startswith('!--') and not tag.startswith('?'):
                    depth += 1
                i = skip + 1
            else:
                i += 1
        elif html[i:i+5] == '</div':
            close_gt = html.find('>', i)
            if close_gt > 0:
                depth -= 1
                i = close_gt + 1
            else:
                i += 1
        else:
            i += 1
    if depth == 0:
        return html[start:i-6].strip()
    return ''


def fetch_list(page):
    """Fetch a list page. Page 0 = no suffix, page 1+ = index_N.html"""
    if page == 0:
        url = f'{BASE}/xwzx/tzgg/'
    else:
        url = f'{BASE}/xwzx/tzgg/index_{page}.html'
    html = os.popen(f'curl -sL --max-time 20 -H "User-Agent: {UA}" "{url}"').read()
    return html


def fetch_detail(url):
    """Fetch a detail page."""
    html = os.popen(f'curl -sL --max-time 20 -H "User-Agent: {UA}" "{url}"').read()
    return html


page = 0
count = 0
seen_urls = set()

while page < (_MAX_PG or 100):  # safety limit
    html = fetch_list(page)
    if not html.strip():
        print(f'Page {page}: empty, stop')
        break

    # Extract list items from <ul class="newslist">
    m = re.search(r'<ul class="newslist">(.*?)</ul>', html, re.DOTALL)
    if not m:
        print(f'Page {page}: no newslist, stop')
        break

    items = m.group(1)
    lis = re.findall(r'<li>(.*?)</li>', items, re.DOTALL)
    if not lis:
        print(f'Page {page}: no items, stop')
        break

    print(f'Page {page}: {len(lis)} items')
    all_before = True

    for li in lis:
        # Extract link
        m_link = re.search(r'href=["\']([^"\']+)["\']', li)
        if not m_link:
            continue
        rel_path = m_link.group(1)
        # Resolve relative URL
        if rel_path.startswith('./'):
            full_url = BASE + '/xwzx/tzgg/' + rel_path[2:]
        elif rel_path.startswith('/'):
            full_url = BASE + rel_path
        elif not rel_path.startswith('http'):
            full_url = BASE + '/xwzx/tzgg/' + rel_path
        else:
            full_url = rel_path

        # Extract list date (MM-DD) and estimate year from URL path
        m_date = re.search(r'<span>(\d{2})-(\d{2})</span>', li)
        list_date_str = ''
        if m_date:
            mm, dd = m_date.group(1), m_date.group(2)
            # Try to get year from URL path (e.g., /202606/t20260615_xxx.html)
            ym = re.search(r'/(20\d{2})(\d{2})/', full_url)
            if ym:
                y, m = ym.group(1), ym.group(2)
                # If month matches, use path year
                if m == mm:
                    list_date_str = f'{y}-{mm}-{dd}'
                else:
                    # Cross-year case, use path year
                    list_date_str = f'{y}-{mm}-{dd}'
            else:
                # Fallback: use current year, but check if month suggests previous year
                now = datetime.now(tz)
                year = now.year
                if int(mm) > now.month:
                    year -= 1
                list_date_str = f'{year}-{mm}-{dd}'

        # Skip if already seen
        if full_url in seen_urls:
            continue
        seen_urls.add(full_url)

        # Skip external links (WeChat, nxpta, etc.) — no content to extract
        if not full_url.startswith(BASE):
            print(f'  Skip external: {full_url[:60]}')
            continue

        # Fetch detail
        time.sleep(0.8)
        detail_html = fetch_detail(full_url)
        if not detail_html or len(detail_html) < 500:
            print(f'  Skip empty detail: {full_url}')
            continue

        # Extract date from detail page (more reliable)
        pub_date = ''
        m_pub = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', detail_html)
        if m_pub:
            pub_date = m_pub.group(1)
        else:
            m_pub2 = re.search(r'<b>(\d{4}-\d{2}-\d{2})</b>', detail_html)
            if m_pub2:
                pub_date = m_pub2.group(1)
            else:
                pub_date = list_date_str

        if not pub_date:
            print(f'  No date, skip: {full_url}')
            continue

        if pub_date < cutoff_str:
            continue
        all_before = False

        # Extract title from meta (cleanest)
        title = ''
        m_t = re.search(r'<meta name="ArticleTitle" content="([^"]+)"', detail_html)
        if m_t:
            title = m_t.group(1).strip()
        else:
            m_t2 = re.search(r'<div class="article-title">\s*([^<]+?)\s*</div>', detail_html)
            if m_t2:
                title = m_t2.group(1).strip()

        # Extract content from zoomcon
        content_html = extract_zoomcon(detail_html)
        if not content_html or len(content_html) < 20:
            print(f'  No content: {full_url[:60]}')
            continue

        # Clean up scripts and styles
        content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL)
        content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.DOTALL)
        content_html = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content_html, flags=re.DOTALL)
        content_html = content_html.strip()
        if not content_html or len(content_html) < 20:
            print(f'  Empty after cleanup: {full_url[:60]}')
            continue

        # Save to DB
        try:
            conn = sqlite3.connect(DB, timeout=60)
            c = conn.cursor()
            c.execute('INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary, script_name) VALUES (?,?,?,?,?,?,?,?, \'crawl_dwk.py\')',
                      (None, title, content_html, pub_date, full_url, full_url, SITE_NAME, title[:200]))
            conn.commit()
            conn.close()
            count += 1
            print(f'  OK [{pub_date}] {title[:40]}')
        except Exception as e:
            print(f'  DB error: {e}')

    if all_before:
        print('All before cutoff, stop')
        break
    page += 1
    time.sleep(2)

print(f'\nDone. Total: {count} articles imported.')