#!/usr/bin/env python3
"""
东营区-通知公告 爬虫
http://www.dyq.gov.cn/col/col39314/index.html
TRS CMS, JPage 分页系统, 首页内嵌45条数据
"""
import re
import json
import time
import argparse
import requests
from bs4 import BeautifulSoup, Tag

SITE_NAME = "东营区-通知公告"
GROUP = "生态环境"
BASE_URL = "http://www.dyq.gov.cn"
LIST_URL = "http://www.dyq.gov.cn/col/col39314/index.html"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
DB_PATH = "/root/search.db"


def fetch(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            return r.text
        except Exception as e:
            print(f"  Error: {e}, retry {attempt+1}")
        time.sleep(2)
    return None


def parse_list_page(html):
    """Parse records from the datastore embedded in the HTML."""
    items = []
    ds_m = re.search(r"<datastore>(.*?)</datastore>", html, re.DOTALL)
    if not ds_m:
        print("  [WARN] No datastore found")
        return items
    ds_raw = ds_m.group(1)
    records = re.findall(r"<record>(.*?)</record>", ds_raw, re.DOTALL)
    print(f"  Found {len(records)} records in datastore")
    for rec in records:
        cdata_m = re.search(r"<!\[CDATA\[(.*?)\]\]>", rec, re.DOTALL)
        if not cdata_m:
            continue
        cdata = cdata_m.group(1)
        soup = BeautifulSoup(cdata, 'html.parser')
        li = soup.find('li')
        if not li:
            continue
        a = li.find('a')
        span = li.find('span')
        if not a:
            continue
        href = a.get('href', '')
        if not href:
            continue
        if not href.startswith('http'):
            href = requests.compat.urljoin(BASE_URL, href)
        title = a.get_text(strip=True)
        date = span.get_text(strip=True) if span else ""
        m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', date)
        if m:
            date = m.group(1).replace('/', '-')
        if title and href:
            items.append((href, title, date))
    return items


def is_attachment_url(href):
    low = href.lower()
    return any(low.endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar', '.wps', '.ofd'])


def process_image_to_parts(img_tag, base_path, content_parts):
    src = img_tag.get('src', '')
    if not src:
        return
    full_src = src if src.startswith('http') else requests.compat.urljoin(base_path, src)
    alt = img_tag.get('alt', '')
    if alt:
        content_parts.append(f"![{alt}]({full_src})")
    else:
        content_parts.append(f"![]({full_src})")


def process_attachment_to_parts(a_tag, base_path, content_parts, attachments):
    href = a_tag.get('href', '')
    if not href or not is_attachment_url(href):
        return
    full_url = href if href.startswith('http') else requests.compat.urljoin(base_path, href)
    link_text = a_tag.get_text(strip=True) or full_url.split('/')[-1]
    link_md = f"[{link_text}]({full_url})"
    attachments.append((link_text, link_md, full_url))
    content_parts.append(link_md)


def render_table(table):
    """保留 HTML 表格结构"""
    return str(table)
def extract_content_from_div(content_div, base_path):
    """Extract content parts, attachments from a content div."""
    content_parts = []
    attachments = []

    for child in content_div.children:
        if not isinstance(child, Tag):
            continue
        tag = child.name.lower()

        if tag == 'p':
            imgs = child.find_all('img')
            table_inner = child.find('table')
            if table_inner:
                content_parts.append(render_table(table_inner))
            elif imgs and not child.get_text(strip=True):
                for img in imgs:
                    process_image_to_parts(img, base_path, content_parts)
            else:
                text = child.get_text('\n', strip=True)
                if text:
                    text = re.sub(r'^视力保护色[：:]\s*', '', text)
                    text = re.sub(r'字体\[大中小\]', '', text).strip()
                    if text:
                        content_parts.append(text)

        elif tag == 'table':
            content_parts.append(render_table(child))

        elif tag in ('div', 'span'):
            cls = child.get('class', [])
            cls_str = ' '.join(cls) if isinstance(cls, list) else str(cls)
            if 'bdshare' in cls_str.lower():
                continue
            for inner in child.children:
                if isinstance(inner, Tag):
                    itag = inner.name.lower()
                    if itag == 'p':
                        txt = inner.get_text('\n', strip=True)
                        if txt:
                            content_parts.append(txt)
                    elif itag == 'table':
                        content_parts.append(render_table(inner))
                    elif itag == 'img':
                        process_image_to_parts(inner, base_path, content_parts)
                    elif itag == 'a':
                        process_attachment_to_parts(inner, base_path, content_parts, attachments)

        elif tag == 'img':
            process_image_to_parts(child, base_path, content_parts)

        elif tag == 'a':
            process_attachment_to_parts(child, base_path, content_parts, attachments)

    # Find all attachments not yet captured
    for a in content_div.find_all('a', href=True):
        href = a['href']
        if is_attachment_url(href):
            full_url = href if href.startswith('http') else requests.compat.urljoin(base_path, href)
            link_text = a.get_text(strip=True) or full_url.split('/')[-1]
            link_md = f"[{link_text}]({full_url})"
            if link_md not in content_parts and link_md not in [a[1] for a in attachments]:
                attachments.append((link_text, link_md, full_url))
                content_parts.append(link_md)

    return content_parts, attachments


def parse_detail(html, url):
    """Parse detail page and return (title, content_text, date, attachments_list)."""
    soup = BeautifulSoup(html, 'html.parser')

    # Title from <title>, stripping site prefix
    title_tag = soup.find('title')
    title = ""
    if title_tag:
        raw = title_tag.get_text(strip=True)
        for prefix in ["东营区人民政府 通知公告 ", "东营区人民政府 ", "通知公告 "]:
            if raw.startswith(prefix):
                title = raw[len(prefix):]
                break
        if not title:
            title = raw

    # Content area
    content_div = soup.find('div', class_='wenzhang') or soup.find('div', class_='bt_content')
    if not content_div:
        return title, "", "", []

    # Date
    date = ""
    ct = content_div.get_text()
    dm = re.search(r'发布日期[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', ct)
    if dm:
        date = dm.group(1).replace('/', '-')

    # Base path for relative URLs
    base_path = '/'.join(url.split('/')[:-1]) + '/'

    content_parts, attachments = extract_content_from_div(content_div, base_path)

    # Join content
    content = '\n\n'.join(content_parts)
    content = re.sub(r'[ \t]+', ' ', content)
    content = re.sub(r'\n{3,}', '\n\n', content)
    content = content.strip()

    # If content is empty, check for image-only page
    if not content:
        img_parts = []
        for img in content_div.find_all('img'):
            src = img.get('src', '')
            if src:
                full_src = src if src.startswith('http') else requests.compat.urljoin(base_path, src)
                img_parts.append(f"![]({full_src})")
        if img_parts:
            content = "\n".join(img_parts)

    # Build attachment links
    attachment_links = [a[1] for a in attachments]

    # Embed attachments
    if not content and attachment_links:
        content = "\n".join(attachment_links)
    elif content and attachment_links:
        content += "\n\n**附件:**\n" + "\n".join(attachment_links)

    return title, content, date, attachments


def crawl(dry_run=False):
    """Main crawl function."""
    print(f"=== {SITE_NAME} 爬虫 ===")
    print(f"目标: {LIST_URL}")
    print()

    html = fetch(LIST_URL)
    if not html:
        print("  [FAIL] Failed to fetch list page")
        return []

    items = parse_list_page(html)
    print(f"总计: {len(items)} 条")

    if dry_run:
        print("\n=== DRY RUN ===")
        for url, title, date in items:
            print(f"  {date} | {title[:60]} | {url}")
        return items

    print("\n=== 获取详情 ===")
    import sqlite3

    for i, (url, list_title, list_date) in enumerate(items):
        print(f"  [{i+1}/{len(items)}] {url.split('/')[-1][:20]}... {list_title[:40]}...")
        html = fetch(url)
        if not html:
            print(f"    [SKIP] Failed to fetch detail")
            continue

        title, content, date, attachments = parse_detail(html, url)
        final_title = title or list_title
        final_date = date or list_date

        if not content:
            print(f"    [SKIP] Empty content")
            continue

        try:
            conn = sqlite3.connect(DB_PATH)
            c = conn.cursor()
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (page_url, title, publish_date, content, summary, site_name, group_name) VALUES (?, ?, ?, ?, ?, ?, ?)",
                (url, final_title, final_date, content, final_title, SITE_NAME, GROUP)
            )
            new = c.rowcount
            conn.commit()
            conn.close()
            print(f"    {'[NEW]' if new else '[skip]'} {final_title[:50]}... | {final_date}")
        except Exception as e:
            print(f"    [DB ERROR] {e}")

    print(f"\n=== 完成 ===")
    return items


if __name__ == '__main__':
    parser = argparse.ArgumentParser(description=f'{SITE_NAME} 爬虫')
    parser.add_argument('--dry-run', action='store_true', help='仅列出不入库')
    parser.add_argument('--pages', type=int, help='别名 (此站仅1页首页数据)')
    args = parser.parse_args()
    crawl(dry_run=args.dry_run)
