#!/usr/bin/env python3
"""
大连高新技术产业园区 - 通知公告爬虫
URL: https://www.dlhitech.gov.cn/news/list/1.html?pageNumber=N
CMS: 大连高新区政府门户
列表: ul.WorkPicList1 > li > p > b > a[title] + span.WorkDate
分页: pageNumber=N (10条/页, 共265页)
详情: h2标题 + "发布时间："日期 + 正文段落
"""

import re
import sys
import json
import time
import hashlib
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.dlhitech.gov.cn'
LIST_URL = BASE_URL + '/news/list/1.html?pageNumber={}'
DB_PATH = '/root/search.db'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
    'Referer': BASE_URL + '/news/list/1.html',
}

MAX_PAGES = 5       # 默认增量5页
MAX_PAGES_FULL = 265  # 全量265页


def get_soup(url, session=None):
    s = session or requests.Session()
    resp = s.get(url, headers=HEADERS, timeout=30)
    resp.encoding = 'utf-8'
    if resp.status_code != 200:
        print(f'  WARNING: HTTP {resp.status_code} for {url}')
        return None
    return BeautifulSoup(resp.text, 'html.parser')


def extract_list_page(soup):
    """从列表页提取文章标题、链接、日期"""
    items = []
    ul = soup.find('ul', class_='WorkPicList1')
    if not ul:
        return items
    for li in ul.find_all('li', recursive=False):
        p = li.find('p')
        if not p:
            continue
        a = p.find('a')
        if not a or not a.get('href'):
            continue
        
        title = a.get('title', '').strip() or a.get_text(strip=True)
        href = a['href'].strip()
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)
        
        # 日期
        date_span = p.find('span', class_='WorkDate')
        date = date_span.get_text(strip=True) if date_span else ''
        
        if title:
            items.append({'title': title, 'url': href, 'date': date})
    return items


def extract_detail(soup, url):
    """提取详情页内容"""
    # 标题 - h2
    title = ''
    h2 = soup.find('h2')
    if h2:
        title = h2.get_text(strip=True)

    # 发布信息行: "信息来源：XXX | 发布时间：2026-07-20 15:05:43 | 浏览次数 XX次"
    publish_date = ''
    content_source = ''
    
    # 查找包含"发布时间"的文本
    info_ps = soup.find_all('p')
    for p in info_ps:
        txt = p.get_text(strip=True)
        if '发布时间' in txt:
            m = re.search(r'发布时间[：:]\s*(\d{4}[-/]\d{2}[-/]\d{2})', txt)
            if m:
                publish_date = m.group(1)
            src_m = re.search(r'信息来源[：:]\s*([^|]+)', txt)
            if src_m:
                content_source = src_m.group(1).strip()
            break

    # 正文 - 找到内容主区域
    # 大连高新区: div.fen-article-con > ul.con
    content_div = soup.find('div', class_='fen-article-con')
    if not content_div:
        content_div = soup.find('div', class_='mobile-fen-article-con')
    if not content_div:
        content_div = soup.find('div', class_='newscontnet')
    if not content_div:
        content_div = soup.find('div', class_='contentbox')

    content_parts = []
    images = []

    if content_div:
        # 正文在 ul.con 中
        con_ul = content_div.find('ul', class_='con')
        if not con_ul:
            con_ul = content_div  # fallback: 整个div

        for child in con_ul.find_all(['p', 'table', 'img', 'a'], recursive=True):
            if child.name == 'p':
                # p中的内容可能按<br>分段
                segments = []
                for seg in child.children:
                    if seg.name == 'br':
                        continue
                    txt = seg.get_text(strip=True) if hasattr(seg, 'get_text') else str(seg).strip()
                    if txt and len(txt) > 2:
                        segments.append(txt)
                if segments:
                    content_parts.append('\n\n'.join(segments))
                else:
                    txt = child.get_text(strip=True)
                    if txt and len(txt) > 2:
                        content_parts.append(txt)
            elif child.name == 'table':
                content_parts.append(str(child))
            elif child.name == 'img':
                src = child.get('src', '') or child.get('filepath', '')
                alt = child.get('alt', '')
                if src and 'icon16' not in src:  # 跳过小图标
                    if not src.startswith('http'):
                        src = urljoin(BASE_URL, src)
                    images.append({'src': src, 'alt': alt})
                    content_parts.append('![{0}]({1})'.format(alt, src))
            elif child.name == 'a':
                href = child.get('href', '')
                if href.endswith('.pdf') or href.endswith('.doc') or href.endswith('.docx'):
                    if not href.startswith('http'):
                        href = urljoin(BASE_URL, href)
                    fname = child.get_text(strip=True) or href.split('/')[-1]
                    content_parts.append('[附件: {0}]({1})'.format(fname, href))

    content = '\n\n'.join(content_parts)

    # 如果正文内容很少，可能是纯附件/PDF页
    if len(content.strip()) < 20:
        content = '<p><a href="{1}">{0}</a></p>'.format(title, url)
        if images:
            for img in images:
                content += '\n<p><a href="{1}">查看图片</a></p>'.format(img['alt'], img['src'])

    summary = content[:300] if len(content) > 300 else content

    return {
        'title': title,
        'content': content,
        'summary': summary,
        'publish_date': publish_date,
        'source_url': content_source,
        'attachments': '',
    }


def main():
    import argparse
    parser = argparse.ArgumentParser(description='大连高新区-通知公告爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES, help='最大爬取页数')
    parser.add_argument('--full', action='store_true', help='全量爬取（265页）')
    args = parser.parse_args()

    max_pages = MAX_PAGES_FULL if args.full else args.max_pages

    session = requests.Session()
    all_items = []
    seen_urls = set()
    total_pages = 0
    empty_pages = 0

    for page in range(1, max_pages + 1):
        url = LIST_URL.format(page)
        print(f'列表页 {page}/{max_pages}: {url}')
        soup = get_soup(url, session)
        if not soup:
            print(f'  FAILED: 无法获取页面')
            empty_pages += 1
            if empty_pages >= 3:
                print('  连续3页失败，停止')
                break
            continue

        items = extract_list_page(soup)
        if not items:
            print(f'  空列表，停止')
            break

        # 去重
        new_count = 0
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
                new_count += 1

        total_pages += 1
        empty_pages = 0
        print(f'  本页{len(items)}条，新增{new_count}条，累计{len(all_items)}条')

        if new_count == 0:
            print('  无新内容，停止')
            break

        time.sleep(0.5)

    print(f'\n共获取{len(all_items)}篇文章，开始抓取详情...')
    conn = sqlite3.connect(DB_PATH, timeout=60)
    inserted = 0
    updated = 0
    errors = 0

    site_name = '大连高新技术产业园区'

    for idx, item in enumerate(all_items):
        try:
            print(f'  [{idx+1}/{len(all_items)}] {item["title"][:40]}...')
            soup = get_soup(item['url'], session)
            if not soup:
                errors += 1
                continue

            detail = extract_detail(soup, item['url'])
            if not detail['title']:
                detail['title'] = item['title']

            page_url = item['url']
            publish_date = detail['publish_date'] or item['date'][:10]
            content = detail['content']
            summary = detail['summary']
            title = detail['title']
            source_url = detail['source_url']
            attachments = detail['attachments']

            # 查重
            cur = conn.execute(
                'SELECT id FROM gov_raw WHERE page_url=? AND site_name=?',
                (page_url, site_name)
            )
            row = cur.fetchone()

            if row:
                conn.execute('''
                    UPDATE gov_raw SET title=?, content=?, summary=?, publish_date=?,
                    source_url=?, attachments=? WHERE id=?
                ''', (title, content, summary, publish_date, source_url, attachments, row[0]))
                updated += 1
            else:
                conn.execute('''
                    INSERT INTO gov_raw (title, content, summary, site_name,
                    page_url, publish_date, source_url, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                ''', (title, content, summary, site_name,
                      page_url, publish_date, source_url, attachments))
                inserted += 1

            time.sleep(0.3)

        except Exception as e:
            print(f'  ERROR: {e}')
            errors += 1

    conn.commit()
    conn.close()
    print(f'\n完成！新增{inserted}条，更新{updated}条，错误{errors}条')


if __name__ == '__main__':
    import sqlite3
    main()
