#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
阳江高新区（阳江滨海新区） - 环境保护信息公开 爬虫
NFCMS
List: http://www.yjgx.gov.cn/zwgk/zdlyxxgk/hjbhxxgk/index.html (+ index_2.html ~ index_20.html)
Detail: /zwgk/zdlyxxgk/hjbhxxgk/content/post_XXXXX.html
"""
import requests, re, json, sqlite3, time, os, sys
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
SCRIPT_NAME = 'crawl_yjgx_hjbh.py'
SITE_NAME = '阳江高新区-环境保护'
PROVINCE = '广东'
BASE_URL = 'http://www.yjgx.gov.cn'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

conn = sqlite3.connect(DB_PATH, timeout=60)
cur = conn.cursor()
total_new = 0


def save_article(title, page_url, publish_date, content_text, attachments):
    global total_new
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    try:
        cur.execute("""
            INSERT OR IGNORE INTO gov_raw (title, page_url, publish_date, content, site_name, category, attachments)
            VALUES (?, ?, ?, ?, ?, ?, ?)
        """, (title.strip(), page_url, publish_date, content_text.strip(), SITE_NAME, '环境保护信息公开', attachments_json))
        if cur.rowcount > 0:
            total_new += 1
            return True
    except Exception as e:
        pass
    return False


def table_to_md(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except:
        return None, []

    soup = BeautifulSoup(html, 'html.parser')

    content_div = soup.find('div', class_='is-contentbox')
    if not content_div:
        content_div = soup.find('div', id='zoom')
    if not content_div:
        content_div = soup.find('div', class_='content_nr')
    if not content_div:
        zoom = soup.find('div', id=lambda i: i and 'zoom' in i.lower() if i else None)
        if zoom:
            content_div = zoom
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'content' in c.lower() if c else None)
    if not content_div:
        return None, []

    for s in content_div.find_all('script'):
        s.decompose()
    for s in content_div.find_all('style'):
        s.decompose()

    table_tags = content_div.find_all('table')
    table_mds = []
    for t in table_tags:
        all_text = t.get_text(strip=True)
        if len(all_text) < 15:
            t.decompose()
            continue
        md = table_to_md(t)
        if md:
            table_mds.append(md)
        t.decompose()

    raw = content_div.get_text()
    lines = [l.strip() for l in raw.split('\n') if l.strip() and len(l.strip()) > 3]

    skip_kw = ['【打印本页】', '【关闭窗口】', '打印本页', '关闭窗口', '转载分享：', '浏览量：', '相关附件：', '相关稿件：']
    filtered = [l for l in lines if not any(s in l for s in skip_kw)]

    result = []
    for line in filtered:
        if any(k in line for k in ['来源：', '发布日期：']):
            continue
        result.append(line)

    if table_mds:
        result.append('')
        result.extend(table_mds)

    content_text = '\n\n'.join(result)

    attachments = []
    for a in soup.find_all('a', href=True):
        href = a['href']
        name = a.get_text(strip=True) or href.split('/')[-1].split('?')[0]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            if href.startswith('/'):
                href = BASE_URL + href
            attachments.append({'name': name, 'url': href})

    seen = set()
    unique = []
    for a in attachments:
        if a['url'] not in seen:
            seen.add(a['url'])
            unique.append(a)

    return content_text, unique


# Main
print('Scraping 阳江高新区-环境保护...')
total_pages = 20

for page in range(1, total_pages + 1):
    if page == 1:
        url = f'{BASE_URL}/zwgk/zdlyxxgk/hjbhxxgk/index.html'
    else:
        url = f'{BASE_URL}/zwgk/zdlyxxgk/hjbhxxgk/index_{page}.html'

    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print(f'  Page {page} error: {e}')
        if page > 1:
            break
        continue

    soup = BeautifulSoup(html, 'html.parser')
    articles = []

    for ul in soup.find_all('ul'):
        lis = ul.find_all('li')
        if len(lis) > 20:
            for li in lis:
                a = li.find('a', href=True)
                if not a:
                    continue
                href = a['href']
                title = a.get('title', '') or a.get_text(strip=True)
                if len(title) < 10:
                    continue

                if href.startswith('/'):
                    full_url = BASE_URL + href
                elif not href.startswith('http'):
                    full_url = BASE_URL + '/' + href.lstrip('/')
                else:
                    full_url = href

                date_span = li.find('span', class_='date')
                date = date_span.get_text(strip=True) if date_span else ''
                if not date:
                    dm = re.search(r'\d{4}-\d{2}-\d{2}', li.get_text())
                    date = dm.group(0) if dm else ''

                articles.append((title, full_url, date))
            break

    if not articles:
        print(f'  Page {page}: no articles, stopping')
        break

    for title, page_url, date in articles:
        if not date:
            continue
        if date < '2020-01-01':
            continue

        content_text, attachments = fetch_detail(page_url)
        if not content_text:
            content_text = f'<p><a href="{page_url}">{title}</a></p>'

        save_article(title, page_url, date, content_text, attachments)

    print(f'  Page {page}/{total_pages}: {len(articles)} articles, new so far: {total_new}')

conn.commit()
conn.close()
print(f'\nDone! New: {total_new}')
