#!/usr/bin/env python3
"""crawl_zrzy_tx_pqgs.py - 泰兴市自然资源和规划局-建设项目批前公示"""
import sys, os, json, re, time, requests, warnings, argparse, urllib.parse
from datetime import datetime
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

warnings.filterwarnings('ignore', category=requests.packages.urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = '泰兴市自然资源和规划局-建设项目批前公示'
BASE_URL = 'https://zrzy.jiangsu.gov.cn/tztx/gtzx/tzgg_10466/tzgg_10469'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}

def clean_title_suffix(title):
    """Strip suffix like '_建设项目批前公示_泰兴市自然资源和规划局'"""
    title = re.sub(r'_.*$', '', title).strip()
    return title

def get_page_url(page_num):
    if page_num == 1:
        return BASE_URL + '/index.htm'
    return f'{BASE_URL}/index_{page_num}.htm'

def parse_list_page(html):
    """Parse list page HTML using BeautifulSoup"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for tr in soup.find_all('tr'):
        tds = tr.find_all('td')
        if len(tds) >= 2:
            a = tds[0].find('a')
            if a and a.get('href', '').startswith('./20'):
                href = a.get('href', '')
                title = (a.get('title') or a.get_text(strip=True)).strip()
                if not title:
                    continue
                date_text = tds[1].get_text(strip=True)
                date_m = re.search(r'\d{4}-\d{2}-\d{2}', date_text)
                date_str = date_m.group(0) if date_m else ''
                url_abs = BASE_URL + '/' + href[2:]
                items.append({
                    'title': title,
                    'url': url_abs,
                    'pub_date': date_str
                })
    return items

def fetch_detail(detail_url, item):
    """Fetch detail page and extract content"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = r.apparent_encoding or 'utf-8'
    except Exception as e:
        return item['title'], item['pub_date'], f'[抓取失败] {e}', []
    
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Title: from <title>, strip suffix
    title_tag = soup.find('title')
    title = clean_title_suffix(title_tag.get_text(strip=True)) if title_tag else item['title']
    
    # Date: from meta PubDate
    pub_date = item['pub_date']
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date:
        d = meta_date.get('content', '')[:10]
        if re.match(r'\d{4}-\d{2}-\d{2}', d):
            pub_date = d
    
    # Content: <div id="fontzoom"> > <div class="TRS_Editor">
    content_parts = []
    attachments = []
    
    fontzoom = soup.find('div', id='fontzoom')
    if fontzoom:
        trs = fontzoom.find('div', class_='TRS_Editor')
        if not trs:
            trs = fontzoom
    else:
        trs = soup.find('div', class_='TRS_Editor')
    
    if trs:
        for elem in trs.find_all(['p', 'div', 'img', 'a']):
            style = elem.get('style', '') or ''
            if 'display:none' in style.replace(' ', ''):
                continue
            
            if elem.name == 'img':
                src = elem.get('src', '')
                alt = elem.get('alt', '图片') or '图片'
                if src and not src.startswith('data:'):
                    abs_src = urllib.parse.urljoin(detail_url, src)
                    content_parts.append(f'![{alt}]({abs_src})')
            
            elif elem.name == 'a':
                href = elem.get('href', '')
                text = elem.get_text(strip=True) or '附件'
                if href:
                    abs_href = urllib.parse.urljoin(detail_url, href)
                    ext = os.path.splitext(abs_href)[1].lower()
                    if ext in ('.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar'):
                        attachments.append({'name': text, 'url': abs_href})
                        content_parts.append(f'[附件：{text}]({abs_href})')
                    elif ext in ('.jpg', '.jpeg', '.png', '.gif', '.bmp'):
                        content_parts.append(f'![{text}]({abs_href})')
                    elif text:
                        content_parts.append(f'[{text}]({abs_href})')
            
            elif elem.name in ('p', 'div'):
                if elem.find('div') and not elem.get_text(strip=True):
                    continue
                text = elem.get_text(strip=True)
                if text and text != '\xa0':
                    text = re.sub(r'[ \t]+(?=[\u4e00-\u9fff])', '', text)
                    text = re.sub(r'(?<=[\u4e00-\u9fff])[ \t]+', '', text)
                    text = re.sub(r'\u00a0', ' ', text)
                    content_parts.append(text)
    
    # If no text content, extract images directly
    if not content_parts and trs:
        for img in trs.find_all('img'):
            src = img.get('src', '')
            alt = img.get('alt', '图片') or '图片'
            if src and not src.startswith('data:'):
                abs_src = urllib.parse.urljoin(detail_url, src)
                content_parts.append(f'![{alt}]({abs_src})')
    
    content = '\n\n'.join(content_parts) if content_parts else '[该文档为图片/PDF文件]'
    return title, pub_date, content, attachments

def get_db_count(site_name):
    import sqlite3
    try:
        db = sqlite3.connect('/root/search.db', timeout=10)
        cur = db.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (site_name,))
        cnt = cur.fetchone()[0]
        db.close()
        return cnt
    except:
        return -1

def run(args):
    max_pages = args.pages
    all_items = []
    
    for pn in range(1, max_pages + 1):
        url = get_page_url(pn)
        print(f"  List page {pn}: {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
            r.encoding = r.apparent_encoding or 'utf-8'
        except Exception as e:
            print(f"  Error page {pn}: {e}")
            break
        
        items = parse_list_page(r.text)
        if not items:
            print(f"  No items on page {pn}, stopping")
            break
        
        print(f"  Page {pn}: {len(items)} items")
        all_items.extend(items)
        time.sleep(0.5)
    
    print(f"\n  Total list items: {len(all_items)}")
    
    records = []
    total = len(all_items)
    for i, item in enumerate(all_items):
        if i % 5 == 0:
            print(f"  Detail {i+1}/{total}...")
        try:
            title, pub_date, content, attachments = fetch_detail(item['url'], item)
        except Exception as e:
            print(f"  Error {item['url']}: {e}")
            title, pub_date, content, attachments = item['title'], item['pub_date'], '', []
        
        records.append({
            'title': title,
            'url': item['url'],
            'source_url': item['url'],
            'pub_date': pub_date,
            'site_name': SITE_NAME,
            'content': content,
            'summary': content[:500] if content else title[:200],
            'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
        })
    
    print(f"\n  Total records: {len(records)}")
    
    if records:
        count_before = get_db_count(SITE_NAME)
        push_to_searchdb(records, "zrzy_tx_pqgs")
        count_after = get_db_count(SITE_NAME)
        print(f"  DB: {count_before} → {count_after} (+{count_after - count_before})")
    
    return len(records)

if __name__ == '__main__':
    parser = argparse.ArgumentParser(description='泰兴市自然资源和规划局-建设项目批前公示爬虫')
    parser.add_argument('--pages', type=int, default=5, help='pages to crawl')
    args = parser.parse_args()
    print(f"泰兴市自然资源和规划局-建设项目批前公示: crawling {args.pages} pages")
    cnt = run(args)
    print(f"Done: {cnt} records")
