
#!/usr/bin/env python3
import http.client, json, sqlite3, re, sys
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '\u4e30\u57ce\u5e02\u516c\u793a\u516c\u544a'
API_HOST = 'www.jxfc.gov.cn'
API_PORT = 443
API_PATH = '/queryList'
CUTOFF_DATE = '2023-06-01'
CHANNEL = 'gsgg'

HEADERS = {
    'Host': 'www.jxfc.gov.cn',
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Content-Type': 'application/json',
    'Accept': 'application/json',
}

def fetch_page(pn, ps=15):
    conn = http.client.HTTPSConnection(API_HOST, API_PORT, timeout=30)
    conn.request('POST', API_PATH, body=json.dumps({"channelCode":[CHANNEL],"pageSize":ps,"webSiteCode":["fcsrmzf"],"current":pn}), headers=HEADERS)
    return json.loads(conn.getresponse().read().decode('utf-8'))['data']

def clean_content(html):
    if not html: return html
    html = re.sub(r'<script[^>]*>.*?</script>', '', html, flags=re.DOTALL|re.I)
    html = re.sub(r'<style[^>]*>.*?</style>', '', html, flags=re.DOTALL|re.I)
    html = re.sub(r'width\s*:\s*\d+px', 'max-width:100%', html, flags=re.I)
    return html.strip()

def build_attachment_html(article_files, domain='http://www.jxfc.gov.cn'):
    try:
        if isinstance(article_files, str):
            files = json.loads(article_files)
        else:
            files = article_files
        if not files or not isinstance(files, list):
            return ''
        parts = ['<div class="article-attachments"><h3>\u9644\u4ef6</h3><ul>']
        for f in files:
            if not isinstance(f, dict):
                continue
            fname = f.get('fileName', '') or f.get('name', '') or ''
            fpath = f.get('filePath', '') or ''
            domain_name = f.get('domainName', '') or domain
            if fpath:
                if fpath.startswith('/'):
                    url = domain_name + fpath
                elif fpath.startswith('http'):
                    url = fpath
                else:
                    url = domain + '/' + fpath
                parts.append(f'<li><a href="{url}" target="_blank" download>{fname}</a></li>')
            else:
                parts.append(f'<li>{fname}</li>')
        parts.append('</ul></div>')
        return '\n'.join(parts)
    except Exception:
        return ''

def main():
    print(f'=== {SITE_NAME} (v2 - with attachments) ===')
    d = fetch_page(1, 1)
    total = d['total']
    total_pages = (total + 14) // 15
    print(f'Total: {total}, Pages: {total_pages}')

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted, skipped, old = 0, 0, 0
    content_filled = 0
    max_pg = 0
    for pg in range(1, total_pages + 1):
        d = fetch_page(pg, 15)
        results = d.get('results', [])
        if not results:
            break
        dates = [r['source'].get('pubDate', '')[:10] for r in results]
        if dates and min(dates) < CUTOFF_DATE:
            max_pg = pg
            break
        max_pg = pg
    print(f'Fetching: {max_pg} pages')

    for pg in range(1, max_pg + 1):
        try:
            d = fetch_page(pg, 15)
        except Exception as e:
            print(f'  [ERROR] Page {pg}: {e}')
            continue
        results = d.get('results', [])
        if not results:
            break
        for art in results:
            src = art.get('source', {})
            title = (src.get('title', '') or '').strip()
            pub_date = (src.get('pubDate', '') or '')[:10]
            if pub_date and pub_date < CUTOFF_DATE:
                old += 1
                continue
            try:
                urls = json.loads(src.get('urls', '{}'))
                page_url = urls.get('pc', '') or ''
            except:
                page_url = ''
            co = src.get('content', {}) or {}
            raw = ''
            if isinstance(co, dict): raw = co.get('content', '') or ''
            elif isinstance(co, str): raw = co
            content_html = clean_content(raw)

            if not content_html:
                af = src.get('articleFiles', '')
                if af and af != '[{}]':
                    attachment_html = build_attachment_html(af)
                    if attachment_html:
                        content_html = attachment_html
                        content_filled += 1

            aid = art.get('id') or src.get('id', '')
            if not title or not aid:
                skipped += 1
                continue
            try:
                c.execute('INSERT OR IGNORE INTO gov_raw(id,site_name,source_url,page_url,title,publish_date,content,date_rank) VALUES(?,?,?,?,?,?,?,?)',
                          (int(aid), SITE_NAME,
                           f'http://www.jxfc.gov.cn{page_url}' if page_url else '',
                           f'http://www.jxfc.gov.cn{page_url}' if page_url else '',
                           title, pub_date, content_html,
                           int(pub_date.replace('-','')) if pub_date else 0))
                if c.rowcount > 0: inserted += 1
                else: skipped += 1
            except Exception as e:
                print(f'  [DB ERR] {e}')
                skipped += 1
        print(f'  Page {pg}/{max_pg}: +{inserted}, -{skipped}, out{old}')

    conn.commit()
    print(f'\nAttachment content filled: {content_filled}')
    print(f'\n=== Done === Inserted {inserted}, Skipped {skipped}, Old {old}')

if __name__ == '__main__':
    main()
