#!/usr/bin/env python3
"""
东营经济技术开发区 - 审批公示 (dyedz_hpsp)
CMS: SupeSite (jPage AJAX)
totalrecord: 5535, perpage: 15, pages: 369
API returns ~46 records per request (groupSize=3 pages)
Strategy: iterate all 369 pages, deduplicate by URL
"""

import urllib.request, urllib.parse, re, json, os, sys, time, sqlite3

SITE_NAME = "dyedz_hpsp"
BASE_URL = "http://www.dyedz.gov.cn"
API_URL = f"{BASE_URL}/module/web/jpage/dataproxy.jsp"
TEMP_DB = "/root/search.db"
MAX_RETRIES = 3

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch_list(startrecord, endrecord):
    """Fetch a page from the dataproxy API"""
    params = {
        "startrecord": startrecord, "endrecord": endrecord, "perpage": 15,
        "unitid": 129990, "webid": 278, "path": BASE_URL + "/",
        "webname": "东营经济技术开发区", "col": 1, "columnid": 39899,
        "sourceContentType": "", "permissiontype": 0,
    }
    query = urllib.parse.urlencode(params)
    url = API_URL + "?" + query
    data = urllib.parse.urlencode(params).encode("utf-8")
    req = urllib.request.Request(url, data=data, headers={
        **HEADERS, "Content-Type": "application/x-www-form-urlencoded",
        "Referer": f"{BASE_URL}/col/col39899/index.html"
    })
    resp = urllib.request.urlopen(req, timeout=30)
    xml = resp.read().decode("utf-8")
    
    articles = []
    for m in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>', xml):
        href = m.group(1)
        title = m.group(2).strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        # Date from the next span
        after = xml[m.end():]
        date_m = re.search(r'<span>(\d{2}-\d{2}-\d{2})</span>', after)
        date_str = date_m.group(1) if date_m else ""
        if date_str:
            parts = date_str.split("-")
            date_str = f"20{parts[0]}-{parts[1]}-{parts[2]}"
        articles.append({"title": title, "url": href, "date": date_str})
    return articles

def fetch_detail(url):
    """Fetch detail page content"""
    for attempt in range(MAX_RETRIES):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=30)
            html = resp.read().decode("utf-8", errors="replace")
            
            # Try zoom first, then bt_content, then ZJEG
            content = ""
            m = re.search(r'id="zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
            if m:
                content = m.group(1)
            else:
                m = re.search(r'class="bt_content"[^>]*>(.*?)</div>', html, re.DOTALL)
                if m:
                    content = m.group(1)
                else:
                    m = re.search(r'<!--ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end-->', html, re.DOTALL)
                    if m:
                        content = m.group(1)
            
            if content:
                content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
                content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
                text_len = len(re.sub(r'<[^>]+>', '', content).strip())
                if text_len >= 50:
                    return content
                # Too short - check for attachment-only page
                if '.pdf' in html or '/download/' in html:
                    atts = re.findall(r'<a[^>]*href="([^"]*\.pdf)"[^>]*>(.*?)</a>', html, re.DOTALL | re.I)
                    if atts:
                        pdf_links = []
                        for ahref, atext in atts:
                            if not ahref.startswith("http"):
                                ahref = BASE_URL + ahref
                            pdf_links.append(f'<a href="{ahref}" target="_blank">{atext.strip()}</a>')
                        return "<p>附件：</p>" + "".join(pdf_links)
            
            return ""  # Empty content - skip
        except urllib.error.HTTPError as e:
            if e.code == 404:
                return ""  # Deleted page
            if attempt < MAX_RETRIES - 1:
                time.sleep(2)
        except Exception as e:
            if attempt < MAX_RETRIES - 1:
                time.sleep(2)
            else:
                return ""
    return ""

def main():
    os.chdir("/root")
    mode = sys.argv[1] if len(sys.argv) > 1 else "list"
    
    if mode == "list":
        # Phase 1: Collect all unique article URLs
        print("=== Phase 1: Collecting all article URLs ===")
        seen_urls = set()
        all_articles = []
        
        for page in range(1, 370):
            startrecord = (page - 1) * 15 + 1
            endrecord = page * 15
            try:
                articles = fetch_list(startrecord, endrecord)
                new = 0
                for art in articles:
                    if art["url"] not in seen_urls:
                        seen_urls.add(art["url"])
                        all_articles.append(art)
                        new += 1
                print(f"  第{page}页: {len(articles)}条, 新增{new}, 累计{len(all_articles)}")
            except Exception as e:
                print(f"  第{page}页失败: {e}")
            time.sleep(0.3)
        
        print(f"\n总共{len(all_articles)}条唯一文章")
        
        # Save list to JSON for phase 2
        with open("/tmp/dyedz_articles.json", "w") as f:
            json.dump(all_articles, f, ensure_ascii=False)
        print("已保存到 /tmp/dyedz_articles.json")
        
    elif mode == "detail":
        # Phase 2: Fetch details
        print("=== Phase 2: Fetching article details ===")
        
        # Load article list
        with open("/tmp/dyedz_articles.json") as f:
            all_articles = json.load(f)
        
        # Resume from where we left off
        start_from = 0
        if os.path.exists("/tmp/dyedz_progress.txt"):
            with open("/tmp/dyedz_progress.txt") as f:
                start_from = int(f.read().strip())
            print(f"从第{start_from}条继续")
        
        conn = sqlite3.connect(TEMP_DB, timeout=60)
        c = conn.cursor()
        c.execute("""
            CREATE TABLE IF NOT EXISTS gov_raw (
                id INTEGER PRIMARY KEY AUTOINCREMENT,
                title TEXT,
                content TEXT,
                source_url TEXT UNIQUE,
                site_name TEXT,
                publish_date TEXT
            )
        """)
        conn.commit()
        
        total = len(all_articles)
        inserted = 0
        skipped = 0
        empty = 0
        
        for i, art in enumerate(all_articles[start_from:], start_from):
            # Check if already exists
            c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url = ?", (art["url"],))
            if c.fetchone()[0] > 0:
                skipped += 1
                if i % 100 == 0:
                    print(f"  [{i}/{total}] {skipped}已跳过/inserted={inserted}/empty={empty}")
                continue
            
            # Fetch content
            art["content"] = fetch_detail(art["url"])
            
            if not art["content"]:
                empty += 1
                # Skip - don't save empty content
            else:
                try:
                    c.execute(
                        "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                        (art["title"], art["content"], art["url"], SITE_NAME, art["date"])
                    )
                    conn.commit()
                    inserted += 1
                except Exception as e:
                    print(f"  DB error: {e}")
            
            if i % 20 == 0:
                print(f"  [{i}/{total}] done={inserted} skipped={skipped} empty={empty}")
                with open("/tmp/dyedz_progress.txt", "w") as f:
                    f.write(str(i + 1))
        
        conn.close()
        print(f"\n=== 完成 ===")
        print(f"总计: {total}")
        print(f"新增: {inserted}")
        print(f"已存在: {skipped}")
        print(f"空正文(跳过): {empty}")
        
if __name__ == "__main__":
    main()
