#!/usr/bin/env python3
"""Continue remaining crawlers - cooperative approach"""
import json
import subprocess
import os
import sys
import time

BASE_DIR = os.path.expanduser('~/gov_crawler')
VENV_PYTHON = os.path.join(BASE_DIR, 'venv', 'bin', 'python3')

def load_config():
    with open(os.path.join(BASE_DIR, 'daily_crawl_config.json'), encoding='utf-8') as f:
        return json.load(f)['crawlers']

def run_one(script, args='', timeout=300, retry=1):
    script_path = os.path.join(BASE_DIR, script)
    cmd = f'cd "{BASE_DIR}" && "{VENV_PYTHON}" -u "{script_path}" {args}'
    for attempt in range(retry + 1):
        try:
            r = subprocess.run(cmd, shell=True, capture_output=True, text=True, timeout=timeout)
            out = r.stdout + r.stderr
            code = r.returncode
            if code == 0 or attempt == retry:
                return out, code
            print(f"  Retry {attempt+1}/{retry}...", flush=True)
        except subprocess.TimeoutExpired:
            if attempt == retry:
                return "[TIMEOUT]", -1
            print(f"  Timeout, retry {attempt+1}/{retry}...", flush=True)
    return "[FAIL]", -1

def get_increment(out):
    """Extract new item count from output"""
    import re
    new = 0
    for line in out.split('\n'):
        m = re.search(r'(?:新[增加]|新数据|新增记录)\s*[：:]\s*(\d+)', line)
        if not m:
            m = re.search(r'(?:入库|新增记录)\s*(\d+)\s*条', line)
        if not m:
            m = re.search(r'新增[：:]?(\d+)', line)
        if m:
            new = max(new, int(m.group(1)))
    return new

print(f"{'='*60}")
print("REMAINING DIRECT SERVER CRAWLERS (42 script names)")
print(f"{'='*60}")

# Scripts that already ran: crawl_xx_sthj, crawl_nqs_sthj, crawl_hg_hjbh, 
# crawl_xyhpgk, crawl_cedz_gs, crawl_jieyang_gsgg, crawl_taihe_yhjg,
# crawl_baiyinqu, crawl_boxing, crawl_cz_sthjj, crawl_debaoenv, crawl_eiacloud,
# crawl_hipac, crawl_huainan, crawl_pingshuo, crawl_qingdao, crawl_taixing,
# crawl_whchem, crawl_xinhui, crawl_sthj_xz, crawl_xjws, crawl_yanshougs

remaining_direct = [
    ('crawl_ylnh.py', '', '延长中煤榆林能源化工', 120),
    ('crawl_czeia.py', '1', '常州环评网', 120),
    ('crawl_cqbdhb.py', '1', '重庆环科源博达', 120),
    ('crawl_haoyuan.py', '', '安徽昊源化工', 120),
    ('crawl_chinataier.py', '', '临沂泰尔化工', 120),
    ('crawl_sjzdaily.py', '1', '石家庄新闻网', 120),
    ('crawl_dezhoudaily.py', '', '德州新闻网', 120),
    ('crawl_ruilinhuagong.py', '', '睿霖化工', 120),
    ('crawl_zpc.py', '', '浙石化', 120),
    ('crawl_sinopec.py', '--incremental', '中石化统一爬虫', 300),
    ('crawl_lanshantunhe.py', '1', '蓝山屯河-专题专栏', 120),
    ('crawl_hky.py', '1', '天津环科源-信息公告', 120),
    ('crawl_ynws_sthj.py', '', '文山州生态环境局', 120),
    ('crawl_zge_hjpj.py', '1', '准格尔旗环保审批', 120),
    ('crawl_zgwn_tzgg.py', '', '万年县通知公告', 120),
    ('crawl_ypx_sthj.py', '', '永平县生态环境', 120),
    ('crawl_yuanqu_sthj.py', '', '垣曲县生态环境', 120),
    ('crawl_yuwangtai_tzgg.py', '', '禹王台区通知公告', 120),
    ('crawl_zgda_wsjk.py', '', '大安区卫生健康', 120),
    ('crawl_yx_gsgg.py', '', '阳新县公示公告', 120),
    ('crawl_zuoyun.py', '', '左云县人民政府', 120),
    ('crawl_xiangyin.py', '', '湘阴县人民政府', 120),
    ('crawl_longchang.py', '--test', '隆昌市人民政府', 120),
    ('crawl_yishui.py', '', '沂水县人民政府', 120),
    ('crawl_yiyang.py', '', '益阳市人民政府', 120),
    ('crawl_gdee_spqgs.py', '', '广东省生态环境厅-环评审批公示', 120),
    ('crawl_zzhkgq_hpgs.py', '', '郑州航空港区-环评公示', 120),
    ('crawl_zx_tzgg.py', '', '镇雄县通知公告', 120),
    ('crawl_zhangwu_tzgg.py', '1', '彰武县通知通告', 120),
    ('crawl_jiangyin_tzgg.py', '', '江阴临港开发区-通知公告', 120),
    ('crawl_zhanyi_hpgl.py', '', '沾益区环评管理', 120),
    ('crawl_zhostar_hp.py', '', '广东奥思特环保-项目公示', 120),
    ('crawl_cngy_hjbh.py', '', '广元经开区-环境保护', 120),
    ('crawl_hetang_tzgg.py', '1', '荷塘区通知公告', 120),
    ('crawl_suyu_gsgg.py', '', '宿豫区公示公告', 120),
    ('crawl_longchang_gsgg.py', '', '隆昌市公示公告', 120),
    ('crawl_teda_hpgs.py', '1', '天津经开区-行政许可', 120),
    ('crawl_ahjs_hpgs.py', '1', '安徽界首市-环评公示', 120),
    ('crawl_ahys_hpgs.py', '1', '安徽颍上县-环评公示', 120),
    ('crawl_daying_gk.py', '1', '大英县-信息公开', 120),
    ('crawl_dongyue_news.py', '', '东岳化工-新闻公告', 120),
    ('crawl_gzcsx_tzgg.py', '1', '贵州长顺县-通知公告', 120),
]

results = []
for script, args, name, to in remaining_direct:
    print(f"\n[{results.count(True)+results.count(False)+1}/{len(remaining_direct)}] {name}...", end=' ', flush=True)
    start = time.time()
    out, code = run_one(script, args, timeout=to, retry=1)
    elapsed = time.time() - start
    new = get_increment(out)
    status = '✅' if code == 0 else '❌'
    results.append(code == 0)
    
    # Summary line
    summary_parts = []
    if new > 0:
        summary_parts.append(f"+{new}条")
    if code != 0:
        # Grab last 3 lines
        tail = out.strip().split('\n')[-3:]
        summary_parts.append(' '.join(t.strip() for t in tail if t.strip()))
    s = ', '.join(summary_parts) if summary_parts else ''
    print(f'{status} ({elapsed:.0f}s) {s}', flush=True)

direct_ok = sum(1 for r in results if r)
direct_total = len(results)
print(f"\n{'='*60}")
print(f"DIRECT SERVER CONTINUED: {direct_ok}/{direct_total} succeeded")
print(f"{'='*60}")
