# -*- coding: utf-8 -*-
"""对源目录里卷商原带的 PDF 去广告（陆老师 2026-09-09 拍板「要」覆盖原件）。
   与 strip_ads.py 同一套 redact()（页眉/页脚带 12% 内含卷商关键词的整块），区别：
   - 对象是 源题文件/中考一模二模真题 下所有仍带广告的 PDF（扫前两页）；
   - 覆盖前先按相对路径备份到 vendor_orig_backup/；已有备份不重复备（不覆盖第一次的原件）；
   - 页数不变 + pdftotext 无残留 才覆盖，否则原样保留记 failed。"""
import os, re, sys, json, shutil, subprocess, concurrent.futures as cf
import fitz
sys.argv = ['x', '--dry', '--limit', '0']
HERE = os.path.dirname(os.path.abspath(__file__))
exec(open(os.path.join(HERE, 'strip_ads.py'), encoding='utf-8').read().split('hits = {')[0])   # KEY/BAND/redact/text_has_ad
SRC_ROOT = os.path.join(os.path.dirname(os.path.dirname(HERE)), '源题文件')
TREE = os.path.join(SRC_ROOT, '中考一模二模真题')
BK = os.path.join(HERE, 'vendor_orig_backup')
dry = os.environ.get('STRIP_DRY') == '1'   # 用法：STRIP_DRY=1 python strip_ads_orig.py 为演练

pdfs = [os.path.join(r, f) for r, _, fs in os.walk(TREE) for f in fs if f.lower().endswith('.pdf') and not f.startswith('.')]
def scan(p):
    try: return p if KEY.search(subprocess.run(['pdftotext', '-l', '2', p, '-'], capture_output=True, text=True, timeout=30).stdout) else None
    except Exception: return None
with cf.ThreadPoolExecutor(4) as ex: hits = [x for x in ex.map(scan, pdfs) if x]
print('扫', len(pdfs), '带广告', len(hits))
tmp = os.path.join(HERE, '_strip_tmp2.pdf'); done = failed = 0; log = []
for p in hits:
    rel = os.path.relpath(p, SRC_ROOT); bk = os.path.join(BK, rel)
    try:
        n0 = len(fitz.open(p)); n1, n_rect, sb = redact(p, tmp)
        ok = n1 == n0 and n_rect > 0 and not text_has_ad(tmp)
        if not ok:
            failed += 1; log.append({'src': p, 'ok': False, 'rects': n_rect, 'pages': (n0, n1), 'still_ad': text_has_ad(tmp), 'body_hits': sb}); continue
        if not dry:
            if not os.path.exists(bk):
                os.makedirs(os.path.dirname(bk), exist_ok=True); shutil.copy2(p, bk)
            shutil.copyfile(tmp, p)
        done += 1; log.append({'src': p, 'backup': bk, 'ok': True, 'rects': n_rect, 'pages': n0, 'body_hits': sb})
    except Exception as e:
        failed += 1; log.append({'src': p, 'ok': False, 'err': str(e)[:200]})
if os.path.exists(tmp): os.remove(tmp)
print('%s处理 %d | 失败 %d | 正文区命中被保护 %d' % ('[dry] ' if dry else '', done, failed, sum(x.get('body_hits', 0) for x in log)))
for x in [l for l in log if not l['ok']][:12]: print('  ✗', os.path.basename(x['src']), {k: v for k, v in x.items() if k not in ('src',)})
json.dump(log, open(os.path.join(HERE, 'strip_ads_orig_report.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
