#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""并库后核对：直接把生成好的 `组卷台.html` 里的 DATA/COURSE/FIG 抠出来验。

判据（一条不过就别发布）：
  1. **id 全局唯一** —— 前端按 id 建索引，撞 id 会被后者静默覆盖、丢题
  2. 题数 = 旧版题数 + 新增（人工填 --base 核对）
  3. 新教材的每条记录 tb/gg/cd/n/q 齐全，q 非空
  4. 内联图 key 全部能在 FIG 里查到（查不到的题应已被标 ⟦图缺失⟧，不许留空引用）
  5. **零答案**：新教材题面不含【答案】【解析】故选/故答案为
  6. 内嵌 JS 能过 `node --check`（三个大常量 + 主脚本）

用法：python3 verify_import_cn.py [组卷台.html] [--base 57299] [--tb 全国中考真题]
"""
import argparse
import json
import os
import re
import subprocess
import sys
import tempfile

ap = argparse.ArgumentParser()
ap.add_argument('html', nargs='?', default=os.path.join(
    os.path.dirname(os.path.abspath(__file__)), '组卷台.html'))
ap.add_argument('--base', type=int, default=None, help='并库前的题数，用来核对增量')
ap.add_argument('--tb', default='全国中考真题')
a = ap.parse_args()

src = open(a.html, encoding='utf-8').read()
print('文件 %s  %.0f MB' % (os.path.basename(a.html), os.path.getsize(a.html) / 1048576))

def grab(name):
    # ⚠ 三个大常量写在同一行 `const DATA=…, COURSE=…, FIG=…`，别要求前面紧跟 const
    ch = r'\[' if name == 'DATA' else r'\{'
    m = re.search(r'(?:^|[,\s])%s\s*=\s*(?=%s)' % (name, ch), src)
    if not m:
        return None
    i = m.end()
    dec = json.JSONDecoder()
    obj, _ = dec.raw_decode(src, i)
    return obj

DATA, COURSE, FIG = grab('DATA'), grab('COURSE'), grab('FIG')
if DATA is None:
    sys.exit('✗ 抠不出 DATA，先确认生成器真的跑完了')
print('DATA %d 题 · COURSE %d 教材 · FIG %d 图' % (len(DATA), len(COURSE or {}), len(FIG or {})))

bad = 0

# 1 id 唯一
seen, dup = set(), []
for d in DATA:
    if d['id'] in seen:
        dup.append(d['id'])
    seen.add(d['id'])
print(('✗ id 重复 %d 条：%s' % (len(dup), dup[:5])) if dup else '✓ id 全局唯一')
bad += bool(dup)

# 2 题数增量
new = [d for d in DATA if d.get('tb') == a.tb]
print('✓ 新教材「%s」%d 题 · %d 个年级轴' % (a.tb, len(new), len(COURSE.get(a.tb, {}))))
if a.base is not None:
    got = len(DATA) - a.base
    ok = got == len(new)
    print(('✓' if ok else '✗') + ' 题数增量 %d（旧 %d → 今 %d），%s新教材题数'
          % (got, a.base, len(DATA), '等于' if ok else '≠'))
    bad += (not ok)

# 3 字段齐全
miss = [d['id'] for d in new if not (d.get('tb') and d.get('gg') and d.get('cd')
                                     and str(d.get('n', '')) and (d.get('q') or '').strip())]
print(('✗ 字段缺失 %d 条：%s' % (len(miss), miss[:5])) if miss else '✓ 新教材字段齐全')
bad += bool(miss)

# 4 内联图引用全部可查
# ⚠ 判据写错过一次（2026-09-05）：题面**本来就该**留着 `![](asset://id)` 标记，
#   前端 inlImg() 渲染时才就地换成 <img>（上海批 4656 题同样如此）。
#   按「题面还有 asset:// 就算错」判，会把 31831 道正常题报成失败。
#   真判据是**每个 asset id 都能在 FIG 里解析出来**（两个前缀任一命中）。
dangling = [d['id'] for d in new if any(k not in FIG for k in (d.get('im') or []))]
# ⚠ 资产 id 的字符集**按教材不同**：中考批是 sha256-<hex>（只有字母数字和短横），
#   AMC8 是 `AMC8_2000_04_0`（**带下划线**）。漏了下划线的后果不是报错，是这项检查
#   **一条都匹配不到、然后打勾通过** —— 2026-09-05 并 AMC8 时实测「0 道题带内联图」，
#   而真实是 209 道。数字一位不差、检查全绿，其实根本没测到（见 feedback: 先证判据）。
AST = re.compile(r'!\[[^\]]*\]\(asset://([A-Za-z0-9_\-]+)\)')
# ⚠ 前缀表**从被校验对象里读**，别在这儿手抄一份：手抄的必然落后。
#   2026-09-05 并 AMC8 时，前端 FIG_PFX 加了 'am/'、这里没加，于是 228 张好图被报成
#   「解析不出」——校验器落后于被校验对象，把「我不认识」报成「你错了」。
_m = re.search(r'const FIG_PFX\s*=\s*(\[[^\]]*\])', src)
PFX = tuple(json.loads(_m.group(1).replace("'", '"'))) if _m else ('ex/', 'cn/', 'am/')
unres = [d['id'] for d in new for aid in AST.findall(d.get('q') or '')
         if not any((p + aid) in FIG for p in PFX)]
print(('✗ im 指向不存在的图 %d 条' % len(dangling)) if dangling else '✓ 内联图引用全部可查')
n_inline = sum(1 for d in new if AST.search(d.get('q') or ''))
n_im = sum(1 for d in new if d.get('im'))
# ⚠ 覆盖度自检：有 im 却一张都没数到 = 正则没匹配上，这项检查是**空跑**，不能算通过
uncovered = n_im and not n_inline
print(('✗ 题面 asset:// 解析不出图 %d 条：%s' % (len(unres), unres[:5])) if unres
      else ('✗ 检查空跑：%d 道题挂着 im，却一条 asset:// 都没匹配到（正则字符集不对？）' % n_im
            if uncovered else
            '✓ 题面 asset:// 全部能解析出图（%d 道题带内联图，im 字段 %d 道）' % (n_inline, n_im)))
bad += bool(dangling) + bool(unres) + bool(uncovered)

# 5 零答案
ANS = re.compile(r'【\s*(答案|解析|详解|点睛|解答|点评)\s*】|故选\s*[:：]|故答案为')
leak = [d['id'] for d in new if ANS.search(d.get('q') or '')]
print(('✗ 题面含答案痕迹 %d 条：%s' % (len(leak), leak[:5])) if leak else '✓ 零答案（题面无答案痕迹）')
bad += bool(leak)

# 6 内嵌 JS 语法
blocks = re.findall(r'<script>(.*?)</script>', src, re.S)
if subprocess.run(['which', 'node'], capture_output=True).returncode:
    print('— 没装 node，跳过 JS 语法校验')
else:
    nerr = 0
    for i, b in enumerate(blocks, 1):
        with tempfile.NamedTemporaryFile('w', suffix='.js', delete=False, encoding='utf-8') as f:
            f.write(b); fn = f.name
        r = subprocess.run(['node', '--check', fn], capture_output=True, text=True)
        os.unlink(fn)
        if r.returncode:
            nerr += 1
            print('  ✗ 第 %d 个 <script> 语法错：%s' % (i, r.stderr.strip().splitlines()[:2]))
    print(('✗ JS 语法错 %d 段' % nerr) if nerr else '✓ %d 段内嵌 JS 全部通过 node --check' % len(blocks))
    bad += bool(nerr)

print('\n%s' % ('全部通过' if not bad else '✗ %d 项不通过，别发布' % bad))
sys.exit(1 if bad else 0)
