# -*- coding: utf-8 -*-
"""把 v2 录入的初中沪教版同步培优讲义（六上~九下 / 381 专题 / 16675 题）
拍平成「组卷台」归一化 schema，作为**一个新教材「初中培优讲义」并入同一个组卷台**。

由 gen_组卷台.py 调用（两处 hook）：
  data_recs, course_sub, fig_dict = zujuan_v2.load(HERE)
  data += data_recs                        # 并入题目
  course['初中培优讲义'] = course_sub       # 并入课程树
  FIG.update(fig_dict)                      # 并入配图(已 base64)

映射（v2 draft_v2.json → 组卷台 schema）
  tb 教材   = 初中培优讲义
  gg 年级   = 册（六上…九下）          lv = 册 key
  cs 节号   = 该册内专题序号(按 units.json 顺序)
  cd 节显示 = 专题标题(同名变体卷追加卷种后缀)
  s/src     = 讲义 / 册名
  n 题号    = qid_local              q = stem
  t 题型    = 选择/填空/解答(据 section header + options 推断)
  d 难度    = 0(v2 无难度标注)
  f/mf      = figures[0]/figures[1:] (base64 内嵌, key=v2fig/<unit>/<file>)
  k 知识点  = 所属 section 的「知识点N」(若有)
"""
import os, re, io, json, base64
from collections import OrderedDict
from PIL import Image

# 2026-09-07：新增「九上·新教材」（2024 五四制新教材，第27章二次函数/第28章相似三角形），
#   与老教材九上（第24-26章）章号内容都不同，单独一册，排在九上之后（surface:3 定）。
BOOK_ORDER = ['六上', '六下', '七上', '七下', '八上', '八上·期末汇编', '八下', '八下·期中汇编', '九上', '九上·新教材', '九下', '中考·分类汇编']
NEW_BOOKS = {'九上·新教材', '八上·期末汇编', '八下·期中汇编', '中考·分类汇编'}      # 这些册：分组名取 units.json 的 chapter；k 只填真名

# 2026-09-08：「中考·分类汇编」（5 年上海中考真题分类汇编 17 专题 + 2020-2024 版 12 专题，学生版 docx）。
#   每题题首带来源标签「（2025·上海崇明·二模）」。中考真题 / 一模 / 二模 三类组卷台已按整卷收全（上海中考批 666 卷），
#   这册只放库里没有的：三模 / 四模 / 校级模拟 / 模拟预测 / 期末 等。按题首标签过滤，标签原样保留在题面里（忠实转录）。
#   ⚠ 短标签「（2022•上海）」也是中考真题（无字样），一并剔。
SRC_DROP = {
    '中考·分类汇编': re.compile(r'^\s*[（(]{1,2}\s*(?:20\d\d|\d\d-\d\d)[^）)]{0,30}?(?:中考真题|中考|一模|二模)[^）)]{0,12}[）)]'
                             r'|^\s*[（(]{1,2}\s*20\d\d\s*[•·.]\s*上海\s*[）)]'),      # 「（（2025·上海杨浦·二模）」源件双括号
}

# ── 单元分组（章 / 类型）────────────────────────────────────────────
# 这套书每册里三类东西混在一起，平铺会很乱：
#   专题X.Y …………… 逐节讲义（X = 第X章，编号本身就编码了归属）
#   第X章 …………… 章级综合讲义
#   第X章 专题0N ……… 章内难点专项
#   …强化卷/提升卷/模拟卷/月考/单元测试 …… 测评卷
# 按「第X章」聚成组，卷类和期中期末专项各自成组，前端用 <optgroup> 呈现两级。
RE_SEC = re.compile(r'专题\s*(\d+)\s*\.')           # 专题1.2 → 第1章
# ⚠ 章号有**阿拉伯数字和中文数字两种写法**：六上「第1章」、八下「第二十三章」。
#   只认阿拉伯会让高年级整册掉进"其他"（踩过：八下 18 个单元没归组）。
RE_CHAP = re.compile(r'第\s*([0-9]+|[一二三四五六七八九十]+)\s*章')
CN_NUM = {c: i for i, c in enumerate('零一二三四五六七八九', 0)}


def _cn2int(s):
    """中文数字 → int（够用到「三十几章」）。纯阿拉伯直接转。"""
    if s.isdigit():
        return int(s)
    if s == '十':
        return 10
    if '十' in s:
        a, _, b = s.partition('十')
        return (CN_NUM.get(a, 1) if a else 1) * 10 + (CN_NUM.get(b, 0) if b else 0)
    return CN_NUM.get(s, 0)
RE_EXAM = re.compile(r'强化卷|提升卷|培优卷|模拟卷|月考|单元测试|单元自测|学情自测')
RE_MID = re.compile(r'期中')
RE_FIN = re.compile(r'期末')


RE_MONTH = re.compile(r'月考')


def _group_of(title, chapter=''):
    """→ 分组名（<optgroup> 的 label）。新册 units.json 带 chapter 的直接用它。"""
    if chapter:
        return chapter
    t = title or ''
    if RE_MID.search(t):
        return '期中复习'
    if RE_FIN.search(t):
        return '期末复习'
    # ⚠ 月考先判：标题里带「第1章~3章」这类范围，用章号正则会被错归到第1章
    if RE_MONTH.search(t):
        return '月考卷'
    # 学情自测卷跨章（如「第5~6章」），用章号正则会被错归到第5章 → 归测评卷
    if '学情自测' in t:
        return '测评卷'
    m = RE_SEC.search(t) or RE_CHAP.search(t)
    if m:
        return f'第{_cn2int(m.group(1))}章'
    if RE_EXAM.search(t):
        return '测评卷'
    return '其他'


RE_SUBNO = re.compile(r'专题\s*\d+\s*\.\s*(\d+)')      # 专题1.2 → 节号 2
RE_ZHUAN = re.compile(r'专题\s*0*(\d+)(?!\s*\.)')       # 第X章 专题01 → 专项序号 1


def _within_key(title):
    """章内排序：逐节讲义 → 章级综合/单元复习 → 章内专项 → 卷（强化→提升→培优）。

    按教学流程排：先学每一节，再看整章综合，再攻难点专项，最后做卷子。
    原始顺序（units.json 的文件顺序）是乱的——章级综合曾排在两份卷之后。
    """
    t = title or ''
    # 新教材册（2026-09-07）：「27.2（1） …」逐节讲义 → 「考点01 …」考点培优 → 「第27章 单元测试」
    m_new = re.match(r'\s*(\d+)\.(\d+)(?:（(\d)）)?', t)
    if m_new:
        return (0, int(m_new.group(2)), int(m_new.group(3) or 0), '')
    m_b = re.search(r'必刷(\d+)题|百题', t)
    if m_b:
        return (2, int(m_b.group(1)) if m_b.group(1) else 100, 0, t)
    m_k = re.match(r'\s*考点\s*0*(\d+)', t)
    if m_k:
        return (2, int(m_k.group(1)), 0, '')
    if '单元测试' in t:
        return (3, 9, 0, t)
    m = RE_SUBNO.search(t)
    if m:                                   # ① 逐节讲义 专题X.Y
        return (0, int(m.group(1)), 0, '')
    if RE_EXAM.search(t):                   # ④ 卷：先按场次(第一次/第三次月考)，再强化→提升→培优
        rank = 0 if '强化' in t else (1 if '提升' in t else 2)
        m2 = re.search(r'第\s*([一二三四五六七八九十]+|\d+)\s*次', t)
        occ = _cn2int(m2.group(1)) if m2 else 0
        return (3, occ, rank, t)
    m = RE_ZHUAN.search(t)
    if m:                                   # ③ 章内专项 专题01/02…
        return (2, int(m.group(1)), 0, '')
    return (1, 0, 0, t)                     # ② 章级综合讲义 / 单元复习


def _group_key(g):
    """分组排序：第N章 → 期中 → 期末 → 测评卷 → 其他。"""
    m = re.match(r'第(\d+)章', g)
    if m:
        return (0, int(m.group(1)))
    return {'月考卷': (1, 0), '期中复习': (2, 0), '期末复习': (3, 0),
            '测评卷': (4, 0),
            # 2026-09-07 八年级汇编两册的分组（surface:3 定）：八上一组；八下 模拟卷 → 必刷题 → 专题
            '期末真题汇编（新教材）': (0, 0), '模拟卷': (0, 1), '必刷题': (0, 2), '专题': (0, 3)}.get(g, (5, 0))


def _qtype(header, q):
    h = header or ''
    if '选择' in h or '单选' in h:
        return '选择题'
    if '填空' in h:
        return '填空题'
    if '解答' in h or '证明' in h:
        return '解答题'
    return '选择题' if q.get('options') else '解答题'


def _kp(header):
    m = re.match(r'\s*【?\s*知识点\s*(\d+)', header or '')
    return f'知识点{m.group(1)}' if m else ''


def _question_text(q):
    """把一道题**学生看得到的部分**完整拼成题面（零答案卷）。
    v2 draft 把一道题拆成多个字段，只取 stem 会漏掉两类内容：
      ① options（选择题选项）——独立字段，漏了就是"没有选项的选择题"；
      ② sub_questions（小问 (1)(2)(3)）——26% 的题有，漏了就只剩引子/阅读材料，看不到实际要问什么。
    **只拼题面**：主干 stem + 各小问的 no+stem + 选项；
    **不拼** analysis/answer/solution/insight —— 那些是答案侧，零答案卷不能出现。

    选项排版：前端 renderStem 靠 markdown 竖线 `|...|` 识别表格，故把每行选项
    转成 `| A．x | B．y |` 管道行 → 渲染成整齐的 2 列 A/B、C/D 网格。"""
    parts = [q.get('stem', '') or '']
    for sq in (q.get('sub_questions') or []):
        no = (sq.get('no') or '').strip()
        st = (sq.get('stem') or '').strip()
        line = (no + ' ' + st).strip()
        if line:
            parts.append(line)
    text = '\n'.join(p for p in parts if p)
    rows = []
    for line in (q.get('options') or []):
        cells = [c.strip() for c in str(line).split('\t') if c.strip()]
        if cells:
            rows.append('| ' + ' | '.join(cells) + ' |')
    if rows:
        text += '\n' + '\n'.join(rows)
    return text


_banner_cache = {}
def _is_banner(fp):
    """判定装饰性栏头图（强化训练/题型精讲/知识点N/方法技巧… 这类彩色圆角横幅），
    它们在原书里是**图片形式的小节标题**，v2 解析按段落就近把它们错挂成了题目插图。
    可靠信号：栏头是**彩色**(绿/蓝/紫)+**扁宽**+**矮}，数学插图是黑白线稿。
    实测 12143 张唯一图命中 294 张，抽验 40/40 全为栏头，零误伤。"""
    if fp in _banner_cache:
        return _banner_cache[fp]
    try:
        with Image.open(fp) as im:
            im = im.convert('RGB')
            W, H = im.size
            ar = W / H if H else 0
            sm = im.resize((min(W, 120), min(H, 60)))
            px = list(sm.getdata())
            col = sum(1 for r, g, b in px if max(r, g, b) - min(r, g, b) > 40) / len(px)
            res = (col > 0.12) and (ar > 1.8) and (H < 110)
    except Exception:
        res = False
    _banner_cache[fp] = res
    return res


def _fig_b64(fp, maxdim=600, colors=32):
    with Image.open(fp) as im:
        if im.mode == 'P':
            im = im.convert('RGBA')
        if im.mode == 'RGBA':
            bg = Image.new('RGB', im.size, (255, 255, 255))
            bg.paste(im, mask=im.split()[-1]); im = bg
        elif im.mode != 'RGB':
            im = im.convert('RGB')
        W, H = im.size
        sc = min(1.0, maxdim / max(W, H))
        im2 = im.resize((max(1, round(W * sc)), max(1, round(H * sc))), Image.LANCZOS) if sc < 1 else im
        W2, H2 = im2.size
        try:
            qz = im2.quantize(colors=colors, method=Image.FASTOCTREE)
        except Exception:
            qz = im2
        buf = io.BytesIO(); qz.save(buf, 'PNG', optimize=True)
    return 'data:image/png;base64,' + base64.b64encode(buf.getvalue()).decode(), W2, H2


def _find_v2(HERE):
    """从 HERE 向上找 数学学习系统/data/v2（两份生成器分别在 组卷系统/ 与 成果/胡小群题库/，
    到 课程设计机器人 根的层数不同，逐级探测最稳）。"""
    for up in ('.', '..', '../..', '../../..'):
        cand = os.path.normpath(os.path.join(HERE, up, '数学学习系统', 'data', 'v2'))
        if os.path.exists(os.path.join(cand, 'units.json')):
            return cand
    return None


def _sol_figs(V2, key):
    """该单元「解析图混入」排除集 {(file, para)} —— 来自确定性审计 audit_xml/<key>.json。
    这些图是解题辅助线/解析图，挂在题面会泄解法（零答案题库不能显示）。
    按 (文件名, para) 精确匹配（para 唯一，不受 qid 逐小节重置撞名影响）。
    审计文件缺失 → 空集，退化为不过滤（安全）。"""
    p = os.path.join(V2, 'audit_xml', f'{key}.json')
    if not os.path.exists(p):
        return set()
    try:
        r = json.load(open(p, encoding='utf-8'))
    except Exception:
        return set()
    return {(fd.get('file'), fd.get('fig_para')) for fd in r.get('findings', [])
            if fd.get('type') == '解析图混入'}


def load(HERE, verbose=True):
    """返回 (data_records, course_sublist_by_book, fig_base64_dict)。"""
    V2 = _find_v2(HERE)
    if not V2:
        if verbose:
            print('  [初中] 未找到 数学学习系统/data/v2/units.json，跳过')
        return [], OrderedDict(), {}
    UNITS = json.load(open(os.path.join(V2, 'units.json'), encoding='utf-8'))

    by_book = OrderedDict((b, []) for b in BOOK_ORDER)
    for u in UNITS:
        by_book.setdefault(u['book'], []).append(u)

    data, fig_refs = [], {}
    n_banner = [0]
    n_solfig = [0]
    n_srcdrop = [0]
    for book in BOOK_ORDER:
        for cs, u in enumerate(by_book.get(book, []), 1):
            key = u['key']
            dp = os.path.join(V2, 'all', key, 'draft_v2.json')
            if not os.path.exists(dp):
                continue
            d = json.load(open(dp, encoding='utf-8'))
            excl = _sol_figs(V2, key)    # 解析图混入排除集 {(file, para)}
            cd = u['title'].strip()      # 同册内 title 无撞名，直接用作专题显示名
            chapter = (u.get('chapter') or '').strip()   # 新册才有；老册为空 → 走 _group_of 的正则
            qseq = 0                     # 单元内题序，保证 id 唯一（qid_local 逐小节重置会撞车）
            for s in d.get('sections') or []:
                hdr = s.get('header', '')
                kp = _kp(hdr)
                if book in NEW_BOOKS:
                    # 新册 k 一律留空：组装器把「题型 二次函数的辨析」切成了「题型二｜次函数的辨析」，
                    #   小节头本身不可靠；surface:3 说好回头用知识点标注管线统一打 KP。
                    kp = ''
                for q in s.get('items') or []:
                    if q.get('type') != 'question':
                        continue
                    if book in SRC_DROP and SRC_DROP[book].match(q.get('stem') or ''):
                        n_srcdrop[0] += 1        # 来源标签属于库里已有整卷的（中考/一模/二模）→ 不重复上架
                        continue
                    fkeys = []
                    for fg in (q.get('figures') or []):
                        fn = fg.get('file')
                        if not fn:
                            continue
                        if (fn, fg.get('para')) in excl:   # 剔除解析区图（泄解法）
                            n_solfig[0] += 1
                            continue
                        fp = os.path.join(V2, 'all', key, 'assets', 'figures', fn)
                        if _is_banner(fp):       # 剔除误挂的小节标题栏头图
                            n_banner[0] += 1
                            continue
                        rel = f'v2fig/{key}/{fn}'
                        fig_refs[rel] = fp
                        fkeys.append(rel)
                    qid = q.get('qid_local', '')
                    qseq += 1
                    # 新册：解不出的矢量图（EMF 几何图/函数图，没有 MTEF 源码）在草稿里是 ⟦公式:imageN.emf⟧ 占位，
                    #   不能把占位符露给老师。docx_assets 已把它栅格化成 assets/formulas/imageN(.png|_t.png)，
                    #   这里从题面摘掉占位符、把那张图当配图挂上（f/mf），几何图仍然看得见。
                    if book in NEW_BOOKS:
                        _ph = re.findall(r'⟦公式:([^⟧]+)⟧', q.get('stem') or '')
                        for _f in _ph:
                            _stem0 = os.path.splitext(_f)[0]
                            for _cand in (_stem0 + '_t.png', _stem0 + '.png'):
                                _fp = os.path.join(V2, 'all', key, 'assets', 'formulas', _cand)
                                if os.path.exists(_fp):
                                    _rel = f'v2fml/{key}/{_cand}'
                                    fig_refs[_rel] = _fp
                                    fkeys.append(_rel)
                                    break
                        if _ph:
                            q = dict(q, stem=re.sub(r'⟦公式:[^⟧]+⟧', '', q.get('stem') or ''))
                    data.append({
                        'id': f'V2-{key}-{qseq}',
                        'tb': '初中培优讲义', 'gg': book, 'lv': book,
                        'cs': cs, 'cd': cd, 'src': f'{book}讲义', 's': '讲义',
                        'n': qid, 'q': _question_text(q),
                        't': _qtype(hdr, q), 'd': 0,
                        'f': fkeys[0] if fkeys else '', 'mf': fkeys[1:],
                        'p': [], 'k': [kp] if kp else [],
                        'ch': chapter,
                    })

    # 课程树节点 = [序号, 标题, 题数, 分组名]
    # ⚠ 第 4 位是新增的分组名（第N章/期中复习/期末复习/测评卷），前端用 <optgroup> 呈现两级；
    #   老前端只解构前 3 位，加一位不破坏兼容。
    course_sub = OrderedDict()
    for book in BOOK_ORDER:
        nodes = OrderedDict()
        for x in data:
            if x['gg'] != book:
                continue
            nodes.setdefault(x['cs'], [x['cd'], 0, x.get('ch', '')])
            nodes[x['cs']][1] += 1
        if nodes:
            rows = [[s, nd[0], nd[1], _group_of(nd[0], nd[2])] for s, nd in sorted(nodes.items())]
            # 先按分组排（第1章→第N章→期中→期末→测评卷），组内保持原节号顺序
            rows.sort(key=lambda r: (_group_key(r[3]), _within_key(r[1]), r[0]))
            course_sub[book] = rows

    FIG, skipped = {}, 0
    for i, (rel, fp) in enumerate(sorted(fig_refs.items()), 1):
        if not os.path.exists(fp):
            skipped += 1; continue
        try:
            d64, W, H = _fig_b64(fp)
            FIG[rel] = {'d': d64, 'w': W, 'h': H}
        except Exception:
            skipped += 1
        if verbose and i % 3000 == 0:
            print(f'  [初中] 配图 {i}/{len(fig_refs)}')
    if verbose:
        print('  [初中] 题目 %d，配图 %d（跳过缺失 %d，剔栏头 %d，剔解析图 %d，来源标签属已收整卷剔 %d），FIG %.1f MB'
              % (len(data), len(FIG), skipped, n_banner[0], n_solfig[0], n_srcdrop[0], len(json.dumps(FIG)) / 1e6))
    return data, course_sub, FIG


if __name__ == '__main__':
    d, c, f = load(os.path.dirname(os.path.abspath(__file__)))
    print('自测：', len(d), '题', list(c.keys()), len(f), '图')
