#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""将《知识库问答库》生成的 Markdown 问答库转换为超星问答库导入模板 qa_import_template.xlsx 同构文件。
列：规则分类 | 标准问题 | 答案 | 规则状态 | 匹配模式 | 相似问法1..10（请参考模板）
零依赖：zipfile 手写 OOXML（inline strings）。"""
import os, re, zipfile, html

BASE = "/home/zyw/Downloads/dl-hub/01-AI课程设计项目/知识库问答库"
QB_ROOT = os.path.join(BASE, "问答库")
OUT = "/home/zyw/Downloads/dl-hub/01-AI课程设计项目/知识库问答库/超星问答库导入-知识库问答库.xlsx"

# 章目录 → 一级分类（≤8 字符）+ 二级分类前缀（节号）
CH_MAP = {
    "第1章 人工智能概述": "人工智能概述",
    "第2章 Python程序设计基础": "Python基础",
    "第3章 NumPy数值分析库": "NumPy分析",
    "第4章 pandas数据分析库": "pandas分析",
    "第5章 计算机视觉技术与应用": "计算机视觉",
    "第6章 智能语音处理与应用": "智能语音",
    "第7章 自然语言处理与应用": "自然语言处理",
    "第8章 生成式大模型应用": "生成式大模型",
    "实验篇": "实验篇",
}
STATUS = "已启用"
MODE = "精确匹配"

def esc(s): return html.escape(str(s), quote=False)

def parse_questions(text):
    """解析问答库 md → [{type,diff,q,options,answer,analysis}]"""
    qs = []
    # 按题号切块：^数字. 【题型·难度】
    blocks = re.split(r'\n(?=\d+\.\s*【)', text)
    for blk in blocks:
        m = re.match(r'(\d+)\.\s*【([^】·]+)·([^】]+)】\s*(.*)', blk, re.S)
        if not m: continue
        _, qtype, diff, rest = m.groups()
        # 选项
        opts = []
        om = re.search(r'(?s)((?:^[ \t]*[A-D][\.、)．].*\n?)+)', rest)
        # 题干 = rest 去掉选项、答案、解析
        rest_body = rest
        opt_re = re.compile(r'^[ \t]*([A-D])[\.、)．]\s*(.*)$', re.M)
        opt_lines = opt_re.findall(rest_body)
        # 提取题干（在第一个选项或"**答案"之前）
        epos = [m2.start() for m2 in [re.search(r'\n[ \t]*[A-D][\.、)．]', rest_body), re.search(r'\*\*答案', rest_body)] if m2]
        cut = min(epos) if epos else len(rest_body)
        qtext = re.sub(r'\s+', ' ', rest_body[:cut]).strip()
        # 选项
        for letter, opt in opt_lines:
            opts.append(f"{letter}. {opt.strip()}")
        # 答案
        am = re.search(r'\*\*答案[：:]\s*([^*\n]+)', rest_body)
        answer = am.group(1).strip() if am else ""
        # 解析
        anm = re.search(r'\*\*解析[：:]\s*(.+?)(?=\n\d+\.|\Z)', rest_body, re.S)
        analysis = re.sub(r'\s+', ' ', anm.group(1)).strip() if anm else ""
        qs.append({"type": qtype, "diff": diff, "q": qtext, "opts": opts, "answer": answer, "analysis": analysis})
    return qs

# 收集所有题目
rows = []  # (分类, 标准问题, 答案)
for ch_dir in sorted(os.listdir(QB_ROOT)):
    ch_path = os.path.join(QB_ROOT, ch_dir)
    if not os.path.isdir(ch_path): continue
    ch = ch_dir
    cat1 = CH_MAP.get(ch, ch)
    for fn in sorted(os.listdir(ch_path)):
        if not fn.endswith(".md"): continue
        # 文件名: 节号-节名.md；实验: 实验N-名称.md
        m = re.match(r'^(.+?)-\s*(.+)\.md$', fn)
        if not m: continue
        nid, title = m.group(1), m.group(2)
        text = open(os.path.join(ch_path, fn), encoding='utf-8').read()
        for q in parse_questions(text):
            # 规则分类：一级/二级（≤8字/级）；节号用作二级
            cat = f"{cat1}/{nid}"
            # 标准问题：题干（选择题含选项？模板80字限制，题干单独，选项并入答案）
            std_q = q['q']
            # 答案：选项 + 答案 + 解析
            ans_parts = []
            if q['opts']:
                ans_parts.append("；".join(q['opts']))
            if q['answer']:
                ans_parts.append("【答案】" + q['answer'])
            if q['analysis']:
                ans_parts.append("【解析】" + q['analysis'])
            ans = "\n".join(ans_parts) if ans_parts else ""
            ans = ans.replace("**", "").strip()
            rows.append((cat, std_q, ans))

print(f"解析到题目行: {len(rows)}")

# 生成 xlsx（inline strings）
def cell(v):
    v = "" if v is None else str(v)
    if v == "":
        return '<c r="" t="inlineStr"><is><t></t></is></c>'
    return f'<c t="inlineStr"><is><t xml:space="preserve">{esc(v)}</t></is></c>'

def row_xml(r, idx):
    c1, c2, c3 = cell(r[0]), cell(r[1]), cell(r[2])
    refs = ['A','B','C']
    cells = "".join(f'<c r="{refs[i]}{idx}">{v}</c>' if False else v for i, v in enumerate([c1, c2, c3]))
    # 更稳妥：不带 r 属性
    cells = c1 + c2 + c3
    return f'<row r="{idx}">{cells}</row>'

# 组装 sheet：第1行提示(跳过)、第2行表头、第3行起数据
header = ["规则分类", "标准问题", "答案", "规则状态", "匹配模式", "相似问法1", "相似问法2", "相似问法3", "相似问法4", "相似问法5", "相似问法6", "相似问法7", "相似问法8", "相似问法9", "相似问法10"]
hdr_cells = "".join(f'<c t="inlineStr"><is><t xml:space="preserve">{esc(h)}</t></is></c>' for h in header)
rows_xml = []
rows_xml.append(f'<row r="2">{hdr_cells}</row>')
for i, r in enumerate(rows, start=3):
    c1 = f'<c t="inlineStr"><is><t xml:space="preserve">{esc(r[0])}</t></is></c>'
    c2 = f'<c t="inlineStr"><is><t xml:space="preserve">{esc(r[1])}</t></is></c>'
    c3 = f'<c t="inlineStr"><is><t xml:space="preserve">{esc(r[2])}</t></is></c>'
    c4 = f'<c t="inlineStr"><is><t xml:space="preserve">已启用</t></is></c>'
    c5 = f'<c t="inlineStr"><is><t xml:space="preserve">模糊匹配</t></is></c>'
    c6 = f'<c t="inlineStr"><is><t xml:space="preserve"> </t></is></c>'
    rows_xml.append(f'<row r="{i}">{c1}{c2}{c3}{c4}{c5}{c6}</row>')

sheet = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
         '<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">'
         '<sheetData>' + "".join(rows_xml) + '</sheetData></worksheet>')

CT = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
      '<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">'
      '<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>'
      '<Default Extension="xml" ContentType="application/xml"/>'
      '<Override PartName="/xl/workbook.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"/>'
      '<Override PartName="/xl/worksheets/sheet1.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml"/>'
      '</Types>')
RELS = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
        '<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">'
        '<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="xl/workbook.xml"/>'
        '</Relationships>')
WB = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
      '<workbook xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main" '
      'xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">'
      '<sheets><sheet name="问答库" sheetId="1" r:id="rId1"/></sheets></workbook>')
WB_RELS = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
           '<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">'
           '<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/worksheet" Target="worksheets/sheet1.xml"/>'
           '</Relationships>')

with zipfile.ZipFile(OUT, 'w', zipfile.ZIP_DEFLATED) as z:
    z.writestr('[Content_Types].xml', CT)
    z.writestr('_rels/.rels', RELS)
    z.writestr('xl/workbook.xml', WB)
    z.writestr('xl/_rels/workbook.xml.rels', WB_RELS)
    z.writestr('xl/worksheets/sheet1.xml', sheet)

print("已生成:", OUT, os.path.getsize(OUT), "字节")

# 校验：分类去重样本
cats = sorted(set(r[0] for r in rows))
print("分类数:", len(cats), "| 样例:", cats[:6])