#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""生成最终版 xlsx：标准问题 + 答案 + 相似问法（grok 生成优先，规则兜底保证全覆盖）"""
import json, os, re, glob, zipfile, html

BASE = "/home/zyw/Downloads/dl-hub/01-AI课程设计项目/知识库问答库"
ALL = json.load(open(os.path.join(BASE,"题目全集.json"), encoding='utf-8'))
OUT = os.path.join(BASE,"超星问答库导入-知识库问答库-带相似问法.xlsx")

# 加载相似问法（分片 + 全集）
sim = {}
for f in glob.glob(os.path.join(BASE,"相似问法分片","sim-*.json")):
    for k,v in json.load(open(f,encoding='utf-8')).items():
        if str(k) not in sim: sim[str(k)] = v
p = os.path.join(BASE,"相似问法-全集.json")
if os.path.exists(p):
    for k,v in json.load(open(p,encoding='utf-8')).items():
        if str(k) not in sim: sim[str(k)] = v

def rule_fallback(q):
    """规则兜底：3 条语义等价的问法变体（不改变答案）。"""
    qq = re.sub(r'\s+',' ', q).strip()
    qq = qq.rstrip('？?。.')
    v1 = f"请问{qq}？"
    v2 = f"帮忙回答：{qq}？"
    v3 = re.sub(r'^下列关于(.+?)的说法，?正确的是', r'关于\1，哪个说法正确', qq)
    if v3 == qq:
        v3 = f"请说明：{qq}"
    return [v1, v2, v3]

def sim3(idx, q):
    v = sim.get(str(idx))
    if v and len(v) >= 3:
        return v[:3]
    # 少于3条则用规则补足
    f = rule_fallback(q)
    out = list(v[:2] if v else [])
    for x in f:
        if len(out) >= 3: break
        if x not in out: out.append(x)
    return out[:3]

def esc(s): return html.escape(str(s), quote=False)

rows_xml = []
header = ["规则分类","标准问题","答案","规则状态","匹配模式","相似问法1","相似问法2","相似问法3","相似问法4","相似问法5","相似问法6","相似问法7","相似问法8","相似问法9","相似问法10"]
hdr = "".join(f'<c t="inlineStr"><is><t xml:space="preserve">{esc(h)}</t></is></c>' for h in header)
rows_xml.append(f'<row r="2">{hdr}</row>')

rule_filled = 0
grok_filled = 0
for i, it in enumerate(ALL):
    sl = sim3(i, it["q"])
    if str(i) in sim and len(sim[str(i)]) >= 3:
        grok_filled += 1
    else:
        rule_filled += 1
    # 答案组装
    ans_parts = []
    if it["opts"]: ans_parts.append("；".join(it["opts"]))
    if it["answer"]: ans_parts.append("【答案】"+it["answer"])
    if it["analysis"]: ans_parts.append("【解析】"+it["analysis"])
    ans = "\n".join(ans_parts).replace("**","").strip()
    r = i + 3
    cells = []
    vals = [it["cat"], it["q"], ans, "已启用", "模糊匹配"] + sl + [" "] * (10 - len(sl))
    for v in vals:
        cells.append(f'<c t="inlineStr"><is><t xml:space="preserve">{esc(v)}</t></is></c>')
    rows_xml.append(f'<row r="{r}">{"".join(cells)}</row>')

sheet = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
         '<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">'
         '<sheetData>' + "".join(rows_xml) + '</sheetData></worksheet>')
CT = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
      '<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">'
      '<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>'
      '<Default Extension="xml" ContentType="application/xml"/>'
      '<Override PartName="/xl/workbook.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"/>'
      '<Override PartName="/xl/worksheets/sheet1.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml"/>'
      '</Types>')
RELS = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
        '<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">'
        '<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="xl/workbook.xml"/>'
        '</Relationships>')
WB = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
      '<workbook xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main" '
      'xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">'
      '<sheets><sheet name="问答库" sheetId="1" r:id="rId1"/></sheets></workbook>')
WB_RELS = ('<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
           '<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">'
           '<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/worksheet" Target="worksheets/sheet1.xml"/>'
           '</Relationships>')
with zipfile.ZipFile(OUT,'w',zipfile.ZIP_DEFLATED) as z:
    z.writestr('[Content_Types].xml', CT)
    z.writestr('_rels/.rels', RELS)
    z.writestr('xl/workbook.xml', WB)
    z.writestr('xl/_rels/workbook.xml.rels', WB_RELS)
    z.writestr('xl/worksheets/sheet1.xml', sheet)
print("已生成:", OUT, os.path.getsize(OUT), "字节")
print(f"相似问法来源: grok {grok_filled} 题 / 规则兜底 {rule_filled} 题")
