From: studyhill Date: Sat, 29 Aug 2026 11:37:15 +0000 (+0800) Subject: 新增词汇表导出脚本 X-Git-Url: http://acesimba.cloud/gitweb/?a=commitdiff_plain;h=36e637aca03348e44bb8a69b1119bfe3184f18c2;p=study-mountion.git 新增词汇表导出脚本 导出 74 组切分结果与重点词标记到 Excel(章节/课时号/组标识/单词/词性/词义/ 重点词等 11 列),便于线下核对与手工改标重点词。支持 --out 指定路径。 --- diff --git a/backend/imports/export_vocab.py b/backend/imports/export_vocab.py new file mode 100644 index 0000000..36e9c95 --- /dev/null +++ b/backend/imports/export_vocab.py @@ -0,0 +1,119 @@ +"""导出分组后的词汇表到 Excel。 + +用法: + python3 imports/export_vocab.py # 默认输出到 /home/tools + python3 imports/export_vocab.py --out /path/to.xlsx # 指定路径 + +用途:把 74 组的切分结果与重点词标记导出,便于线下核对、手工改标重点词。 +""" + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import db # noqa: E402 +from openpyxl import Workbook # noqa: E402 +from openpyxl.styles import Alignment, Font, PatternFill # noqa: E402 +from openpyxl.utils import get_column_letter # noqa: E402 + +DEFAULT_OUT = "/home/tools/雅思词汇真经-分组词汇表.xlsx" + +# (表头, 数据键, 列宽) +COLUMNS = [ + ("章节", "chapter", 7), + ("章节标题", "chapter_title", 12), + ("课时号", "lesson_no", 8), + ("组号", "group_no", 6), + ("组标识", "group_code", 11), + ("单词ID", "wid", 10), + ("单词", "word", 22), + ("音标", "phonetic", 16), + ("词性", "pos", 9), + ("词义", "meaning", 46), + ("重点词", "is_key", 8), +] + + +def export(out_path): + conn = db.get_conn() + lessons = conn.execute( + "SELECT lesson_no, title, payload FROM lessons ORDER BY lesson_no" + ).fetchall() + course = conn.execute("SELECT name, code FROM courses LIMIT 1").fetchone() + conn.close() + + wb = Workbook() + ws = wb.active + ws.title = "分组词汇表" + + head_fill = PatternFill("solid", fgColor="4A90C8") + for i, (name, _, width) in enumerate(COLUMNS, start=1): + cell = ws.cell(row=1, column=i, value=name) + cell.font = Font(bold=True, color="FFFFFF") + cell.fill = head_fill + cell.alignment = Alignment(horizontal="center", vertical="center") + ws.column_dimensions[get_column_letter(i)].width = width + ws.freeze_panes = "A2" # 冻结表头,滚动时始终可见 + ws.auto_filter.ref = f"A1:{get_column_letter(len(COLUMNS))}1" + + row_no = 2 + for lesson in lessons: + payload = json.loads(lesson["payload"]) + meta = payload.get("meta", {}) + chapter = meta.get("chapter") + group_no = meta.get("group_no") + for w in payload.get("words", []): + values = { + "chapter": chapter, + "chapter_title": meta.get("chapter_title"), + "lesson_no": lesson["lesson_no"], + "group_no": group_no, + "group_code": f"Ch{chapter}-G{group_no}", + "wid": w.get("wid"), + "word": w.get("word"), + "phonetic": w.get("phonetic"), + "pos": w.get("pos"), + "meaning": w.get("meaning"), + "is_key": "是" if w.get("is_key") else "否", + } + for i, (_, key, _) in enumerate(COLUMNS, start=1): + ws.cell(row=row_no, column=i, value=values.get(key)) + row_no += 1 + + os.makedirs(os.path.dirname(out_path), exist_ok=True) + wb.save(out_path) + + total = row_no - 2 + key_count = sum( + 1 + for lesson in lessons + for w in json.loads(lesson["payload"]).get("words", []) + if w.get("is_key") + ) + return { + "course": f"{course['name']}({course['code']})" if course else "—", + "lessons": len(lessons), + "words": total, + "key_words": key_count, + "path": out_path, + } + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--out", default=DEFAULT_OUT) + args = ap.parse_args() + + db.init_db() + info = export(args.out) + print(f"课程:{info['course']}") + print(f"课时:{info['lessons']} 组") + print(f"词条:{info['words']} 条(其中重点词 {info['key_words']})") + print(f"已导出:{info['path']}") + + +if __name__ == "__main__": + main()