From: studyhill Date: Sat, 29 Aug 2026 09:26:54 +0000 (+0800) Subject: 雅思词汇真经课程:3073 词导入 + 五阶段训练 + TTS X-Git-Url: http://acesimba.cloud/gitweb/?a=commitdiff_plain;h=9b9b54c7c0e93d22a1a183bc484bf0a2923a3506;p=study-mountion.git 雅思词汇真经课程:3073 词导入 + 五阶段训练 + TTS 数据: - 新增 imports/ielts_vocab.py:xlsx(3073 词 / 22 章 / 19 列)按 ceil(n/50) 切为 74 组;含词性归一化 474 条、缺例句 16 词与无搭配 2173 词标记跳过、 每组自动选 10 个 is_key - 删除旧的占位课程,改为真实词库 训练: - 新增内容类型 vocab_drill_group:学词/拼写/选择/听写/错题重做五阶段, 错题池清空才允许打卡;复习走缩水版(仅选择 + 听写) - 新增循环播放器:单词×2 → 逐字母 → 中文 → 例句 → 例句中文 → 各搭配逐一, 整组无限循环,支持上下词/语速/从第 N 词开始 - 新增 audio.js 单例播放器,修复 StrictMode 下 useEffect 双执行导致的叠音; 支持分段语速(例句 0.92)与序列 token 取代 TTS: - 新增 tts.py:单词走有道 type=1 英音,长句与中文走 edge-tts (en-GB-SoniaNeural / zh-CN-XiaoxiaoNeural),百度兜底;本地缓存, edge 偶发抖动故超时单独放宽到 30s,避免回退到机械音 计划: - plan_items 加 interval_days,学习计划页支持每天/隔天/每 N 天 + 生效日多选 - auto 类课时不再允许一键打卡,改为打开训练器 其他: - .gitignore 排除 backend/db/tts_cache 音频缓存目录 --- diff --git a/.gitignore b/.gitignore index 33b48e5..de68418 100644 --- a/.gitignore +++ b/.gitignore @@ -13,3 +13,6 @@ frontend/dist/ # Database & Logs *.db backend/db/*.db + +# TTS 音频缓存(按需生成,不入库) +backend/db/tts_cache/ diff --git a/backend/db.py b/backend/db.py index 663d6c6..7bedbb6 100644 --- a/backend/db.py +++ b/backend/db.py @@ -140,6 +140,7 @@ CREATE TABLE IF NOT EXISTS plan_items ( daily_new INTEGER NOT NULL DEFAULT 1, daily_review INTEGER NOT NULL DEFAULT 1, weekdays TEXT NOT NULL DEFAULT '1,2,3,4,5,6,7', + interval_days INTEGER NOT NULL DEFAULT 1, UNIQUE(plan_id, course_id) ); @@ -191,11 +192,23 @@ def _ensure_log_columns(conn): conn.execute(f"ALTER TABLE study_logs ADD COLUMN {col} {decl}") +# interval_days 是后续加的「每 N 天学一组」,存量库按需补齐 +_PLAN_EXTRA_COLUMNS = (("interval_days", "INTEGER NOT NULL DEFAULT 1"),) + + +def _ensure_plan_columns(conn): + existing = {r["name"] for r in conn.execute("PRAGMA table_info(plan_items)")} + for col, decl in _PLAN_EXTRA_COLUMNS: + if col not in existing: + conn.execute(f"ALTER TABLE plan_items ADD COLUMN {col} {decl}") + + def init_db(): os.makedirs(os.path.join(BASE_DIR, "db"), exist_ok=True) conn = get_conn() conn.executescript(LEGACY_SCHEMA) conn.executescript(SCHEMA) _ensure_log_columns(conn) + _ensure_plan_columns(conn) conn.commit() conn.close() diff --git a/backend/imports/ielts_vocab.py b/backend/imports/ielts_vocab.py new file mode 100644 index 0000000..83a25b9 --- /dev/null +++ b/backend/imports/ielts_vocab.py @@ -0,0 +1,314 @@ +"""雅思词汇真经 xlsx → 74 个课时。 + +数据源:雅思词汇真经_最终校正版.xlsx · sheet「最终单词表」· 3073 词 · 22 ç«  · 19 列 +切组规则(不跨章):k = ceil(n / 50),章内均分,余数摊到前几组。 +产出:1 门课程 code=IELTS-VOCAB-01 + 74 个 lessons(content_type=vocab_drill_group)。 + +用法: + python3 imports/ielts_vocab.py # 用默认 xlsx 路径 + python3 imports/ielts_vocab.py --xlsx 路径 # 指定词库 + python3 imports/ielts_vocab.py --dry-run # 只打印切组结果,不写库 +""" + +import argparse +import json +import math +import os +import re +import sys + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import db # noqa: E402 + +DEFAULT_XLSX = "/home/Codebuddy-web/data/pastes/雅思词汇真经_最终校正版.xlsx" +SHEET = "最终单词表" + +COURSE = { + "code": "IELTS-VOCAB-01", + "name": "雅思词汇真经", + "category": "vocab", + "locale": "en", + "description": "22 ç«  3073 词,切分为 74 组;每组五阶段训练(学词/拼写/选择/听写/错题重做)", + "icon": "A", + "color": "#4a9e7a", + "is_open": 1, + "allow_skip": 0, + "sort_order": 1, +} + +# 列索引(0-based,对应 xlsx 表头顺序) +C_CHAPTER, C_CHAPTER_TITLE, C_WID, C_WORD, C_PHONETIC, C_POS, C_MEANING = range(7) +C_EXAMPLE, C_EXAMPLE_ZH = 7, 8 +C_COL_EN = (9, 11, 13) # 搭配1/2/3 +C_COL_ZH = (10, 12, 14) # 搭配1/2/3 中文 + +GROUP_TARGET = 50 +KEY_WORDS_PER_GROUP = 10 + +# 词性归一化顺序:名词 > 动词 > 形容词 > 副词 +POS_ORDER = {"n.": 0, "v.": 1, "adj.": 2, "adv.": 3} +# 抽象/学术词后缀,用于挑写作重点词时加权 +ABSTRACT_SUFFIX = re.compile(r"(tion|sion|ment|ness|ity|ism|ance|ence|ship|hood|cy|al)$") + + +# ---------- 读取与清洗 ---------- + +def _clean(v): + if v is None: + return None + s = str(v).strip() + return s or None + + +def normalize_pos(pos): + """词性归一化:'v./n.' 与 'n./v.' 统一为 'n./v.',去重后按固定顺序排。""" + s = _clean(pos) + if not s: + return None + parts = [p.strip() for p in re.split(r"[/、,;]", s) if p.strip()] + norm, seen = [], set() + for p in parts: + p = p.rstrip(".") + "." + if p not in seen: + seen.add(p) + norm.append(p) + norm.sort(key=lambda p: POS_ORDER.get(p, 99)) + return "/".join(norm) + + +def load_rows(xlsx_path): + import openpyxl + + wb = openpyxl.load_workbook(xlsx_path, data_only=True) + if SHEET not in wb.sheetnames: + raise SystemExit(f"找不到 sheet「{SHEET}」,现有:{wb.sheetnames}") + ws = wb[SHEET] + + rows = [] + for r in ws.iter_rows(min_row=2, values_only=True): + if not _clean(r[C_WORD]): + continue + collocations = [] + for i, j in zip(C_COL_EN, C_COL_ZH): + en, zh = _clean(r[i]), _clean(r[j]) + if en: + collocations.append({"en": en, "zh": zh}) + rows.append( + { + "chapter": int(r[C_CHAPTER]), + "chapter_title": _clean(r[C_CHAPTER_TITLE]), + "wid": _clean(r[C_WID]), + "word": _clean(r[C_WORD]), + "phonetic": _clean(r[C_PHONETIC]), + "pos": normalize_pos(r[C_POS]), + "meaning": _clean(r[C_MEANING]), + "example": _clean(r[C_EXAMPLE]), + "example_zh": _clean(r[C_EXAMPLE_ZH]), + "collocations": collocations, + } + ) + return rows + + +# ---------- 切组 ---------- + +def split_sizes(n): + """章内均分成 k 组,余数摊到前几组。""" + k = math.ceil(n / GROUP_TARGET) + base, rem = divmod(n, k) + return [base + (1 if i < rem else 0) for i in range(k)] + + +def group_chapters(rows): + """返回 [(chapter, chapter_title, group_no, [word...]), ...],lesson_no 全书连续。""" + by_chapter = {} + for r in rows: + by_chapter.setdefault(r["chapter"], {"title": r["chapter_title"], "words": []}) + by_chapter[r["chapter"]]["words"].append(r) + + groups = [] + for ch in sorted(by_chapter): + data = by_chapter[ch] + words = data["words"] + offset = 0 + for g_no, size in enumerate(split_sizes(len(words)), start=1): + groups.append((ch, data["title"], g_no, words[offset : offset + size])) + offset += size + return groups + + +# ---------- is_key 选取 ---------- + +def key_score(w): + """优先有搭配的词 > 学术抽象词 > 词长。""" + s = 0 + if w["collocations"]: + s += 100 + if ABSTRACT_SUFFIX.search(w["word"].lower()): + s += 20 + s += min(len(w["word"]), 15) + return s + + +def mark_key_words(words): + ranked = sorted(range(len(words)), key=lambda i: -key_score(words[i])) + for i in ranked[:KEY_WORDS_PER_GROUP]: + words[i]["is_key"] = True + return words + + +# ---------- payload ---------- + +def build_payload(chapter, chapter_title, group_no, words): + mark_key_words(words) + return { + "version": 1, + "content_type": "vocab_drill_group", + "meta": { + "chapter": chapter, + "chapter_title": chapter_title, + "group_no": group_no, + "size": len(words), + }, + "drill_config": { + "stages": ["learn", "spell", "choice", "dictation", "error_redo"], + "choice_options": 4, + "key_words_per_group": KEY_WORDS_PER_GROUP, + "pass_rule": "error_pool_empty", + # 数据源音标为 OCR 转写,非标准 IPA,前台默认不展示 + "phonetic_visible": False, + }, + "words": [ + { + "wid": w["wid"], + "word": w["word"], + "phonetic": w["phonetic"], + "phonetic_reliable": False, + "pos": w["pos"], + "meaning": w["meaning"], + "example": w["example"], + "example_zh": w["example_zh"], + "collocations": w["collocations"], + "is_key": w.get("is_key", False), + "tts": True, + } + for w in words + ], + } + + +def duration_of(size): + """一组 40-50 词走完五阶段,按每词约 0.6 分钟估算。""" + return max(20, round(size * 0.6)) + + +# ---------- 写库 ---------- + +def purge_course(conn, code): + """按 code 删除课程及其全部关联数据(进度/完成/复习/任务/计划条目)。""" + row = conn.execute("SELECT id FROM courses WHERE code=?", (code,)).fetchone() + if not row: + return 0 + cid = row["id"] + conn.execute("DELETE FROM daily_tasks WHERE course_id=?", (cid,)) + conn.execute("DELETE FROM review_queue WHERE course_id=?", (cid,)) + conn.execute("DELETE FROM lesson_completions WHERE course_id=?", (cid,)) + conn.execute("DELETE FROM course_progress WHERE course_id=?", (cid,)) + conn.execute("DELETE FROM plan_items WHERE course_id=?", (cid,)) + conn.execute("DELETE FROM lessons WHERE course_id=?", (cid,)) + conn.execute("DELETE FROM courses WHERE id=?", (cid,)) + return cid + + +def import_course(rows, dry_run=False): + groups = group_chapters(rows) + total_words = sum(len(g[3]) for g in groups) + print(f"词库:{len(rows)} 词 · {len({g[0] for g in groups})} ç«  · 切为 {len(groups)} 组") + + if dry_run: + for ch, title, g_no, words in groups: + mark_key_words(words) + keys = [w["word"] for w in words if w.get("is_key")] + print(f" Ch{ch:<2} {title:<6} G{g_no} {len(words):>2} 词 key={len(keys)}: {','.join(keys[:4])}…") + return + + conn = db.get_conn() + try: + old = purge_course(conn, COURSE["code"]) + if old: + print(f"已删除旧课程(id={old})及其关联数据") + conn.execute("DELETE FROM daily_tasks") # 今日任务基于旧课程,重建 + + ts = db.now() + cur = conn.execute( + """ + INSERT INTO courses (code, name, category, locale, description, icon, color, + is_open, allow_skip, sort_order, created_at, updated_at) + VALUES (:code, :name, :category, :locale, :description, :icon, :color, + :is_open, :allow_skip, :sort_order, :created_at, :updated_at) + """, + {**COURSE, "created_at": ts, "updated_at": ts}, + ) + course_id = cur.lastrowid + + for lesson_no, (ch, title, g_no, words) in enumerate(groups, start=1): + payload = build_payload(ch, title, g_no, words) + conn.execute( + """ + INSERT INTO lessons (course_id, lesson_no, title, content_type, payload, + completion_type, duration_min, created_at, updated_at) + VALUES (?,?,?,?,?,?,?,?,?) + """, + ( + course_id, + lesson_no, + f"Ch{ch} {title} · G{g_no}", + "vocab_drill_group", + json.dumps(payload, ensure_ascii=False), + "auto", + duration_of(len(words)), + ts, + ts, + ), + ) + + conn.execute( + """ + INSERT INTO course_progress (user_id, course_id, next_lesson_no, completed_count, + total_lessons, updated_at) + VALUES (?,?,1,0,?,?) + """, + (1, course_id, len(groups), ts), + ) + + plan = conn.execute( + "SELECT id FROM study_plans WHERE user_id=1 AND is_active=1 ORDER BY id LIMIT 1" + ).fetchone() + if plan: + conn.execute( + "INSERT OR REPLACE INTO plan_items (plan_id, course_id, daily_new, daily_review) VALUES (?,?,?,?)", + (plan["id"], course_id, 1, 1), + ) + + conn.commit() + print(f"已写入:course_id={course_id} · {len(groups)} 课时 · {total_words} 词") + finally: + conn.close() + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--xlsx", default=DEFAULT_XLSX) + ap.add_argument("--dry-run", action="store_true") + args = ap.parse_args() + + if not os.path.exists(args.xlsx): + raise SystemExit(f"找不到词库文件:{args.xlsx}") + + db.init_db() + import_course(load_rows(args.xlsx), dry_run=args.dry_run) + + +if __name__ == "__main__": + main() diff --git a/backend/main.py b/backend/main.py index cdd99aa..16a2e53 100644 --- a/backend/main.py +++ b/backend/main.py @@ -8,10 +8,12 @@ import json from fastapi import FastAPI, HTTPException from fastapi.middleware.cors import CORSMiddleware +from fastapi.responses import Response from pydantic import BaseModel import db import scheduler +import tts from seed import DEMO_USER_ID app = FastAPI(title="学习山峰 API", version="0.2.0") @@ -39,6 +41,8 @@ def on_startup(): class PlanItemIn(BaseModel): daily_new: int = 0 daily_review: int = 0 + weekdays: str = None # '1,2,3,4,5,6,7' + interval_days: int = None # 1=每天,2=隔天 class CompleteIn(BaseModel): @@ -114,7 +118,7 @@ def get_today(): rows = conn.execute( """ SELECT dt.id, dt.course_id, dt.lesson_id, dt.task_type, dt.status, - l.lesson_no, l.title, l.content_type, l.duration_min, + l.lesson_no, l.title, l.content_type, l.completion_type, l.duration_min, c.name AS course_name, c.category, c.color, c.icon FROM daily_tasks dt JOIN lessons l ON l.id = dt.lesson_id @@ -159,6 +163,7 @@ def get_today(): "task_type": r["task_type"], "status": r["status"], "content_type": r["content_type"], + "completion_type": r["completion_type"], "duration_min": r["duration_min"], } ) @@ -337,7 +342,7 @@ def get_plan(): return {"plan": None, "items": []} rows = conn.execute( """ - SELECT pi.course_id, pi.daily_new, pi.daily_review, pi.weekdays, + SELECT pi.course_id, pi.daily_new, pi.daily_review, pi.weekdays, pi.interval_days, c.name AS course_name, c.color, c.icon, c.is_open FROM plan_items pi JOIN courses c ON c.id=pi.course_id WHERE pi.plan_id=? ORDER BY c.sort_order, c.id @@ -368,14 +373,31 @@ def update_plan_item(course_id: int, body: PlanItemIn): ).fetchone() if not plan: raise HTTPException(status_code=404, detail="no active plan") + # 只更新传入的字段,未传的保持原值 + sets = ["daily_new=?", "daily_review=?"] + vals = [max(0, body.daily_new), max(0, body.daily_review)] + if body.weekdays: + sets.append("weekdays=?") + vals.append(body.weekdays) + if body.interval_days is not None: + sets.append("interval_days=?") + vals.append(max(1, body.interval_days)) + cur = conn.execute( - "UPDATE plan_items SET daily_new=?, daily_review=? WHERE plan_id=? AND course_id=?", - (max(0, body.daily_new), max(0, body.daily_review), plan["id"], course_id), + f"UPDATE plan_items SET {','.join(sets)} WHERE plan_id=? AND course_id=?", + vals + [plan["id"], course_id], ) if cur.rowcount == 0: conn.execute( - "INSERT INTO plan_items (plan_id, course_id, daily_new, daily_review) VALUES (?,?,?,?)", - (plan["id"], course_id, max(0, body.daily_new), max(0, body.daily_review)), + """ + INSERT INTO plan_items (plan_id, course_id, daily_new, daily_review, weekdays, interval_days) + VALUES (?,?,?,?,?,?) + """, + ( + plan["id"], course_id, + max(0, body.daily_new), max(0, body.daily_review), + body.weekdays or "1,2,3,4,5,6,7", max(1, body.interval_days or 1), + ), ) conn.commit() return {"ok": True} @@ -432,6 +454,18 @@ def admin_stats(): conn.close() +# ---------- TTS ---------- + +@app.get("/api/tts") +def get_tts(text: str = "", lang: str = "en"): + """合成语音并缓存。lang: en(单词/例句/字母) / zh(中文含义)。""" + try: + data = tts.synthesize(text, lang) + except tts.TTSError as e: + raise HTTPException(status_code=502, detail=str(e)) + return Response(content=data, media_type="audio/mpeg") + + # ---------- 其他 ---------- @app.get("/api/me") diff --git a/backend/requirements.txt b/backend/requirements.txt index c9b6004..da322d9 100644 --- a/backend/requirements.txt +++ b/backend/requirements.txt @@ -1,3 +1,5 @@ fastapi uvicorn pydantic +edge-tts +openpyxl diff --git a/backend/scheduler.py b/backend/scheduler.py index 672ebb1..1336b63 100644 --- a/backend/scheduler.py +++ b/backend/scheduler.py @@ -27,7 +27,7 @@ def ensure_daily_tasks(conn, user_id: int, date: str = None) -> int: weekday = str(datetime.date.fromisoformat(date).isoweekday()) plan = conn.execute( - "SELECT id FROM study_plans WHERE user_id=? AND is_active=1 ORDER BY id LIMIT 1", + "SELECT id, start_date FROM study_plans WHERE user_id=? AND is_active=1 ORDER BY id LIMIT 1", (user_id,), ).fetchone() if not plan: @@ -35,7 +35,7 @@ def ensure_daily_tasks(conn, user_id: int, date: str = None) -> int: items = conn.execute( """ - SELECT pi.course_id, pi.daily_new, pi.daily_review, pi.weekdays, c.allow_skip + SELECT pi.course_id, pi.daily_new, pi.daily_review, pi.weekdays, pi.interval_days, c.allow_skip FROM plan_items pi JOIN courses c ON c.id = pi.course_id WHERE pi.plan_id=? AND c.is_open=1 @@ -48,7 +48,7 @@ def ensure_daily_tasks(conn, user_id: int, date: str = None) -> int: for item in items: if weekday not in (item["weekdays"] or "").split(","): continue - created += _gen_new(conn, user_id, date, item) + created += _gen_new(conn, user_id, date, item, plan["start_date"]) created += _gen_review(conn, user_id, date, item) conn.commit() return created @@ -66,8 +66,26 @@ def _remaining_quota(conn, user_id, date, course_id, task_type, quota) -> int: return max(0, (quota or 0) - used) -def _gen_new(conn, user_id, date, item) -> int: +def _interval_due(start_date, date, interval) -> bool: + """每 N 天学一组:按 (date - start_date) % interval 判断当天是否该出新学。 + + 复习不受此限制——积压的复习本来就该及时清掉。 + """ + interval = interval or 1 + if interval <= 1 or not start_date: + return True + try: + d0 = datetime.date.fromisoformat(start_date) + d1 = datetime.date.fromisoformat(date) + except ValueError: + return True + return (d1 - d0).days % interval == 0 + + +def _gen_new(conn, user_id, date, item, start_date=None) -> int: course_id = item["course_id"] + if not _interval_due(start_date, date, item["interval_days"]): + return 0 want = _remaining_quota(conn, user_id, date, course_id, "new", item["daily_new"]) if want <= 0: return 0 diff --git a/backend/tts.py b/backend/tts.py new file mode 100644 index 0000000..81aab3a --- /dev/null +++ b/backend/tts.py @@ -0,0 +1,146 @@ +"""TTS 代理:有道(英文单词) / edge-tts(长句·中文) / 百度兜底,带本地缓存。 + +源选择依据 2026-08-29 实测: + 有道 EN 单词 ✓ 0.24s,词典真人发音,最快 + 有道 EN 长句 ✗ HTTP 500 + edge-tts 长句/中文 ✓ ~2s,微软神经网络音,自然度明显优于百度 + 百度 EN/ZH ✓ 合成音偏机械,仅作兜底 + +策略:单词走有道(快且准);长句与中文走 edge-tts(自然度优先);全部源失败再落百度。 + +缓存:db/tts_cache/<前缀>_.mp3,按需生成,重复播放零请求。 +""" + +import asyncio +import hashlib +import os +import re +import urllib.error +import urllib.parse +import urllib.request + +import db + +CACHE_DIR = os.path.join(db.BASE_DIR, "db", "tts_cache") +os.makedirs(CACHE_DIR, exist_ok=True) + +UA = {"User-Agent": "Mozilla/5.0"} +BAIDU_HEADERS = {"User-Agent": "Mozilla/5.0", "Referer": "https://fanyi.baidu.com/"} + +TIMEOUT = 12 # HTTP 源(有道 / 百度) +EDGE_TIMEOUT = 30 # edge-tts 偶发抖动到 10s+,放宽些避免轻易回退到机械音 +# 超过该长度的英文按长句处理,走 edge-tts(有道对长句返回 500) +SHORT_EN_MAXLEN = 24 +MIN_AUDIO_BYTES = 500 + +# 微软神经网络语音:英式女声 Sonia / 中文女声晓晓 +VOICE_EN = "en-GB-SoniaNeural" +VOICE_ZH = "zh-CN-XiaoxiaoNeural" +# 有道 type=1 英音 / type=2 美音。与 VOICE_EN 的英音保持一致,避免同一课里英美混读 +YOUDAO_TYPE = 1 + + +class TTSError(Exception): + pass + + +def _cache_path(text: str, lang: str) -> str: + safe = re.sub(r"[^0-9A-Za-z\u4e00-\u9fff]+", "_", text.strip().lower())[:24].strip("_") + digest = hashlib.md5(f"{lang}:{text}".encode("utf-8")).hexdigest()[:12] + name = f"{safe}_{digest}" if safe else digest + return os.path.join(CACHE_DIR, f"{name}.mp3") + + +def _fetch(url: str, headers: dict) -> bytes: + req = urllib.request.Request(url, headers=headers) + with urllib.request.urlopen(req, timeout=TIMEOUT) as resp: + if resp.headers.get("Content-Type", "").startswith("application/json"): + raise TTSError("upstream returned json instead of audio") + data = resp.read() + if len(data) < MIN_AUDIO_BYTES: + raise TTSError(f"audio too small ({len(data)} bytes)") + return data + + +def _youdao_en(text: str) -> bytes: + url = ( + "https://dict.youdao.com/dictvoice?audio=" + + urllib.parse.quote(text) + + f"&type={YOUDAO_TYPE}" + ) + return _fetch(url, UA) + + +def _edge(text: str, lang: str) -> bytes: + """微软 Edge 神经网络 TTS,自然度最好。同步上下文里跑一次事件循环。""" + import edge_tts + + voice = VOICE_ZH if lang == "zh" else VOICE_EN + + async def synth() -> bytes: + chunks = [] + async for chunk in edge_tts.Communicate(text, voice).stream(): + if chunk["type"] == "audio": + chunks.append(chunk["data"]) + data = b"".join(chunks) + if len(data) < MIN_AUDIO_BYTES: + raise TTSError(f"edge-tts audio too small ({len(data)} bytes)") + return data + + return asyncio.run(asyncio.wait_for(synth(), timeout=EDGE_TIMEOUT)) + + +def _baidu(text: str, lang: str) -> bytes: + lan = "zh" if lang == "zh" else "en" + url = ( + "https://fanyi.baidu.com/gettts?lan=" + + lan + + "&text=" + + urllib.parse.quote(text) + + "&spd=3" + ) + return _fetch(url, BAIDU_HEADERS) + + +def synthesize(text: str, lang: str = "en") -> bytes: + """返回 mp3 字节。命中缓存直接读盘,未命中按源优先级合成后写缓存。""" + text = (text or "").strip() + if not text: + raise TTSError("empty text") + + path = _cache_path(text, lang) + if os.path.exists(path): + with open(path, "rb") as f: + return f.read() + + is_short_en = lang == "en" and len(text) <= SHORT_EN_MAXLEN + if lang == "zh": + # 中文:edge 最自然,百度兜底 + sources = [(lambda t: _edge(t, "zh")), (lambda t: _baidu(t, "zh"))] + elif is_short_en: + # 英文单词:有道最快且为词典真人发音,edge 兜底 + sources = [_youdao_en, (lambda t: _edge(t, "en")), (lambda t: _baidu(t, "en"))] + else: + # 英文长句:有道会 500,edge 自然度优先,百度兜底 + sources = [(lambda t: _edge(t, "en")), (lambda t: _baidu(t, "en"))] + + last_err = None + data = None + for src in sources: + try: + data = src(text) + break + except (urllib.error.URLError, TTSError, OSError, asyncio.TimeoutError) as e: + last_err = f"{type(e).__name__}: {e}" + data = None + + if not data: + raise TTSError(f"all TTS sources failed -> {last_err}") + + try: + with open(path, "wb") as f: + f.write(data) + except OSError as e: + print(f"TTS cache write failed: {e}") + + return data diff --git a/frontend/src/api.js b/frontend/src/api.js index 46a4aad..cdfe0a8 100644 --- a/frontend/src/api.js +++ b/frontend/src/api.js @@ -7,6 +7,10 @@ const http = axios.create({ baseURL: API_BASE }) const unwrap = (p) => p.then((r) => r.data) +/** TTS 音频地址,直接喂给