From: lexicon Date: Fri, 21 Aug 2026 12:06:01 +0000 (+0800) Subject: V1: TTS 切换 Edge 神经嗓音(Sonia)+缩写展开; 修复 off_ed 同步误删进度; 补回 8/20 丢失记录 X-Git-Url: http://acesimba.cloud/gitweb/?a=commitdiff_plain;h=b9d4077883dec8b0c588af5c84bc5de334d0ff6d;p=words.git V1: TTS 切换 Edge 神经嗓音(Sonia)+缩写展开; 修复 off_ed 同步误删进度; 补回 8/20 丢失记录 - backend/main.py: /api/tts 引擎由有道/百度改为 Edge 神经嗓音(en-GB-SoniaNeural)首选, 有道/百度作兜底; 合成前做无歧义缩写展开(sth->something 等); 缓存键加 edge_sonia_ 前缀 - frontend/public/sw.js: CACHE_VERSION v1->v2, 强制重新拉取 TTS 音频避免旧难听嗓音残留 - frontend/src/sync.js: 删除 phase 2b 破坏性删除逻辑, 同步改为只补不删(布尔并集), 修复 Review2 等已完成进度被静默丢弃的问题 - backend/db/bcd.db: 修正 8/20 云端记录(G54 review3 补回; G56 review2/G57 review1 撤销) --- diff --git a/.gitignore b/.gitignore index aabb8b3..8b60d53 100644 --- a/.gitignore +++ b/.gitignore @@ -20,3 +20,7 @@ backend/db/tts_cache/ # certbot 一次性 DNS 挑战值(临时文件,不入库) deploy/dns-txt-value.txt + +# 导出/打包产物(供下载,不入库) +words_package.zip +words_export.xlsx diff --git a/backend/db/bcd.db b/backend/db/bcd.db index b54b828..ed30f86 100644 Binary files a/backend/db/bcd.db and b/backend/db/bcd.db differ diff --git a/backend/main.py b/backend/main.py index 4646339..fd91b1f 100644 --- a/backend/main.py +++ b/backend/main.py @@ -313,18 +313,51 @@ async def get_picture(pic_id: int): except Exception as e: raise HTTPException(status_code=500, detail=str(e)) -# --- TTS 代理(有道美式发音 + 本地缓存) --- +# --- TTS 代理(Edge 神经嗓音首选 + 有道/百度兜底 + 缩写展开) --- TTS_CACHE_DIR = os.path.join(BASE_DIR, "db", "tts_cache") os.makedirs(TTS_CACHE_DIR, exist_ok=True) -def _build_tts_response(word: str): +# 选用嗓音:英式 RP(接近高考听力风格,免费、无需 key) +TTS_VOICE = "en-GB-SoniaNeural" +# 缓存文件名前缀:换嗓音/换引擎时旧缓存不会被误用(避免仍播放旧嗓音) +TTS_CACHE_PREFIX = "edge_sonia_" + +# 常见无歧义缩写 -> 完整读音(合成前展开,避免把 sth 拼读成三个字母) +_ABBREV = { + "sth": "something", "sb": "somebody", "sbdy": "somebody", + "esp": "especially", "etc": "et cetera", "eg": "for example", + "ie": "that is", "approx": "approximately", "info": "information", + "num": "number", "max": "maximum", "min": "minimum", + "ext": "extension", "dept": "department", "gov": "government", + "lab": "laboratory", "exam": "examination", "gym": "gymnasium", + "vs": "versus", "pref": "prefix", "suff": "suffix", "adb": "adverb", +} +import re as _re +def _expand_abbrev(text: str) -> str: + def _rep(m): + w = m.group(0) + return _ABBREV.get(w.lower(), w) + # 仅整词匹配替换,避免误改正常单词(如 clothes 不会被拆) + return _re.sub(r"[A-Za-z]+", _rep, text) + +async def _gen_edge_tts(text: str, voice: str) -> bytes: + import io, edge_tts + buf = io.BytesIO() + communicate = edge_tts.Communicate(text, voice) + async for chunk in communicate.stream(): + if chunk.get("type") == "audio": + buf.write(chunk["data"]) + return buf.getvalue() + +async def _build_tts_response(word: str): word = word.strip() if not word: raise HTTPException(status_code=400, detail="empty word") - # 缓存文件名:小写 + 下划线,避免特殊字符 - safe_name = "".join(c if c.isalnum() or c in "-_" else "_" for c in word.lower()) - cache_path = os.path.join(TTS_CACHE_DIR, f"{safe_name}.mp3") + # 缩写展开后再合成(如 sth -> something),缓存也按展开后的文本 + spoken = _expand_abbrev(word) + safe_name = "".join(c if c.isalnum() or c in "-_" else "_" for c in spoken.lower()) + cache_path = os.path.join(TTS_CACHE_DIR, f"{TTS_CACHE_PREFIX}{safe_name}.mp3") # 命中缓存直接返回 if os.path.exists(cache_path): @@ -332,29 +365,33 @@ def _build_tts_response(word: str): data = f.read() return Response(content=data, media_type="audio/mpeg") - # 多源 TTS:先有道(type=2 美式),失败回退百度翻译(本机可达,可覆盖有道无法合成的短语,无需 key) - sources = [ - ("youdao", "https://dict.youdao.com/dictvoice?audio=" + urllib.parse.quote(word) + "&type=2", - {"User-Agent": "Mozilla/5.0"}), - ("baidu", "https://fanyi.baidu.com/gettts?lan=en&text=" + urllib.parse.quote(word) + "&spd=3", - {"User-Agent": "Mozilla/5.0", "Referer": "https://fanyi.baidu.com/"}), - ] data = None last_err = None - for name, url, headers in sources: + # 1) Edge 神经嗓音(首选) + try: + data = await _gen_edge_tts(spoken, TTS_VOICE) + except Exception as e: + last_err = f"edge: {e}" + # 2) 兜底: 有道(美式) + if not data: try: - req = urllib.request.Request(url, headers=headers) + url = "https://dict.youdao.com/dictvoice?audio=" + urllib.parse.quote(spoken) + "&type=2" + req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"}) with urllib.request.urlopen(req, timeout=10) as resp: - ct = resp.headers.get("Content-Type", "") - if ct.startswith("application/json"): - last_err = f"{name} returned json (no audio)" - continue - data = resp.read() - if data: - break - except urllib.error.URLError as e: - last_err = f"{name}: {e}" - continue + if not resp.headers.get("Content-Type", "").startswith("application/json"): + data = resp.read() + except Exception as e: + last_err = f"youdao: {e}" + # 3) 兜底: 百度 + if not data: + try: + url = "https://fanyi.baidu.com/gettts?lan=en&text=" + urllib.parse.quote(spoken) + "&spd=3" + req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0", "Referer": "https://fanyi.baidu.com/"}) + with urllib.request.urlopen(req, timeout=10) as resp: + if not resp.headers.get("Content-Type", "").startswith("application/json"): + data = resp.read() + except Exception as e: + last_err = f"baidu: {e}" if not data: raise HTTPException(status_code=502, detail=f"TTS upstream error: {last_err}") @@ -371,13 +408,13 @@ def _build_tts_response(word: str): @app.get("/api/tts/{word}") async def get_tts(word: str): - return _build_tts_response(word) + return await _build_tts_response(word) @app.get("/api/tts") async def get_tts_query(w: str = ""): # 含 / 的短语(如 "in the/a form of")走查询参数,避免与 URL 路径分隔符冲突 - return _build_tts_response(w) + return await _build_tts_response(w) @app.post("/api/progress/update") async def update_progress(data: ProgressUpdate): diff --git a/frontend/public/sw.js b/frontend/public/sw.js index beef00e..94ba4c5 100644 --- a/frontend/public/sw.js +++ b/frontend/public/sw.js @@ -5,7 +5,7 @@ // - navigation (HTML) → network-first, 失败回退 cached index.html // - 静态资源 (JS/CSS/img/font) → cache-first -const CACHE_VERSION = 'words-pwa-v1'; +const CACHE_VERSION = 'words-pwa-v2'; self.addEventListener('install', (event) => { event.waitUntil( diff --git a/frontend/src/sync.js b/frontend/src/sync.js index 7009d23..99df80e 100644 --- a/frontend/src/sync.js +++ b/frontend/src/sync.js @@ -6,7 +6,7 @@ import { api } from './api'; import { - getDirtyProgress, clearDirty, getAllProgress, putProgress, deleteProgress, + getDirtyProgress, clearDirty, getAllProgress, putProgress, putWords, setMeta, getMeta, setLearnableGroups, getAllWordsFromIDB, getWordsCount, } from './db'; @@ -79,17 +79,12 @@ export async function syncNow() { } } - // 阶段2b: 服务端已删除(且本地已确认/非 dirty)的进度, 本地同步删除。 - // 仅在成功拉取到非空集合时才执行, 避免服务端异常返回空时误删本地进度; - // 仅删非 dirty 记录(本地未推送完成的进度绝不删, 保护离线工作)。 - if (serverEvents.length > 0) { - const serverKeys = new Set(serverEvents.map(e => `${e.group_id}:${e.op_type}`)); - for (const p of localAll) { - if (!p.dirty && !serverKeys.has(p.key)) { - await deleteProgress(p.group_id, p.op_type); - } - } - } + // 阶段2b: 不再自动删除本地进度。 + // 原逻辑会把"服务端 pull 中不存在的本地非 dirty 记录"删掉,用于"撤销"回写; + // 但布尔并集语义下,服务端 pull 缺失≠被撤销(可能是推送时序/服务端异常), + // 这会静默丢掉已完成的进度(如 Review2)。回归离线方案原始设计「不删本地」, + // 同步只做布尔并集(只补不删),保证任何已完成的 (group,op) 永不丢失。 + // 如需撤销,应在 /api/sync/pull 显式返回"已删除清单"再据此删除,而非按缺失推断。 // 阶段3: 首次 warmup 词库 (仅 IDB 空时, 避免每次重拉 1852 词) if ((await getWordsCount()) === 0) {