From: Codebuddy Date: Sat, 22 Aug 2026 15:35:26 +0000 (+0800) Subject: fix(backend): 模型切换真正重启会话 + 新增后台进程回收机制 X-Git-Url: http://acesimba.cloud/gitweb/?a=commitdiff_plain;h=d190cd0b6da049ff11b85e6e34c71a132ae85e15;p=codebuddy-web.git fix(backend): 模型切换真正重启会话 + 新增后台进程回收机制 - 模型切换(api_set_model)改为服务端 SIGTERM+SIGKILL 杀旧进程并立刻移除 RUNNING entry,重连按新模型拉起新会话。修复 Station AI Demo 卡死无法输入: codebuddy 忽略 SIGTERM 导致旧进程僵死、重连 attach 回死进程 - 放宽 /ws 连接守卫,允许 prev is None 时重连,避免竞态误拒"已在另一窗口运行" - api_kill 同样补 SIGKILL 兜底 - 新增 _reap_loop 回收协程(startup 即启动),每 30s 清理已退出/僵死且已断开 连接的 RUNNING entry;不触碰后台常驻会话与仍连接的会话 - 前端:模型切换成功后延迟 800ms 重连并移除 killCurrent,不再依赖不可靠的前端 pid Co-Authored-By: CodeBuddy --- diff --git a/backend/app.py b/backend/app.py index 5177e49..cae45d0 100644 --- a/backend/app.py +++ b/backend/app.py @@ -92,6 +92,13 @@ if not PASSWORD: app = FastAPI(title="Codebuddy Web Console") +@app.on_event("startup") +async def _startup(): + # 启动后台回收协程,定期清理已退出/僵死且断开连接的任务 entry + asyncio.create_task(_reap_loop()) + logger.info("已启动后台任务回收协程") + + def create_token() -> str: expire = datetime.datetime.utcnow() + datetime.timedelta(hours=JWT_EXPIRE_HOURS) return jwt.encode({"exp": expire, "rand": secrets.token_hex(8)}, JWT_SECRET, algorithm=JWT_ALGORITHM) @@ -415,14 +422,27 @@ def _launch_process(task: dict, sid: str, model: str, cwd: str, codebuddy_path: def _kill_entry(task_id: str) -> bool: - """真正终止任务进程(进程组)并从 RUNNING 移除。""" + """真正终止任务进程(进程组),先 SIGTERM 再补 SIGKILL 兜底,并从 RUNNING 移除。 + + 注意:codebuddy 进程会忽略 SIGTERM,若只发 SIGTERM 旧进程不会死, + 重连时 /ws 会 attach 回这个僵死的旧进程,导致终端卡住、无法输入。 + 因此这里必须在 SIGTERM 之后立刻补 SIGKILL。 + """ entry = RUNNING.pop(task_id, None) if not entry: return False proc = entry["proc"] try: pgid = os.getpgid(proc.pid) - os.killpg(pgid, signal.SIGTERM) + try: + os.killpg(pgid, signal.SIGTERM) + except ProcessLookupError: + return True + # 兜底:SIGKILL 确保进程一定退出(codebuddy 忽略 SIGTERM) + try: + os.killpg(pgid, signal.SIGKILL) + except ProcessLookupError: + pass except Exception: try: proc.kill() @@ -431,6 +451,53 @@ def _kill_entry(task_id: str) -> bool: return True +def _proc_state(pid: int) -> str: + """读取 /proc//stat 的进程状态字符(R/S/D/Z/T 等);读不到返回空串。""" + try: + with open(f"/proc/{pid}/stat") as _f: + s = _f.read() + # 进程名 comm 可能含空格且被包在括号里,状态位在最后一个 ')' 之后 + rp = s.rfind(")") + return s[rp + 2:rp + 3] + except Exception: + return "" + + +async def _reap_loop(): + """后台回收:周期性清理 RUNNING 中已退出/僵死且已断开连接的任务 entry。 + + 目的: + - 避免“陈旧 entry 仍指向已退出的进程”,导致前端重连时 /ws 误以为进程存活、 + 从而 attach 回死进程(表现为终端卡死、无法输入)。 + - 释放已结束且无人观看的任务占用的内存与缓冲。 + + 注意:不会触碰“已断开连接但进程仍活着”的 entry——那是“断开后继续在后台跑” + 的预期行为,必须保留;也不会触碰“仍连接着”的 entry。 + """ + while True: + await asyncio.sleep(30) + try: + for tid, entry in list(RUNNING.items()): + proc = entry["proc"] + # 仍连着且进程活着:正常会话,跳过 + if entry.get("ws") is not None and proc.poll() is None: + continue + dead = proc.poll() is not None + if not dead: + # 进程可能处于僵尸态(poll 仍返回 None,需 wait 回收) + if _proc_state(proc.pid) == "Z": + try: + proc.wait(timeout=1) + except Exception: + pass + dead = True + if dead and entry.get("ws") is None: + RUNNING.pop(tid, None) + logger.info(f"回收已退出的任务 entry: name={entry.get('name')}, pid={proc.pid}") + except Exception as _e: + logger.warning(f"回收任务时出错: {_e}") + + # ==================== 前端 HTML ==================== # 前后端分离:前端页面单独放在 frontend/index.html,便于维护 FRONTEND_DIR = Path(__file__).resolve().parent.parent / "frontend" @@ -682,6 +749,7 @@ async def api_set_model(task_id: str, request: Request): t = find_task(data, task_id) if not t: return JSONResponse({"ok": False, "error": "任务不存在"}, status_code=404) + old_model = (t.get("model") or "") if model: # 校验是否在已知模型列表中(未知则拒绝,避免无效模型) with MODELS_CACHE_LOCK: @@ -694,6 +762,18 @@ async def api_set_model(task_id: str, request: Request): t["model"] = "" save_tasks(data) logger.info(f"设置模型: id={task_id}, name={t.get('name')}, model='{t['model']}'") + + # 模型实际变化且进程在运行:服务端直接终止进程并从 RUNNING 移除 entry, + # 前端重连时 /ws 会按新保存的模型重新拉起一个全新会话。 + # 关键:必须用 SIGKILL 兜底(codebuddy 忽略 SIGTERM),且立刻 pop RUNNING entry, + # 否则重连会 attach 回僵死的旧进程导致终端卡死、无法输入。 + # 不依赖前端 pid——从非挂载窗口(如任务列表页)切换时前端 pid 不可靠,曾导致"切换无效"。 + if model != old_model: + entry = RUNNING.get(task_id) + if entry is not None and entry["proc"].poll() is None: + old_pid = entry["proc"].pid + _kill_entry(task_id) + logger.info(f"模型变更,终止旧任务进程等待重启: name={t.get('name')}, old_pid={old_pid}") return JSONResponse({"ok": True, "task": t}) @@ -707,14 +787,20 @@ async def api_kill(pid: int, request: Request): RUNNING.pop(tid, None) break # 杀掉整个进程组(codebuddy 用 setsid 独立成组,子进程也一起杀) + # codebuddy 会忽略 SIGTERM,必须再补 SIGKILL 兜底 try: pgid = os.getpgid(pid) - os.killpg(pgid, signal.SIGTERM) - except ProcessLookupError: - pass + try: + os.killpg(pgid, signal.SIGTERM) + except ProcessLookupError: + pass + try: + os.killpg(pgid, signal.SIGKILL) + except ProcessLookupError: + pass except Exception: try: - os.kill(pid, signal.SIGTERM) + os.kill(pid, signal.SIGKILL) except Exception: pass return JSONResponse({"ok": True}) @@ -1284,9 +1370,11 @@ async def websocket_terminal(websocket: WebSocket): if task_id in ACTIVE_TASKS: # 可能是上一次连接刚断开、finally 还没来得及 discard 的瞬间竞态; - # 若该任务的 ws 已经置空(已 detach),说明旧连接已结束,允许本次重连(会 attach 到同一进程) + # - 若该任务的 ws 已置空(已 detach),说明旧连接已结束,允许重连(attach 到同一进程) + # - 若 RUNNING 中已无该任务(进程刚被终止,如切换模型/重启会话),同样放行, + # 由后续逻辑按新状态启动,避免"已在另一窗口运行"误拒 prev = RUNNING.get(task_id) - if prev is not None and prev.get("ws") is None: + if prev is None or prev.get("ws") is None: ACTIVE_TASKS.discard(task_id) else: await websocket.accept() diff --git a/frontend/index.html b/frontend/index.html index 86806b9..b12986a 100644 --- a/frontend/index.html +++ b/frontend/index.html @@ -832,10 +832,12 @@ body { position: fixed; top: 0; left: 0; right: 0; bottom: 0; width: 100%; heigh const data = await res.json(); currentTask.model = (data.task && data.task.model) || ''; term.writeln('\x1b[33m⏳ 切换模型为 "' + (currentTask.model || '默认·最新GLM') + '",正在重启会话...\x1b[0m'); - // 重启以应用新模型(resume 同样会采用新 --model) - killCurrent(); - if (term) term.clear(); - connectWS(); + // 后端保存模型时已终止运行中的进程;这里延迟重连, + // 等进程退出与旧 WS 清理完成后再按新模型拉起(避免"已在另一窗口运行"误拒) + setTimeout(() => { + if (term) term.clear(); + connectWS(); + }, 800); } catch (e) { alert('设置模型失败: ' + e.message); modelSelect.value = (currentTask.model) || '';