]> acesimba.cloud Git - codebuddy-web.git/commitdiff
fix(backend): 模型切换真正重启会话 + 新增后台进程回收机制
authorCodebuddy <codebuddy@localhost>
Sat, 22 Aug 2026 15:35:26 +0000 (23:35 +0800)
committerCodebuddy <codebuddy@localhost>
Sat, 22 Aug 2026 15:35:26 +0000 (23:35 +0800)
- 模型切换(api_set_model)改为服务端 SIGTERM+SIGKILL 杀旧进程并立刻移除
  RUNNING entry,重连按新模型拉起新会话。修复 Station AI Demo 卡死无法输入:
  codebuddy 忽略 SIGTERM 导致旧进程僵死、重连 attach 回死进程
- 放宽 /ws 连接守卫,允许 prev is None 时重连,避免竞态误拒"已在另一窗口运行"
- api_kill 同样补 SIGKILL 兜底
- 新增 _reap_loop 回收协程(startup 即启动),每 30s 清理已退出/僵死且已断开
  连接的 RUNNING entry;不触碰后台常驻会话与仍连接的会话
- 前端:模型切换成功后延迟 800ms 重连并移除 killCurrent,不再依赖不可靠的前端 pid

Co-Authored-By: CodeBuddy <noreply@codebuddy.ai>
backend/app.py
frontend/index.html

index 5177e492f97827fbe8dae484fccac11ebdf40cd3..cae45d079965d77b4e562b7a3c0ff6de7b4c3803 100644 (file)
@@ -92,6 +92,13 @@ if not PASSWORD:
 app = FastAPI(title="Codebuddy Web Console")
 
 
+@app.on_event("startup")
+async def _startup():
+    # 启动后台回收协程,定期清理已退出/僵死且断开连接的任务 entry
+    asyncio.create_task(_reap_loop())
+    logger.info("已启动后台任务回收协程")
+
+
 def create_token() -> str:
     expire = datetime.datetime.utcnow() + datetime.timedelta(hours=JWT_EXPIRE_HOURS)
     return jwt.encode({"exp": expire, "rand": secrets.token_hex(8)}, JWT_SECRET, algorithm=JWT_ALGORITHM)
@@ -415,14 +422,27 @@ def _launch_process(task: dict, sid: str, model: str, cwd: str, codebuddy_path:
 
 
 def _kill_entry(task_id: str) -> bool:
-    """真正终止任务进程(进程组)并从 RUNNING 移除。"""
+    """真正终止任务进程(进程组),先 SIGTERM 再补 SIGKILL 兜底,并从 RUNNING 移除。
+
+    注意:codebuddy 进程会忽略 SIGTERM,若只发 SIGTERM 旧进程不会死,
+    重连时 /ws 会 attach 回这个僵死的旧进程,导致终端卡住、无法输入。
+    因此这里必须在 SIGTERM 之后立刻补 SIGKILL。
+    """
     entry = RUNNING.pop(task_id, None)
     if not entry:
         return False
     proc = entry["proc"]
     try:
         pgid = os.getpgid(proc.pid)
-        os.killpg(pgid, signal.SIGTERM)
+        try:
+            os.killpg(pgid, signal.SIGTERM)
+        except ProcessLookupError:
+            return True
+        # 兜底:SIGKILL 确保进程一定退出(codebuddy 忽略 SIGTERM)
+        try:
+            os.killpg(pgid, signal.SIGKILL)
+        except ProcessLookupError:
+            pass
     except Exception:
         try:
             proc.kill()
@@ -431,6 +451,53 @@ def _kill_entry(task_id: str) -> bool:
     return True
 
 
+def _proc_state(pid: int) -> str:
+    """读取 /proc/<pid>/stat 的进程状态字符(R/S/D/Z/T 等);读不到返回空串。"""
+    try:
+        with open(f"/proc/{pid}/stat") as _f:
+            s = _f.read()
+        # 进程名 comm 可能含空格且被包在括号里,状态位在最后一个 ')' 之后
+        rp = s.rfind(")")
+        return s[rp + 2:rp + 3]
+    except Exception:
+        return ""
+
+
+async def _reap_loop():
+    """后台回收:周期性清理 RUNNING 中已退出/僵死且已断开连接的任务 entry。
+
+    目的:
+    - 避免“陈旧 entry 仍指向已退出的进程”,导致前端重连时 /ws 误以为进程存活、
+      从而 attach 回死进程(表现为终端卡死、无法输入)。
+    - 释放已结束且无人观看的任务占用的内存与缓冲。
+
+    注意:不会触碰“已断开连接但进程仍活着”的 entry——那是“断开后继续在后台跑”
+    的预期行为,必须保留;也不会触碰“仍连接着”的 entry。
+    """
+    while True:
+        await asyncio.sleep(30)
+        try:
+            for tid, entry in list(RUNNING.items()):
+                proc = entry["proc"]
+                # 仍连着且进程活着:正常会话,跳过
+                if entry.get("ws") is not None and proc.poll() is None:
+                    continue
+                dead = proc.poll() is not None
+                if not dead:
+                    # 进程可能处于僵尸态(poll 仍返回 None,需 wait 回收)
+                    if _proc_state(proc.pid) == "Z":
+                        try:
+                            proc.wait(timeout=1)
+                        except Exception:
+                            pass
+                        dead = True
+                if dead and entry.get("ws") is None:
+                    RUNNING.pop(tid, None)
+                    logger.info(f"回收已退出的任务 entry: name={entry.get('name')}, pid={proc.pid}")
+        except Exception as _e:
+            logger.warning(f"回收任务时出错: {_e}")
+
+
 # ==================== 前端 HTML ====================
 # 前后端分离:前端页面单独放在 frontend/index.html,便于维护
 FRONTEND_DIR = Path(__file__).resolve().parent.parent / "frontend"
@@ -682,6 +749,7 @@ async def api_set_model(task_id: str, request: Request):
     t = find_task(data, task_id)
     if not t:
         return JSONResponse({"ok": False, "error": "任务不存在"}, status_code=404)
+    old_model = (t.get("model") or "")
     if model:
         # 校验是否在已知模型列表中(未知则拒绝,避免无效模型)
         with MODELS_CACHE_LOCK:
@@ -694,6 +762,18 @@ async def api_set_model(task_id: str, request: Request):
         t["model"] = ""
     save_tasks(data)
     logger.info(f"设置模型: id={task_id}, name={t.get('name')}, model='{t['model']}'")
+
+    # 模型实际变化且进程在运行:服务端直接终止进程并从 RUNNING 移除 entry,
+    # 前端重连时 /ws 会按新保存的模型重新拉起一个全新会话。
+    # 关键:必须用 SIGKILL 兜底(codebuddy 忽略 SIGTERM),且立刻 pop RUNNING entry,
+    # 否则重连会 attach 回僵死的旧进程导致终端卡死、无法输入。
+    # 不依赖前端 pid——从非挂载窗口(如任务列表页)切换时前端 pid 不可靠,曾导致"切换无效"。
+    if model != old_model:
+        entry = RUNNING.get(task_id)
+        if entry is not None and entry["proc"].poll() is None:
+            old_pid = entry["proc"].pid
+            _kill_entry(task_id)
+            logger.info(f"模型变更,终止旧任务进程等待重启: name={t.get('name')}, old_pid={old_pid}")
     return JSONResponse({"ok": True, "task": t})
 
 
@@ -707,14 +787,20 @@ async def api_kill(pid: int, request: Request):
             RUNNING.pop(tid, None)
             break
     # 杀掉整个进程组(codebuddy 用 setsid 独立成组,子进程也一起杀)
+    # codebuddy 会忽略 SIGTERM,必须再补 SIGKILL 兜底
     try:
         pgid = os.getpgid(pid)
-        os.killpg(pgid, signal.SIGTERM)
-    except ProcessLookupError:
-        pass
+        try:
+            os.killpg(pgid, signal.SIGTERM)
+        except ProcessLookupError:
+            pass
+        try:
+            os.killpg(pgid, signal.SIGKILL)
+        except ProcessLookupError:
+            pass
     except Exception:
         try:
-            os.kill(pid, signal.SIGTERM)
+            os.kill(pid, signal.SIGKILL)
         except Exception:
             pass
     return JSONResponse({"ok": True})
@@ -1284,9 +1370,11 @@ async def websocket_terminal(websocket: WebSocket):
 
     if task_id in ACTIVE_TASKS:
         # 可能是上一次连接刚断开、finally 还没来得及 discard 的瞬间竞态;
-        # 若该任务的 ws 已经置空(已 detach),说明旧连接已结束,允许本次重连(会 attach 到同一进程)
+        # - 若该任务的 ws 已置空(已 detach),说明旧连接已结束,允许重连(attach 到同一进程)
+        # - 若 RUNNING 中已无该任务(进程刚被终止,如切换模型/重启会话),同样放行,
+        #   由后续逻辑按新状态启动,避免"已在另一窗口运行"误拒
         prev = RUNNING.get(task_id)
-        if prev is not None and prev.get("ws") is None:
+        if prev is None or prev.get("ws") is None:
             ACTIVE_TASKS.discard(task_id)
         else:
             await websocket.accept()
index 86806b923e2abcb9fd117161a05727df6751b517..b12986aab38c7860e65db15c53136f27799ae286 100644 (file)
@@ -832,10 +832,12 @@ body { position: fixed; top: 0; left: 0; right: 0; bottom: 0; width: 100%; heigh
         const data = await res.json();
         currentTask.model = (data.task && data.task.model) || '';
         term.writeln('\x1b[33m⏳ 切换模型为 "' + (currentTask.model || '默认·最新GLM') + '",正在重启会话...\x1b[0m');
-        // 重启以应用新模型(resume 同样会采用新 --model)
-        killCurrent();
-        if (term) term.clear();
-        connectWS();
+        // 后端保存模型时已终止运行中的进程;这里延迟重连,
+        // 等进程退出与旧 WS 清理完成后再按新模型拉起(避免"已在另一窗口运行"误拒)
+        setTimeout(() => {
+          if (term) term.clear();
+          connectWS();
+        }, 800);
       } catch (e) {
         alert('设置模型失败: ' + e.message);
         modelSelect.value = (currentTask.model) || '';