# -*- coding: utf-8 -*- """bg_jobs.py - run_command 后台任务(2026-09-10,对齐 Hermes terminal background + process_manage) 状态文件化:{workspace}/.bg/{job_id}/ 下 meta.json {job_id, command, pid, status(running/done/failed/timeout/killed), started_at, ended_at, rc, sandbox} output.log stdout+stderr 合并追加 为什么文件化而不是内存注册表:web/worker 多进程(未来多主机 NFS 共享 workspace), poll/log 请求可能落到没启动该任务的进程上——文件是唯一跨进程事实源。 进程句柄注册表(_PROCS)只用于本进程 kill 加速,缺失时回退 meta.json 的 pid。 隔离:任务目录在发起者的 workspace 内(项目工作空间 / _general/{uid}), process 工具只解析 self.workspace_dir 下的 .bg——跨用户/跨项目天然不可达; job_id 白名单正则防路径穿越。 沙箱:与前台 run_command 完全同一条路径(bwrap 参数同款),generic 会话 strict 档照旧;无 bwrap 时 generic 拒绝(不降级裸 shell)。 兜底:后台任务硬上限 MAX_BG_SECONDS(默认1小时),watcher 超时 SIGTERM 进程组; worker 崩溃后 meta 残留 running——poll 时用 pid 存活检测标记 stale。 """ import asyncio import json import logging import os import re import signal import time logger = logging.getLogger("pipeline.bg_jobs") BG_DIR = ".bg" MAX_BG_SECONDS = 3600 _JOB_ID_RE = re.compile(r"^[A-Za-z0-9_-]{1,32}$") # 本进程启动的任务句柄(跨进程 kill 回退 pid) _PROCS = {} # watcher 协程引用(asyncio.create_task 不保引用会被 GC 中途回收) _WATCHERS = set() class BgJobError(Exception): pass def _bg_root(workspace_dir: str) -> str: return os.path.join(workspace_dir, BG_DIR) def _validate_job_id(job_id: str) -> str: jid = (job_id or "").strip() if not _JOB_ID_RE.match(jid): raise BgJobError(f"非法 job_id: {job_id!r}") return jid def job_dir(workspace_dir: str, job_id: str) -> str: jid = _validate_job_id(job_id) return os.path.join(_bg_root(workspace_dir), jid) def _write_meta(jdir: str, meta: dict): tmp = os.path.join(jdir, "meta.json.tmp") with open(tmp, "w", encoding="utf-8") as f: json.dump(meta, f, ensure_ascii=False) os.replace(tmp, os.path.join(jdir, "meta.json")) def _read_meta(jdir: str) -> dict: path = os.path.join(jdir, "meta.json") if not os.path.exists(path): raise BgJobError("任务不存在(meta.json 缺失)") with open(path, "r", encoding="utf-8") as f: return json.load(f) def _pid_alive(pid: int) -> bool: if not pid: return False try: os.kill(pid, 0) return True except OSError: return False async def start_bg_job(command: str, workspace_dir: str, strict: bool = False, timeout: int = MAX_BG_SECONDS) -> str: """启动后台命令,返回 job_id。沙箱/目录安全校验与前台 _run_shell 完全一致。""" from .agent_loop import (_find_bwrap, _build_agent_bwrap_cmd, _sandbox_writable_root, _is_safe_workdir_async) command = (command or "").strip() if not command: raise BgJobError("命令为空") cwd = os.path.abspath(workspace_dir) if not await _is_safe_workdir_async(cwd): raise BgJobError(f"安全限制:目录 {cwd} 不在允许范围") if not os.path.isdir(cwd): raise BgJobError(f"目录不存在: {cwd}") timeout = max(10, min(int(timeout or MAX_BG_SECONDS), MAX_BG_SECONDS)) bwrap = _find_bwrap() if strict and not bwrap: raise BgJobError("通用会话的后台命令需要 bwrap 沙箱(当前服务器不可用),已拒绝执行") from appPublic.uniqueID import getID job_id = getID()[:12] jdir = job_dir(workspace_dir, job_id) os.makedirs(jdir, exist_ok=True) log_path = os.path.join(jdir, "output.log") if bwrap: writable_root = cwd if strict else await _sandbox_writable_root(cwd) cmd = _build_agent_bwrap_cmd(bwrap, cwd, writable_root, command, include_platform_ro=not strict) else: cmd = ["/bin/bash", "-c", command] log_f = open(log_path, "ab") try: proc = await asyncio.create_subprocess_exec( *cmd, stdout=log_f, stderr=asyncio.subprocess.STDOUT, cwd=None if bwrap else cwd, start_new_session=True, # 独立进程组,kill 时整组终止 ) finally: log_f.close() meta = { "job_id": job_id, "command": command[:2000], "pid": proc.pid, "status": "running", "started_at": time.time(), "ended_at": None, "rc": None, "sandbox": bool(bwrap), "timeout": timeout, "strict": bool(strict), } _write_meta(jdir, meta) _PROCS[job_id] = proc w = asyncio.create_task(_watch(job_id, jdir, proc, timeout)) _WATCHERS.add(w) w.add_done_callback(_WATCHERS.discard) return job_id async def _watch(job_id: str, jdir: str, proc: asyncio.subprocess.Process, timeout: int): """等待进程结束写 rc;超时 SIGTERM 进程组。worker 崩溃则本协程消失, meta 残留 running,由 poll 的 pid 存活检测兜底标记。""" status, rc = "done", None try: rc = await asyncio.wait_for(proc.wait(), timeout=timeout) if rc != 0: status = "failed" except asyncio.TimeoutError: status = "timeout" _terminate(proc) try: await asyncio.wait_for(proc.wait(), timeout=10) except Exception: pass except asyncio.CancelledError: status = "killed" raise except Exception as e: logger.error(f"bg watch {job_id} error: {e}") status = "failed" finally: try: meta = _read_meta(jdir) if meta.get("status") == "running": # kill_job 可能已改写 meta.update({"status": status, "rc": rc, "ended_at": time.time()}) _write_meta(jdir, meta) except Exception: pass _PROCS.pop(job_id, None) def _terminate(proc): try: os.killpg(os.getpgid(proc.pid), signal.SIGTERM) except Exception: try: proc.terminate() except Exception: pass def poll_job(workspace_dir: str, job_id: str, offset: int = 0) -> dict: """查状态 + 新增输出(offset 之后的部分)。跨进程安全(纯文件读)。""" jdir = job_dir(workspace_dir, job_id) meta = _read_meta(jdir) # stale 检测:meta 说 running 但 pid 已不在(启动它的 worker 崩了) if meta.get("status") == "running" and not _pid_alive(meta.get("pid") or 0) \ and job_id not in _PROCS: # 宽限 5 秒(进程刚退出、watcher 还没写 meta 的窗口) if time.time() - (meta.get("started_at") or 0) > 5: meta.update({"status": "failed", "rc": None, "ended_at": time.time()}) try: _write_meta(jdir, meta) except Exception: pass new_output = "" log_path = os.path.join(jdir, "output.log") total = 0 if os.path.exists(log_path): total = os.path.getsize(log_path) if offset < total: with open(log_path, "r", encoding="utf-8", errors="replace") as f: f.seek(offset) new_output = f.read()[-8000:] return {"meta": meta, "new_output": new_output, "log_size": total} def read_log(workspace_dir: str, job_id: str, offset: int = 0, limit: int = 20000) -> dict: jdir = job_dir(workspace_dir, job_id) meta = _read_meta(jdir) log_path = os.path.join(jdir, "output.log") content, total = "", 0 if os.path.exists(log_path): total = os.path.getsize(log_path) with open(log_path, "r", encoding="utf-8", errors="replace") as f: if offset: f.seek(offset) content = f.read(limit) return {"meta": meta, "content": content, "offset": offset, "log_size": total, "truncated": offset + len(content.encode('utf-8', 'replace')) < total} async def wait_job(workspace_dir: str, job_id: str, max_wait: int = 120) -> dict: """阻塞等待结束(最多 max_wait 秒),超时返回当前状态+部分输出。""" jdir = job_dir(workspace_dir, job_id) deadline = time.time() + min(max_wait, 120) while time.time() < deadline: meta = _read_meta(jdir) if meta.get("status") != "running": return poll_job(workspace_dir, job_id) proc = _PROCS.get(job_id) if proc is None and not _pid_alive(meta.get("pid") or 0): await asyncio.sleep(0.5) # 让 watcher 落 meta return poll_job(workspace_dir, job_id) await asyncio.sleep(1) return poll_job(workspace_dir, job_id) def kill_job(workspace_dir: str, job_id: str) -> dict: jdir = job_dir(workspace_dir, job_id) meta = _read_meta(jdir) if meta.get("status") != "running": return {"meta": meta, "message": f"任务已结束({meta.get('status')}),无需终止"} proc = _PROCS.get(job_id) if proc is not None: _terminate(proc) else: # 跨进程:按 meta 的 pid 杀进程组 pid = meta.get("pid") or 0 if _pid_alive(pid): try: os.killpg(os.getpgid(pid), signal.SIGTERM) except Exception: try: os.kill(pid, signal.SIGTERM) except Exception as e: raise BgJobError(f"终止失败: {e}") else: meta.update({"status": "failed", "ended_at": time.time()}) _write_meta(jdir, meta) return {"meta": meta, "message": "进程已不在(可能随启动进程退出),状态已标记"} meta.update({"status": "killed", "ended_at": time.time()}) _write_meta(jdir, meta) return {"meta": meta, "message": "已发送终止信号"}