1. memory 工具(add/list/remove): 多租户写入门禁——org_id/user_id强制注入、 scope白名单user/project/pipeline(global/org种子域禁写)、无身份拒写、 remove仅本人条目; 记忆注入改可见性过滤版(修跨机构泄漏) 2. manage_skill(create/patch/write_file/remove_file/delete): skill_live扩展, 只落本租户orgs/users目录, global原版fork-on-write(校验通过才fork,失败零残留), org+user双副本同步改, 产线层拒改, delete只删本租户副本+审计留痕 3. run_command background=true + process工具(poll/log/wait/kill): bg_jobs.py 状态文件化(workspace/.bg/,跨worker可见), 沙箱档位与前台一致(generic强制 strict+无bwrap拒绝), 超时SIGTERM进程组+stale心跳判活 4. delegate_subtask background + subagent工具(list/steer/stop/result): subagents.py 并行上限3/workspace, 深度限制1(子agent禁再委派), 子会话session_isolation=none(不读父历史不写会话表), steer/stop文件传递 每轮tool-loop边界消费, 结果流式落盘(stop即部分结果)
274 lines
9.9 KiB
Python
274 lines
9.9 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""bg_jobs.py - run_command 后台任务(2026-09-10,对齐 Hermes terminal background + process_manage)
|
||
|
||
状态文件化:{workspace}/.bg/{job_id}/ 下
|
||
meta.json {job_id, command, pid, status(running/done/failed/timeout/killed),
|
||
started_at, ended_at, rc, sandbox}
|
||
output.log stdout+stderr 合并追加
|
||
|
||
为什么文件化而不是内存注册表:web/worker 多进程(未来多主机 NFS 共享 workspace),
|
||
poll/log 请求可能落到没启动该任务的进程上——文件是唯一跨进程事实源。
|
||
进程句柄注册表(_PROCS)只用于本进程 kill 加速,缺失时回退 meta.json 的 pid。
|
||
|
||
隔离:任务目录在发起者的 workspace 内(项目工作空间 / _general/{uid}),
|
||
process 工具只解析 self.workspace_dir 下的 .bg——跨用户/跨项目天然不可达;
|
||
job_id 白名单正则防路径穿越。
|
||
|
||
沙箱:与前台 run_command 完全同一条路径(bwrap 参数同款),generic 会话
|
||
strict 档照旧;无 bwrap 时 generic 拒绝(不降级裸 shell)。
|
||
|
||
兜底:后台任务硬上限 MAX_BG_SECONDS(默认1小时),watcher 超时 SIGTERM 进程组;
|
||
worker 崩溃后 meta 残留 running——poll 时用 pid 存活检测标记 stale。
|
||
"""
|
||
import asyncio
|
||
import json
|
||
import logging
|
||
import os
|
||
import re
|
||
import signal
|
||
import time
|
||
|
||
logger = logging.getLogger("pipeline.bg_jobs")
|
||
|
||
BG_DIR = ".bg"
|
||
MAX_BG_SECONDS = 3600
|
||
_JOB_ID_RE = re.compile(r"^[A-Za-z0-9_-]{1,32}$")
|
||
|
||
# 本进程启动的任务句柄(跨进程 kill 回退 pid)
|
||
_PROCS = {}
|
||
# watcher 协程引用(asyncio.create_task 不保引用会被 GC 中途回收)
|
||
_WATCHERS = set()
|
||
|
||
|
||
class BgJobError(Exception):
|
||
pass
|
||
|
||
|
||
def _bg_root(workspace_dir: str) -> str:
|
||
return os.path.join(workspace_dir, BG_DIR)
|
||
|
||
|
||
def _validate_job_id(job_id: str) -> str:
|
||
jid = (job_id or "").strip()
|
||
if not _JOB_ID_RE.match(jid):
|
||
raise BgJobError(f"非法 job_id: {job_id!r}")
|
||
return jid
|
||
|
||
|
||
def job_dir(workspace_dir: str, job_id: str) -> str:
|
||
jid = _validate_job_id(job_id)
|
||
return os.path.join(_bg_root(workspace_dir), jid)
|
||
|
||
|
||
def _write_meta(jdir: str, meta: dict):
|
||
tmp = os.path.join(jdir, "meta.json.tmp")
|
||
with open(tmp, "w", encoding="utf-8") as f:
|
||
json.dump(meta, f, ensure_ascii=False)
|
||
os.replace(tmp, os.path.join(jdir, "meta.json"))
|
||
|
||
|
||
def _read_meta(jdir: str) -> dict:
|
||
path = os.path.join(jdir, "meta.json")
|
||
if not os.path.exists(path):
|
||
raise BgJobError("任务不存在(meta.json 缺失)")
|
||
with open(path, "r", encoding="utf-8") as f:
|
||
return json.load(f)
|
||
|
||
|
||
def _pid_alive(pid: int) -> bool:
|
||
if not pid:
|
||
return False
|
||
try:
|
||
os.kill(pid, 0)
|
||
return True
|
||
except OSError:
|
||
return False
|
||
|
||
|
||
async def start_bg_job(command: str, workspace_dir: str, strict: bool = False,
|
||
timeout: int = MAX_BG_SECONDS) -> str:
|
||
"""启动后台命令,返回 job_id。沙箱/目录安全校验与前台 _run_shell 完全一致。"""
|
||
from .agent_loop import (_find_bwrap, _build_agent_bwrap_cmd,
|
||
_sandbox_writable_root, _is_safe_workdir_async)
|
||
|
||
command = (command or "").strip()
|
||
if not command:
|
||
raise BgJobError("命令为空")
|
||
cwd = os.path.abspath(workspace_dir)
|
||
if not await _is_safe_workdir_async(cwd):
|
||
raise BgJobError(f"安全限制:目录 {cwd} 不在允许范围")
|
||
if not os.path.isdir(cwd):
|
||
raise BgJobError(f"目录不存在: {cwd}")
|
||
timeout = max(10, min(int(timeout or MAX_BG_SECONDS), MAX_BG_SECONDS))
|
||
|
||
bwrap = _find_bwrap()
|
||
if strict and not bwrap:
|
||
raise BgJobError("通用会话的后台命令需要 bwrap 沙箱(当前服务器不可用),已拒绝执行")
|
||
|
||
from appPublic.uniqueID import getID
|
||
job_id = getID()[:12]
|
||
jdir = job_dir(workspace_dir, job_id)
|
||
os.makedirs(jdir, exist_ok=True)
|
||
log_path = os.path.join(jdir, "output.log")
|
||
|
||
if bwrap:
|
||
writable_root = cwd if strict else await _sandbox_writable_root(cwd)
|
||
cmd = _build_agent_bwrap_cmd(bwrap, cwd, writable_root, command,
|
||
include_platform_ro=not strict)
|
||
else:
|
||
cmd = ["/bin/bash", "-c", command]
|
||
|
||
log_f = open(log_path, "ab")
|
||
try:
|
||
proc = await asyncio.create_subprocess_exec(
|
||
*cmd, stdout=log_f, stderr=asyncio.subprocess.STDOUT,
|
||
cwd=None if bwrap else cwd,
|
||
start_new_session=True, # 独立进程组,kill 时整组终止
|
||
)
|
||
finally:
|
||
log_f.close()
|
||
|
||
meta = {
|
||
"job_id": job_id, "command": command[:2000], "pid": proc.pid,
|
||
"status": "running", "started_at": time.time(), "ended_at": None,
|
||
"rc": None, "sandbox": bool(bwrap), "timeout": timeout,
|
||
"strict": bool(strict),
|
||
}
|
||
_write_meta(jdir, meta)
|
||
_PROCS[job_id] = proc
|
||
|
||
w = asyncio.create_task(_watch(job_id, jdir, proc, timeout))
|
||
_WATCHERS.add(w)
|
||
w.add_done_callback(_WATCHERS.discard)
|
||
return job_id
|
||
|
||
|
||
async def _watch(job_id: str, jdir: str, proc: asyncio.subprocess.Process,
|
||
timeout: int):
|
||
"""等待进程结束写 rc;超时 SIGTERM 进程组。worker 崩溃则本协程消失,
|
||
meta 残留 running,由 poll 的 pid 存活检测兜底标记。"""
|
||
status, rc = "done", None
|
||
try:
|
||
rc = await asyncio.wait_for(proc.wait(), timeout=timeout)
|
||
if rc != 0:
|
||
status = "failed"
|
||
except asyncio.TimeoutError:
|
||
status = "timeout"
|
||
_terminate(proc)
|
||
try:
|
||
await asyncio.wait_for(proc.wait(), timeout=10)
|
||
except Exception:
|
||
pass
|
||
except asyncio.CancelledError:
|
||
status = "killed"
|
||
raise
|
||
except Exception as e:
|
||
logger.error(f"bg watch {job_id} error: {e}")
|
||
status = "failed"
|
||
finally:
|
||
try:
|
||
meta = _read_meta(jdir)
|
||
if meta.get("status") == "running": # kill_job 可能已改写
|
||
meta.update({"status": status, "rc": rc, "ended_at": time.time()})
|
||
_write_meta(jdir, meta)
|
||
except Exception:
|
||
pass
|
||
_PROCS.pop(job_id, None)
|
||
|
||
|
||
def _terminate(proc):
|
||
try:
|
||
os.killpg(os.getpgid(proc.pid), signal.SIGTERM)
|
||
except Exception:
|
||
try:
|
||
proc.terminate()
|
||
except Exception:
|
||
pass
|
||
|
||
|
||
def poll_job(workspace_dir: str, job_id: str, offset: int = 0) -> dict:
|
||
"""查状态 + 新增输出(offset 之后的部分)。跨进程安全(纯文件读)。"""
|
||
jdir = job_dir(workspace_dir, job_id)
|
||
meta = _read_meta(jdir)
|
||
# stale 检测:meta 说 running 但 pid 已不在(启动它的 worker 崩了)
|
||
if meta.get("status") == "running" and not _pid_alive(meta.get("pid") or 0) \
|
||
and job_id not in _PROCS:
|
||
# 宽限 5 秒(进程刚退出、watcher 还没写 meta 的窗口)
|
||
if time.time() - (meta.get("started_at") or 0) > 5:
|
||
meta.update({"status": "failed", "rc": None,
|
||
"ended_at": time.time()})
|
||
try:
|
||
_write_meta(jdir, meta)
|
||
except Exception:
|
||
pass
|
||
new_output = ""
|
||
log_path = os.path.join(jdir, "output.log")
|
||
total = 0
|
||
if os.path.exists(log_path):
|
||
total = os.path.getsize(log_path)
|
||
if offset < total:
|
||
with open(log_path, "r", encoding="utf-8", errors="replace") as f:
|
||
f.seek(offset)
|
||
new_output = f.read()[-8000:]
|
||
return {"meta": meta, "new_output": new_output, "log_size": total}
|
||
|
||
|
||
def read_log(workspace_dir: str, job_id: str, offset: int = 0,
|
||
limit: int = 20000) -> dict:
|
||
jdir = job_dir(workspace_dir, job_id)
|
||
meta = _read_meta(jdir)
|
||
log_path = os.path.join(jdir, "output.log")
|
||
content, total = "", 0
|
||
if os.path.exists(log_path):
|
||
total = os.path.getsize(log_path)
|
||
with open(log_path, "r", encoding="utf-8", errors="replace") as f:
|
||
if offset:
|
||
f.seek(offset)
|
||
content = f.read(limit)
|
||
return {"meta": meta, "content": content, "offset": offset,
|
||
"log_size": total, "truncated": offset + len(content.encode('utf-8', 'replace')) < total}
|
||
|
||
|
||
async def wait_job(workspace_dir: str, job_id: str, max_wait: int = 120) -> dict:
|
||
"""阻塞等待结束(最多 max_wait 秒),超时返回当前状态+部分输出。"""
|
||
jdir = job_dir(workspace_dir, job_id)
|
||
deadline = time.time() + min(max_wait, 120)
|
||
while time.time() < deadline:
|
||
meta = _read_meta(jdir)
|
||
if meta.get("status") != "running":
|
||
return poll_job(workspace_dir, job_id)
|
||
proc = _PROCS.get(job_id)
|
||
if proc is None and not _pid_alive(meta.get("pid") or 0):
|
||
await asyncio.sleep(0.5) # 让 watcher 落 meta
|
||
return poll_job(workspace_dir, job_id)
|
||
await asyncio.sleep(1)
|
||
return poll_job(workspace_dir, job_id)
|
||
|
||
|
||
def kill_job(workspace_dir: str, job_id: str) -> dict:
|
||
jdir = job_dir(workspace_dir, job_id)
|
||
meta = _read_meta(jdir)
|
||
if meta.get("status") != "running":
|
||
return {"meta": meta, "message": f"任务已结束({meta.get('status')}),无需终止"}
|
||
proc = _PROCS.get(job_id)
|
||
if proc is not None:
|
||
_terminate(proc)
|
||
else:
|
||
# 跨进程:按 meta 的 pid 杀进程组
|
||
pid = meta.get("pid") or 0
|
||
if _pid_alive(pid):
|
||
try:
|
||
os.killpg(os.getpgid(pid), signal.SIGTERM)
|
||
except Exception:
|
||
try:
|
||
os.kill(pid, signal.SIGTERM)
|
||
except Exception as e:
|
||
raise BgJobError(f"终止失败: {e}")
|
||
else:
|
||
meta.update({"status": "failed", "ended_at": time.time()})
|
||
_write_meta(jdir, meta)
|
||
return {"meta": meta, "message": "进程已不在(可能随启动进程退出),状态已标记"}
|
||
meta.update({"status": "killed", "ended_at": time.time()})
|
||
_write_meta(jdir, meta)
|
||
return {"meta": meta, "message": "已发送终止信号"}
|