ymq 1c04efe1be feat(agent): 会话agent四项Hermes能力对齐(记忆写入/技能管理/后台进程/并行委派)
1. memory 工具(add/list/remove): 多租户写入门禁——org_id/user_id强制注入、
   scope白名单user/project/pipeline(global/org种子域禁写)、无身份拒写、
   remove仅本人条目; 记忆注入改可见性过滤版(修跨机构泄漏)
2. manage_skill(create/patch/write_file/remove_file/delete): skill_live扩展,
   只落本租户orgs/users目录, global原版fork-on-write(校验通过才fork,失败零残留),
   org+user双副本同步改, 产线层拒改, delete只删本租户副本+审计留痕
3. run_command background=true + process工具(poll/log/wait/kill): bg_jobs.py
   状态文件化(workspace/.bg/,跨worker可见), 沙箱档位与前台一致(generic强制
   strict+无bwrap拒绝), 超时SIGTERM进程组+stale心跳判活
4. delegate_subtask background + subagent工具(list/steer/stop/result):
   subagents.py 并行上限3/workspace, 深度限制1(子agent禁再委派),
   子会话session_isolation=none(不读父历史不写会话表), steer/stop文件传递
   每轮tool-loop边界消费, 结果流式落盘(stop即部分结果)
2026-09-10 17:09:45 +08:00

274 lines
9.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# -*- coding: utf-8 -*-
"""bg_jobs.py - run_command 后台任务(2026-09-10,对齐 Hermes terminal background + process_manage)
状态文件化:{workspace}/.bg/{job_id}/ 下
meta.json {job_id, command, pid, status(running/done/failed/timeout/killed),
started_at, ended_at, rc, sandbox}
output.log stdout+stderr 合并追加
为什么文件化而不是内存注册表:web/worker 多进程(未来多主机 NFS 共享 workspace),
poll/log 请求可能落到没启动该任务的进程上——文件是唯一跨进程事实源。
进程句柄注册表(_PROCS)只用于本进程 kill 加速,缺失时回退 meta.json 的 pid。
隔离:任务目录在发起者的 workspace 内(项目工作空间 / _general/{uid}),
process 工具只解析 self.workspace_dir 下的 .bg——跨用户/跨项目天然不可达;
job_id 白名单正则防路径穿越。
沙箱:与前台 run_command 完全同一条路径(bwrap 参数同款),generic 会话
strict 档照旧;无 bwrap 时 generic 拒绝(不降级裸 shell)。
兜底:后台任务硬上限 MAX_BG_SECONDS(默认1小时),watcher 超时 SIGTERM 进程组;
worker 崩溃后 meta 残留 running——poll 时用 pid 存活检测标记 stale。
"""
import asyncio
import json
import logging
import os
import re
import signal
import time
logger = logging.getLogger("pipeline.bg_jobs")
BG_DIR = ".bg"
MAX_BG_SECONDS = 3600
_JOB_ID_RE = re.compile(r"^[A-Za-z0-9_-]{1,32}$")
# 本进程启动的任务句柄(跨进程 kill 回退 pid)
_PROCS = {}
# watcher 协程引用(asyncio.create_task 不保引用会被 GC 中途回收)
_WATCHERS = set()
class BgJobError(Exception):
pass
def _bg_root(workspace_dir: str) -> str:
return os.path.join(workspace_dir, BG_DIR)
def _validate_job_id(job_id: str) -> str:
jid = (job_id or "").strip()
if not _JOB_ID_RE.match(jid):
raise BgJobError(f"非法 job_id: {job_id!r}")
return jid
def job_dir(workspace_dir: str, job_id: str) -> str:
jid = _validate_job_id(job_id)
return os.path.join(_bg_root(workspace_dir), jid)
def _write_meta(jdir: str, meta: dict):
tmp = os.path.join(jdir, "meta.json.tmp")
with open(tmp, "w", encoding="utf-8") as f:
json.dump(meta, f, ensure_ascii=False)
os.replace(tmp, os.path.join(jdir, "meta.json"))
def _read_meta(jdir: str) -> dict:
path = os.path.join(jdir, "meta.json")
if not os.path.exists(path):
raise BgJobError("任务不存在(meta.json 缺失)")
with open(path, "r", encoding="utf-8") as f:
return json.load(f)
def _pid_alive(pid: int) -> bool:
if not pid:
return False
try:
os.kill(pid, 0)
return True
except OSError:
return False
async def start_bg_job(command: str, workspace_dir: str, strict: bool = False,
timeout: int = MAX_BG_SECONDS) -> str:
"""启动后台命令,返回 job_id。沙箱/目录安全校验与前台 _run_shell 完全一致。"""
from .agent_loop import (_find_bwrap, _build_agent_bwrap_cmd,
_sandbox_writable_root, _is_safe_workdir_async)
command = (command or "").strip()
if not command:
raise BgJobError("命令为空")
cwd = os.path.abspath(workspace_dir)
if not await _is_safe_workdir_async(cwd):
raise BgJobError(f"安全限制:目录 {cwd} 不在允许范围")
if not os.path.isdir(cwd):
raise BgJobError(f"目录不存在: {cwd}")
timeout = max(10, min(int(timeout or MAX_BG_SECONDS), MAX_BG_SECONDS))
bwrap = _find_bwrap()
if strict and not bwrap:
raise BgJobError("通用会话的后台命令需要 bwrap 沙箱(当前服务器不可用),已拒绝执行")
from appPublic.uniqueID import getID
job_id = getID()[:12]
jdir = job_dir(workspace_dir, job_id)
os.makedirs(jdir, exist_ok=True)
log_path = os.path.join(jdir, "output.log")
if bwrap:
writable_root = cwd if strict else await _sandbox_writable_root(cwd)
cmd = _build_agent_bwrap_cmd(bwrap, cwd, writable_root, command,
include_platform_ro=not strict)
else:
cmd = ["/bin/bash", "-c", command]
log_f = open(log_path, "ab")
try:
proc = await asyncio.create_subprocess_exec(
*cmd, stdout=log_f, stderr=asyncio.subprocess.STDOUT,
cwd=None if bwrap else cwd,
start_new_session=True, # 独立进程组,kill 时整组终止
)
finally:
log_f.close()
meta = {
"job_id": job_id, "command": command[:2000], "pid": proc.pid,
"status": "running", "started_at": time.time(), "ended_at": None,
"rc": None, "sandbox": bool(bwrap), "timeout": timeout,
"strict": bool(strict),
}
_write_meta(jdir, meta)
_PROCS[job_id] = proc
w = asyncio.create_task(_watch(job_id, jdir, proc, timeout))
_WATCHERS.add(w)
w.add_done_callback(_WATCHERS.discard)
return job_id
async def _watch(job_id: str, jdir: str, proc: asyncio.subprocess.Process,
timeout: int):
"""等待进程结束写 rc;超时 SIGTERM 进程组。worker 崩溃则本协程消失,
meta 残留 running,由 poll 的 pid 存活检测兜底标记。"""
status, rc = "done", None
try:
rc = await asyncio.wait_for(proc.wait(), timeout=timeout)
if rc != 0:
status = "failed"
except asyncio.TimeoutError:
status = "timeout"
_terminate(proc)
try:
await asyncio.wait_for(proc.wait(), timeout=10)
except Exception:
pass
except asyncio.CancelledError:
status = "killed"
raise
except Exception as e:
logger.error(f"bg watch {job_id} error: {e}")
status = "failed"
finally:
try:
meta = _read_meta(jdir)
if meta.get("status") == "running": # kill_job 可能已改写
meta.update({"status": status, "rc": rc, "ended_at": time.time()})
_write_meta(jdir, meta)
except Exception:
pass
_PROCS.pop(job_id, None)
def _terminate(proc):
try:
os.killpg(os.getpgid(proc.pid), signal.SIGTERM)
except Exception:
try:
proc.terminate()
except Exception:
pass
def poll_job(workspace_dir: str, job_id: str, offset: int = 0) -> dict:
"""查状态 + 新增输出(offset 之后的部分)。跨进程安全(纯文件读)。"""
jdir = job_dir(workspace_dir, job_id)
meta = _read_meta(jdir)
# stale 检测:meta 说 running 但 pid 已不在(启动它的 worker 崩了)
if meta.get("status") == "running" and not _pid_alive(meta.get("pid") or 0) \
and job_id not in _PROCS:
# 宽限 5 秒(进程刚退出、watcher 还没写 meta 的窗口)
if time.time() - (meta.get("started_at") or 0) > 5:
meta.update({"status": "failed", "rc": None,
"ended_at": time.time()})
try:
_write_meta(jdir, meta)
except Exception:
pass
new_output = ""
log_path = os.path.join(jdir, "output.log")
total = 0
if os.path.exists(log_path):
total = os.path.getsize(log_path)
if offset < total:
with open(log_path, "r", encoding="utf-8", errors="replace") as f:
f.seek(offset)
new_output = f.read()[-8000:]
return {"meta": meta, "new_output": new_output, "log_size": total}
def read_log(workspace_dir: str, job_id: str, offset: int = 0,
limit: int = 20000) -> dict:
jdir = job_dir(workspace_dir, job_id)
meta = _read_meta(jdir)
log_path = os.path.join(jdir, "output.log")
content, total = "", 0
if os.path.exists(log_path):
total = os.path.getsize(log_path)
with open(log_path, "r", encoding="utf-8", errors="replace") as f:
if offset:
f.seek(offset)
content = f.read(limit)
return {"meta": meta, "content": content, "offset": offset,
"log_size": total, "truncated": offset + len(content.encode('utf-8', 'replace')) < total}
async def wait_job(workspace_dir: str, job_id: str, max_wait: int = 120) -> dict:
"""阻塞等待结束(最多 max_wait 秒),超时返回当前状态+部分输出。"""
jdir = job_dir(workspace_dir, job_id)
deadline = time.time() + min(max_wait, 120)
while time.time() < deadline:
meta = _read_meta(jdir)
if meta.get("status") != "running":
return poll_job(workspace_dir, job_id)
proc = _PROCS.get(job_id)
if proc is None and not _pid_alive(meta.get("pid") or 0):
await asyncio.sleep(0.5) # 让 watcher 落 meta
return poll_job(workspace_dir, job_id)
await asyncio.sleep(1)
return poll_job(workspace_dir, job_id)
def kill_job(workspace_dir: str, job_id: str) -> dict:
jdir = job_dir(workspace_dir, job_id)
meta = _read_meta(jdir)
if meta.get("status") != "running":
return {"meta": meta, "message": f"任务已结束({meta.get('status')}),无需终止"}
proc = _PROCS.get(job_id)
if proc is not None:
_terminate(proc)
else:
# 跨进程:按 meta 的 pid 杀进程组
pid = meta.get("pid") or 0
if _pid_alive(pid):
try:
os.killpg(os.getpgid(pid), signal.SIGTERM)
except Exception:
try:
os.kill(pid, signal.SIGTERM)
except Exception as e:
raise BgJobError(f"终止失败: {e}")
else:
meta.update({"status": "failed", "ended_at": time.time()})
_write_meta(jdir, meta)
return {"meta": meta, "message": "进程已不在(可能随启动进程退出),状态已标记"}
meta.update({"status": "killed", "ended_at": time.time()})
_write_meta(jdir, meta)
return {"meta": meta, "message": "已发送终止信号"}