1044 lines
47 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""pipeline_service - 通用产线执行引擎模块。
把 Hermes Agent 验证过的业务流程固化为可重复、可并发的产线业务环境。
支持多租户隔离、DAG 步骤调度、可插拔步骤处理器、artifact 版本管理。
支持人工交互步骤human_task/approval_gate人机协作产线。
任何宿主应用都可以通过 load_pipeline_service() 加载本模块。
"""
import json
import asyncio
import os
from ahserver.serverenv import ServerEnv
from appPublic.uniqueID import getID
from appPublic.log import debug
from .state import (
TASK_SUBMITTED, TASK_RUNNING, TASK_COMPLETED, TASK_FAILED, TASK_PAUSED, TASK_CANCELLED, TASK_WAITING,
TASK_REVIEW, TASK_APPROVED,
build_step_graph, get_cascade_rerun_steps, get_rerun_from_next,
)
from .storage import (
create_task, init_task_steps, get_task, get_task_steps,
get_artifact, get_all_artifacts, list_tasks,
update_task_state, update_task_version,
get_pipeline_steps, reset_steps, save_artifact,
get_human_task,
)
from .executor import start_task, resume_task, stop_task, is_running
from .handler import register_handler, list_handlers, register_default_handler
from .step_registry import (
register_step_type, get_step_type, list_step_types,
unregister_step_type, load_builtin_types,
)
from .human import human_complete, approval_approve, approval_reject, human_list
from .questions import (
agent_ask, answer_question, forward_question,
get_question, list_questions, get_task_qna,
)
from .communication import (
raise_problem, resolve_problem, escalate_problem,
list_problems_for, get_task_qa,
)
from .audit import record_audit
from .task_capability import (
claim_task, submit_task, approve_task, reject_task,
complete_task, mark_failed, retry_task, suspend_task,
revive_task, set_task_state,
)
from .agent_loop import role_agent_run, role_agent_loop, agent_loop, run_agent_loop
from .agent_loop import pm_review_run, pm_review_loop, qc_review_run, handle_failed_task
MODULE_NAME = "pipeline_service"
MODULE_VERSION = "3.4.0"
async def pipeline_submit(tenant_id, pipeline_id, owner_id, title, params=None):
"""提交新产线任务。
Args:
tenant_id: 租户ID由宿主应用提供可以是 org_id、user_id 等)
pipeline_id: 产线定义ID来自 pipelines 表)
owner_id: 提交人ID
title: 任务标题
params: 提交参数dict
Returns:
JSON string with status, task_id
"""
result = {"success": False}
try:
if not tenant_id or not pipeline_id:
result["message"] = "缺少 tenant_id 或 pipeline_id"
return json.dumps(result, ensure_ascii=False)
params = params or {}
task_id = await create_task(tenant_id, pipeline_id, owner_id, title, params)
# Read step definitions from pipeline_steps table
step_records = await get_pipeline_steps(pipeline_id)
if not step_records:
result["message"] = f"产线 {pipeline_id} 没有步骤定义"
return json.dumps(result, ensure_ascii=False)
# Create step execution records
await init_task_steps(task_id, step_records)
# Start execution
await start_task(task_id)
result["success"] = True
result["task_id"] = task_id
result["message"] = "任务已提交并开始执行"
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False)
async def pipeline_role_submit(tenant_id, pipeline_id, owner_id, title, params=None, role=""):
"""提交角色任务主agent → 角色agent 模式v3.2.0)。
与 pipeline_submit 的区别:
- 带 role 字段由角色agent循环role_agent_loop认领执行
- 不创建步骤记录、不启动 DAG executor
Args:
role: 目标角色requirement/design/dev/test/ops 等)。必填。
Returns:
JSON string with success, task_id
"""
result = {"success": False}
try:
if not tenant_id:
result["message"] = "缺少 tenant_id"
return json.dumps(result, ensure_ascii=False)
if not role:
result["message"] = "缺少 role角色任务必须指定目标角色"
return json.dumps(result, ensure_ascii=False)
params = params or {}
task_id = await create_task(tenant_id, pipeline_id or "", owner_id, title, params, role=role)
result["success"] = True
result["task_id"] = task_id
result["role"] = role
result["message"] = f"角色任务已提交role={role}等待角色agent认领执行"
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False)
async def pipeline_list(tenant_id, pipeline_id=None, limit=100):
"""查询租户的任务列表。"""
result = {"success": False}
try:
tasks = await list_tasks(tenant_id, pipeline_id, limit)
result["success"] = True
result["tasks"] = tasks
result["total"] = len(tasks)
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False, default=str)
async def pipeline_detail(tenant_id, task_id):
"""获取任务详情 + 步骤状态树。"""
result = {"success": False}
try:
task = await get_task(tenant_id, task_id)
if not task:
result["message"] = "任务不存在"
return json.dumps(result, ensure_ascii=False)
steps = await get_task_steps(task_id)
# Enrich steps with human task info for interactive steps
for step in steps:
if step.get('state') == 'waiting':
ht = await get_human_task(task_id, step['step_name'])
if ht:
step['human_task'] = ht
task["steps"] = steps
task["is_running"] = is_running(task_id)
result["success"] = True
result["task"] = task
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False, default=str)
async def pipeline_node(tenant_id, task_id, step_name, version=None):
"""获取某节点某版本的 input/output artifact。"""
result = {"success": False}
try:
task = await get_task(tenant_id, task_id)
if not task:
result["message"] = "任务不存在"
return json.dumps(result, ensure_ascii=False)
v = version or task.get("current_version", 1)
if isinstance(v, str):
v = int(v)
input_data = await get_artifact(task_id, v, step_name, "input")
output_data = await get_artifact(task_id, v, step_name, "output")
result["success"] = True
result["step_name"] = step_name
result["version"] = v
result["input"] = input_data
result["output"] = output_data
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False, default=str)
async def pipeline_modify(tenant_id, task_id, updates, rerun_from="node"):
"""修改节点 artifact 并触发级联重跑。
Args:
updates: {step_name: {content: {...}}, ...}
rerun_from: "node" = 从该节点重跑, "next" = 从下游节点重跑
"""
result = {"success": False}
try:
task = await get_task(tenant_id, task_id)
if not task:
result["message"] = "任务不存在"
return json.dumps(result, ensure_ascii=False)
if is_running(task_id):
result["message"] = "任务正在执行中,请先暂停"
return json.dumps(result, ensure_ascii=False)
pipeline_id = task.get("pipeline_id", task.get("Pipeline_id", ""))
current_version = task.get("current_version", task.get("current_Version", 1))
if isinstance(current_version, str):
current_version = int(current_version)
# Load step graph
step_records = await get_pipeline_steps(pipeline_id)
step_graph = build_step_graph(step_records)
# Calculate affected steps
all_rerun = set()
for step_name in updates:
if step_name not in step_graph:
result["message"] = f"未知步骤: {step_name}"
return json.dumps(result, ensure_ascii=False)
if rerun_from == "node":
affected = get_cascade_rerun_steps(step_graph, step_name)
else:
affected = get_rerun_from_next(step_graph, step_name)
all_rerun.update(affected)
# Create new version
new_version = current_version + 1
await update_task_version(task_id, new_version)
# Save modified artifacts
for step_name, step_update in updates.items():
content = step_update.get("content", step_update)
io_type = "input" if rerun_from == "node" else "output"
await save_artifact(task_id, new_version, step_name, io_type, content)
# Reset affected steps
all_rerun_list = sorted(all_rerun, key=lambda s: step_graph.get(s, {}).get("order", 999))
await reset_steps(task_id, all_rerun_list)
# Update task state and resume
await update_task_state(task_id, TASK_RUNNING)
await resume_task(task_id)
result["success"] = True
result["new_version"] = new_version
result["rerun_steps"] = all_rerun_list
result["message"] = f"创建 v{new_version},重跑 {len(all_rerun_list)} 个步骤"
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False, default=str)
async def pipeline_pause(tenant_id, task_id):
"""暂停任务。"""
result = {"success": False}
try:
task = await get_task(tenant_id, task_id)
if not task:
result["message"] = "任务不存在"
return json.dumps(result, ensure_ascii=False)
await stop_task(task_id)
await update_task_state(task_id, TASK_PAUSED)
result["success"] = True
result["message"] = "任务已暂停"
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False)
async def pipeline_resume(tenant_id, task_id):
"""恢复任务。"""
result = {"success": False}
try:
task = await get_task(tenant_id, task_id)
if not task:
result["message"] = "任务不存在"
return json.dumps(result, ensure_ascii=False)
await update_task_state(task_id, TASK_RUNNING)
await resume_task(task_id)
result["success"] = True
result["message"] = "任务已恢复"
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False)
async def pipeline_cancel(tenant_id, task_id):
"""取消任务。"""
result = {"success": False}
try:
task = await get_task(tenant_id, task_id)
if not task:
result["message"] = "任务不存在"
return json.dumps(result, ensure_ascii=False)
await stop_task(task_id)
await update_task_state(task_id, TASK_CANCELLED)
result["success"] = True
result["message"] = "任务已取消"
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False, default=str)
async def pipeline_restart(tenant_id, task_id):
"""重启已完成/失败/取消的任务 — 重置所有步骤为pending新版本继续。"""
result = {"success": False}
try:
task = await get_task(tenant_id, task_id)
if not task:
result["message"] = "任务不存在"
return json.dumps(result, ensure_ascii=False)
if is_running(task_id):
result["message"] = "任务正在执行中,请先暂停"
return json.dumps(result, ensure_ascii=False)
# Stop any lingering executor
await stop_task(task_id)
# Get step names to reset
steps = await get_task_steps(task_id)
step_names = [s['step_name'] for s in steps]
# New version
current_version = task.get("current_version", task.get("current_Version", 1))
if isinstance(current_version, str):
current_version = int(current_version)
new_version = current_version + 1
await update_task_version(task_id, new_version)
# Reset all steps to pending
await reset_steps(task_id, step_names)
# Restart execution
await update_task_state(task_id, TASK_RUNNING)
await start_task(task_id)
result["success"] = True
result["new_version"] = new_version
result["message"] = f"任务已重新启动 v{new_version}"
except Exception as e:
result["message"] = str(e)
return json.dumps(result, ensure_ascii=False, default=str)
def pipeline_handlers():
"""查看已注册的步骤处理器。"""
return json.dumps(list_handlers(), ensure_ascii=False)
def pipeline_step_types():
"""查看所有注册的步骤类型(含元数据)。"""
return json.dumps(list_step_types(), ensure_ascii=False)
def pipeline_register_step_type(step_type, metadata):
"""注册步骤类型(可装卸)。"""
register_step_type(step_type, metadata)
return json.dumps({"success": True, "step_type": step_type}, ensure_ascii=False)
def pipeline_unregister_step_type(step_type):
"""卸载步骤类型。"""
unregister_step_type(step_type)
return json.dumps({"success": True, "step_type": step_type}, ensure_ascii=False)
# ── Git / Shell 操作v3.2.1: 主agent devops 能力)──
_SHELL_BASE_DIR = '/d/pipeline/workspaces'
_WORKDIR_FALLBACK = '/tmp/pipeline_workspaces'
_PIPELINE_WS = os.path.expanduser('~/pipeline_ws')
def _resolve_workdir():
"""选择可写的工作目录根。"""
parent = os.path.dirname(_SHELL_BASE_DIR)
if os.path.isdir(parent) and os.access(parent, os.W_OK):
os.makedirs(_SHELL_BASE_DIR, exist_ok=True)
return _SHELL_BASE_DIR
os.makedirs(_WORKDIR_FALLBACK, exist_ok=True)
return _WORKDIR_FALLBACK
async def shell_exec(command: str, workdir: str = None, timeout: int = 60):
"""安全外壳执行。限制在 workspace 目录下,超时自动终止。
Returns: {"rc": exit_code, "stdout": "...", "stderr": "..."}
"""
cwd = workdir or _resolve_workdir()
cwd = os.path.abspath(cwd)
# 如果指定目录不存在,回落到可写基础目录
if not os.path.isdir(cwd):
cwd = _resolve_workdir()
cwd = os.path.abspath(cwd)
base = os.path.abspath(_SHELL_BASE_DIR)
fallback = os.path.abspath(_WORKDIR_FALLBACK)
pipeline_ws = os.path.abspath(_PIPELINE_WS)
if not (cwd.startswith(base) or cwd.startswith(fallback) or cwd.startswith(pipeline_ws)):
return {"rc": -1, "stdout": "", "stderr": f"安全限制:工作目录必须在允许范围内"}
try:
proc = await asyncio.create_subprocess_shell(
command, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE,
cwd=cwd, executable='/bin/bash')
try:
stdout, stderr = await asyncio.wait_for(
proc.communicate(), timeout=timeout)
except asyncio.TimeoutError:
proc.kill()
await proc.wait()
return {"rc": -1, "stdout": "", "stderr": f"命令超时({timeout}s"}
return {"rc": proc.returncode or 0, "stdout": stdout.decode('utf-8', 'replace')[-8000:],
"stderr": stderr.decode('utf-8', 'replace')[-4000:]}
except Exception as e:
return {"rc": -1, "stdout": "", "stderr": str(e)[:500]}
async def skill_import_git(repo_url: str, skills_dir: str = 'skills'):
"""克隆 git 仓库,扫描指定目录下所有子目录(每个=一个企业技能)。
Returns: {"success": bool, "repo":..., "target":..., "skills": [{"name","path","description"}]}
"""
from appPublic.uniqueID import getID
repo_name = repo_url.rstrip('/').split('/')[-1].replace('.git', '') or f"repo_{getID()[:8]}"
target = os.path.join(_resolve_workdir(), repo_name)
if not os.path.isdir(target):
r = await shell_exec(f'git clone {repo_url} {target}', workdir=_resolve_workdir())
if r['rc'] != 0:
return {"success": False, "error": f"clone 失败: {r['stderr'][:500]}"}
full_skills = os.path.join(target, skills_dir)
if not os.path.isdir(full_skills):
entries = ', '.join(os.listdir(target)[:10]) if os.path.isdir(target) else '(dir not found)'
return {"success": False, "error": f"skills目录不存在: {full_skills},仓库内容: {entries}"}
found = []
for entry in sorted(os.listdir(full_skills)):
entry_path = os.path.join(full_skills, entry)
if os.path.isdir(entry_path):
desc = ''
skill_file = os.path.join(entry_path, 'SKILL.md')
if os.path.isfile(skill_file):
try:
with open(skill_file) as f:
content = f.read(500)
for line in content.split('\n'):
if line.startswith('description:'):
desc = line.split(':', 1)[1].strip()
break
except:
pass
found.append({"name": entry, "path": entry_path, "description": desc})
return {"success": True, "repo": repo_url, "target": target,
"skills_dir": skills_dir, "skills": found}
def load_pipeline_service():
"""注册所有函数到 ServerEnv。任何宿主应用调用此函数即可使用产线引擎。"""
import os as _os
# 运行模式(进程级,由 start.sh 设 PIPELINE_MODE
# all(默认) = HTTP + pollerweb = 仅 HTTPworker = 仅 poller分布式多主机 worker 节点)
_run_mode = (_os.environ.get('PIPELINE_MODE', 'all') or 'all').strip().lower()
_run_pollers = _run_mode in ('all', 'worker') # web 模式不注册 poller
env = ServerEnv()
# Task lifecycle
env.pipeline_submit = pipeline_submit
env.pipeline_role_submit = pipeline_role_submit
env.pipeline_list = pipeline_list
env.pipeline_detail = pipeline_detail
env.pipeline_node = pipeline_node
env.pipeline_modify = pipeline_modify
env.pipeline_pause = pipeline_pause
env.pipeline_resume = pipeline_resume
env.pipeline_cancel = pipeline_cancel
env.pipeline_restart = pipeline_restart
# Handler management
env.pipeline_register_handler = register_handler
env.pipeline_handlers = pipeline_handlers
# Step type registry (pluggable)
env.pipeline_step_types = pipeline_step_types
env.pipeline_register_step_type = pipeline_register_step_type
env.pipeline_unregister_step_type = pipeline_unregister_step_type
# Human task operations
env.human_task_complete = human_complete
env.approval_approve = approval_approve
env.approval_reject = approval_reject
env.human_task_list = human_list
# SDLC 项目级人类任务清单 + owner 校验
from .human_task_capability import (
create_human_task, complete_human_task, qc_human_task, confirm_stage_gate,
list_project_human_tasks, count_my_human_tasks, list_my_human_todos, bug_accept)
from .project_capability import check_project_owner, check_task_owner, check_tenant_owner
env.create_human_task = create_human_task
env.complete_human_task = complete_human_task
env.qc_human_task = qc_human_task
env.confirm_stage_gate = confirm_stage_gate
env.list_project_human_tasks = list_project_human_tasks
env.count_my_human_tasks = count_my_human_tasks
env.list_my_human_todos = list_my_human_todos
env.bug_accept = bug_accept
env.check_project_owner = check_project_owner
env.check_task_owner = check_task_owner
env.check_tenant_owner = check_tenant_owner
# 知识库入库建议PM 建议 → owner 待办批准 → rag API 上传)
from .kb_ingest_capability import suggest_kb_ingest, decide_kb_ingest
env.suggest_kb_ingest = suggest_kb_ingest
env.decide_kb_ingest = decide_kb_ingest
# 产线流程裁剪(提案/确认/驳回/查询,确认端点 flow_plan_confirm.dspy 调用)
from .flow_plan_capability import (
propose_flow_plan, confirm_flow_plan, reject_flow_plan,
get_active_plan, get_latest_plan, get_flow_template,
)
env.flow_plan_propose = propose_flow_plan
env.flow_plan_confirm = confirm_flow_plan
env.flow_plan_reject = reject_flow_plan
env.flow_plan_active = get_active_plan
env.flow_plan_latest = get_latest_plan
env.flow_template = get_flow_template
# Register default handler
register_default_handler()
# Load built-in interactive step types
load_builtin_types()
# 钉钉审批适配器:注册 step_type='dingtalk_approval' + 审批结果回调钩子
# 软依赖 dingdingflow未安装时步骤仍可挂起等人工审批只是不推钉钉
try:
from .dingtalk_approval import load_dingtalk_approval
load_dingtalk_approval()
except Exception as e:
debug(f'dingtalk_approval 加载失败(不影响产线其他功能): {e}')
# Register workspace file-management functions (shared by workspace_*.dspy)
from .workspace import load_workspace
load_workspace()
# 内部 LLM 代理:短期 token 鉴权,真 api_key 永不下发到运行环境
from .llm_proxy import load_llm_proxy
load_llm_proxy()
# Register intent classifier + LLM bridge (shared across all pipelines)
from .intent_classifier import intent_classify
from .llm_bridge import llm_call
env.intent_classify = intent_classify
env.pipeline_llm_call = llm_call
# Role-agent loop (v3.2.0: 主agent + 角色agent 认领执行)
env.role_agent_run = role_agent_run
env.role_agent_loop = role_agent_loop
env.agent_loop = agent_loop # 向后兼容
env.run_agent_loop = run_agent_loop # 向后兼容
# v2 Agent Executor (v3.5.0: 对照 Hermes Agent 架构)
from .agent_loop_v2 import AgentExecutor, run_agent
env.AgentExecutor = AgentExecutor
env.run_agent = run_agent
# gateway统一消息入口Web AgentIO / 微信通道共用)
from .gateway import get_gateway, Gateway
env.Gateway = Gateway
env.get_gateway = get_gateway
env.gateway = get_gateway()
# PM agent (v3.3.0: 项目经理审核 + 任务链)
env.pm_review_run = pm_review_run
env.pm_review_loop = pm_review_loop
env.qc_review_run = qc_review_run
# Question loop (角色agent ↔ 主agent ↔ 客户)
env.agent_ask = agent_ask
env.question_answer = answer_question
env.question_forward = forward_question
env.question_detail = get_question
env.question_list = list_questions
env.question_qna = get_task_qna
# 团队沟通(通用问题冒泡引擎,多 agent 感知;规范在 team-communication skill
env.raise_problem = raise_problem
env.resolve_problem = resolve_problem
env.escalate_problem = escalate_problem
env.problem_list_for = list_problems_for
env.problem_qa = get_task_qa
# 任务能力(状态机语义化迁移;状态机在 task skill+ 审计原语
env.record_audit = record_audit
env.claim_task = claim_task
env.submit_task = submit_task
env.approve_task = approve_task
env.reject_task = reject_task
env.complete_task = complete_task
env.mark_failed = mark_failed
env.retry_task = retry_task
env.suspend_task = suspend_task
env.revive_task = revive_task
env.set_task_state = set_task_state
# DevOps: git/shell operations (v3.2.1)
env.shell_exec = shell_exec
env.skill_import_git = skill_import_git
# Background poller: auto-dispatch submitted role_tasks to role agents
async def _role_poller(app):
"启动后台轮询器spawn 后立即返回(不阻塞 aiohttp 启动)。"
from sqlor.dbpools import DBPools
poll_db = DBPools()
_dispatched = set() # 防止重复分发
async def _poll_once(poll_db):
"""执行一轮 agent poll。可能因连接获取/SQL 锁等待卡住,由外层 wait_for 超时保护。"""
async with poll_db.sqlorContext("pipeline") as sor:
# 回收僵尸 running 任务:进程崩溃/协程挂起遗留(心跳超时未更新)。
# 阈值 20 分钟 > LLM 单轮最坏时长3 次 × 300s 重试 ≈ 15 分钟),避免误杀慢任务。
await sor.sqlExe(
"UPDATE pipeline_tasks SET state='submitted', claimed_by=NULL, updated_at=NOW() "
"WHERE state='running' AND pipeline_id='role_task' "
"AND updated_at < (NOW() - INTERVAL 20 MINUTE)", {})
# 全局并发 agent 数限制max_concurrent_agentsappbase params 表,默认 3
# 以 DB state='running' 计数为准(权威),超出上限时本轮回合不再派发。
from .workspace import get_max_concurrent_agents
max_n = await get_max_concurrent_agents(sor)
running_rows = await sor.sqlExe(
"SELECT COUNT(*) as c FROM pipeline_tasks "
"WHERE state='running' AND pipeline_id='role_task'", {})
cur = getattr(running_rows[0], 'c', 0) if running_rows else 0
avail = max(0, max_n - cur)
if avail <= 0:
# 达到并发上限,本轮不派发(退出 context 后末尾 sleep 再 poll
return
# 每项目/租户并发上限max_agents_per_projectappbase params 表,默认 4
# 改小后 poller 不再给已达上限的项目派新任务,超出的 running 任务自然跑完(不强制杀)——
# 即「改小后超出任务等其完成」。
from .workspace import get_param
per_max = 4
try:
per_max = max(1, int(float(str(await get_param(sor, 'max_agents_per_project', '4')))))
except (ValueError, TypeError):
per_max = 4
# 每项目 running 计数GROUP BY tenant_id供每项目上限门控用
per_running = {}
_per_rows = await sor.sqlExe(
"SELECT tenant_id, COUNT(*) as c FROM pipeline_tasks "
"WHERE state='running' AND pipeline_id='role_task' GROUP BY tenant_id", {})
for _r in (_per_rows or []):
per_running[getattr(_r, 'tenant_id', '')] = getattr(_r, 'c', 0) or 0
await sor.sqlExe("COMMIT", {})
def _deps_of(depends_on_raw):
"""解析 depends_onJSON 数组字符串或 list→ 依赖任务 ID 列表。空/非法 → []"""
if not depends_on_raw:
return []
try:
d = json.loads(depends_on_raw) if isinstance(depends_on_raw, str) else depends_on_raw
return [str(x) for x in d if x] if isinstance(d, list) else []
except (json.JSONDecodeError, TypeError, ValueError):
return []
# 候选任务(含 depends_onLIMIT 放大以容纳「依赖未满足排在前面」的僵尸任务)
# 排除 paused 项目paused = 完全暂停推进(不派发/不审查/不审核),三个 poller 语义统一,
# 否则会出现「任务照常派发执行、却卡在 PM 审核」的半吊子状态2026-08 实测)。
recs = await sor.sqlExe(
"SELECT t.id, t.tenant_id, t.role, t.depends_on, t.params FROM pipeline_tasks t "
"WHERE t.state='submitted' AND t.pipeline_id='role_task' "
"AND t.claimed_by IS NULL "
"AND NOT EXISTS (SELECT 1 FROM sd_projects p WHERE p.id=t.tenant_id AND p.status='paused') "
# 排除已删除项目的悬空任务tenant_id 非空非全局但 sd_projects 已无此行)——
# 否则孤儿任务永远处理不完、每轮再生产 fault_report 灌给 owner.superuser2026-09-15 实测 38 条)。
# 全局任务tenant_id='' 或 '0')不受影响,照常处理。
"AND (t.tenant_id IN ('','0') OR EXISTS (SELECT 1 FROM sd_projects p2 WHERE p2.id=t.tenant_id)) "
"AND NOT EXISTS ("
" SELECT 1 FROM pipeline_human_tasks ht WHERE ht.iteration_id = ("
" SELECT i.id FROM sd_iterations i WHERE i.project_id=t.tenant_id "
" AND i.status='in_progress' ORDER BY i.seq_no DESC LIMIT 1) "
" AND ((ht.task_type='general' AND (ht.status='pending' OR (ht.status='done' AND ht.qc_status != 'passed'))) "
" OR (ht.task_type='bug_acceptance' AND ht.status='pending') "
" OR (ht.task_type IN ('requirement_confirmation','design_confirmation') AND ht.status='pending'))"
") "
"ORDER BY t.created_at ASC LIMIT 200",
{})
# 依赖门控提前到候选筛选:批量查 depends_on 里所有依赖任务的状态,
# 排除依赖未满足的「僵尸 submitted」任务——否则它们占住队列头、饿死其他就绪任务
# (一项目卡死 → 全平台瘫痪)。
# 求值统一走 agent_loop._eval_deps_policy纯函数支持 all/any/at_least 启动策略)——
# 此前这里是手写 all 语义,与 _task_deps_satisfied 并存构成两处实现,
# 语义漂移过2026-08-25 修复注释可查)。现在单一来源,两处必一致。
from .agent_loop import _eval_deps_policy
all_dep_ids = set()
for rec in (recs or []):
all_dep_ids.update(_deps_of(getattr(rec, 'depends_on', '') or ''))
dep_state = {}
if all_dep_ids:
dep_ids = ",".join("'" + x.replace("'", "''") + "'" for x in all_dep_ids)
dep_rows = await sor.sqlExe(
"SELECT id, state FROM pipeline_tasks WHERE id IN (" + dep_ids + ")", {})
dep_state = {getattr(r_, 'id', ''): getattr(r_, 'state', '') for r_ in (dep_rows or [])}
await sor.sqlExe("COMMIT", {})
ready = []
blocked_dead = [] # 依赖永不可达 → 冒泡人工(逃逸阀)
for rec in (recs or []):
_rid = getattr(rec, 'id', '')
deps = _deps_of(getattr(rec, 'depends_on', '') or '')
if not deps:
ready.append(rec)
continue
if _rid in deps:
blocked_dead.append((_rid, getattr(rec, 'tenant_id', ''),
"依赖自身(" + _rid[:8] + "),永久阻塞,需人工修正 depends_on"))
continue
try:
_pp = json.loads(getattr(rec, 'params', '{}') or '{}')
except (json.JSONDecodeError, TypeError):
_pp = {}
_ok, _why, _kind = _eval_deps_policy(deps, dep_state, _pp.get('dep_policy'))
if _ok:
ready.append(rec)
elif _kind == 'dead':
blocked_dead.append((_rid, getattr(rec, 'tenant_id', ''), _why))
continue # 永不放行,交人工
# 逃逸阀:依赖永不可达的任务冒泡人工(去重由 _bubble_blocked_dependency 内部做)
if blocked_dead:
from .agent_loop import _bubble_blocked_dependency
for _bid, _bpid, _bwhy in blocked_dead:
if not (_bid and _bpid):
continue
try:
_bt = await sor.sqlExe(
"SELECT title FROM pipeline_tasks WHERE id=${i}$", {"i": _bid})
await sor.sqlExe("COMMIT", {})
_btitle = getattr(_bt[0], 'title', '') if _bt else ''
await _bubble_blocked_dependency(sor, _bpid, _bid, _btitle, _bwhy)
except Exception as _e:
debug(f"agent_poller 依赖阻塞冒泡失败 task={_bid}: {_e}")
# 项目轮询round-robin + 每项目并发上限):每个项目轮流取一个就绪任务,
# 保证项目间公平 + 单项目不占满全局并发名额、饿死其他项目。dict 保持插入序)
by_project = {}
for rec in ready:
by_project.setdefault(getattr(rec, 'tenant_id', ''), []).append(rec)
selected = []
while len(selected) < avail and by_project:
progressed = False
for pid in list(by_project.keys()):
if len(selected) >= avail:
break
if per_running.get(pid, 0) >= per_max:
continue # 该项目已达每项目上限,跳过(等其 running 任务完成再派)
tasks = by_project[pid]
if tasks:
selected.append(tasks.pop(0))
per_running[pid] = per_running.get(pid, 0) + 1 # 计入本轮即将派发的名额
progressed = True
if not tasks:
del by_project[pid]
if not progressed:
break
for rec in selected:
tid = getattr(rec, 'id', '')
pid = getattr(rec, 'tenant_id', '')
r = getattr(rec, 'role', '')
if not (tid and pid and r) or tid in _dispatched:
continue
_dispatched.add(tid)
debug(f"agent_poller dispatching task={tid} role={r}")
async def _dispatch(tid, pid, r):
try:
from .agent_instance import resolve_agent_instance
_aid = await resolve_agent_instance(pid, r)
await role_agent_loop(pid, r, agent_id=_aid)
except Exception as e:
debug(f"agent_poller dispatch error: task={tid} err={e}")
finally:
_dispatched.discard(tid)
asyncio.ensure_future(_dispatch(tid, pid, r))
async def _poll_loop():
while True:
try:
# watchdog每轮 poll 最多 60 秒,超时则跳过本轮,防止连接池/MDL 锁
# 卡死导致 poller 永久停摆2026-08 生产事故poller 停摆 15.7h
await asyncio.wait_for(_poll_once(poll_db), timeout=60)
except Exception as e:
debug(f"agent_poller error/timeout: {e}")
await asyncio.sleep(10)
asyncio.create_task(_poll_loop())
debug("agent poller started")
from ahserver.configuredServer import add_startup
add_startup(_role_poller) if _run_pollers else None
# PM review poller (v3.3.0): auto-dispatch review-state tasks to PM agent
async def _pm_poller(app):
from sqlor.dbpools import DBPools as _DBP
pm_db = _DBP()
_pm_dispatched = set()
async def _pm_poll_once(pm_db):
"""执行一轮 PM poll。可能因连接获取/SQL 锁等待卡住,由外层 wait_for 超时保护。"""
async with pm_db.sqlorContext("pipeline") as sor:
# 回收僵尸 review 任务PM 审核进程崩溃后 claimed_by 残留,重新放回审核队列。
await sor.sqlExe(
"UPDATE pipeline_tasks SET claimed_by=NULL, updated_at=NOW() "
"WHERE state='review' AND claimed_by IS NOT NULL "
"AND updated_at < (NOW() - INTERVAL 20 MINUTE)", {})
# 候选任务:排除 paused 项目的任务(暂停门控提前到筛选,避免 paused 项目僵尸任务
# 卡死队列头、饿死 active 项目的审核。LIMIT 放大容纳「paused 任务排前面」的僵尸任务。
recs = await sor.sqlExe(
"SELECT t.id, t.tenant_id, t.role FROM pipeline_tasks t "
"WHERE t.state='review' AND t.claimed_by IS NULL "
"AND NOT EXISTS (SELECT 1 FROM sd_projects p WHERE p.id=t.tenant_id AND p.status='paused') "
# 排除已删除项目的悬空任务tenant_id 非空非全局但 sd_projects 已无此行)——
# 否则孤儿任务永远处理不完、每轮再生产 fault_report 灌给 owner.superuser2026-09-15 实测 38 条)。
# 全局任务tenant_id='' 或 '0')不受影响,照常处理。
"AND (t.tenant_id IN ('','0') OR EXISTS (SELECT 1 FROM sd_projects p2 WHERE p2.id=t.tenant_id)) "
"AND NOT EXISTS ("
" SELECT 1 FROM pipeline_human_tasks ht WHERE ht.iteration_id = ("
" SELECT i.id FROM sd_iterations i WHERE i.project_id=t.tenant_id "
" AND i.status='in_progress' ORDER BY i.seq_no DESC LIMIT 1) "
" AND ((ht.task_type='general' AND (ht.status='pending' OR (ht.status='done' AND ht.qc_status != 'passed'))) "
" OR (ht.task_type='bug_acceptance' AND ht.status='pending') "
" OR (ht.task_type IN ('requirement_confirmation','design_confirmation') AND ht.status='pending'))"
") "
"ORDER BY t.created_at ASC LIMIT 200",
{})
await sor.sqlExe("COMMIT", {})
# 项目轮询round-robin每个项目轮流取一个任务保证项目间公平
# 避免单项目占满窗口、饿死其他项目的审核(与 agent_poller 队列饥饿修复同款)。
by_project = {}
for rec in (recs or []):
by_project.setdefault(getattr(rec, 'tenant_id', ''), []).append(rec)
selected = []
while by_project:
progressed = False
for pid in list(by_project.keys()):
tasks = by_project[pid]
if tasks:
selected.append(tasks.pop(0))
progressed = True
if not tasks:
del by_project[pid]
if not progressed:
break
for rec in selected:
tid = getattr(rec, 'id', '')
pid = getattr(rec, 'tenant_id', '')
if not (tid and pid) or tid in _pm_dispatched:
continue
_pm_dispatched.add(tid)
debug(f"pm_poller dispatching review task={tid}")
async def _pm_dispatch(tid, pid):
try:
from .agent_instance import resolve_agent_instance
_aid = await resolve_agent_instance(pid, "agent.pm")
await pm_review_run(pid, agent_id=_aid)
except Exception as e:
debug(f"pm_poller dispatch error: task={tid} err={e}")
finally:
_pm_dispatched.discard(tid)
asyncio.ensure_future(_pm_dispatch(tid, pid))
async def _pm_poll_loop():
while True:
try:
# watchdog每轮 PM poll 最多 60 秒,超时则跳过本轮,防止 poller 永久停摆
await asyncio.wait_for(_pm_poll_once(pm_db), timeout=60)
except Exception as e:
debug(f"pm_poller error/timeout: {e}")
await asyncio.sleep(15)
asyncio.create_task(_pm_poll_loop())
debug("pm poller started")
add_startup(_pm_poller) if _run_pollers else None
# QC review poller: auto-dispatch qc_review-state tasks to QC agent合规/质量门禁)
async def _qc_poller(app):
from sqlor.dbpools import DBPools as _DBP2
qc_db = _DBP2()
_qc_dispatched = set()
async def _qc_poll_once(qc_db):
"""执行一轮 QC poll。可能因连接获取/SQL 锁等待卡住,由外层 wait_for 超时保护。"""
async with qc_db.sqlorContext("pipeline") as sor:
# 回收僵尸 qc_review 任务QC 检查进程崩溃后 claimed_by 残留,重新放回检查队列。
await sor.sqlExe(
"UPDATE pipeline_tasks SET claimed_by=NULL, updated_at=NOW() "
"WHERE state='qc_review' AND claimed_by IS NOT NULL "
"AND updated_at < (NOW() - INTERVAL 20 MINUTE)", {})
recs = await sor.sqlExe(
"SELECT t.id, t.tenant_id, t.role FROM pipeline_tasks t "
"WHERE t.state='qc_review' AND t.claimed_by IS NULL "
"AND NOT EXISTS (SELECT 1 FROM sd_projects p WHERE p.id=t.tenant_id AND p.status='paused') "
# 排除已删除项目的悬空任务(同 submitted/review poller2026-09-15
"AND (t.tenant_id IN ('','0') OR EXISTS (SELECT 1 FROM sd_projects p2 WHERE p2.id=t.tenant_id)) "
"ORDER BY t.created_at ASC LIMIT 5",
{})
for rec in (recs or []):
tid = getattr(rec, 'id', '')
pid = getattr(rec, 'tenant_id', '')
if tid and pid and tid not in _qc_dispatched:
_qc_dispatched.add(tid)
debug(f"qc_poller dispatching qc_review task={tid}")
async def _qc_dispatch(tid, pid):
try:
from .agent_instance import resolve_agent_instance
_aid = await resolve_agent_instance(pid, "agent.qc")
await qc_review_run(pid, agent_id=_aid)
except Exception as e:
debug(f"qc_poller dispatch error: task={tid} err={e}")
finally:
_qc_dispatched.discard(tid)
asyncio.ensure_future(_qc_dispatch(tid, pid))
async def _qc_poll_loop():
while True:
try:
# watchdog每轮 QC poll 最多 60 秒,超时则跳过本轮,防止 poller 永久停摆
await asyncio.wait_for(_qc_poll_once(qc_db), timeout=60)
except Exception as e:
debug(f"qc_poller error/timeout: {e}")
await asyncio.sleep(15)
asyncio.create_task(_qc_poll_loop())
debug("qc poller started")
add_startup(_qc_poller) if _run_pollers else None
# Failed task poller: 失败任务自动重跑(最多3次),超限报故障给用户
async def _failed_poller(app):
from sqlor.dbpools import DBPools as _DBP3
fd_db = _DBP3()
_fd_dispatched = set()
async def _fd_poll_loop():
while True:
try:
async with fd_db.sqlorContext("pipeline") as sor:
recs = await sor.sqlExe(
"SELECT t.id, t.tenant_id FROM pipeline_tasks t "
"WHERE t.state='failed' AND t.pipeline_id='role_task' "
# 排除已删除项目的悬空任务——孤儿 failed 任务反复触发 fault_report
# 每轮灌给 owner.superuseradmin是"待办越来越多"的直接源头2026-09-15 实测)。
# 全局任务tenant_id='' 或 '0')照常处理。
"AND (t.tenant_id IN ('','0') OR EXISTS (SELECT 1 FROM sd_projects p2 WHERE p2.id=t.tenant_id)) "
"ORDER BY t.created_at ASC LIMIT 10",
{})
for rec in (recs or []):
tid = getattr(rec, 'id', '')
pid = getattr(rec, 'tenant_id', '')
if tid and pid and tid not in _fd_dispatched:
_fd_dispatched.add(tid)
debug(f"failed_poller handling task={tid}")
async def _fd_dispatch(tid, pid):
try:
await handle_failed_task(tid, pid)
except Exception as e:
debug(f"failed_poller dispatch error: task={tid} err={e}")
finally:
_fd_dispatched.discard(tid)
asyncio.ensure_future(_fd_dispatch(tid, pid))
# SELECT-only 事务必须显式 COMMIT否则连接回到池后仍持有元数据锁(MDL)
# 会阻塞后续 ALTER TABLE / DDLsqlor 只在有写入时才自动提交)。
await sor.sqlExe("COMMIT", {})
except Exception as e:
debug(f"failed_poller error: {e}")
await asyncio.sleep(60)
asyncio.create_task(_fd_poll_loop())
debug("failed poller started")
add_startup(_failed_poller) if _run_pollers else None
# Bug 生命周期 poller: 独立的 bug 循环(与任务循环平行的第二套循环)。
# 扫 sd_bugs 表驱动 bug 状态机open→confirmed→fixing→fixed→verified→closed
# 不寄生在任务循环的任何环节PM 审核/角色 agent 执行都不再顺手处理 bug
async def _bug_poller(app):
from sqlor.dbpools import DBPools as _DBP4
from .bug_flow import bug_flow_poll_once
bug_db = _DBP4()
async def _bug_poll_loop():
while True:
try:
# watchdog每轮 bug poll 最多 120 秒(含 LLM 确认 open bug超时跳过本轮
await asyncio.wait_for(bug_flow_poll_once(bug_db), timeout=120)
except Exception as e:
debug(f"bug_poller error/timeout: {e}")
await asyncio.sleep(20)
asyncio.create_task(_bug_poll_loop())
debug("bug poller started")
add_startup(_bug_poller) if _run_pollers else None
# 铁律:运行期不做任何 schema 变更。v2 引擎的建表/加列已全部迁到
# 部署期 scripts/create_tables.pybuild.sh 会调用),此处不再启动时建表。
debug(f"[{MODULE_NAME}] v{MODULE_VERSION} loaded — pipeline engine with role-agent + question loop")
return True