From 5f98cd79a7ff7032ab0170ccb9370a4cd42c3396 Mon Sep 17 00:00:00 2001 From: ymq Date: Wed, 16 Sep 2026 17:58:14 +0800 Subject: [PATCH] =?UTF-8?q?feat(timeout):=20LLM=E8=B6=85=E6=97=B6=E9=A2=84?= =?UTF-8?q?=E7=AE=97=E7=BB=9F=E4=B8=80=E6=94=B6=E6=95=9B=E5=88=B0bridge.?= =?UTF-8?q?=5Fresolve=5Fbudget=E5=8D=95=E7=82=B9(2026-09-16=E7=94=A8?= =?UTF-8?q?=E6=88=B7=E8=A3=81=E5=AE=9A:=E7=A6=81=E6=95=A3=E8=90=BD?= =?UTF-8?q?=E5=90=84=E4=BA=A7=E7=BA=BF=E5=90=84=E8=A7=92=E8=89=B2)?= =?UTF-8?q?=E2=80=94=E2=80=94timeout=E5=8F=82=E6=95=B0=E8=AF=AD=E4=B9=89?= =?UTF-8?q?=3D=E4=B8=80=E6=AC=A1=E8=B0=83=E7=94=A8=E6=80=BB=E9=A2=84?= =?UTF-8?q?=E7=AE=97(0=3D=E5=B9=B3=E5=8F=B0=E7=BC=BA=E7=9C=81chat300/gen90?= =?UTF-8?q?0,=E5=B0=81=E9=A1=B6900=E4=B8=8B=E9=99=9060),payload.=5Ftimeout?= =?UTF-8?q?=E6=81=92=E4=BC=A0+=E5=AE=A2=E6=88=B7=E7=AB=AF=3D=E9=A2=84?= =?UTF-8?q?=E7=AE=97+60=E6=81=92=E8=A6=86=E7=9B=96=E4=B8=8A=E6=B8=B8;?= =?UTF-8?q?=E5=9B=9B=E4=B8=AA=E5=85=AC=E5=BC=80=E5=85=A5=E5=8F=A3(llm=5Fca?= =?UTF-8?q?ll/llm=5Fcall=5Fmsgs/llm=5Fcall=5Fmsgs=5Fnative/llm=5Finfer)?= =?UTF-8?q?=E5=BA=9F=E5=BC=83=E5=90=84=E8=87=AA=E4=B8=BA=E6=94=BF=E7=9A=84?= =?UTF-8?q?=E7=AD=89=E5=BE=85=E5=85=AC=E5=BC=8F(330=E7=BC=BA=E7=9C=81/+60/?= =?UTF-8?q?=E5=8E=9F=E5=80=BC-60/min+60)=E5=85=A8=E8=B5=B0=E7=BB=9F?= =?UTF-8?q?=E4=B8=80=E8=A7=A3=E6=9E=90;=E8=B0=83=E7=94=A8=E6=96=B9?= =?UTF-8?q?=E5=BF=98=E4=BC=A0timeout=E4=B9=9F=E7=BB=93=E6=9E=84=E6=80=A7?= =?UTF-8?q?=E5=AE=89=E5=85=A8=E2=80=94=E2=80=94=E6=B6=88=E7=81=AD09-05?= =?UTF-8?q?=E6=8F=90=E5=8F=96/09-14develop/09-16QC=E4=B8=89=E6=AC=A1'?= =?UTF-8?q?=E5=BF=98=E4=BC=A0=E2=86=92=E7=AB=AF=E7=82=B9=E6=8C=89=E4=BE=9B?= =?UTF-8?q?=E5=BA=94=E5=95=86120s=E6=8E=90=E6=96=AD=E9=95=BF=E7=94=9F?= =?UTF-8?q?=E6=88=90'=E5=90=8C=E6=AC=BE=E4=BA=8B=E6=95=85;=E9=A1=BA?= =?UTF-8?q?=E5=B8=A6=E4=BF=AEllm=5Finfer=E6=BD=9C=E4=BC=8Fbug(=E4=B8=8D?= =?UTF-8?q?=E4=BC=A0timeout=E5=AE=A2=E6=88=B7=E7=AB=AF=E7=BC=BA=E7=9C=8133?= =?UTF-8?q?0<=E5=BC=82=E6=AD=A5=E8=BD=AE=E8=AF=A2=E9=A2=84=E7=AE=97600?= =?UTF-8?q?=E5=AE=A2=E6=88=B7=E7=AB=AF=E5=85=88=E6=96=AD);agent=5Floop.=5F?= =?UTF-8?q?LLM=5FCLIENT=5FTIMEOUT=E8=AF=AD=E4=B9=89=E5=90=8C=E6=AD=A5?= =?UTF-8?q?=E4=B8=BA=E6=80=BB=E9=A2=84=E7=AE=97510(510=E2=86=92=E5=AE=A2?= =?UTF-8?q?=E6=88=B7=E7=AB=AF570=E2=86=92wait=5Ffor600=E4=B8=89=E5=B1=82?= =?UTF-8?q?=E5=AF=B9=E9=BD=90),v2=E5=BC=95=E6=93=8E=5FMAIN=5FTIMEOUT=3D180?= =?UTF-8?q?=E5=9C=A8=E6=96=B0=E8=AF=AD=E4=B9=89=E4=B8=8B=E6=80=BB=E6=97=B6?= =?UTF-8?q?=E9=95=BF=E5=8F=97=E9=A2=84=E7=AE=97=E7=BA=A6=E6=9D=9F=E9=A1=BA?= =?UTF-8?q?=E5=B8=A6=E4=BF=AE=E5=A4=8D=E6=97=A73x=E6=94=BE=E5=A4=A7?= =?UTF-8?q?=E8=B6=85=E5=AE=A2=E6=88=B7=E7=AB=AF=E7=9A=84=E6=BD=9C=E4=BC=8F?= =?UTF-8?q?bug?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- pipeline_service/agent_loop.py | 15 +++--- pipeline_service/llm_bridge.py | 94 ++++++++++++++++++++++++---------- 2 files changed, 74 insertions(+), 35 deletions(-) diff --git a/pipeline_service/agent_loop.py b/pipeline_service/agent_loop.py index c55f4e5..1786e58 100644 --- a/pipeline_service/agent_loop.py +++ b/pipeline_service/agent_loop.py @@ -2309,16 +2309,13 @@ async def _exec_agent_tool(tool, params, workspace_dir, ctx=None): # 30 轮全耗在 list_files/read_file/run_shell 反复「了解现状 + 验证已有内容」,从不 write_file/deliver。 _FORCE_PRODUCE_TURN = 5 -# LLM 调用硬超时(外层 asyncio.wait_for 兜底)。llm_bridge 内部 300s×3 重试最坏 15 分钟, -# 期间 role_agent_run 心跳不更新(心跳在调用前 touch),观察上像任务挂起/僵尸。 -# 外层硬上限:① 防 aiohttp total 超时因 DNS 等场景失效导致的真正无限挂起 ② 给 llm_bridge 至少 2 轮完整重试机会 -# (420s 只够 ~1.4 轮,网关瞬时挂起时重试被提前掐断,可自愈的瞬时故障也 TimeoutError 超限暂停任务链)。 +# LLM 调用超时预算(2026-09-16 统一收敛到 bridge._resolve_budget 单点后, +# 本常量语义 = 一次调用的【总预算】,bridge 自动派生客户端等待 = 预算+60): +# 预算 510 → 客户端 570 → 外层硬超时 600(恒覆盖,防真正无限挂起)。 +# 历史:2026-09-14 前此处是「客户端超时」且只挂 develop native 路径,QC/PM/ +# 复盘漏传 → 端点按供应商配置掐断(三次事故);现忘传也走平台缺省,不再掉洞。 _LLM_HARD_TIMEOUT = 600 - -# 单次 LLM 调用客户端超时(aiohttp total),略小于外层硬超时,让长生成用满外层 -# 预算而不被缺省 330s 提前掐断(2026-09-14 pbls:需求分析生成 ~368s > 330s 缺省 -# → 客户端 TimeoutError,旧代码误报「端点不可达」触发任务链暂停)。 -_LLM_CLIENT_TIMEOUT = 570 +_LLM_CLIENT_TIMEOUT = 510 # 总预算(inference deadline 约束含内部重试) _FORCE_PRODUCE_HINT = ( "⚠️ 你已经探索了足够多轮(已超过 5 轮)。现在必须立即产出并交付:\n" diff --git a/pipeline_service/llm_bridge.py b/pipeline_service/llm_bridge.py index 527e0e1..bde9364 100644 --- a/pipeline_service/llm_bridge.py +++ b/pipeline_service/llm_bridge.py @@ -18,6 +18,43 @@ import time logger = logging.getLogger("pipeline.llm_bridge") +# ── 平台统一超时预算(单点治理,2026-09-16 用户裁定收敛)───────────────── +# 背景:超时逻辑曾散落三层——①各调用方自传 timeout(忘传 → payload 无 +# _timeout → inference 按供应商端点配置(百炼 120s)掐断长生成:2026-09-05 +# 提取、2026-09-14 develop、2026-09-16 QC 三次事故同一根因);②bridge 四个 +# 公开函数客户端等待公式各不相同(330 缺省 / +60 / 原值 / min+60);③ +# inference 3 attempt × 单attempt超时最坏 3× 总预算,超过客户端等待 → 客户端 +# 先断连报模糊 TimeoutError(分类器再误判永久错误)。 +# +# 统一语义(只有这一个函数 + inference._post_upstream 的 deadline 两处): +# timeout 参数 = 一次调用的【总预算】秒数(含 inference 内部重试)。 +# 不传 → 平台缺省(chat 类 300 / 生成类 900),结构性消除「忘传掉洞」。 +# 三层预算对齐公式: +# payload._timeout = total (inference 端 deadline 约束,最坏≈total) +# client aiohttp = total + 60 (恒覆盖上游:断连只可能是端点侧 +# 结构化错误,永不模糊 TimeoutError) +# 调用方外层 wait_for(若有)须 ≥ client(agent_loop 600 > 510+60 ✓) +_DEFAULT_CHAT_BUDGET = 300 # 对话类缺省总预算(对齐 inference _TOTAL_TIMEOUT) +_DEFAULT_GEN_BUDGET = 900 # 生成类(llm_infer:t2i/t2v/tts 等慢任务)缺省 +_BUDGET_CAP = 900 # inference 端 _timeout 封顶(一致) +_BUDGET_FLOOR = 60 +_CLIENT_MARGIN = 60 + + +def _resolve_budget(timeout, default): + """统一预算解析 → (payload._timeout 总预算, 客户端 aiohttp total)。 + + 全平台唯一公式:任何产线/角色/模块经本 bridge 调用都走这里, + 调用方不需要(也不应该)自己算客户端等待。 + """ + try: + t = int(timeout or 0) + except (TypeError, ValueError): + t = 0 + total = t if t > 0 else default + total = max(_BUDGET_FLOOR, min(total, _BUDGET_CAP)) + return total, total + _CLIENT_MARGIN + # 内部自调用 token 缓存:(org_id, user_id, model_name) -> {token, calls, expires_at} # 每次 LLM 调用都签发新 token 会让 tokens 表膨胀,故按上下文缓存复用; # 接近调用上限或临近过期时换新(上限/过期由签发侧强制,本地计数只是提前量)。 @@ -74,8 +111,8 @@ async def _http_chat(payload, org_id, user_id, model_name, timeout: int = 0, project_id: str = ''): """POST 本进程推理端点。返回上游响应 dict;失败抛 ValueError(消息真实可行动)。 - timeout:客户端等待秒数(0=缺省 330)。异步生成模型(视频等)端点侧最长 - 等 900 秒,调用方须同步放大客户端超时,否则客户端先断连。 + timeout:客户端等待秒数(0=缺省 chat 预算+60 余量)。公开入口(llm_call*) + 一律经 _resolve_budget 算好后传入,调用方无需自己算(单点治理)。 """ import aiohttp @@ -85,7 +122,8 @@ async def _http_chat(payload, org_id, user_id, model_name, timeout: int = 0, # (2026-09-04 实测:Authorization 头变成 "***plk-..." 致端点校验失败) _BEARER = 'Bea' + 'rer ' headers = {"Authorization": _BEARER + token, "Content-Type": "application/json"} - _total = int(timeout) if timeout and int(timeout) > 0 else 330 + _total = int(timeout) if timeout and int(timeout) > 0 \ + else _DEFAULT_CHAT_BUDGET + _CLIENT_MARGIN try: async with aiohttp.ClientSession() as session: async with session.post( @@ -139,6 +177,9 @@ async def llm_call(prompt: str, model: str = None, temperature: float = 0.7, org_id 为空 = 系统级('0'),与旧语义(不过滤机构)等价。 purpose='utility':辅助任务(分类/选择/摘要),治理层按用途选模型链 (机构策略配置辅助模型优先),经 payload 的 _purpose 键透传到端点。 + timeout:一次调用的【总预算】秒数(0=平台缺省 chat 300;上限 900)。 + 预算分配/客户端等待由 _resolve_budget 单点计算(2026-09-16 统一收敛), + 调用方忘传也结构性安全,不再按供应商端点小配置被掐断。 session_id(2026-09-11 用户定夺):会话粘性键——同会话固定同一账号, 上游 KV/前缀缓存按账号隔离,固定账号命中缓存省钱;经 payload._session_id 透传到端点(不进 token 缓存键——token 与账号选择解耦)。 @@ -163,12 +204,15 @@ async def llm_call(prompt: str, model: str = None, temperature: float = 0.7, } if purpose: payload["_purpose"] = purpose - if timeout: - payload["_timeout"] = int(timeout) if session_id: payload["_session_id"] = session_id + # 统一预算(单点):payload._timeout 恒传 → inference 端按总预算 deadline + # 约束重试;客户端 = 预算+60 恒覆盖上游。不传 timeout 走平台缺省,结构性 + # 消除「忘传 → 端点按供应商小配置掐断」(2026-09-05/09-14/09-16 三事故根因)。 + _budget, _client = _resolve_budget(timeout, _DEFAULT_CHAT_BUDGET) + payload["_timeout"] = _budget data = await _http_chat(payload, org_id or '0', user_id or '', model or '', - project_id=project_id or '') + timeout=_client, project_id=project_id or '') try: return data["choices"][0]["message"]["content"] except (KeyError, IndexError, TypeError) as e: @@ -187,20 +231,20 @@ async def llm_call_msgs(messages: list, model: str = None, temperature: float = """Call LLM with full message array (system/user/assistant). purpose='utility':辅助任务(分类/选择/摘要),治理层按用途选模型链。 - timeout:单次上游调用超时秒数(0=用端点默认;上限 900,超长文本提取用)。 + timeout:一次调用的【总预算】秒数(0=平台缺省 chat 300;上限 900)。 + 预算分配/客户端等待由 _resolve_budget 单点计算,调用方只管声明预算。 session_id(2026-09-11):会话粘性账号键,经 payload._session_id 透传。 """ payload = {"model": model or '', "messages": messages, "temperature": temperature} if purpose: payload["_purpose"] = purpose - if timeout: - payload["_timeout"] = int(timeout) if session_id: payload["_session_id"] = session_id - # 客户端等待须覆盖端点侧预算 + 余量(同 llm_infer 模式),否则客户端先断连、 - # 报模糊 TimeoutError 而非端点的结构化错误 + # 统一预算(单点,同 llm_call) + _budget, _client = _resolve_budget(timeout, _DEFAULT_CHAT_BUDGET) + payload["_timeout"] = _budget data = await _http_chat(payload, org_id or '0', user_id or '', model or '', - timeout=(int(timeout) + 60 if timeout else 0), + timeout=_client, project_id=project_id or '') try: return data["choices"][0]["message"]["content"] @@ -221,6 +265,9 @@ async def llm_call_msgs_native(messages: list, tools: list = None, model: str = 当模型返回 tool_calls 时,content 通常为空字符串。 purpose='utility':辅助任务(分类/选择/摘要),治理层按用途选模型链。 session_id(2026-09-11):会话粘性账号键,经 payload._session_id 透传。 + timeout:一次调用的【总预算】秒数(0=平台缺省 chat 300);预算分配/客户端 + 等待由 _resolve_budget 单点计算(2026-09-15 pbls design 三连超时、 + 2026-09-16 QC 事故后统一收敛,调用方不再自己算 -60/+60)。 """ payload = {"model": model or '', "messages": messages, "temperature": temperature} if tools: @@ -228,18 +275,12 @@ async def llm_call_msgs_native(messages: list, tools: list = None, model: str = payload["tool_choice"] = "auto" if purpose: payload["_purpose"] = purpose - if timeout: - # 端点侧上游预算必须随调用方透传(2026-09-15 pbls design 三连超时根因): - # 原来 native 只把 timeout 用作客户端等待,payload 不带 _timeout → 端点按 - # 供应商端点配置(百炼 timeout=120)掐断上游,而 design 收尾大请求天然 - # 72~115s,一半概率 120s 被杀 → 内部 3 重试全灭 → 任务 failed。 - # 端点预算 = 客户端等待 - 60s 余量:让端点先掐并返回结构化错误(可分类 - # 重试),而不是客户端 aiohttp 断连报模糊 TimeoutError。 - payload["_timeout"] = max(int(timeout) - 60, 60) if session_id: payload["_session_id"] = session_id + _budget, _client = _resolve_budget(timeout, _DEFAULT_CHAT_BUDGET) + payload["_timeout"] = _budget data = await _http_chat(payload, org_id or '0', user_id or '', model or '', - timeout=timeout, project_id=project_id or '') + timeout=_client, project_id=project_id or '') try: msg = data["choices"][0]["message"] except (KeyError, IndexError, TypeError) as e: @@ -269,11 +310,12 @@ async def llm_infer(payload: dict, model: str = None, org_id: str = None, body = dict(payload or {}) if model: body["model"] = model - if timeout: - body["_timeout"] = int(timeout) if session_id and not body.get('_session_id'): body["_session_id"] = session_id - # 客户端等待须覆盖端点侧预算(_timeout 端点封顶 900)+ 余量,否则客户端先断连 - client_timeout = min(int(timeout or 0), 900) + 60 if timeout else 0 + # 统一预算(单点,同 llm_call*):timeout=总预算(0=生成类缺省 900), + # payload._timeout 恒传 + 客户端=预算+60。旧实现不传 timeout 时客户端缺省 + # 330 而端点异步轮询预算 600 → 客户端先断连(潜伏 bug,一并根治)。 + _budget, _client = _resolve_budget(timeout, _DEFAULT_GEN_BUDGET) + body["_timeout"] = _budget return await _http_chat(body, org_id or '0', user_id or '', model or '', - timeout=client_timeout, project_id=project_id or '') + timeout=_client, project_id=project_id or '')