fix: poller 加 watchdog(wait_for 60s 超时),防连接池/MDL 锁卡死导致停摆

agent_poller 和 pm_poller 的 poll_loop 改为 _poll_once + asyncio.wait_for(timeout=60)
包裹,每轮 poll 卡住超过 60s 就跳过本轮继续,不再因连接池耗尽/MDL 锁等待
(lock_wait_timeout 默认 24h)而永久停摆(2026-08 生产事故:poller 停摆 15.7h)。
This commit is contained in:
ymq 2026-08-18 11:14:23 +08:00
parent 0b75f10731
commit b819a665f6

View File

@ -581,9 +581,8 @@ def load_pipeline_service():
poll_db = DBPools() poll_db = DBPools()
_dispatched = set() # 防止重复分发 _dispatched = set() # 防止重复分发
async def _poll_loop(): async def _poll_once(poll_db):
while True: """执行一轮 agent poll。可能因连接获取/SQL 锁等待卡住,由外层 wait_for 超时保护。"""
try:
async with poll_db.sqlorContext("pipeline") as sor: async with poll_db.sqlorContext("pipeline") as sor:
# 回收僵尸 running 任务:进程崩溃/协程挂起遗留(心跳超时未更新)。 # 回收僵尸 running 任务:进程崩溃/协程挂起遗留(心跳超时未更新)。
# 阈值 20 分钟 > LLM 单轮最坏时长(3 次 × 300s 重试 ≈ 15 分钟),避免误杀慢任务。 # 阈值 20 分钟 > LLM 单轮最坏时长(3 次 × 300s 重试 ≈ 15 分钟),避免误杀慢任务。
@ -603,7 +602,7 @@ def load_pipeline_service():
avail = max(0, max_n - cur) avail = max(0, max_n - cur)
if avail <= 0: if avail <= 0:
# 达到并发上限,本轮不派发(退出 context 后末尾 sleep 再 poll) # 达到并发上限,本轮不派发(退出 context 后末尾 sleep 再 poll)
continue return
recs = await sor.sqlExe( recs = await sor.sqlExe(
"SELECT id, tenant_id, role FROM pipeline_tasks " "SELECT id, tenant_id, role FROM pipeline_tasks "
@ -627,8 +626,15 @@ def load_pipeline_service():
_dispatched.discard(tid) _dispatched.discard(tid)
asyncio.ensure_future(_dispatch(tid, pid, r)) asyncio.ensure_future(_dispatch(tid, pid, r))
async def _poll_loop():
while True:
try:
# watchdog:每轮 poll 最多 60 秒,超时则跳过本轮,防止连接池/MDL 锁
# 卡死导致 poller 永久停摆(2026-08 生产事故:poller 停摆 15.7h)
await asyncio.wait_for(_poll_once(poll_db), timeout=60)
except Exception as e: except Exception as e:
debug(f"agent_poller error: {e}") debug(f"agent_poller error/timeout: {e}")
await asyncio.sleep(10) await asyncio.sleep(10)
asyncio.create_task(_poll_loop()) asyncio.create_task(_poll_loop())
@ -643,9 +649,8 @@ def load_pipeline_service():
pm_db = _DBP() pm_db = _DBP()
_pm_dispatched = set() _pm_dispatched = set()
async def _pm_poll_loop(): async def _pm_poll_once(pm_db):
while True: """执行一轮 PM poll。可能因连接获取/SQL 锁等待卡住,由外层 wait_for 超时保护。"""
try:
async with pm_db.sqlorContext("pipeline") as sor: async with pm_db.sqlorContext("pipeline") as sor:
# 回收僵尸 review 任务:PM 审核进程崩溃后 claimed_by 残留,重新放回审核队列。 # 回收僵尸 review 任务:PM 审核进程崩溃后 claimed_by 残留,重新放回审核队列。
await sor.sqlExe( await sor.sqlExe(
@ -673,8 +678,14 @@ def load_pipeline_service():
_pm_dispatched.discard(tid) _pm_dispatched.discard(tid)
asyncio.ensure_future(_pm_dispatch(tid, pid)) asyncio.ensure_future(_pm_dispatch(tid, pid))
async def _pm_poll_loop():
while True:
try:
# watchdog:每轮 PM poll 最多 60 秒,超时则跳过本轮,防止 poller 永久停摆
await asyncio.wait_for(_pm_poll_once(pm_db), timeout=60)
except Exception as e: except Exception as e:
debug(f"pm_poller error: {e}") debug(f"pm_poller error/timeout: {e}")
await asyncio.sleep(15) await asyncio.sleep(15)
asyncio.create_task(_pm_poll_loop()) asyncio.create_task(_pm_poll_loop())