test: load_test.py upgraded — 4min×3(50/100/200) + HTTP status + host CPU/MEM monitoring
This commit is contained in:
parent
9c363bc617
commit
06e14af6ab
210
load_test.py
210
load_test.py
@ -1,137 +1,187 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""并发压力测试 llmage /v1/chat/completions,统计 TTFB / 完成时间 / QPM"""
|
"""并发压力测试 llmage /v1/chat/completions — 4分钟×3组(50/100/200),TTFB/QPM/500/主机资源"""
|
||||||
|
|
||||||
import asyncio
|
import asyncio, aiohttp, time, json, sys, statistics, subprocess
|
||||||
import aiohttp
|
|
||||||
import time
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
import statistics
|
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from typing import List
|
from typing import List
|
||||||
|
|
||||||
URL = "https://token.opencomputing.cn/llmage/v1/chat/completions"
|
URL = "https://token.opencomputing.cn/llmage/v1/chat/completions"
|
||||||
TOKEN = "V9J41PngWBUU6gdHWJWDJ"
|
TOKEN = "V9J41PngWBUU6gdHWJWDJ"
|
||||||
MODEL = "qwen3.6-35b-a3b"
|
MODEL = "qwen3.6-35b-a3b"
|
||||||
DURATION = 180 # 3 分钟
|
DURATION = 240 # 4 minutes
|
||||||
CONCURRENCIES = [10, 50, 100, 200]
|
CONCURRENCIES = [50, 100, 200]
|
||||||
|
|
||||||
|
# Host monitoring via SSH
|
||||||
|
HOST_SSH = "token@token.opencomputing.cn"
|
||||||
|
HOST_CPU_CMD = "top -bn1 | grep 'Cpu(s)' | awk '{print $2+$4}'"
|
||||||
|
HOST_MEM_CMD = "free -m | awk '/Mem:/{printf \"%.1f\", $3/$2*100}'"
|
||||||
|
HOST_LOAD_CMD = "uptime | awk -F'[a-z]:' '{print $2}' | awk '{print $1,$2,$3}'"
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class ReqStat:
|
class ReqStat:
|
||||||
prompt_idx: int
|
idx: int
|
||||||
start_ts: float
|
start_ts: float
|
||||||
first_byte_ts: float | None = None
|
first_byte_ts: float | None = None
|
||||||
end_ts: float | None = None
|
end_ts: float | None = None
|
||||||
|
http_status: int = 0
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class HostSnap:
|
||||||
|
ts: float
|
||||||
|
cpu: float
|
||||||
|
mem: float
|
||||||
|
load: str
|
||||||
|
|
||||||
|
|
||||||
|
async def host_snapshot() -> HostSnap:
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
try:
|
||||||
|
cpu = float((await loop.run_in_executor(
|
||||||
|
None, lambda: subprocess.run(
|
||||||
|
["ssh", "-o", "ConnectTimeout=3", HOST_SSH, HOST_CPU_CMD],
|
||||||
|
capture_output=True, text=True, timeout=5
|
||||||
|
).stdout.strip()
|
||||||
|
)) or 0)
|
||||||
|
except:
|
||||||
|
cpu = 0
|
||||||
|
try:
|
||||||
|
mem = float((await loop.run_in_executor(
|
||||||
|
None, lambda: subprocess.run(
|
||||||
|
["ssh", "-o", "ConnectTimeout=3", HOST_SSH, HOST_MEM_CMD],
|
||||||
|
capture_output=True, text=True, timeout=5
|
||||||
|
).stdout.strip()
|
||||||
|
)) or 0)
|
||||||
|
except:
|
||||||
|
mem = 0
|
||||||
|
try:
|
||||||
|
load = (await loop.run_in_executor(
|
||||||
|
None, lambda: subprocess.run(
|
||||||
|
["ssh", "-o", "ConnectTimeout=3", HOST_SSH, HOST_LOAD_CMD],
|
||||||
|
capture_output=True, text=True, timeout=5
|
||||||
|
).stdout.strip()
|
||||||
|
)) or "N/A"
|
||||||
|
except:
|
||||||
|
load = "N/A"
|
||||||
|
return HostSnap(ts=time.monotonic(), cpu=cpu, mem=mem, load=load)
|
||||||
|
|
||||||
|
|
||||||
async def worker(session: aiohttp.ClientSession, idx: int, stats_out: list):
|
async def worker(session: aiohttp.ClientSession, idx: int, stats_out: list):
|
||||||
"""单个请求:发送 stream 请求,记录首字时间和完成时间"""
|
|
||||||
prompt = f"请用一句话介绍你自己,编号{idx}"
|
|
||||||
payload = {
|
payload = {
|
||||||
"model": MODEL,
|
"model": MODEL,
|
||||||
"stream": True,
|
"stream": True,
|
||||||
"messages": [{"role": "user", "content": prompt}],
|
"messages": [{"role": "user", "content": f"你是谁? 请用一句话回答,编号{idx}"}],
|
||||||
}
|
}
|
||||||
stat = ReqStat(prompt_idx=idx, start_ts=time.monotonic())
|
stat = ReqStat(idx=idx, start_ts=time.monotonic())
|
||||||
try:
|
try:
|
||||||
async with session.post(
|
async with session.post(
|
||||||
URL,
|
URL, json=payload,
|
||||||
json=payload,
|
headers={"Content-Type": "application/json", "Authorization": f"Bearer {TOKEN}"},
|
||||||
headers={
|
|
||||||
"Content-Type": "application/json",
|
|
||||||
"Authorization": f"Bearer {TOKEN}",
|
|
||||||
},
|
|
||||||
timeout=aiohttp.ClientTimeout(total=120),
|
timeout=aiohttp.ClientTimeout(total=120),
|
||||||
) as resp:
|
) as resp:
|
||||||
|
stat.http_status = resp.status
|
||||||
first = True
|
first = True
|
||||||
async for line in resp.content:
|
async for line in resp.content:
|
||||||
if first:
|
if first:
|
||||||
stat.first_byte_ts = time.monotonic()
|
stat.first_byte_ts = time.monotonic()
|
||||||
first = False
|
first = False
|
||||||
# 读完所有 chunk 才算完成
|
|
||||||
stat.end_ts = time.monotonic()
|
stat.end_ts = time.monotonic()
|
||||||
except Exception as e:
|
except Exception:
|
||||||
# 异常请求也记录(TTFB=None 表示失败)
|
|
||||||
stat.end_ts = time.monotonic()
|
stat.end_ts = time.monotonic()
|
||||||
stats_out.append(stat)
|
stats_out.append(stat)
|
||||||
|
|
||||||
|
|
||||||
async def run_concurrency(concurrency: int):
|
async def run_concurrency(concurrency: int):
|
||||||
"""以固定并发运行 DURATION 秒,持续发起新请求"""
|
|
||||||
stats: List[ReqStat] = []
|
stats: List[ReqStat] = []
|
||||||
|
host_snaps: List[HostSnap] = []
|
||||||
idx = 0
|
idx = 0
|
||||||
stop_at = time.monotonic() + DURATION
|
stop_at = time.monotonic() + DURATION
|
||||||
|
|
||||||
connector = aiohttp.TCPConnector(limit=concurrency + 20, force_close=True)
|
connector = aiohttp.TCPConnector(limit=concurrency + 50, force_close=True)
|
||||||
async with aiohttp.ClientSession(connector=connector) as session:
|
async with aiohttp.ClientSession(connector=connector) as session:
|
||||||
tasks: list[asyncio.Task] = []
|
tasks: list[asyncio.Task] = []
|
||||||
|
last_snap = 0
|
||||||
|
|
||||||
while time.monotonic() < stop_at:
|
while time.monotonic() < stop_at:
|
||||||
# 保持并发数:补满到 concurrency
|
# Fill to concurrency
|
||||||
while len(tasks) < concurrency and time.monotonic() < stop_at:
|
while len(tasks) < concurrency and time.monotonic() < stop_at:
|
||||||
idx += 1
|
idx += 1
|
||||||
tasks.append(
|
tasks.append(asyncio.create_task(worker(session, idx, stats)))
|
||||||
asyncio.create_task(worker(session, idx, stats))
|
|
||||||
)
|
|
||||||
|
|
||||||
if not tasks:
|
if not tasks:
|
||||||
break
|
break
|
||||||
|
|
||||||
# 等待任意一个完成,腾出槽位
|
# Host snapshot every 15s
|
||||||
done, tasks = await asyncio.wait(
|
now = time.monotonic()
|
||||||
tasks, return_when=asyncio.FIRST_COMPLETED, timeout=0.5
|
if now - last_snap > 15:
|
||||||
)
|
host_snaps.append(await host_snapshot())
|
||||||
# 清理已完成的
|
last_snap = now
|
||||||
|
sys.stdout.write(f"\r [{concurrency}] {len(stats)} req | CPU:{host_snaps[-1].cpu:.0f}% MEM:{host_snaps[-1].mem:.0f}% LOAD:{host_snaps[-1].load}")
|
||||||
|
sys.stdout.flush()
|
||||||
|
|
||||||
|
done, tasks = await asyncio.wait(tasks, return_when=asyncio.FIRST_COMPLETED, timeout=0.5)
|
||||||
tasks = list(tasks)
|
tasks = list(tasks)
|
||||||
|
|
||||||
# 时间到,等待所有进行中的请求完成
|
# Drain remaining
|
||||||
if tasks:
|
if tasks:
|
||||||
await asyncio.wait(tasks)
|
await asyncio.wait(tasks)
|
||||||
|
|
||||||
return stats
|
# Final snapshot
|
||||||
|
host_snaps.append(await host_snapshot())
|
||||||
|
|
||||||
|
return stats, host_snaps
|
||||||
|
|
||||||
|
|
||||||
def analyze(name: str, stats: List[ReqStat]):
|
def analyze(concurrency: int, stats: List[ReqStat], host_snaps: List[HostSnap]):
|
||||||
"""分析并打印统计"""
|
|
||||||
ttfb_list = [s.first_byte_ts - s.start_ts for s in stats if s.first_byte_ts]
|
ttfb_list = [s.first_byte_ts - s.start_ts for s in stats if s.first_byte_ts]
|
||||||
total_list = [s.end_ts - s.start_ts for s in stats if s.end_ts and s.first_byte_ts]
|
total_list = [s.end_ts - s.start_ts for s in stats if s.end_ts and s.first_byte_ts]
|
||||||
failed = sum(1 for s in stats if s.first_byte_ts is None)
|
failed = sum(1 for s in stats if s.first_byte_ts is None)
|
||||||
|
status_500 = sum(1 for s in stats if s.http_status >= 500)
|
||||||
|
status_errors = sum(1 for s in stats if s.http_status >= 400 and s.http_status != 200)
|
||||||
|
|
||||||
total_req = len(stats)
|
total_req = len(stats)
|
||||||
elapsed = DURATION
|
qpm = total_req / (DURATION / 60)
|
||||||
qpm = total_req / (elapsed / 60)
|
|
||||||
|
|
||||||
print(f"\n{'='*60}")
|
print(f"\n{'='*65}")
|
||||||
print(f" 并发={name} | 运行{DURATION}s | 总请求={total_req} | 失败={failed}")
|
print(f" 并发={concurrency} | {DURATION}s | 请求={total_req} | 失败={failed} | 5xx={status_500}")
|
||||||
print(f"{'='*60}")
|
print(f"{'='*65}")
|
||||||
if ttfb_list:
|
if ttfb_list:
|
||||||
print(f" TTFB (s): min={min(ttfb_list):.3f} avg={statistics.mean(ttfb_list):.3f} "
|
print(f" TTFB(s): min={min(ttfb_list):.3f} avg={statistics.mean(ttfb_list):.3f} "
|
||||||
f"p50={statistics.median(ttfb_list):.3f} p95={_pct(ttfb_list, 95):.3f} p99={_pct(ttfb_list, 99):.3f}")
|
f"p50={statistics.median(ttfb_list):.3f} p95={_pct(ttfb_list,95):.3f} p99={_pct(ttfb_list,99):.3f}")
|
||||||
if total_list:
|
if total_list:
|
||||||
print(f" 完成 (s): min={min(total_list):.3f} avg={statistics.mean(total_list):.3f} "
|
print(f" 完成(s): min={min(total_list):.3f} avg={statistics.mean(total_list):.3f} "
|
||||||
f"p50={statistics.median(total_list):.3f} p95={_pct(total_list, 95):.3f} p99={_pct(total_list, 99):.3f}")
|
f"p50={statistics.median(total_list):.3f} p95={_pct(total_list,95):.3f} p99={_pct(total_list,99):.3f}")
|
||||||
print(f" QPM: {qpm:.1f}")
|
print(f" QPM: {qpm:.1f} | QPS: {total_req/DURATION:.1f} | HTTP错误: {status_errors}")
|
||||||
print(f" QPS: {total_req / elapsed:.1f}")
|
|
||||||
|
|
||||||
# 按分钟分段统计
|
# Per-minute breakdown
|
||||||
for minute in range(int(elapsed / 60)):
|
for minute in range(int(DURATION / 60)):
|
||||||
win_start = minute * 60
|
win_start = minute * 60
|
||||||
win_end = (minute + 1) * 60
|
win_end = (minute + 1) * 60
|
||||||
cnt = sum(1 for s in stats if s.end_ts and (s.end_ts - s.start_ts) >= 0
|
pm = sum(1 for s in stats if s.end_ts and s.first_byte_ts
|
||||||
and win_start <= (s.start_ts - stats[0].start_ts) < win_end)
|
and win_start <= (s.start_ts - stats[0].start_ts) < win_end)
|
||||||
print(f" 第{minute+1}分钟完成请求数: {cnt}")
|
print(f" 第{minute+1}分钟完成: {pm}")
|
||||||
|
|
||||||
return {
|
# Host stats
|
||||||
"concurrency": name,
|
if host_snaps:
|
||||||
"total": total_req,
|
cpus = [s.cpu for s in host_snaps if s.cpu > 0]
|
||||||
"failed": failed,
|
mems = [s.mem for s in host_snaps if s.mem > 0]
|
||||||
"ttfb_avg": statistics.mean(ttfb_list) if ttfb_list else None,
|
print(f" 主机: CPU avg={statistics.mean(cpus):.1f}% max={max(cpus):.1f}% "
|
||||||
"ttfb_p50": statistics.median(ttfb_list) if ttfb_list else None,
|
f"MEM avg={statistics.mean(mems):.1f}% max={max(mems):.1f}% "
|
||||||
"ttfb_p95": _pct(ttfb_list, 95) if ttfb_list else None,
|
f"LOAD max={max((s.load for s in host_snaps if s.load!='N/A'), default='N/A')}")
|
||||||
"total_avg": statistics.mean(total_list) if total_list else None,
|
|
||||||
"total_p50": statistics.median(total_list) if total_list else None,
|
return {"concurrency": concurrency, "total": total_req, "failed": failed,
|
||||||
"qpm": qpm,
|
"status_500": status_500, "status_errors": status_errors,
|
||||||
}
|
"ttfb_avg": statistics.mean(ttfb_list) if ttfb_list else None,
|
||||||
|
"ttfb_p50": statistics.median(ttfb_list) if ttfb_list else None,
|
||||||
|
"ttfb_p95": _pct(ttfb_list, 95) if ttfb_list else None,
|
||||||
|
"ttfb_p99": _pct(ttfb_list, 99) if ttfb_list else None,
|
||||||
|
"total_avg": statistics.mean(total_list) if total_list else None,
|
||||||
|
"total_p50": statistics.median(total_list) if total_list else None,
|
||||||
|
"qpm": qpm,
|
||||||
|
"host_cpu_avg": statistics.mean(cpus) if cpus else None,
|
||||||
|
"host_cpu_max": max(cpus) if cpus else None,
|
||||||
|
"host_mem_avg": statistics.mean(mems) if mems else None}
|
||||||
|
|
||||||
|
|
||||||
def _pct(data, p):
|
def _pct(data, p):
|
||||||
@ -141,22 +191,28 @@ def _pct(data, p):
|
|||||||
async def main():
|
async def main():
|
||||||
results = []
|
results = []
|
||||||
for c in CONCURRENCIES:
|
for c in CONCURRENCIES:
|
||||||
print(f"\n>>> 开始测试 并发={c} ...")
|
print(f"\n>>> 开始 并发={c} [{DURATION}s] ...")
|
||||||
stats = await run_concurrency(c)
|
stats, snaps = await run_concurrency(c)
|
||||||
r = analyze(str(c), stats)
|
r = analyze(c, stats, snaps)
|
||||||
results.append(r)
|
results.append(r)
|
||||||
|
|
||||||
# 汇总表格
|
print(f"\n{'='*65}")
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(" 汇总对比")
|
print(" 汇总对比")
|
||||||
print(f"{'='*60}")
|
print(f"{'='*65}")
|
||||||
print(f" {'并发':>6} {'总请求':>8} {'失败':>5} {'TTFB_avg':>9} {'TTFB_p50':>9} {'TTFB_p95':>9} {'完成_avg':>9} {'QPM':>8}")
|
print(f" {'并发':>5} {'请求':>7} {'失败':>5} {'5xx':>5} "
|
||||||
|
f"{'TTFB_avg':>9} {'TTFB_p50':>9} {'TTFB_p95':>9} {'TTFB_p99':>9} "
|
||||||
|
f"{'完成_avg':>9} {'QPM':>8} {'CPU_avg':>7} {'CPU_max':>7}")
|
||||||
for r in results:
|
for r in results:
|
||||||
print(f" {r['concurrency']:>6} {r['total']:>8} {r['failed']:>5} "
|
t_avg = f'{r["ttfb_avg"]:.3f}' if r['ttfb_avg'] else 'N/A'
|
||||||
f"{r['ttfb_avg']:.3f}s" if r['ttfb_avg'] else "N/A".rjust(9) + " "
|
t_p50 = f'{r["ttfb_p50"]:.3f}' if r['ttfb_p50'] else 'N/A'
|
||||||
f"{(r['ttfb_p50'] or 0):.3f}s".rjust(9) + " "
|
t_p95 = f'{r["ttfb_p95"]:.3f}' if r['ttfb_p95'] else 'N/A'
|
||||||
f"{(r['ttfb_p95'] or 0):.3f}s".rjust(9) + " "
|
t_p99 = f'{r["ttfb_p99"]:.3f}' if r['ttfb_p99'] else 'N/A'
|
||||||
f"{r['total_avg']:.3f}s".rjust(9) if r['total_avg'] else "N/A".rjust(9))
|
c_avg = f'{r["total_avg"]:.3f}' if r['total_avg'] else 'N/A'
|
||||||
|
cpu = f'{r["host_cpu_avg"]:.0f}%' if r['host_cpu_avg'] else 'N/A'
|
||||||
|
cmax = f'{r["host_cpu_max"]:.0f}%' if r['host_cpu_max'] else 'N/A'
|
||||||
|
print(f" {r['concurrency']:>5} {r['total']:>7} {r['failed']:>5} {r['status_500']:>5} "
|
||||||
|
f"{t_avg:>9} {t_p50:>9} {t_p95:>9} {t_p99:>9} "
|
||||||
|
f"{c_avg:>9} {r['qpm']:>8.0f} {cpu:>7} {cmax:>7}")
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user