Initial commit of LLM speed test app
This commit is contained in:
+18
@@ -0,0 +1,18 @@
|
||||
# Python
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*.egg-info/
|
||||
.pytest_cache/
|
||||
|
||||
# Virtual env
|
||||
.venv/
|
||||
venv/
|
||||
|
||||
# IDE
|
||||
.idea/
|
||||
.vscode/
|
||||
*.iml
|
||||
|
||||
# OS
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
@@ -0,0 +1,114 @@
|
||||
# 大模型速度测试助手 (Ollama 版)
|
||||
|
||||
一个用于测试**本地 Ollama 大模型**响应速度的工具,提供 Web 界面,可测量
|
||||
**TTFT(首 Token 延迟)、Prefill 时间、解码速度(tokens/s)、端到端延迟**等真实推理指标。
|
||||
|
||||
## 目录结构
|
||||
|
||||
```
|
||||
llm_speed_test_app/
|
||||
├── run.py # 启动入口
|
||||
├── requirements.txt # Python 依赖
|
||||
├── backend/ # 后端(Python)
|
||||
│ ├── config.py # 配置(Ollama 地址、模型、端口)
|
||||
│ ├── ollama_client.py # Ollama API 客户端(流式调用 + 指标测量)
|
||||
│ ├── engine.py # 测试引擎(编排用例、统计、落盘)
|
||||
│ ├── report.py # HTML 报告生成
|
||||
│ ├── server.py # HTTP 服务器(API + 前端静态托管)
|
||||
│ ├── stats.py # 统计计算(均值/标准差/百分位)
|
||||
│ └── test_cases/ # 7 个测试用例(真实调用 Ollama)
|
||||
│ ├── base.py
|
||||
│ ├── case_01_generation.py 纯文本生成基准
|
||||
│ ├── case_02_simple_tool.py 简单工具调用延迟
|
||||
│ ├── case_03_file_analysis.py 文件读取+分析
|
||||
│ ├── case_04_parallel_tool.py 并行工具调用
|
||||
│ ├── case_05_long_context.py 长上下文处理
|
||||
│ ├── case_06_reasoning.py 复杂推理任务
|
||||
│ └── case_07_multiturn.py 多轮对话累积
|
||||
├── frontend/ # 前端(静态页面)
|
||||
│ ├── index.html
|
||||
│ ├── css/style.css
|
||||
│ └── js/app.js
|
||||
├── results/ # 测试结果 JSON(运行时自动创建)
|
||||
└── report/ # HTML 报告(运行时自动创建)
|
||||
```
|
||||
|
||||
## 环境要求
|
||||
|
||||
- Python 3.8+
|
||||
- [Ollama](https://ollama.com/) 已安装并启动
|
||||
- 已拉取至少一个模型,例如:`ollama pull gemma2:2b`
|
||||
|
||||
## 快速开始
|
||||
|
||||
```bash
|
||||
# 1. 启动 Ollama(如未启动)
|
||||
ollama serve
|
||||
# 或直接运行 ollama(Windows 桌面版一般自启)
|
||||
|
||||
# 2. 拉取模型(任选其一,或使用已拉取的模型)
|
||||
ollama pull deepseek-r1:latest # 约 5GB,轻量
|
||||
|
||||
# 3. 安装依赖
|
||||
pip install -r requirements.txt
|
||||
|
||||
# 4. 启动应用
|
||||
python run.py
|
||||
```
|
||||
|
||||
然后浏览器打开 **http://localhost:8000**。
|
||||
|
||||
> 注意:如果前端下拉框显示"未找到已拉取的模型",请确认 Ollama 服务已启动,
|
||||
> 刷新页面即可重新加载模型列表。
|
||||
|
||||
## 使用说明
|
||||
|
||||
1. 选择测试用例("推荐组合"勾选快速用例,耗时用例用于压力测试)
|
||||
2. 选择模型(自动从本地 Ollama 拉取已安装模型)
|
||||
3. 设置重复次数
|
||||
4. 点击"开始测试",观察实时进度
|
||||
5. 测试完成后自动展示:推理核心指标、延迟对比图表、详细数据表
|
||||
6. 可导出 CSV / JSON,或打开完整 HTML 报告
|
||||
|
||||
## 测试用例说明
|
||||
|
||||
| 用例 | 难度 | 说明 |
|
||||
|---|---|---|
|
||||
| 纯文本生成基准 | 简单 | 基础生成速度基准测试 |
|
||||
| 简单工具调用延迟 | 简单 | 结构化工具输出的调用延迟 |
|
||||
| 文件读取+分析 | 中等 | 文件读取与分析完整链路 |
|
||||
| 并行工具调用 | 中等 | 串行 vs 并行请求的加速比 |
|
||||
| 长上下文处理 | 耗时 | 500~4000 token 下 prefill 延迟变化 |
|
||||
| 复杂推理任务 | 耗时 | 不同推理复杂度的延迟对比 |
|
||||
| 多轮对话累积 | 耗时 | 上下文累积对延迟的影响 |
|
||||
|
||||
## 测量指标说明
|
||||
|
||||
| 指标 | 来源 |
|
||||
|---|---|
|
||||
| TTFT (首 Token 延迟) | 发起请求到首个生成 token 返回的时间 |
|
||||
| Prefill Time | Ollama 返回的 `prompt_eval_duration`(prompt 预填充耗时) |
|
||||
| Decode Speed | `eval_count / eval_duration`(生成速度 tokens/s) |
|
||||
| E2E | 请求到完整回复的总耗时 |
|
||||
|
||||
## 配置(可选)
|
||||
|
||||
通过环境变量覆盖默认配置:
|
||||
|
||||
| 变量 | 默认值 | 说明 |
|
||||
|---|---|---|
|
||||
| `OLLAMA_BASE_URL` | `http://localhost:11434` | Ollama 服务地址 |
|
||||
| `DEFAULT_MODEL` | `qwen3.6:35b-a3b` | 默认模型 |
|
||||
| `SPEED_TEST_PORT` | `8000` | Web 服务端口 |
|
||||
|
||||
## API 一览
|
||||
|
||||
| 方法 | 路径 | 说明 |
|
||||
|---|---|---|
|
||||
| GET | `/api/models` | 本地已拉取的模型列表 |
|
||||
| GET | `/api/cases` | 用例注册表 |
|
||||
| POST | `/api/run` | 开始测试(后台线程执行),body: `{cases, repeats, model, skip_heavy}` |
|
||||
| GET | `/api/status` | 当前运行进度 |
|
||||
| POST | `/api/stop` | 停止当前测试 |
|
||||
| GET | `/api/results/latest.json` | 最近一次测试结果 |
|
||||
| GET | `/report.html` | 完整 HTML 报告 |
|
||||
@@ -0,0 +1,35 @@
|
||||
"""客户端公共基类:推理指标与错误类型
|
||||
|
||||
Ollama 原生客户端与 OpenAI 兼容客户端共享同一数据模型,
|
||||
测试用例通过统一接口(generate/chat)调用,无需关心后端差异。
|
||||
"""
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
class ClientError(Exception):
|
||||
"""推理服务调用相关错误(统一异常基类)"""
|
||||
|
||||
|
||||
@dataclass
|
||||
class InferenceResult:
|
||||
"""单次推理的完整指标(两套后端共用)"""
|
||||
ttft_ms: float = 0.0 # 首 Token 延迟(毫秒)
|
||||
prefill_ms: float = 0.0 # 预填充/处理 prompt 时间(毫秒)
|
||||
decode_speed_tok_s: float = 0.0 # 解码速度(tokens/s)
|
||||
total_tokens: int = 0 # 总 token 数
|
||||
prompt_tokens: int = 0 # prompt token 数
|
||||
completion_tokens: int = 0 # 生成 token 数
|
||||
e2e_ms: float = 0.0 # 端到端耗时(毫秒)
|
||||
response: str = "" # 完整回复文本
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"ttft_ms": round(self.ttft_ms, 2),
|
||||
"prefill_ms": round(self.prefill_ms, 2),
|
||||
"decode_speed_tok_s": round(self.decode_speed_tok_s, 2),
|
||||
"total_tokens": self.total_tokens,
|
||||
"prompt_tokens": self.prompt_tokens,
|
||||
"completion_tokens": self.completion_tokens,
|
||||
"e2e_ms": round(self.e2e_ms, 2),
|
||||
"response_length": len(self.response),
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
"""全局配置"""
|
||||
import os
|
||||
|
||||
# ---- 推理后端选择 ----
|
||||
# 可选值: "ollama"(本地 Ollama)| "openai"(任意 OpenAI 兼容 /v1 服务)
|
||||
LLM_BACKEND = os.environ.get("LLM_BACKEND", "openai")
|
||||
|
||||
# ---- Ollama 服务(backend=ollama)----
|
||||
# 本地 Ollama 默认监听端口 11434,可通过环境变量覆盖
|
||||
OLLAMA_BASE_URL = os.environ.get("OLLAMA_BASE_URL", "http://localhost:11434")
|
||||
# 默认模型(前端会从 /api/models 动态加载实际可用的模型列表)
|
||||
DEFAULT_MODEL = os.environ.get("DEFAULT_MODEL", "qwen3.6:35b-a3b")
|
||||
|
||||
# ---- OpenAI 兼容服务(backend=openai)----
|
||||
OPENAI_BASE_URL = os.environ.get("OPENAI_BASE_URL", "https://ai.lebiztrips.com/v1")
|
||||
OPENAI_API_KEY = os.environ.get("OPENAI_API_KEY", "123")
|
||||
OPENAI_DEFAULT_MODEL = os.environ.get("OPENAI_DEFAULT_MODEL", "Q3.6-35B-A3B-Orig-Thi")
|
||||
|
||||
# ---- 并发压力测试 ----
|
||||
DEFAULT_CONCURRENCY = int(os.environ.get("DEFAULT_CONCURRENCY", "100"))
|
||||
|
||||
# ---- HTTP 服务 ----
|
||||
SERVER_HOST = os.environ.get("SPEED_TEST_HOST", "0.0.0.0")
|
||||
SERVER_PORT = int(os.environ.get("SPEED_TEST_PORT", "8000"))
|
||||
|
||||
# ---- 目录结构 ----
|
||||
ROOT_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
BACKEND_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
FRONTEND_DIR = os.path.join(ROOT_DIR, "frontend")
|
||||
RESULTS_DIR = os.path.join(ROOT_DIR, "results")
|
||||
REPORT_DIR = os.path.join(ROOT_DIR, "report")
|
||||
@@ -0,0 +1,240 @@
|
||||
"""测试引擎:编排测试用例、统计指标、保存结果并生成报告"""
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime
|
||||
from glob import glob
|
||||
|
||||
from . import config
|
||||
from . import report as report_module
|
||||
from .stats import mean, std, percentile, percentiles, min_val, max_val
|
||||
from .ollama_client import OllamaClient, OllamaError
|
||||
|
||||
from .test_cases.case_01_generation import run_test as run_case_01
|
||||
from .test_cases.case_02_simple_tool import run_test as run_case_02
|
||||
from .test_cases.case_03_file_analysis import run_test as run_case_03
|
||||
from .test_cases.case_04_parallel_tool import run_test as run_case_04
|
||||
from .test_cases.case_05_long_context import run_test as run_case_05
|
||||
from .test_cases.case_06_reasoning import run_test as run_case_06
|
||||
from .test_cases.case_07_multiturn import run_test as run_case_07
|
||||
from .test_cases.case_08_concurrency import run_test as run_case_08
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# 用例注册表
|
||||
TEST_CASES = {
|
||||
"case_01_generation": {"func": run_case_01, "name": "纯文本生成基准", "difficulty": "简单", "estimated_seconds": 5},
|
||||
"case_02_simple_tool": {"func": run_case_02, "name": "简单工具调用延迟", "difficulty": "简单", "estimated_seconds": 3},
|
||||
"case_03_file_analysis": {"func": run_case_03, "name": "文件读取+分析", "difficulty": "中等", "estimated_seconds": 2},
|
||||
"case_04_parallel_tool": {"func": run_case_04, "name": "并行工具调用", "difficulty": "中等", "estimated_seconds": 5},
|
||||
"case_05_long_context": {"func": run_case_05, "name": "长上下文处理", "difficulty": "耗时", "estimated_seconds": 10},
|
||||
"case_06_reasoning": {"func": run_case_06, "name": "复杂推理任务", "difficulty": "耗时", "estimated_seconds": 15},
|
||||
"case_07_multiturn": {"func": run_case_07, "name": "多轮对话累积", "difficulty": "耗时", "estimated_seconds": 10},
|
||||
"case_08_concurrency": {"func": run_case_08, "name": "并发压力测试", "difficulty": "压力", "estimated_seconds": 30},
|
||||
}
|
||||
|
||||
|
||||
def create_client(model: str):
|
||||
"""根据配置创建对应的推理客户端"""
|
||||
if config.LLM_BACKEND == "openai":
|
||||
from .openai_client import OpenAIClient
|
||||
return OpenAIClient(model=model)
|
||||
from .ollama_client import OllamaClient
|
||||
return OllamaClient(model=model)
|
||||
|
||||
# 保留最近 N 次运行结果,超出则删除旧文件
|
||||
_MAX_RESULTS = 50
|
||||
|
||||
|
||||
def calculate_statistics(results_list: list) -> dict:
|
||||
"""从用例结果列表计算延迟统计(基于 elapsed_ms)"""
|
||||
values = [float(r["elapsed_ms"]) for r in results_list if r.get("elapsed_ms") is not None]
|
||||
if not values:
|
||||
return {}
|
||||
|
||||
pcts = percentiles(values, [50, 95, 99])
|
||||
|
||||
return {
|
||||
"count": len(values),
|
||||
"mean_ms": round(mean(values), 2),
|
||||
"std_ms": round(std(values), 2),
|
||||
"min_ms": round(min_val(values), 2),
|
||||
"max_ms": round(max_val(values), 2),
|
||||
"p50_ms": round(pcts[50], 2),
|
||||
"p95_ms": round(pcts[95], 2),
|
||||
"p99_ms": round(pcts[99], 2),
|
||||
}
|
||||
|
||||
|
||||
def calculate_inference_metrics(results_list: list) -> dict:
|
||||
"""单次遍历聚合推理指标(TTFT / Prefill / Decode / E2E)"""
|
||||
accumulators = {}
|
||||
for r in results_list:
|
||||
for key in ("ttft_ms", "prefill_ms", "e2e_ms"):
|
||||
val = r.get(key, 0)
|
||||
if val and float(val) > 0:
|
||||
accumulators.setdefault(key, []).append(float(val))
|
||||
decode = r.get("decode_speed_tok_s", 0)
|
||||
if decode and float(decode) > 0:
|
||||
accumulators.setdefault("decode_speed_tok_s", []).append(float(decode))
|
||||
total_tok = r.get("total_tokens", 0)
|
||||
if total_tok and int(total_tok) > 0:
|
||||
accumulators.setdefault("total_tokens", []).append(int(total_tok))
|
||||
|
||||
result = {}
|
||||
for key, vals in accumulators.items():
|
||||
result[f"{key}_mean"] = round(mean(vals), 2)
|
||||
result[f"{key}_p95"] = round(percentile(vals, 95), 2)
|
||||
return result
|
||||
|
||||
|
||||
def _cleanup_old_results() -> None:
|
||||
"""清理过期的运行结果,只保留最近 _MAX_RESULTS 个 run_*.json"""
|
||||
pattern = os.path.join(config.RESULTS_DIR, "run_*.json")
|
||||
files = sorted(glob(pattern))
|
||||
while len(files) > _MAX_RESULTS:
|
||||
old = files.pop(0)
|
||||
try:
|
||||
os.remove(old)
|
||||
logger.info("清理旧结果: %s", old)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def run_all_tests(cases: list, repeats: int, model: str, skip_heavy: bool,
|
||||
status: dict = None, stop_event=None) -> dict:
|
||||
"""运行选定的测试用例,返回完整报告
|
||||
|
||||
status: 运行状态字典(由 RunManager 共享,用于向前端汇报进度)
|
||||
stop_event: threading.Event,置位后停止
|
||||
"""
|
||||
client = create_client(model)
|
||||
|
||||
# 预检:确认推理服务可用
|
||||
if not client.ping():
|
||||
backend = config.LLM_BACKEND
|
||||
hint = "请确认已运行 'ollama serve' 并拉取模型" if backend == "ollama" else "请检查 OPENAI_BASE_URL / OPENAI_API_KEY 配置"
|
||||
raise OllamaError(f"无法连接 {backend} 后端服务。{hint}")
|
||||
|
||||
# 过滤耗时用例
|
||||
if skip_heavy:
|
||||
cases = [c for c in cases if TEST_CASES[c]["difficulty"] != "耗时"]
|
||||
|
||||
total_cases = len(cases)
|
||||
all_results = []
|
||||
stopped = False
|
||||
total_start = time.time()
|
||||
|
||||
def _update(**kw):
|
||||
if status is not None:
|
||||
status.update(kw)
|
||||
|
||||
for i, case_id in enumerate(cases, 1):
|
||||
if stop_event and stop_event.is_set():
|
||||
stopped = True
|
||||
_update(message="用户停止测试")
|
||||
break
|
||||
|
||||
case_info = TEST_CASES[case_id]
|
||||
_update(
|
||||
status="running",
|
||||
current_case_index=i - 1,
|
||||
total_cases=total_cases,
|
||||
current_case_name=case_info["name"],
|
||||
progress=round((i - 1) / total_cases * 100, 1),
|
||||
message=f"正在执行: {case_info['name']} ({case_info['difficulty']})",
|
||||
)
|
||||
|
||||
case_start = time.time()
|
||||
try:
|
||||
raw_result = case_info["func"](client=client, repeats=repeats, stop_event=stop_event)
|
||||
|
||||
stats = {}
|
||||
inference_metrics = {}
|
||||
if raw_result.get("results"):
|
||||
stats = calculate_statistics(raw_result["results"])
|
||||
inference_metrics = calculate_inference_metrics(raw_result["results"])
|
||||
|
||||
all_results.append({
|
||||
"case_id": case_id,
|
||||
"case_name": case_info["name"],
|
||||
"difficulty": case_info["difficulty"],
|
||||
"status": "passed",
|
||||
"raw_data": raw_result,
|
||||
"statistics": stats,
|
||||
"inference_metrics": inference_metrics,
|
||||
"elapsed_seconds": round(time.time() - case_start, 2),
|
||||
})
|
||||
_update(
|
||||
current_case_index=i,
|
||||
progress=round(i / total_cases * 100, 1),
|
||||
message=f"完成: {case_info['name']}",
|
||||
)
|
||||
logger.info("用例 %s 完成: %s", case_id, case_info["name"])
|
||||
except Exception as e:
|
||||
all_results.append({
|
||||
"case_id": case_id,
|
||||
"case_name": case_info["name"],
|
||||
"difficulty": case_info["difficulty"],
|
||||
"status": "failed",
|
||||
"error": str(e),
|
||||
"elapsed_seconds": round(time.time() - case_start, 2),
|
||||
})
|
||||
_update(message=f"失败: {case_info['name']} - {e}")
|
||||
logger.error("用例 %s 失败: %s", case_id, e)
|
||||
|
||||
total_elapsed = time.time() - total_start
|
||||
|
||||
report = {
|
||||
"version": "1.0",
|
||||
"timestamp": datetime.now().isoformat(),
|
||||
"run_id": datetime.now().strftime("%Y%m%d-%H%M%S"),
|
||||
"environment": {
|
||||
"python_version": f"{sys.version_info.major}.{sys.version_info.minor}.{sys.version_info.micro}",
|
||||
"platform": sys.platform,
|
||||
"model": model,
|
||||
"backend": config.LLM_BACKEND,
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": cases,
|
||||
"repeats": repeats,
|
||||
"skip_heavy": skip_heavy,
|
||||
"total_cases": len(cases),
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": len(cases),
|
||||
"passed": sum(1 for r in all_results if r["status"] == "passed"),
|
||||
"failed": sum(1 for r in all_results if r["status"] == "failed"),
|
||||
"stopped": stopped,
|
||||
"total_elapsed_seconds": round(total_elapsed, 2),
|
||||
},
|
||||
"results": all_results,
|
||||
}
|
||||
|
||||
# 保存结果 + 生成 HTML 报告
|
||||
save_results(report)
|
||||
|
||||
return report
|
||||
|
||||
|
||||
def save_results(report: dict) -> None:
|
||||
"""保存 JSON 结果并生成 HTML 报告"""
|
||||
os.makedirs(config.RESULTS_DIR, exist_ok=True)
|
||||
|
||||
latest_path = os.path.join(config.RESULTS_DIR, "latest.json")
|
||||
with open(latest_path, "w", encoding="utf-8") as f:
|
||||
json.dump(report, f, ensure_ascii=False, indent=2)
|
||||
|
||||
archive_path = os.path.join(config.RESULTS_DIR, f"run_{report['run_id']}.json")
|
||||
with open(archive_path, "w", encoding="utf-8") as f:
|
||||
json.dump(report, f, ensure_ascii=False, indent=2)
|
||||
|
||||
# 清理旧结果
|
||||
_cleanup_old_results()
|
||||
|
||||
try:
|
||||
report_module.generate_report(report)
|
||||
except Exception as e:
|
||||
logger.warning("生成 HTML 报告失败: %s", e)
|
||||
@@ -0,0 +1,154 @@
|
||||
"""Ollama API 客户端
|
||||
|
||||
通过 HTTP 流式调用本地 Ollama 服务,并测量真实推理指标:
|
||||
- TTFT: 首个 token 到达时间(从发起请求开始计时)
|
||||
- Prefill: prompt 预填充耗时(取自 Ollama 返回的 prompt_eval_duration)
|
||||
- Decode Speed: 解码速度 tokens/s(eval_count / eval_duration)
|
||||
- E2E: 端到端总耗时
|
||||
|
||||
依赖:requests(Ollama 服务需已启动,默认 http://localhost:11434)
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
from . import config
|
||||
from .client_base import InferenceResult, ClientError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class OllamaError(ClientError):
|
||||
"""Ollama 调用相关错误"""
|
||||
|
||||
|
||||
class OllamaClient:
|
||||
"""封装对本地 Ollama 服务的调用"""
|
||||
|
||||
def __init__(self, base_url: str = None, model: str = None, timeout: int = 180):
|
||||
self.base_url = (base_url or config.OLLAMA_BASE_URL).rstrip("/")
|
||||
self.model = model or config.DEFAULT_MODEL
|
||||
self.timeout = timeout
|
||||
self._session = requests.Session()
|
||||
|
||||
# ---------- 服务探测 ----------
|
||||
|
||||
def ping(self) -> bool:
|
||||
"""探测 Ollama 服务是否可用"""
|
||||
try:
|
||||
return self._session.get(f"{self.base_url}/api/tags", timeout=5).ok
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def list_models(self) -> list:
|
||||
"""列出本地已拉取的模型"""
|
||||
try:
|
||||
resp = self._session.get(f"{self.base_url}/api/tags", timeout=5)
|
||||
resp.raise_for_status()
|
||||
return [m["name"] for m in resp.json().get("models", [])]
|
||||
except requests.exceptions.ConnectionError as e:
|
||||
raise OllamaError(
|
||||
f"无法连接 Ollama 服务({self.base_url}),请确认已运行 'ollama serve'"
|
||||
) from e
|
||||
except Exception as e:
|
||||
raise OllamaError(f"获取模型列表失败: {e}") from e
|
||||
|
||||
# ---------- 文本生成(/api/generate)----------
|
||||
|
||||
def generate(self, prompt: str, system: str = None, options: dict = None) -> InferenceResult:
|
||||
"""单轮文本生成"""
|
||||
payload = {
|
||||
"model": self.model,
|
||||
"prompt": prompt,
|
||||
"stream": True,
|
||||
"options": options or {},
|
||||
}
|
||||
if system:
|
||||
payload["system"] = system
|
||||
return self._request(payload)
|
||||
|
||||
# ---------- 多轮对话(/api/chat)----------
|
||||
|
||||
def chat(self, messages: list, options: dict = None) -> InferenceResult:
|
||||
"""多轮对话,messages 为 [{"role": "user"/"assistant", "content": ...}]"""
|
||||
payload = {
|
||||
"model": self.model,
|
||||
"messages": messages,
|
||||
"stream": True,
|
||||
"options": options or {},
|
||||
}
|
||||
return self._request(payload, chat=True)
|
||||
|
||||
# ---------- 核心请求逻辑 ----------
|
||||
|
||||
def _request(self, payload: dict, chat: bool = False) -> InferenceResult:
|
||||
result = InferenceResult()
|
||||
chunks = []
|
||||
final = {}
|
||||
ttft_start = None
|
||||
start = time.perf_counter()
|
||||
|
||||
endpoint = "chat" if chat else "generate"
|
||||
try:
|
||||
resp = self._session.post(
|
||||
f"{self.base_url}/api/{endpoint}",
|
||||
json=payload,
|
||||
stream=True,
|
||||
timeout=self.timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
for line in resp.iter_lines(decode_unicode=True):
|
||||
if not line:
|
||||
continue
|
||||
data = json.loads(line)
|
||||
if chat:
|
||||
piece = (data.get("message") or {}).get("content", "")
|
||||
else:
|
||||
piece = data.get("response", "")
|
||||
done = data.get("done", False)
|
||||
if not done:
|
||||
# 首个非空片段出现时刻记为 TTFT
|
||||
if ttft_start is None and piece:
|
||||
ttft_start = time.perf_counter() - start
|
||||
if piece:
|
||||
chunks.append(piece)
|
||||
else:
|
||||
final = data
|
||||
except requests.exceptions.ConnectionError as e:
|
||||
raise OllamaError(
|
||||
f"无法连接 Ollama 服务({self.base_url}),请确认已运行 'ollama serve' 且模型 {self.model} 已拉取"
|
||||
) from e
|
||||
except requests.exceptions.Timeout as e:
|
||||
raise OllamaError(f"Ollama 请求超时({self.timeout}s): {e}") from e
|
||||
except (json.JSONDecodeError, KeyError) as e:
|
||||
raise OllamaError(f"Ollama 返回数据格式错误: {e}") from e
|
||||
|
||||
result.response = "".join(chunks)
|
||||
result.e2e_ms = (time.perf_counter() - start) * 1000
|
||||
result.ttft_ms = (ttft_start * 1000) if ttft_start else 0.0
|
||||
|
||||
# Ollama 结束块中的指标(单位为纳秒)
|
||||
prompt_eval_count = final.get("prompt_eval_count", 0) or 0
|
||||
prompt_eval_dur_ns = final.get("prompt_eval_duration", 0) or 0
|
||||
eval_count = final.get("eval_count", 0) or 0
|
||||
eval_dur_ns = final.get("eval_duration", 0) or 0
|
||||
|
||||
result.prompt_tokens = int(prompt_eval_count)
|
||||
result.completion_tokens = int(eval_count)
|
||||
result.total_tokens = result.prompt_tokens + result.completion_tokens
|
||||
result.prefill_ms = prompt_eval_dur_ns / 1_000_000
|
||||
if eval_dur_ns:
|
||||
result.decode_speed_tok_s = result.completion_tokens / (eval_dur_ns / 1e9)
|
||||
|
||||
# TTFT 未测到时(如空回复),回退到 prefill 时间
|
||||
if result.ttft_ms <= 0 and result.prefill_ms > 0:
|
||||
result.ttft_ms = result.prefill_ms
|
||||
|
||||
logger.debug(
|
||||
"推理完成: e2e=%.1fms ttft=%.1fms decode=%.1f tok/s tokens=%d/%d",
|
||||
result.e2e_ms, result.ttft_ms, result.decode_speed_tok_s,
|
||||
result.prompt_tokens, result.completion_tokens,
|
||||
)
|
||||
return result
|
||||
@@ -0,0 +1,213 @@
|
||||
"""OpenAI 兼容 API 客户端
|
||||
|
||||
对接任意 OpenAI /v1 兼容服务(OpenAI、Together、硅基流动、llama.cpp server 等),
|
||||
与 OllamaClient 保持相同的 generate/chat 接口,测试用例无需区分后端。
|
||||
|
||||
指标来源:
|
||||
- TTFT: 流式请求中首个 content 片段到达时间(真实测量)
|
||||
- Prefill: 优先取服务端 timings.prompt_ms(llama.cpp server 暴露),否则回退 TTFT
|
||||
- Decode Speed: 优先取服务端 timings.predicted_per_second,否则按 completion_tokens/e2e 估算
|
||||
- E2E: 客户端请求总耗时
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
from . import config
|
||||
from .client_base import InferenceResult, ClientError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class OpenAIError(ClientError):
|
||||
"""OpenAI 兼容 API 调用错误"""
|
||||
|
||||
|
||||
class OpenAIClient:
|
||||
"""封装对 OpenAI 兼容 /v1 服务的调用"""
|
||||
|
||||
def __init__(self, base_url: str = None, model: str = None, api_key: str = None,
|
||||
timeout: int = 300):
|
||||
self.base_url = (base_url or config.OPENAI_BASE_URL).rstrip("/")
|
||||
self.model = model or config.OPENAI_DEFAULT_MODEL
|
||||
self.api_key = api_key if api_key not in (None, "") else config.OPENAI_API_KEY
|
||||
self.timeout = timeout
|
||||
self._session = requests.Session()
|
||||
|
||||
# ---------- 请求头 ----------
|
||||
|
||||
@property
|
||||
def _headers(self) -> dict:
|
||||
headers = {"Content-Type": "application/json"}
|
||||
if self.api_key:
|
||||
headers["Authorization"] = f"Bearer {self.api_key}"
|
||||
return headers
|
||||
|
||||
# ---------- 服务探测 ----------
|
||||
|
||||
def ping(self) -> bool:
|
||||
"""探测服务是否可用"""
|
||||
try:
|
||||
return self._session.get(
|
||||
f"{self.base_url}/models", headers=self._headers, timeout=5
|
||||
).ok
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def list_models(self) -> list:
|
||||
"""列出可用模型"""
|
||||
try:
|
||||
resp = self._session.get(
|
||||
f"{self.base_url}/models", headers=self._headers, timeout=10
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json().get("data", [])
|
||||
models = []
|
||||
for m in data:
|
||||
mid = m.get("id") or m.get("name")
|
||||
if mid:
|
||||
models.append(mid)
|
||||
return models
|
||||
except requests.exceptions.ConnectionError as e:
|
||||
raise OpenAIError(
|
||||
f"无法连接服务({self.base_url}),请检查地址与网络"
|
||||
) from e
|
||||
except requests.exceptions.HTTPError as e:
|
||||
if resp.status_code in (401, 403):
|
||||
raise OpenAIError("API Key 无效或被拒绝(401/403)") from e
|
||||
raise OpenAIError(f"获取模型列表失败: {resp.status_code} {resp.text[:200]}") from e
|
||||
except Exception as e:
|
||||
raise OpenAIError(f"获取模型列表失败: {e}") from e
|
||||
|
||||
# ---------- 文本生成 ----------
|
||||
|
||||
def generate(self, prompt: str, system: str = None, options: dict = None) -> InferenceResult:
|
||||
"""单轮文本生成(流式),等价于单条 user 消息的 chat"""
|
||||
messages = []
|
||||
if system:
|
||||
messages.append({"role": "system", "content": system})
|
||||
messages.append({"role": "user", "content": prompt})
|
||||
return self.chat(messages, options)
|
||||
|
||||
# ---------- 多轮对话 ----------
|
||||
|
||||
def chat(self, messages: list, options: dict = None) -> InferenceResult:
|
||||
"""多轮对话,messages 为 [{"role": ..., "content": ...}]"""
|
||||
options = options or {}
|
||||
payload = {
|
||||
"model": self.model,
|
||||
"messages": messages,
|
||||
"stream": True,
|
||||
"temperature": options.get("temperature", 0.3),
|
||||
"max_tokens": options.get("num_predict", 256),
|
||||
}
|
||||
|
||||
start = time.perf_counter()
|
||||
ttft_ms = 0.0
|
||||
chunks = []
|
||||
final = {}
|
||||
timings = {}
|
||||
|
||||
try:
|
||||
resp = self._session.post(
|
||||
f"{self.base_url}/chat/completions",
|
||||
headers=self._headers,
|
||||
json=payload,
|
||||
stream=True,
|
||||
timeout=self.timeout,
|
||||
)
|
||||
except requests.exceptions.ConnectionError as e:
|
||||
raise OpenAIError(
|
||||
f"无法连接服务({self.base_url}),请检查地址与网络"
|
||||
) from e
|
||||
except requests.exceptions.Timeout as e:
|
||||
raise OpenAIError(f"请求超时({self.timeout}s): {e}") from e
|
||||
|
||||
if resp.status_code != 200:
|
||||
err_text = resp.text[:300]
|
||||
raise OpenAIError(f"API 错误 {resp.status_code}: {err_text}")
|
||||
|
||||
# 流式解析 SSE
|
||||
try:
|
||||
for line in resp.iter_lines(decode_unicode=True):
|
||||
if not line:
|
||||
continue
|
||||
line = line.strip()
|
||||
if line.startswith("data:"):
|
||||
line = line[5:].strip()
|
||||
if line == "[DONE]":
|
||||
break
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
|
||||
choices = data.get("choices") or []
|
||||
if not choices:
|
||||
continue
|
||||
delta = choices[0].get("delta") or {}
|
||||
# 兼容推理模型:token 可能放在 reasoning_content(思考过程)或 content
|
||||
piece = delta.get("content") or delta.get("reasoning_content") or ""
|
||||
|
||||
if piece:
|
||||
# 首个内容片段到达时间 = TTFT
|
||||
if ttft_ms == 0.0:
|
||||
ttft_ms = (time.perf_counter() - start) * 1000
|
||||
chunks.append(piece)
|
||||
|
||||
# 服务端最终块可能携带 usage / timings
|
||||
if choices[0].get("finish_reason"):
|
||||
final = data
|
||||
if data.get("usage"):
|
||||
final = data
|
||||
except (requests.exceptions.ConnectionError, requests.exceptions.ChunkedEncodingError) as e:
|
||||
raise OpenAIError(f"流式读取中断: {e}") from e
|
||||
|
||||
e2e_ms = (time.perf_counter() - start) * 1000
|
||||
result = InferenceResult()
|
||||
result.response = "".join(chunks)
|
||||
result.e2e_ms = e2e_ms
|
||||
result.ttft_ms = ttft_ms
|
||||
|
||||
# usage: token 计数(部分服务端流式响应不含 usage,回退用 timings)
|
||||
usage = final.get("usage", {}) or {}
|
||||
prompt_tok = int(usage.get("prompt_tokens", 0) or 0)
|
||||
completion_tok = int(usage.get("completion_tokens", 0) or 0)
|
||||
|
||||
# timings: llama.cpp / llama-server 暴露的服务端计时(单位毫秒)
|
||||
timings = final.get("timings", {}) or {}
|
||||
prompt_n = int(timings.get("prompt_n", 0) or 0)
|
||||
predicted_n = int(timings.get("predicted_n", 0) or 0)
|
||||
prompt_ms = float(timings.get("prompt_ms", 0) or 0)
|
||||
predicted_ms = float(timings.get("predicted_ms", 0) or 0)
|
||||
predicted_pps = float(timings.get("predicted_per_second", 0) or 0)
|
||||
|
||||
result.prompt_tokens = prompt_tok or prompt_n
|
||||
result.completion_tokens = completion_tok or predicted_n
|
||||
result.total_tokens = result.prompt_tokens + result.completion_tokens
|
||||
|
||||
result.prefill_ms = prompt_ms
|
||||
|
||||
# 解码速度:优先服务端计时
|
||||
if predicted_pps > 0:
|
||||
result.decode_speed_tok_s = predicted_pps
|
||||
elif result.completion_tokens > 0 and result.e2e_ms > 0:
|
||||
result.decode_speed_tok_s = result.completion_tokens / (result.e2e_ms / 1000)
|
||||
|
||||
# TTFT 回退:流式被缓冲或未测到时,用服务端 prompt_ms
|
||||
if result.ttft_ms <= 0 or result.ttft_ms > e2e_ms * 0.95:
|
||||
if prompt_ms > 0:
|
||||
result.ttft_ms = prompt_ms
|
||||
else:
|
||||
result.ttft_ms = e2e_ms if e2e_ms > 0 else 0.0
|
||||
|
||||
logger.debug(
|
||||
"推理完成(openai): e2e=%.1fms ttft=%.1fms decode=%.1f tok/s tokens=%d/%d",
|
||||
result.e2e_ms, result.ttft_ms, result.decode_speed_tok_s,
|
||||
result.prompt_tokens, result.completion_tokens,
|
||||
)
|
||||
return result
|
||||
@@ -0,0 +1,279 @@
|
||||
"""生成 HTML 可视化报告"""
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
|
||||
from . import config
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def generate_report(report: dict) -> str:
|
||||
"""从报告数据生成 HTML 报告文件,返回输出路径"""
|
||||
html = build_html(report)
|
||||
|
||||
os.makedirs(config.REPORT_DIR, exist_ok=True)
|
||||
output_path = os.path.join(config.REPORT_DIR, "report.html")
|
||||
with open(output_path, "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
|
||||
logger.info("报告已生成: %s", output_path)
|
||||
return output_path
|
||||
|
||||
|
||||
def _diff_color(difficulty: str) -> str:
|
||||
"""难度对应的颜色"""
|
||||
return {
|
||||
"简单": "#22c55e",
|
||||
"中等": "#f59e0b",
|
||||
"耗时": "#ef4444",
|
||||
"压力": "#8b5cf6",
|
||||
}.get(difficulty, "#6b7280")
|
||||
|
||||
|
||||
def _stats_rows(results: list) -> str:
|
||||
"""构建详细数据表格行 HTML"""
|
||||
rows = []
|
||||
for r in results:
|
||||
stats = r.get("statistics", {})
|
||||
if not stats:
|
||||
continue
|
||||
rows.append(
|
||||
f'<tr>'
|
||||
f'<td><strong>{r["case_name"]}</strong></td>'
|
||||
f'<td><span class="diff-badge" style="background:{_diff_color(r["difficulty"])}">{r["difficulty"]}</span></td>'
|
||||
f'<td class="status-{r["status"]}" style="font-weight:bold">{r["status"]}</td>'
|
||||
f'<td class="mono">{stats.get("mean_ms", "N/A")}</td>'
|
||||
f'<td class="mono">{stats.get("std_ms", "N/A")}</td>'
|
||||
f'<td class="mono">{stats.get("min_ms", "N/A")}</td>'
|
||||
f'<td class="mono">{stats.get("max_ms", "N/A")}</td>'
|
||||
f'<td class="mono">{stats.get("p95_ms", "N/A")}</td>'
|
||||
f'</tr>'
|
||||
)
|
||||
return "\n".join(rows)
|
||||
|
||||
|
||||
def _chart_data(results: list) -> tuple:
|
||||
"""提取图表数据:(labels, means, p95s)"""
|
||||
labels, means, p95s = [], [], []
|
||||
for r in results:
|
||||
stats = r.get("statistics", {})
|
||||
if stats:
|
||||
labels.append(r["case_name"])
|
||||
means.append(stats.get("mean_ms", 0))
|
||||
p95s.append(stats.get("p95_ms", 0))
|
||||
return labels, means, p95s
|
||||
|
||||
|
||||
def _performance_analysis(results: list) -> str:
|
||||
"""性能分析摘要"""
|
||||
_, means, _ = _chart_data(results)
|
||||
if len(means) < 2:
|
||||
return ""
|
||||
|
||||
best_idx = means.index(min(means))
|
||||
worst_idx = means.index(max(means))
|
||||
ratio = (means[worst_idx] / means[best_idx]) if means[best_idx] > 0 else "N/A"
|
||||
avg = sum(means) / len(means)
|
||||
|
||||
return (
|
||||
f'<div class="analysis-box">'
|
||||
f'<h3>性能分析</h3>'
|
||||
f'<ul>'
|
||||
f'<li><strong>最快用例:</strong> {results[best_idx]["case_name"]} ({means[best_idx]:.2f}ms)</li>'
|
||||
f'<li><strong>最慢用例:</strong> {results[worst_idx]["case_name"]} ({means[worst_idx]:.2f}ms)</li>'
|
||||
f'<li><strong>最快/最慢比:</strong> {ratio}x</li>'
|
||||
f'<li><strong>平均延迟:</strong> {avg:.2f}ms</li>'
|
||||
f'</ul></div>'
|
||||
)
|
||||
|
||||
|
||||
def _concurrency_summary(results: list) -> str:
|
||||
"""提取并发压力测试摘要(case_08),无则返回空串"""
|
||||
for r in results:
|
||||
raw = r.get("raw_data", {}) or {}
|
||||
if r.get("case_id") == "case_08_concurrency" and raw:
|
||||
def g(key, suffix=""):
|
||||
val = raw.get(key)
|
||||
return "N/A" if val is None else f"{val}{suffix}"
|
||||
|
||||
keep = raw.get("tps_keep_rate_pct", 0)
|
||||
keep_color = "#22c55e" if keep >= 60 else ("#f59e0b" if keep > 0 else "#ef4444")
|
||||
err = raw.get("error_rate_pct", 0)
|
||||
err_color = "#22c55e" if err == 0 else "#ef4444"
|
||||
|
||||
return (
|
||||
f'<div class="section">'
|
||||
f'<h2>⚡ 并发压力测试摘要</h2>'
|
||||
f'<div class="dashboard">'
|
||||
f'<div class="stat-card"><div class="label">并发数</div><div class="value" style="color:#8b5cf6">{g("concurrency")}</div></div>'
|
||||
f'<div class="stat-card"><div class="label">基准 TPS</div><div class="value" style="color:#6366f1">{g("baseline_tps")}</div></div>'
|
||||
f'<div class="stat-card"><div class="label">并发 TPS</div><div class="value" style="color:#6366f1">{g("concurrent_tps")}</div></div>'
|
||||
f'<div class="stat-card"><div class="label">TPS 保持率</div><div class="value" style="color:{keep_color}">{g("tps_keep_rate_pct", "%")}</div></div>'
|
||||
f'<div class="stat-card"><div class="label">P95 延迟</div><div class="value" style="color:#ef4444">{g("e2e_p95_ms", "ms")}</div></div>'
|
||||
f'<div class="stat-card"><div class="label">P99 延迟</div><div class="value" style="color:#ef4444">{g("e2e_p99_ms", "ms")}</div></div>'
|
||||
f'<div class="stat-card"><div class="label">错误率</div><div class="value" style="color:{err_color}">{g("error_rate_pct", "%")}</div></div>'
|
||||
f'</div></div>'
|
||||
)
|
||||
return ""
|
||||
|
||||
|
||||
def build_html(report: dict) -> str:
|
||||
"""构建 HTML 报告"""
|
||||
summary = report.get("summary", {})
|
||||
results = report.get("results", [])
|
||||
env = report.get("environment", {})
|
||||
config_ = report.get("config", {})
|
||||
timestamp = report.get("timestamp", "")
|
||||
run_id = report.get("run_id", "")
|
||||
|
||||
case_names, case_means, case_p95s = _chart_data(results)
|
||||
|
||||
return f"""<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>大模型速度测试报告 - {run_id}</title>
|
||||
<script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.0/dist/chart.umd.min.js"></script>
|
||||
<style>
|
||||
* {{ margin: 0; padding: 0; box-sizing: border-box; }}
|
||||
body {{
|
||||
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', 'PingFang SC', 'Microsoft YaHei', sans-serif;
|
||||
background: #f8fafc; color: #1e293b; line-height: 1.6;
|
||||
}}
|
||||
.container {{ max-width: 1400px; margin: 0 auto; padding: 20px; }}
|
||||
.header {{ background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); color: white; padding: 30px; border-radius: 12px; margin-bottom: 20px; }}
|
||||
.header h1 {{ font-size: 28px; margin-bottom: 10px; }}
|
||||
.header .meta {{ font-size: 14px; opacity: 0.9; }}
|
||||
.dashboard {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(200px, 1fr)); gap: 15px; margin-bottom: 20px; }}
|
||||
.stat-card {{ background: white; padding: 20px; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,0.1); }}
|
||||
.stat-card .label {{ font-size: 12px; color: #64748b; text-transform: uppercase; }}
|
||||
.stat-card .value {{ font-size: 28px; font-weight: bold; margin-top: 5px; }}
|
||||
.section {{ background: white; border-radius: 10px; padding: 25px; margin-bottom: 20px; box-shadow: 0 2px 8px rgba(0,0,0,0.1); }}
|
||||
.section h2 {{ font-size: 20px; margin-bottom: 15px; padding-bottom: 10px; border-bottom: 2px solid #e2e8f0; }}
|
||||
table {{ width: 100%; border-collapse: collapse; }}
|
||||
th, td {{ padding: 12px; text-align: left; border-bottom: 1px solid #e2e8f0; }}
|
||||
th {{ background: #f1f5f9; font-weight: 600; font-size: 13px; text-transform: uppercase; }}
|
||||
tr:hover {{ background: #f8fafc; }}
|
||||
.mono {{ font-family: 'SF Mono', 'Fira Code', monospace; }}
|
||||
.diff-badge {{ padding: 4px 10px; border-radius: 12px; color: white; font-size: 12px; font-weight: 600; }}
|
||||
.status-passed {{ color: #22c55e; }}
|
||||
.status-failed {{ color: #ef4444; }}
|
||||
.chart-container {{ position: relative; height: 400px; margin: 20px 0; }}
|
||||
.analysis-box {{ background: #f0fdf4; border-left: 4px solid #22c55e; padding: 15px; border-radius: 6px; }}
|
||||
.analysis-box ul {{ padding-left: 20px; margin-top: 10px; }}
|
||||
.analysis-box li {{ margin: 5px 0; }}
|
||||
footer {{ text-align: center; padding: 20px; color: #64748b; font-size: 13px; }}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
<div class="header">
|
||||
<h1>大模型速度测试报告</h1>
|
||||
<div class="meta">
|
||||
Run ID: {run_id} | 时间: {timestamp} | 用例: {config_.get('total_cases', 0)} | 重复: {config_.get('repeats', 0)}次
|
||||
<br>模型: {env.get('model', 'N/A')} | 平台: {env.get('platform', 'N/A')}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="dashboard">
|
||||
<div class="stat-card"><div class="label">总用例数</div><div class="value">{summary.get('total_cases', 0)}</div></div>
|
||||
<div class="stat-card"><div class="label">通过</div><div class="value" style="color:#22c55e">{summary.get('passed', 0)}</div></div>
|
||||
<div class="stat-card"><div class="label">失败</div><div class="value" style="color:#ef4444">{summary.get('failed', 0)}</div></div>
|
||||
<div class="stat-card"><div class="label">总耗时</div><div class="value">{summary.get('total_elapsed_seconds', 0):.2f}s</div></div>
|
||||
</div>
|
||||
|
||||
{_concurrency_summary(results)}
|
||||
|
||||
{_performance_analysis(results)}
|
||||
|
||||
<div class="section">
|
||||
<h2>性能对比柱状图</h2>
|
||||
<div class="chart-container"><canvas id="barChart"></canvas></div>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<h2>P95延迟对比</h2>
|
||||
<div class="chart-container"><canvas id="p95Chart"></canvas></div>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<h2>详细数据</h2>
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>用例名称</th><th>难度</th><th>状态</th><th>平均延迟 (ms)</th>
|
||||
<th>标准差</th><th>最小</th><th>最大</th><th>P95</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{_stats_rows(results)}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<h2>环境信息</h2>
|
||||
<table>
|
||||
<tr><td>模型</td><td class="mono">{env.get('model', 'N/A')}</td></tr>
|
||||
<tr><td>运行平台</td><td class="mono">{env.get('platform', 'N/A')}</td></tr>
|
||||
<tr><td>运行ID</td><td class="mono">{run_id}</td></tr>
|
||||
<tr><td>测试时间</td><td class="mono">{timestamp}</td></tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<footer>大模型速度测试助手 | 报告由 report.py 自动生成</footer>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
const barCtx = document.getElementById('barChart').getContext('2d');
|
||||
new Chart(barCtx, {{
|
||||
type: 'bar',
|
||||
data: {{
|
||||
labels: {json.dumps(case_names, ensure_ascii=False)},
|
||||
datasets: [{{
|
||||
label: '平均延迟 (ms)',
|
||||
data: {json.dumps(case_means)},
|
||||
backgroundColor: 'rgba(102, 126, 234, 0.7)',
|
||||
borderColor: 'rgba(102, 126, 234, 1)',
|
||||
borderWidth: 1
|
||||
}}]
|
||||
}},
|
||||
options: {{
|
||||
responsive: true,
|
||||
maintainAspectRatio: false,
|
||||
plugins: {{
|
||||
legend: {{ display: false }},
|
||||
title: {{ display: true, text: '各用例平均延迟对比 (ms)', font: {{ size: 16 }} }}
|
||||
}},
|
||||
scales: {{ y: {{ beginAtZero: true, title: {{ display: true, text: '延迟 (ms)' }} }} }}
|
||||
}}
|
||||
}});
|
||||
|
||||
const p95Ctx = document.getElementById('p95Chart').getContext('2d');
|
||||
new Chart(p95Ctx, {{
|
||||
type: 'bar',
|
||||
data: {{
|
||||
labels: {json.dumps(case_names, ensure_ascii=False)},
|
||||
datasets: [{{
|
||||
label: 'P95延迟 (ms)',
|
||||
data: {json.dumps(case_p95s)},
|
||||
backgroundColor: 'rgba(239, 68, 68, 0.7)',
|
||||
borderColor: 'rgba(239, 68, 68, 1)',
|
||||
borderWidth: 1
|
||||
}}]
|
||||
}},
|
||||
options: {{
|
||||
responsive: true,
|
||||
maintainAspectRatio: false,
|
||||
plugins: {{
|
||||
legend: {{ display: false }},
|
||||
title: {{ display: true, text: '各用例P95延迟 (ms)', font: {{ size: 16 }} }}
|
||||
}},
|
||||
scales: {{ y: {{ beginAtZero: true, title: {{ display: true, text: '延迟 (ms)' }} }} }}
|
||||
}}
|
||||
}});
|
||||
</script>
|
||||
</body>
|
||||
</html>"""
|
||||
@@ -0,0 +1,380 @@
|
||||
"""HTTP 服务器:提供 API + 托管前端静态文件
|
||||
|
||||
启动后访问 http://localhost:8000 即可打开测试界面。
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
import mimetypes
|
||||
import os
|
||||
import threading
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from datetime import datetime
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from . import config
|
||||
from .client_base import ClientError
|
||||
from .engine import TEST_CASES, run_all_tests
|
||||
from .ollama_client import OllamaClient, OllamaError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class RunState:
|
||||
"""测试运行状态(线程安全:所有读写通过 RunManager 加锁)"""
|
||||
status: str = "idle" # idle | running | done | stopped | failed
|
||||
run_id: str = ""
|
||||
model: str = ""
|
||||
repeats: int = 0
|
||||
skip_heavy: bool = False
|
||||
cases: list = field(default_factory=list)
|
||||
current_case_index: int = 0
|
||||
total_cases: int = 0
|
||||
current_case_name: str = ""
|
||||
message: str = "准备中"
|
||||
progress: float = 0.0
|
||||
started_at: str = ""
|
||||
finished_at: str = ""
|
||||
summary: dict = field(default_factory=dict)
|
||||
|
||||
def update(self, values=None, **kwargs) -> None:
|
||||
"""批量更新字段(引擎通过 status.update(kw) 或 status.update(**kw) 汇报进度)"""
|
||||
updates = dict(values or {})
|
||||
updates.update(kwargs)
|
||||
for key, value in updates.items():
|
||||
if hasattr(self, key):
|
||||
setattr(self, key, value)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
class RunManager:
|
||||
"""管理测试运行状态(单实例、同一时间只允许一次运行)"""
|
||||
|
||||
def __init__(self):
|
||||
self._lock = threading.Lock()
|
||||
self._stop_event = threading.Event()
|
||||
self._state = RunState()
|
||||
self._thread = None
|
||||
|
||||
@property
|
||||
def is_running(self) -> bool:
|
||||
return self._state.status == "running"
|
||||
|
||||
def start(self, cases: list, repeats: int, model: str, skip_heavy: bool) -> str:
|
||||
with self._lock:
|
||||
if self.is_running:
|
||||
raise RuntimeError("已有测试正在运行,请等待完成或先停止")
|
||||
self._stop_event.clear()
|
||||
run_id = datetime.now().strftime("%Y%m%d-%H%M%S")
|
||||
self._state = RunState(
|
||||
status="running",
|
||||
run_id=run_id,
|
||||
model=model,
|
||||
repeats=repeats,
|
||||
skip_heavy=skip_heavy,
|
||||
cases=list(cases),
|
||||
total_cases=len(cases),
|
||||
started_at=datetime.now().isoformat(),
|
||||
)
|
||||
self._thread = threading.Thread(
|
||||
target=self._run, args=(cases, repeats, model, skip_heavy), daemon=True
|
||||
)
|
||||
self._thread.start()
|
||||
return run_id
|
||||
|
||||
def stop(self) -> None:
|
||||
"""请求停止(置位事件,引擎在用例之间检查)"""
|
||||
self._stop_event.set()
|
||||
|
||||
def status(self) -> dict:
|
||||
"""返回当前运行状态(对外只读拷贝)"""
|
||||
with self._lock:
|
||||
return self._state.to_dict()
|
||||
|
||||
def _run(self, cases, repeats, model, skip_heavy):
|
||||
try:
|
||||
report = run_all_tests(
|
||||
cases=cases,
|
||||
repeats=repeats,
|
||||
model=model,
|
||||
skip_heavy=skip_heavy,
|
||||
status=self._state,
|
||||
stop_event=self._stop_event,
|
||||
)
|
||||
stopped = report["summary"].get("stopped", False)
|
||||
with self._lock:
|
||||
self._state.status = "stopped" if stopped else "done"
|
||||
self._state.progress = 100
|
||||
self._state.message = "测试已停止" if stopped else "测试完成"
|
||||
self._state.finished_at = datetime.now().isoformat()
|
||||
self._state.summary = report["summary"]
|
||||
except Exception as e:
|
||||
logger.exception("测试引擎异常")
|
||||
with self._lock:
|
||||
self._state.status = "failed"
|
||||
self._state.message = f"测试失败: {e}"
|
||||
self._state.finished_at = datetime.now().isoformat()
|
||||
|
||||
|
||||
manager = RunManager()
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
"""API + 静态文件处理"""
|
||||
|
||||
# ---------- HTTP 基础 ----------
|
||||
|
||||
def _send_json(self, obj, code=200):
|
||||
body = json.dumps(obj, ensure_ascii=False).encode("utf-8")
|
||||
self.send_response(code)
|
||||
self.send_header("Content-Type", "application/json; charset=utf-8")
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def _read_json(self) -> dict:
|
||||
try:
|
||||
length = int(self.headers.get("Content-Length", 0) or 0)
|
||||
except ValueError:
|
||||
length = 0
|
||||
if length <= 0:
|
||||
return {}
|
||||
try:
|
||||
return json.loads(self.rfile.read(length).decode("utf-8"))
|
||||
except json.JSONDecodeError:
|
||||
return {}
|
||||
|
||||
def log_message(self, format, *args):
|
||||
print(f"[API] {args[0]}")
|
||||
|
||||
# ---------- GET ----------
|
||||
|
||||
def do_GET(self):
|
||||
path = urlparse(self.path).path
|
||||
if path == "/api/health":
|
||||
self._send_json({"status": "ok"})
|
||||
elif path == "/api/config":
|
||||
self._handle_config()
|
||||
elif path == "/api/models":
|
||||
self._handle_models()
|
||||
elif path == "/api/cases":
|
||||
cases = {
|
||||
cid: {
|
||||
"name": info["name"],
|
||||
"difficulty": info["difficulty"],
|
||||
"estimated_seconds": info["estimated_seconds"],
|
||||
}
|
||||
for cid, info in TEST_CASES.items()
|
||||
}
|
||||
self._send_json(cases)
|
||||
elif path == "/api/status":
|
||||
self._send_json(manager.status())
|
||||
elif path == "/api/results/latest.json":
|
||||
self._handle_latest()
|
||||
elif path == "/api/results/list":
|
||||
self._handle_results_list()
|
||||
elif path.startswith("/api/results/"):
|
||||
self._handle_result_file(path)
|
||||
elif path == "/report.html":
|
||||
self._serve_file(os.path.join(config.REPORT_DIR, "report.html"))
|
||||
else:
|
||||
self._serve_static(path)
|
||||
|
||||
# ---------- POST ----------
|
||||
|
||||
def do_POST(self):
|
||||
path = urlparse(self.path).path
|
||||
body = self._read_json()
|
||||
if path == "/api/run":
|
||||
self._handle_run(body)
|
||||
elif path == "/api/stop":
|
||||
manager.stop()
|
||||
self._send_json({"ok": True, "message": "已请求停止"})
|
||||
else:
|
||||
self._send_json({"error": "Not found"}, 404)
|
||||
|
||||
# ---------- API 处理器 ----------
|
||||
|
||||
def _handle_config(self):
|
||||
self._send_json({
|
||||
"backend": config.LLM_BACKEND,
|
||||
"ollama_base_url": config.OLLAMA_BASE_URL,
|
||||
"openai_base_url": config.OPENAI_BASE_URL,
|
||||
"default_model": config.DEFAULT_MODEL,
|
||||
"openai_default_model": config.OPENAI_DEFAULT_MODEL,
|
||||
"concurrency": config.DEFAULT_CONCURRENCY,
|
||||
})
|
||||
|
||||
def _handle_models(self):
|
||||
try:
|
||||
if config.LLM_BACKEND == "openai":
|
||||
from .openai_client import OpenAIClient
|
||||
models = OpenAIClient().list_models()
|
||||
else:
|
||||
models = OllamaClient().list_models()
|
||||
self._send_json({"models": models, "error": None})
|
||||
except ClientError as e:
|
||||
self._send_json({"models": [], "error": str(e)})
|
||||
|
||||
def _handle_run(self, body):
|
||||
cases = body.get("cases") or []
|
||||
repeats = int(body.get("repeats", 3) or 3)
|
||||
model = body.get("model") or config.DEFAULT_MODEL
|
||||
skip_heavy = bool(body.get("skip_heavy", False))
|
||||
|
||||
# 校验用例
|
||||
unknown = [c for c in cases if c not in TEST_CASES]
|
||||
if not cases:
|
||||
self._send_json({"error": "未选择测试用例"}, 400)
|
||||
return
|
||||
if unknown:
|
||||
self._send_json({"error": f"未知用例: {', '.join(unknown)}"}, 400)
|
||||
return
|
||||
if repeats < 1 or repeats > 50:
|
||||
self._send_json({"error": "重复次数需在 1-50 之间"}, 400)
|
||||
return
|
||||
|
||||
try:
|
||||
run_id = manager.start(cases, repeats, model, skip_heavy)
|
||||
self._send_json({"run_id": run_id})
|
||||
except RuntimeError as e:
|
||||
self._send_json({"error": str(e)}, 409)
|
||||
|
||||
def _handle_latest(self):
|
||||
"""返回最新测试结果,文件损坏时优雅降级"""
|
||||
latest_path = os.path.join(config.RESULTS_DIR, "latest.json")
|
||||
if not os.path.exists(latest_path):
|
||||
self._send_json({"error": "暂无测试结果,请先运行测试"}, 404)
|
||||
return
|
||||
try:
|
||||
with open(latest_path, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
self._send_json(data)
|
||||
except (json.JSONDecodeError, OSError) as e:
|
||||
logger.error("读取 latest.json 失败: %s", e)
|
||||
self._send_json({"error": "结果文件损坏,请先运行新测试"}, 500)
|
||||
|
||||
def _handle_results_list(self):
|
||||
"""列出所有 run_*.json 结果文件"""
|
||||
import glob as glob_mod
|
||||
pattern = os.path.join(config.RESULTS_DIR, "run_*.json")
|
||||
files = sorted(glob_mod.glob(pattern))
|
||||
result = []
|
||||
for f in files:
|
||||
name = os.path.basename(f)
|
||||
try:
|
||||
with open(f, "r", encoding="utf-8") as fh:
|
||||
data = json.load(fh)
|
||||
summary = data.get("summary", {})
|
||||
result.append({
|
||||
"filename": name,
|
||||
"timestamp": data.get("timestamp", ""),
|
||||
"model": data.get("environment", {}).get("model", ""),
|
||||
"total_cases": summary.get("total_cases", 0),
|
||||
"passed": summary.get("passed", 0),
|
||||
"failed": summary.get("failed", 0),
|
||||
"elapsed_seconds": summary.get("total_elapsed_seconds", 0),
|
||||
})
|
||||
except (json.JSONDecodeError, OSError):
|
||||
result.append({"filename": name, "error": "无法读取"})
|
||||
self._send_json({"results": result})
|
||||
|
||||
def _handle_result_file(self, path):
|
||||
"""返回单个历史结果文件 /api/results/<filename>"""
|
||||
filename = path.rsplit("/", 1)[-1]
|
||||
if not filename.endswith(".json") or "/" in filename or "\\" in filename:
|
||||
self._send_json({"error": "Forbidden"}, 403)
|
||||
return
|
||||
full = os.path.join(config.RESULTS_DIR, filename)
|
||||
if not os.path.isfile(full):
|
||||
self._send_json({"error": "Not found"}, 404)
|
||||
return
|
||||
try:
|
||||
with open(full, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
self._send_json(data)
|
||||
except (json.JSONDecodeError, OSError):
|
||||
self._send_json({"error": "文件损坏"}, 500)
|
||||
|
||||
# ---------- 静态文件 ----------
|
||||
|
||||
def _serve_static(self, path):
|
||||
# 只允许从 frontend/ 目录读取
|
||||
if path == "/" or path == "/index.html":
|
||||
rel = "index.html"
|
||||
else:
|
||||
rel = path.lstrip("/")
|
||||
|
||||
full = os.path.normpath(os.path.join(config.FRONTEND_DIR, rel))
|
||||
frontend_root = os.path.normpath(config.FRONTEND_DIR)
|
||||
|
||||
# 防止路径遍历攻击
|
||||
if not os.path.abspath(full).startswith(os.path.abspath(frontend_root + os.sep)):
|
||||
self._send_json({"error": "Forbidden"}, 403)
|
||||
return
|
||||
|
||||
# 防止符号链接指向外部目录
|
||||
# 逐段检查路径中每个组件是否安全
|
||||
parts = os.path.relpath(full, frontend_root).split(os.sep)
|
||||
current = frontend_root
|
||||
for part in parts:
|
||||
current = os.path.join(current, part)
|
||||
if os.path.islink(current):
|
||||
link_target = os.path.realpath(current)
|
||||
if not link_target.startswith(os.path.realpath(frontend_root) + os.sep):
|
||||
self._send_json({"error": "Forbidden"}, 403)
|
||||
return
|
||||
if not os.path.exists(current):
|
||||
break
|
||||
|
||||
self._serve_file(full)
|
||||
|
||||
def _serve_file(self, full_path):
|
||||
if not os.path.isfile(full_path):
|
||||
self._send_json({"error": "Not found"}, 404)
|
||||
return
|
||||
content_type = mimetypes.guess_type(full_path)[0] or "application/octet-stream"
|
||||
with open(full_path, "rb") as f:
|
||||
data = f.read()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", content_type)
|
||||
self.send_header("Content-Length", str(len(data)))
|
||||
self.end_headers()
|
||||
self.wfile.write(data)
|
||||
|
||||
|
||||
def main():
|
||||
# 配置根日志
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [%(levelname)s] %(name)s: %(message)s",
|
||||
datefmt="%H:%M:%S",
|
||||
)
|
||||
|
||||
server = ThreadingHTTPServer((config.SERVER_HOST, config.SERVER_PORT), Handler)
|
||||
backend = config.LLM_BACKEND
|
||||
print("=" * 60)
|
||||
print(f"大模型速度测试助手 v2.1 ({'OpenAI 兼容' if backend == 'openai' else 'Ollama'})")
|
||||
print("=" * 60)
|
||||
if backend == "openai":
|
||||
print(f"服务地址 : {config.OPENAI_BASE_URL}")
|
||||
print(f"默认模型 : {config.OPENAI_DEFAULT_MODEL}")
|
||||
print(f"并发测试 : {config.DEFAULT_CONCURRENCY} 并发")
|
||||
else:
|
||||
print(f"Ollama 地址 : {config.OLLAMA_BASE_URL}")
|
||||
print(f"默认模型 : {config.DEFAULT_MODEL}")
|
||||
print(f"前端界面 : http://localhost:{config.SERVER_PORT}")
|
||||
print("按 Ctrl+C 停止服务器")
|
||||
print("=" * 60)
|
||||
try:
|
||||
server.serve_forever()
|
||||
except KeyboardInterrupt:
|
||||
print("\n服务器已停止")
|
||||
server.server_close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,61 @@
|
||||
"""统计计算工具"""
|
||||
import math
|
||||
from typing import List
|
||||
|
||||
|
||||
def _percentile_at(sorted_data: list[float], pct: float) -> float:
|
||||
"""在线性插值下,返回排序后数据的指定百分位值。
|
||||
|
||||
注意:调用方必须确保 data 已排序。
|
||||
"""
|
||||
n = len(sorted_data)
|
||||
if n == 0:
|
||||
return 0.0
|
||||
k = (pct / 100.0) * (n - 1)
|
||||
f = math.floor(k)
|
||||
c = math.ceil(k)
|
||||
if f == c:
|
||||
return sorted_data[int(k)]
|
||||
d0 = sorted_data[int(f)] * (c - k)
|
||||
d1 = sorted_data[int(c)] * (k - f)
|
||||
return d0 + d1
|
||||
|
||||
|
||||
def percentile(data: List[float], pct: float) -> float:
|
||||
"""计算指定百分位数的值(线性插值)
|
||||
|
||||
空数据返回 0.0,调用方应通过检查 count 来判断是否有数据。
|
||||
"""
|
||||
return _percentile_at(sorted(data), pct)
|
||||
|
||||
|
||||
def percentiles(data: List[float], pcts: List[float]) -> dict:
|
||||
"""单次排序,批量计算多个百分位,返回 {pct: value}。"""
|
||||
sorted_data = sorted(data)
|
||||
return {p: _percentile_at(sorted_data, p) for p in pcts}
|
||||
|
||||
|
||||
def mean(data: List[float]) -> float:
|
||||
"""算术平均值"""
|
||||
if not data:
|
||||
return 0.0
|
||||
return sum(data) / len(data)
|
||||
|
||||
|
||||
def std(data: List[float]) -> float:
|
||||
"""样本标准差(n-1)"""
|
||||
if len(data) < 2:
|
||||
return 0.0
|
||||
m = mean(data)
|
||||
variance = sum((x - m) ** 2 for x in data) / (len(data) - 1)
|
||||
return math.sqrt(variance)
|
||||
|
||||
|
||||
def min_val(data: List[float]) -> float:
|
||||
"""最小值"""
|
||||
return min(data) if data else 0.0
|
||||
|
||||
|
||||
def max_val(data: List[float]) -> float:
|
||||
"""最大值"""
|
||||
return max(data) if data else 0.0
|
||||
@@ -0,0 +1 @@
|
||||
"""测试用例包:每个用例实现 run_test(client, repeats, stop_event)"""
|
||||
@@ -0,0 +1,32 @@
|
||||
"""测试用例公共工具"""
|
||||
from ..ollama_client import InferenceResult
|
||||
|
||||
|
||||
def make_result(infer: InferenceResult, iteration: int, **extra) -> dict:
|
||||
"""将一次推理结果转换为用例数据条目"""
|
||||
item = {
|
||||
"iteration": iteration,
|
||||
"ttft_ms": round(infer.ttft_ms, 2),
|
||||
"prefill_ms": round(infer.prefill_ms, 2),
|
||||
"decode_speed_tok_s": round(infer.decode_speed_tok_s, 2),
|
||||
"total_tokens": infer.total_tokens,
|
||||
"prompt_tokens": infer.prompt_tokens,
|
||||
"completion_tokens": infer.completion_tokens,
|
||||
"e2e_ms": round(infer.e2e_ms, 2),
|
||||
"elapsed_ms": round(infer.e2e_ms, 2),
|
||||
"response_length": len(infer.response),
|
||||
}
|
||||
item.update(extra)
|
||||
return item
|
||||
|
||||
|
||||
def case_meta(case_id: str, name: str, description: str,
|
||||
difficulty: str, estimated_seconds: int) -> dict:
|
||||
"""构造用例元数据"""
|
||||
return {
|
||||
"case_id": case_id,
|
||||
"name": name,
|
||||
"description": description,
|
||||
"difficulty": difficulty,
|
||||
"estimated_seconds": estimated_seconds,
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
"""用例1:纯文本生成基准 —— 基础生成速度基准测试"""
|
||||
from ..ollama_client import OllamaClient
|
||||
from .base import case_meta, make_result
|
||||
|
||||
PROMPT = "请用200字左右概述人工智能从诞生至今的关键发展阶段,每段不超过50字。"
|
||||
|
||||
|
||||
def run_test(client: OllamaClient, repeats: int = 3, stop_event=None) -> dict:
|
||||
"""执行纯文本生成测试"""
|
||||
results = []
|
||||
for i in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
infer = client.generate(PROMPT, options={"temperature": 0.7, "num_predict": 300})
|
||||
results.append(make_result(infer, i + 1))
|
||||
|
||||
meta = case_meta(
|
||||
"case_01_generation", "纯文本生成基准",
|
||||
"基础生成速度基准测试", "简单", 5,
|
||||
)
|
||||
meta["results"] = results
|
||||
meta["prompt_used"] = PROMPT
|
||||
return meta
|
||||
@@ -0,0 +1,41 @@
|
||||
"""用例2:简单工具调用延迟 —— 测量结构化工具输出的调用延迟"""
|
||||
from ..ollama_client import OllamaClient
|
||||
from .base import case_meta, make_result
|
||||
|
||||
# 模拟"工具调用"场景:要求模型只输出结构化结果(短输出、低复杂度)
|
||||
SYSTEM = (
|
||||
"你是一个工具调用助手。你只能输出一个JSON对象,格式为 "
|
||||
'{"tool": "工具名", "params": {"参数名": "参数值"}},不要输出任何其它文字。'
|
||||
)
|
||||
|
||||
TOOL_PROMPTS = [
|
||||
("查询当前日期", {"tool": "get_date", "params": {}}),
|
||||
("获取CPU使用率", {"tool": "get_cpu_usage", "params": {}}),
|
||||
("列出当前目录文件", {"tool": "list_files", "params": {"path": "/tmp"}}),
|
||||
]
|
||||
|
||||
|
||||
def run_test(client: OllamaClient, repeats: int = 3, stop_event=None) -> dict:
|
||||
"""执行简单工具调用延迟测试"""
|
||||
results = []
|
||||
for i in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
desc, expected = TOOL_PROMPTS[i % len(TOOL_PROMPTS)]
|
||||
infer = client.generate(
|
||||
f"请调用工具:{desc}",
|
||||
system=SYSTEM,
|
||||
options={"temperature": 0.0, "num_predict": 60},
|
||||
)
|
||||
results.append(make_result(
|
||||
infer, i + 1,
|
||||
tool=expected["tool"],
|
||||
matched=(expected["tool"] in infer.response),
|
||||
))
|
||||
|
||||
meta = case_meta(
|
||||
"case_02_simple_tool", "简单工具调用延迟",
|
||||
"测量简单工具调用的额外延迟", "简单", 3,
|
||||
)
|
||||
meta["results"] = results
|
||||
return meta
|
||||
@@ -0,0 +1,72 @@
|
||||
"""用例3:文件读取+分析 —— 测试文件写入、读取、分析的完整链路"""
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
from ..ollama_client import OllamaClient
|
||||
from .base import case_meta, make_result
|
||||
|
||||
SAMPLE_CODE = '''"""示例模块:一个简单的计算器"""
|
||||
class Calculator:
|
||||
"""四则运算计算器"""
|
||||
|
||||
def __init__(self):
|
||||
self.history = []
|
||||
|
||||
def add(self, a, b):
|
||||
self.history.append(("add", a, b))
|
||||
return a + b
|
||||
|
||||
def divide(self, a, b):
|
||||
if b == 0:
|
||||
raise ValueError("除数不能为零")
|
||||
self.history.append(("divide", a, b))
|
||||
return a / b
|
||||
|
||||
|
||||
def main():
|
||||
calc = Calculator()
|
||||
print(calc.add(1, 2))
|
||||
print(calc.divide(10, 2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
'''
|
||||
|
||||
ANALYSIS_PROMPT = "请分析以下Python代码的结构、功能,并指出其中可能存在的问题:\n\n```python\n{code}\n```"
|
||||
|
||||
|
||||
def run_test(client: OllamaClient, repeats: int = 3, stop_event=None) -> dict:
|
||||
"""执行文件读取+分析测试"""
|
||||
results = []
|
||||
for i in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
|
||||
# 写入临时文件,模拟"生成测试文件"环节
|
||||
fd, test_path = tempfile.mkstemp(prefix="llm_speed_test_", suffix=".py")
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as f:
|
||||
f.write(SAMPLE_CODE)
|
||||
|
||||
# 读取文件内容作为模型输入
|
||||
with open(test_path, "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
finally:
|
||||
try:
|
||||
os.remove(test_path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
infer = client.generate(
|
||||
ANALYSIS_PROMPT.format(code=content),
|
||||
options={"temperature": 0.3, "num_predict": 300},
|
||||
)
|
||||
results.append(make_result(infer, i + 1, file_chars=len(content)))
|
||||
|
||||
meta = case_meta(
|
||||
"case_03_file_analysis", "文件读取+分析",
|
||||
"文件读取与分析完整链路计时", "中等", 2,
|
||||
)
|
||||
meta["results"] = results
|
||||
return meta
|
||||
@@ -0,0 +1,64 @@
|
||||
"""用例4:并行工具调用 —— 对比串行 vs 并行的加速比"""
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
|
||||
from ..ollama_client import OllamaClient
|
||||
from .base import case_meta
|
||||
|
||||
# 三个独立、互不依赖的"工具调用"请求
|
||||
PARALLEL_PROMPTS = [
|
||||
"请用一句话说明今天的天气如何。",
|
||||
"请用一句话说明Python的优缺点。",
|
||||
"请用一句话说明如何提高代码质量。",
|
||||
]
|
||||
|
||||
NUM_PARALLEL = len(PARALLEL_PROMPTS)
|
||||
|
||||
|
||||
def run_test(client: OllamaClient, repeats: int = 3, stop_event=None) -> dict:
|
||||
"""执行串行 vs 并行对比测试"""
|
||||
results = []
|
||||
|
||||
def call_one(prompt: str):
|
||||
return client.generate(prompt, options={"temperature": 0.3, "num_predict": 80})
|
||||
|
||||
# 线程池在循环外创建,复用 worker 线程
|
||||
with ThreadPoolExecutor(max_workers=NUM_PARALLEL) as executor:
|
||||
for i in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
|
||||
# 串行:依次执行
|
||||
serial_start = time.perf_counter()
|
||||
serial_results = [call_one(p) for p in PARALLEL_PROMPTS]
|
||||
serial_ms = (time.perf_counter() - serial_start) * 1000
|
||||
|
||||
# 并行:线程池并发执行
|
||||
parallel_start = time.perf_counter()
|
||||
parallel_results = list(executor.map(call_one, PARALLEL_PROMPTS))
|
||||
parallel_ms = (time.perf_counter() - parallel_start) * 1000
|
||||
|
||||
speedup = (serial_ms / parallel_ms) if parallel_ms > 0 else 0.0
|
||||
|
||||
avg_ttft = sum(r.ttft_ms for r in parallel_results) / NUM_PARALLEL
|
||||
avg_decode = sum(r.decode_speed_tok_s for r in parallel_results) / NUM_PARALLEL
|
||||
|
||||
results.append({
|
||||
"iteration": i + 1,
|
||||
"parallel_tasks": NUM_PARALLEL,
|
||||
"serial_ms": round(serial_ms, 2),
|
||||
"parallel_ms": round(parallel_ms, 2),
|
||||
"speedup_ratio": round(speedup, 2),
|
||||
"elapsed_ms": round(parallel_ms, 2), # 用并行耗时作为该用例的代表延迟
|
||||
"avg_ttft_ms": round(avg_ttft, 2),
|
||||
"avg_decode_speed_tok_s": round(avg_decode, 2),
|
||||
"ttft_ms": round(avg_ttft, 2),
|
||||
"decode_speed_tok_s": round(avg_decode, 2),
|
||||
})
|
||||
|
||||
meta = case_meta(
|
||||
"case_04_parallel_tool", "并行工具调用",
|
||||
"对比串行 vs 并行工具调用的延迟差异", "中等", 5,
|
||||
)
|
||||
meta["results"] = results
|
||||
return meta
|
||||
@@ -0,0 +1,45 @@
|
||||
"""用例5:长上下文处理 —— 测试不同上下文大小下的处理延迟"""
|
||||
from ..ollama_client import OllamaClient
|
||||
from .base import case_meta, make_result
|
||||
|
||||
# 约 15 个 token/句的填充句,用于按目标 token 数拼接上下文
|
||||
FILLER_SENTENCE = "这是一个用于测试长上下文处理能力的示例句子,包含一些常见的词汇和表述方式。\n"
|
||||
TOKENS_PER_SENTENCE = 15
|
||||
|
||||
CONTEXT_SIZES = [500, 1000, 2000, 4000] # 目标 prompt token 数
|
||||
|
||||
QUESTION = "请用一句话总结:上面这段文本主要介绍了什么?"
|
||||
|
||||
|
||||
def _build_context(target_tokens: int) -> str:
|
||||
"""按目标 token 数近似生成一段填充文本"""
|
||||
n_sentences = max(1, target_tokens // TOKENS_PER_SENTENCE)
|
||||
return FILLER_SENTENCE * n_sentences
|
||||
|
||||
|
||||
def run_test(client: OllamaClient, repeats: int = 2, stop_event=None) -> dict:
|
||||
"""执行长上下文处理测试"""
|
||||
results = []
|
||||
for context_size in CONTEXT_SIZES:
|
||||
for r in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
context = _build_context(context_size)
|
||||
prompt = f"{context}\n\n{QUESTION}"
|
||||
infer = client.generate(
|
||||
prompt,
|
||||
options={"temperature": 0.3, "num_predict": 80},
|
||||
)
|
||||
results.append(make_result(
|
||||
infer, r + 1,
|
||||
context_tokens_target=context_size,
|
||||
context_chars=len(context),
|
||||
))
|
||||
|
||||
meta = case_meta(
|
||||
"case_05_long_context", "长上下文处理",
|
||||
"不同上下文长度对延迟的影响", "耗时", 10,
|
||||
)
|
||||
meta["results"] = results
|
||||
meta["context_sizes_tested"] = CONTEXT_SIZES
|
||||
return meta
|
||||
@@ -0,0 +1,33 @@
|
||||
"""用例6:复杂推理任务 —— 对比不同难度推理任务的延迟差异"""
|
||||
from ..ollama_client import OllamaClient
|
||||
from .base import case_meta, make_result
|
||||
|
||||
REASONING_LEVELS = [
|
||||
("简单问答", "什么是人工智能?", 80),
|
||||
("中等推理", "解释Transformer架构中的自注意力机制是如何工作的。", 200),
|
||||
("复杂推理", "分析大语言模型在长链推理中的主要局限性,并提出三种改进方案。", 400),
|
||||
]
|
||||
|
||||
|
||||
def run_test(client: OllamaClient, repeats: int = 2, stop_event=None) -> dict:
|
||||
"""执行复杂推理任务测试"""
|
||||
results = []
|
||||
for level_name, prompt, num_predict in REASONING_LEVELS:
|
||||
for i in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
infer = client.generate(
|
||||
prompt,
|
||||
options={"temperature": 0.3, "num_predict": num_predict},
|
||||
)
|
||||
results.append(make_result(
|
||||
infer, i + 1,
|
||||
reasoning_level=level_name,
|
||||
))
|
||||
|
||||
meta = case_meta(
|
||||
"case_06_reasoning", "复杂推理任务",
|
||||
"不同推理复杂度下的延迟对比", "耗时", 15,
|
||||
)
|
||||
meta["results"] = results
|
||||
return meta
|
||||
@@ -0,0 +1,49 @@
|
||||
"""用例7:多轮对话累积 —— 追踪连续多轮对话中延迟随上下文累积的变化"""
|
||||
from ..ollama_client import OllamaClient
|
||||
from .base import case_meta, make_result
|
||||
|
||||
DEBUG_SCENARIOS = [
|
||||
"修复Python中的索引越界错误",
|
||||
"修复JavaScript闭包变量捕获问题",
|
||||
"修复SQL注入漏洞",
|
||||
"修复多线程竞态条件",
|
||||
"修复递归栈溢出",
|
||||
"修复正则表达式灾难回溯",
|
||||
]
|
||||
|
||||
|
||||
def run_test(client: OllamaClient, repeats: int = 1, stop_event=None) -> dict:
|
||||
"""执行多轮对话累积测试(使用 /api/chat 累积上下文)
|
||||
|
||||
repeats 控制总轮次数,每轮切换不同调试场景。
|
||||
"""
|
||||
results = []
|
||||
messages = []
|
||||
|
||||
for i in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
scenario = DEBUG_SCENARIOS[i % len(DEBUG_SCENARIOS)]
|
||||
|
||||
user_msg = f"(对话第{i + 1}轮)请协助:{scenario}。请简要给出你的分析。"
|
||||
messages.append({"role": "user", "content": user_msg})
|
||||
|
||||
infer = client.chat(messages, options={"temperature": 0.3, "num_predict": 120})
|
||||
|
||||
assistant_reply = infer.response
|
||||
messages.append({"role": "assistant", "content": assistant_reply})
|
||||
|
||||
results.append(make_result(
|
||||
infer, i + 1,
|
||||
round=i + 1,
|
||||
scenario=scenario,
|
||||
messages_in_context=len(messages),
|
||||
))
|
||||
|
||||
meta = case_meta(
|
||||
"case_07_multiturn", "多轮对话累积",
|
||||
"追踪连续多轮对话中延迟随上下文累积的变化", "耗时", 10,
|
||||
)
|
||||
meta["results"] = results
|
||||
meta["total_rounds"] = len(results)
|
||||
return meta
|
||||
@@ -0,0 +1,146 @@
|
||||
"""用例8:并发压力测试 —— 测量高并发下的 TPS 保持率、P95/P99 延迟与错误率
|
||||
|
||||
并发数取 config.DEFAULT_CONCURRENCY(默认 100),repeats 控制并发批次数量。
|
||||
"""
|
||||
import concurrent.futures
|
||||
import logging
|
||||
|
||||
from .. import config
|
||||
from ..stats import percentile
|
||||
from .base import case_meta, make_result
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
CONCURRENT_PROMPT = "用一句话回答:1+1等于几?"
|
||||
BASELINE_SAMPLES = 5 # 基准串行采样次数
|
||||
|
||||
|
||||
def _run_one(client) -> dict:
|
||||
"""单次并发请求,失败时返回 None 标记"""
|
||||
try:
|
||||
infer = client.generate(
|
||||
CONCURRENT_PROMPT,
|
||||
options={"temperature": 0.0, "num_predict": 32},
|
||||
)
|
||||
return {
|
||||
"ok": True,
|
||||
"infer": infer,
|
||||
"elapsed_ms": infer.e2e_ms,
|
||||
}
|
||||
except Exception as e:
|
||||
logger.debug("并发请求失败: %s", e)
|
||||
return {"ok": False, "error": str(e)}
|
||||
|
||||
|
||||
def _collect_serial(client, n) -> list:
|
||||
"""串行基准采样"""
|
||||
entries = []
|
||||
for i in range(n):
|
||||
r = _run_one(client)
|
||||
if r["ok"]:
|
||||
entries.append(r)
|
||||
return entries
|
||||
|
||||
|
||||
def _collect_concurrent(client, n, stop_event=None) -> list:
|
||||
"""并发采集(线程池)"""
|
||||
entries = []
|
||||
completed = 0
|
||||
|
||||
def _wrapped():
|
||||
return _run_one(client)
|
||||
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=n) as executor:
|
||||
futures = [executor.submit(_wrapped) for _ in range(n)]
|
||||
for fut in concurrent.futures.as_completed(futures):
|
||||
try:
|
||||
r = fut.result(timeout=300)
|
||||
except Exception as e:
|
||||
r = {"ok": False, "error": str(e)}
|
||||
entries.append(r)
|
||||
completed += 1
|
||||
if stop_event and stop_event.is_set():
|
||||
for f in futures:
|
||||
f.cancel()
|
||||
break
|
||||
|
||||
return entries
|
||||
|
||||
|
||||
def run_test(client, repeats: int = 1, stop_event=None) -> dict:
|
||||
"""执行并发压力测试"""
|
||||
if repeats > 3:
|
||||
repeats = 3 # 并发批次过多会非常耗时,限制上限
|
||||
|
||||
n_concurrent = config.DEFAULT_CONCURRENCY
|
||||
|
||||
# ---- Step 1: 串行基准 ----
|
||||
logger.info("[并发] 计算基准 TPS (串行 %d 次)...", BASELINE_SAMPLES)
|
||||
baseline = _collect_serial(client, BASELINE_SAMPLES)
|
||||
valid_base = [e for e in baseline if e["ok"]]
|
||||
if not valid_base:
|
||||
return case_meta(
|
||||
"case_08_concurrency", "并发压力测试",
|
||||
"基准测试全部失败", "压力", 30,
|
||||
) | {"status": "failed", "error": "基准测试全部失败", "results": []}
|
||||
|
||||
base_tps = sum(e["infer"].decode_speed_tok_s for e in valid_base) / len(valid_base)
|
||||
base_e2e = sorted(e["elapsed_ms"] for e in valid_base)
|
||||
base_p50 = percentile(base_e2e, 50)
|
||||
logger.info("[并发] 基准 TPS: %.2f tok/s, E2E P50: %.0fms", base_tps, base_p50)
|
||||
|
||||
# ---- Step 2: 并发测试 ----
|
||||
logger.info("[并发] 启动 %d 并发请求...", n_concurrent)
|
||||
all_concurrent = []
|
||||
for batch in range(repeats):
|
||||
if stop_event and stop_event.is_set():
|
||||
break
|
||||
batch_results = _collect_concurrent(client, n_concurrent, stop_event)
|
||||
all_concurrent.extend(batch_results)
|
||||
ok_count = sum(1 for r in batch_results if r["ok"])
|
||||
logger.info("[并发] 批次 %d 完成: %d/%d 成功", batch + 1, ok_count, len(batch_results))
|
||||
|
||||
# ---- Step 3: 汇总 ----
|
||||
valid_con = [e for e in all_concurrent if e["ok"]]
|
||||
failed_count = len(all_concurrent) - len(valid_con)
|
||||
|
||||
con_tps = (
|
||||
sum(e["infer"].decode_speed_tok_s for e in valid_con) / len(valid_con)
|
||||
if valid_con else 0.0
|
||||
)
|
||||
e2e_vals = sorted(e["elapsed_ms"] for e in valid_con)
|
||||
e2e_p50 = percentile(e2e_vals, 50)
|
||||
e2e_p95 = percentile(e2e_vals, 95)
|
||||
e2e_p99 = percentile(e2e_vals, 99)
|
||||
|
||||
tps_keep = (con_tps / base_tps * 100) if base_tps > 0 else 0.0
|
||||
total = len(all_concurrent) or 1
|
||||
error_rate = failed_count / total * 100
|
||||
|
||||
# 构造标准结果条目(供引擎统计/前端展示),并附加并发摘要
|
||||
results = [
|
||||
make_result(
|
||||
e["infer"], i + 1,
|
||||
batch="concurrent",
|
||||
error=("" if e["ok"] else e.get("error", "unknown")),
|
||||
)
|
||||
for i, e in enumerate(all_concurrent)
|
||||
if e["ok"]
|
||||
]
|
||||
|
||||
meta = case_meta(
|
||||
"case_08_concurrency", "并发压力测试",
|
||||
"高并发下的 TPS 保持率、P95/P99 延迟、错误率", "压力", 30,
|
||||
)
|
||||
meta["results"] = results
|
||||
meta["concurrency"] = n_concurrent
|
||||
meta["baseline_tps"] = round(base_tps, 2)
|
||||
meta["concurrent_tps"] = round(con_tps, 2)
|
||||
meta["tps_keep_rate_pct"] = round(tps_keep, 2)
|
||||
meta["e2e_p50_ms"] = round(e2e_p50, 2)
|
||||
meta["e2e_p95_ms"] = round(e2e_p95, 2)
|
||||
meta["e2e_p99_ms"] = round(e2e_p99, 2)
|
||||
meta["success_count"] = len(valid_con)
|
||||
meta["failed_count"] = failed_count
|
||||
meta["error_rate_pct"] = round(error_rate, 2)
|
||||
return meta
|
||||
@@ -0,0 +1,440 @@
|
||||
/* 速度测试助手 UI 样式 */
|
||||
:root {
|
||||
--primary: #6366f1;
|
||||
--primary-dark: #4f46e5;
|
||||
--success: #22c55e;
|
||||
--warning: #f59e0b;
|
||||
--danger: #ef4444;
|
||||
--gray-50: #f8fafc;
|
||||
--gray-100: #f1f5f9;
|
||||
--gray-200: #e2e8f0;
|
||||
--gray-300: #cbd5e1;
|
||||
--gray-600: #475569;
|
||||
--gray-700: #334155;
|
||||
--gray-800: #1e293b;
|
||||
--gray-900: #0f172a;
|
||||
--radius: 8px;
|
||||
--shadow: 0 1px 3px rgba(0,0,0,0.1), 0 1px 2px rgba(0,0,0,0.06);
|
||||
--shadow-lg: 0 10px 15px -3px rgba(0,0,0,0.1);
|
||||
}
|
||||
|
||||
* { margin: 0; padding: 0; box-sizing: border-box; }
|
||||
|
||||
body {
|
||||
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', 'PingFang SC', 'Microsoft YaHei', sans-serif;
|
||||
background: var(--gray-50);
|
||||
color: var(--gray-800);
|
||||
line-height: 1.6;
|
||||
}
|
||||
|
||||
.app-container {
|
||||
display: grid;
|
||||
grid-template-columns: 320px 1fr;
|
||||
min-height: 100vh;
|
||||
}
|
||||
|
||||
/* 左侧面板 */
|
||||
.sidebar {
|
||||
background: white;
|
||||
border-right: 1px solid var(--gray-200);
|
||||
padding: 20px;
|
||||
overflow-y: auto;
|
||||
height: 100vh;
|
||||
position: sticky;
|
||||
top: 0;
|
||||
}
|
||||
|
||||
.sidebar-header {
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 15px;
|
||||
border-bottom: 2px solid var(--gray-200);
|
||||
}
|
||||
|
||||
.sidebar-header h1 {
|
||||
font-size: 18px;
|
||||
color: var(--primary);
|
||||
margin-bottom: 5px;
|
||||
}
|
||||
|
||||
.sidebar-header .version {
|
||||
font-size: 12px;
|
||||
color: var(--gray-600);
|
||||
}
|
||||
|
||||
.panel {
|
||||
margin-bottom: 20px;
|
||||
}
|
||||
|
||||
.panel-title {
|
||||
font-size: 14px;
|
||||
font-weight: 600;
|
||||
color: var(--gray-700);
|
||||
margin-bottom: 10px;
|
||||
text-transform: uppercase;
|
||||
letter-spacing: 0.5px;
|
||||
}
|
||||
|
||||
.test-case-item {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
padding: 10px;
|
||||
margin-bottom: 6px;
|
||||
border-radius: var(--radius);
|
||||
cursor: pointer;
|
||||
transition: background 0.2s;
|
||||
border: 1px solid transparent;
|
||||
}
|
||||
|
||||
.test-case-item:hover {
|
||||
background: var(--gray-100);
|
||||
}
|
||||
|
||||
.test-case-item input[type="checkbox"] {
|
||||
margin-right: 10px;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.test-case-info {
|
||||
flex: 1;
|
||||
}
|
||||
|
||||
.test-case-name {
|
||||
font-size: 13px;
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
.test-case-diff {
|
||||
font-size: 11px;
|
||||
color: var(--gray-600);
|
||||
margin-top: 2px;
|
||||
}
|
||||
|
||||
.diff-简单 { color: var(--success); }
|
||||
.diff-中等 { color: var(--warning); }
|
||||
.diff-耗时 { color: var(--danger); }
|
||||
|
||||
.quick-actions {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
margin-bottom: 15px;
|
||||
}
|
||||
|
||||
.quick-actions button {
|
||||
flex: 1;
|
||||
padding: 6px 10px;
|
||||
font-size: 12px;
|
||||
border: 1px solid var(--gray-300);
|
||||
border-radius: var(--radius);
|
||||
background: white;
|
||||
cursor: pointer;
|
||||
transition: all 0.2s;
|
||||
}
|
||||
|
||||
.quick-actions button:hover {
|
||||
background: var(--gray-100);
|
||||
}
|
||||
|
||||
.config-group {
|
||||
margin-bottom: 15px;
|
||||
}
|
||||
|
||||
.config-group label {
|
||||
display: block;
|
||||
font-size: 13px;
|
||||
font-weight: 500;
|
||||
margin-bottom: 5px;
|
||||
color: var(--gray-700);
|
||||
}
|
||||
|
||||
.config-group input, .config-group select {
|
||||
width: 100%;
|
||||
padding: 8px 12px;
|
||||
border: 1px solid var(--gray-300);
|
||||
border-radius: var(--radius);
|
||||
font-size: 13px;
|
||||
}
|
||||
|
||||
.config-group input:focus, .config-group select:focus {
|
||||
outline: none;
|
||||
border-color: var(--primary);
|
||||
box-shadow: 0 0 0 3px rgba(99, 102, 241, 0.1);
|
||||
}
|
||||
|
||||
.btn {
|
||||
display: block;
|
||||
width: 100%;
|
||||
padding: 12px;
|
||||
border: none;
|
||||
border-radius: var(--radius);
|
||||
font-size: 15px;
|
||||
font-weight: 600;
|
||||
cursor: pointer;
|
||||
transition: all 0.2s;
|
||||
margin-bottom: 10px;
|
||||
}
|
||||
|
||||
.btn-primary {
|
||||
background: var(--primary);
|
||||
color: white;
|
||||
}
|
||||
|
||||
.btn-primary:hover { background: var(--primary-dark); }
|
||||
|
||||
.btn-secondary {
|
||||
background: var(--gray-100);
|
||||
color: var(--gray-700);
|
||||
}
|
||||
|
||||
.btn-secondary:hover { background: var(--gray-200); }
|
||||
|
||||
.btn:disabled {
|
||||
opacity: 0.5;
|
||||
cursor: not-allowed;
|
||||
}
|
||||
|
||||
/* 右侧主区域 */
|
||||
.main-content {
|
||||
padding: 20px 30px;
|
||||
overflow-y: auto;
|
||||
}
|
||||
|
||||
.status-panel {
|
||||
background: white;
|
||||
border-radius: var(--radius);
|
||||
padding: 20px;
|
||||
margin-bottom: 20px;
|
||||
box-shadow: var(--shadow);
|
||||
display: none;
|
||||
}
|
||||
|
||||
.status-panel.active {
|
||||
display: block;
|
||||
}
|
||||
|
||||
.status-header {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
margin-bottom: 15px;
|
||||
}
|
||||
|
||||
.status-text {
|
||||
font-size: 16px;
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
.status-text.running { color: var(--primary); }
|
||||
.status-text.complete { color: var(--success); }
|
||||
.status-text.error { color: var(--danger); }
|
||||
|
||||
.progress-bar-container {
|
||||
width: 100%;
|
||||
height: 8px;
|
||||
background: var(--gray-200);
|
||||
border-radius: 4px;
|
||||
overflow: hidden;
|
||||
margin-bottom: 15px;
|
||||
}
|
||||
|
||||
.progress-bar {
|
||||
height: 100%;
|
||||
background: linear-gradient(90deg, var(--primary), #8b5cf6);
|
||||
border-radius: 4px;
|
||||
transition: width 0.3s ease;
|
||||
width: 0%;
|
||||
}
|
||||
|
||||
.step-list {
|
||||
list-style: none;
|
||||
padding: 0;
|
||||
}
|
||||
|
||||
.step-item {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
padding: 8px 0;
|
||||
font-size: 13px;
|
||||
color: var(--gray-600);
|
||||
}
|
||||
|
||||
.step-item.completed { color: var(--success); }
|
||||
.step-item.active { color: var(--primary); font-weight: 500; }
|
||||
|
||||
.step-icon {
|
||||
width: 20px;
|
||||
margin-right: 10px;
|
||||
text-align: center;
|
||||
}
|
||||
|
||||
/* 结果区域 */
|
||||
.results-area {
|
||||
display: none;
|
||||
}
|
||||
|
||||
.results-area.visible {
|
||||
display: block;
|
||||
}
|
||||
|
||||
.report-card {
|
||||
background: white;
|
||||
border-radius: var(--radius);
|
||||
padding: 25px;
|
||||
margin-bottom: 20px;
|
||||
box-shadow: var(--shadow);
|
||||
}
|
||||
|
||||
.report-card h2 {
|
||||
font-size: 18px;
|
||||
margin-bottom: 15px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--gray-200);
|
||||
}
|
||||
|
||||
.metric-grid {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(auto-fit, minmax(180px, 1fr));
|
||||
gap: 15px;
|
||||
margin-bottom: 20px;
|
||||
}
|
||||
|
||||
.metric-card {
|
||||
background: var(--gray-50);
|
||||
padding: 15px;
|
||||
border-radius: var(--radius);
|
||||
border-left: 4px solid var(--primary);
|
||||
}
|
||||
|
||||
.metric-card .label {
|
||||
font-size: 12px;
|
||||
color: var(--gray-600);
|
||||
text-transform: uppercase;
|
||||
}
|
||||
|
||||
.metric-card .value {
|
||||
font-size: 24px;
|
||||
font-weight: bold;
|
||||
margin-top: 5px;
|
||||
color: var(--gray-800);
|
||||
}
|
||||
|
||||
.metric-card-value-desc {
|
||||
font-size: 11px;
|
||||
color: var(--gray-600);
|
||||
margin-top: 4px;
|
||||
}
|
||||
|
||||
.metric-description {
|
||||
background: var(--gray-50);
|
||||
padding: 15px;
|
||||
border-radius: var(--radius);
|
||||
margin-top: 15px;
|
||||
font-size: 13px;
|
||||
line-height: 1.8;
|
||||
color: var(--gray-700);
|
||||
}
|
||||
|
||||
.metric-description p {
|
||||
margin-bottom: 8px;
|
||||
}
|
||||
|
||||
.metric-description p:last-child {
|
||||
margin-bottom: 0;
|
||||
}
|
||||
|
||||
.metric-description strong {
|
||||
color: var(--gray-900);
|
||||
}
|
||||
|
||||
.metric-card.success { border-left-color: var(--success); }
|
||||
.metric-card.warning { border-left-color: var(--warning); }
|
||||
.metric-card.danger { border-left-color: var(--danger); }
|
||||
|
||||
table.data-table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
}
|
||||
|
||||
table.data-table th,
|
||||
table.data-table td {
|
||||
padding: 10px 12px;
|
||||
text-align: left;
|
||||
border-bottom: 1px solid var(--gray-200);
|
||||
font-size: 13px;
|
||||
}
|
||||
|
||||
table.data-table th {
|
||||
background: var(--gray-100);
|
||||
font-weight: 600;
|
||||
font-size: 12px;
|
||||
text-transform: uppercase;
|
||||
color: var(--gray-600);
|
||||
}
|
||||
|
||||
table.data-table tr:hover {
|
||||
background: var(--gray-50);
|
||||
}
|
||||
|
||||
.mono {
|
||||
font-family: 'SF Mono', 'Fira Code', 'Consolas', monospace;
|
||||
font-size: 12px;
|
||||
}
|
||||
|
||||
.chart-wrapper {
|
||||
position: relative;
|
||||
height: 350px;
|
||||
margin: 20px 0;
|
||||
}
|
||||
|
||||
.action-buttons {
|
||||
display: flex;
|
||||
gap: 10px;
|
||||
margin-top: 20px;
|
||||
}
|
||||
|
||||
.action-buttons button {
|
||||
padding: 10px 20px;
|
||||
border: 1px solid var(--gray-300);
|
||||
border-radius: var(--radius);
|
||||
background: white;
|
||||
cursor: pointer;
|
||||
font-size: 13px;
|
||||
font-weight: 500;
|
||||
transition: all 0.2s;
|
||||
}
|
||||
|
||||
.action-buttons button:hover {
|
||||
background: var(--gray-100);
|
||||
}
|
||||
|
||||
.action-buttons button.primary {
|
||||
background: var(--primary);
|
||||
color: white;
|
||||
border-color: var(--primary);
|
||||
}
|
||||
|
||||
.action-buttons button.primary:hover {
|
||||
background: var(--primary-dark);
|
||||
}
|
||||
|
||||
/* 错误提示 */
|
||||
.error-banner {
|
||||
background: #fef2f2;
|
||||
border: 1px solid #fecaca;
|
||||
border-left: 4px solid var(--danger);
|
||||
color: #b91c1c;
|
||||
padding: 12px 15px;
|
||||
border-radius: var(--radius);
|
||||
margin-bottom: 15px;
|
||||
font-size: 13px;
|
||||
}
|
||||
|
||||
/* 响应式 */
|
||||
@media (max-width: 1024px) {
|
||||
.app-container {
|
||||
grid-template-columns: 1fr;
|
||||
}
|
||||
.sidebar {
|
||||
position: static;
|
||||
height: auto;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,173 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>大模型速度测试助手</title>
|
||||
<link rel="stylesheet" href="css/style.css">
|
||||
<script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.0/dist/chart.umd.min.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<div class="app-container">
|
||||
<!-- 左侧控制面板 -->
|
||||
<aside class="sidebar">
|
||||
<div class="sidebar-header">
|
||||
<h1>大模型速度测试助手</h1>
|
||||
<div class="version" id="versionLabel">v2.1</div>
|
||||
</div>
|
||||
|
||||
<!-- 测试用例选择 -->
|
||||
<div class="panel">
|
||||
<div class="panel-title">测试用例选择</div>
|
||||
<div class="quick-actions">
|
||||
<button onclick="selectRecommended()">推荐组合</button>
|
||||
<button onclick="selectAll()">全选</button>
|
||||
<button onclick="clearAll()">清空</button>
|
||||
</div>
|
||||
<div id="testCaseList">
|
||||
<!-- 从 /api/cases 动态加载 -->
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- 参数配置 -->
|
||||
<div class="panel">
|
||||
<div class="panel-title">参数配置</div>
|
||||
<div class="config-group">
|
||||
<label>模型</label>
|
||||
<select id="modelSelect">
|
||||
<option value="">加载中...</option>
|
||||
</select>
|
||||
</div>
|
||||
<div class="config-group">
|
||||
<label>重复次数</label>
|
||||
<input type="number" id="repeatsInput" value="3" min="1" max="20">
|
||||
</div>
|
||||
<div class="config-group">
|
||||
<label>运行模式</label>
|
||||
<select id="runMode">
|
||||
<option value="all">全部用例</option>
|
||||
<option value="skip_heavy">跳过耗时用例</option>
|
||||
</select>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- 操作按钮 -->
|
||||
<div class="panel">
|
||||
<button class="btn btn-primary" id="startBtn" onclick="startTests()" disabled>
|
||||
开始测试
|
||||
</button>
|
||||
<button class="btn btn-secondary" id="stopBtn" onclick="stopTests()" disabled>
|
||||
停止测试
|
||||
</button>
|
||||
<button class="btn btn-secondary" id="loadHistoryBtn" onclick="loadHistory()">
|
||||
加载历史对比
|
||||
</button>
|
||||
</div>
|
||||
</aside>
|
||||
|
||||
<!-- 右侧主内容区 -->
|
||||
<main class="main-content">
|
||||
<!-- 运行状态面板 -->
|
||||
<div class="status-panel" id="statusPanel">
|
||||
<div class="status-header">
|
||||
<div class="status-text running" id="statusText">准备中...</div>
|
||||
<div id="progressPercent">0%</div>
|
||||
</div>
|
||||
<div class="progress-bar-container">
|
||||
<div class="progress-bar" id="progressBar"></div>
|
||||
</div>
|
||||
<div id="currentCaseInfo" style="font-size:13px;margin-bottom:10px;color:#64748b;"></div>
|
||||
<ul class="step-list" id="stepList"></ul>
|
||||
</div>
|
||||
|
||||
<!-- 结果展示区 -->
|
||||
<div class="results-area" id="resultsArea">
|
||||
<!-- 推理指标概览 -->
|
||||
<div class="report-card" id="inferenceCard" style="display:none;">
|
||||
<h2>推理核心指标</h2>
|
||||
<div class="metric-grid" id="inferenceMetricsGrid"></div>
|
||||
<div class="metric-description">
|
||||
<p><strong>TTFT(首 Token 延迟)</strong>:从请求到生成第一个 token 的时间,影响用户感知响应速度</p>
|
||||
<p><strong>Prefill Time(预填充时间)</strong>:处理 prompt 的时间,与上下文长度正相关</p>
|
||||
<p><strong>Decode Speed(解码速度)</strong>:tokens/s,衡量模型生成效率</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- 并发压力摘要 -->
|
||||
<div class="report-card" id="concurrencyCard" style="display:none;">
|
||||
<h2>⚡ 并发压力测试摘要</h2>
|
||||
<div class="metric-grid" id="concurrencyMetricsGrid"></div>
|
||||
</div>
|
||||
|
||||
<!-- 概览卡片 -->
|
||||
<div class="report-card" id="overviewCard" style="display:none;">
|
||||
<h2>测试概览</h2>
|
||||
<div class="metric-grid" id="metricGrid"></div>
|
||||
</div>
|
||||
|
||||
<!-- 图表区 -->
|
||||
<div class="report-card" id="chartCard" style="display:none;">
|
||||
<h2>性能对比图表</h2>
|
||||
<div class="chart-wrapper">
|
||||
<canvas id="barChart"></canvas>
|
||||
</div>
|
||||
<div class="chart-wrapper">
|
||||
<canvas id="p95Chart"></canvas>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- 数据表 -->
|
||||
<div class="report-card" id="tableCard" style="display:none;">
|
||||
<h2>详细数据</h2>
|
||||
<div style="overflow-x:auto;">
|
||||
<table class="data-table" id="dataTable">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>用例名称</th>
|
||||
<th>难度</th>
|
||||
<th>状态</th>
|
||||
<th>平均延迟 (ms)</th>
|
||||
<th>标准差</th>
|
||||
<th>最小</th>
|
||||
<th>最大</th>
|
||||
<th>P95</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody id="dataTableBody"></tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- 操作按钮 -->
|
||||
<div class="action-buttons" id="actionButtons" style="display:none;">
|
||||
<button onclick="openReport()">打开完整报告</button>
|
||||
<button onclick="exportCSV()">导出 CSV</button>
|
||||
<button onclick="exportJSON()">导出 JSON</button>
|
||||
<button class="primary" onclick="runNewTest()">新测试</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- 初始欢迎状态 -->
|
||||
<div class="report-card" id="welcomeCard">
|
||||
<h2>欢迎使用大模型速度测试助手</h2>
|
||||
<p style="color:#64748b;margin:15px 0;">
|
||||
本工具用于测试大模型在不同场景下的响应速度
|
||||
<span id="welcomeBackend">(Ollama 本地 / OpenAI 兼容服务)</span>。
|
||||
<br>请在左侧面板选择测试用例和模型,然后点击"开始测试"。
|
||||
</p>
|
||||
<div style="margin-top:20px;">
|
||||
<h3 style="font-size:14px;margin-bottom:10px;">快速开始:</h3>
|
||||
<ol style="padding-left:20px;color:#64748b;font-size:13px;line-height:2;">
|
||||
<li>左侧选择测试用例("推荐组合"快速勾选)</li>
|
||||
<li>选择模型和重复次数</li>
|
||||
<li>点击"开始测试",观察实时进度</li>
|
||||
<li>完成后导出 CSV / JSON 或打开完整报告</li>
|
||||
</ol>
|
||||
</div>
|
||||
</div>
|
||||
</main>
|
||||
</div>
|
||||
|
||||
<script src="js/app.js"></script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,664 @@
|
||||
// 大模型速度测试助手 - 前端主逻辑 (Ollama 版)
|
||||
|
||||
const RECOMMENDED_CASES = ["case_01_generation", "case_02_simple_tool", "case_03_file_analysis", "case_04_parallel_tool"];
|
||||
|
||||
let charts = {};
|
||||
let currentTestData = null;
|
||||
let pollTimer = null;
|
||||
let backendCases = {}; // 从 /api/cases 动态加载的用例
|
||||
let appConfig = {}; // 从 /api/config 动态加载的后端配置
|
||||
|
||||
// 初始化
|
||||
document.addEventListener('DOMContentLoaded', () => {
|
||||
loadConfig();
|
||||
renderTestCaseList();
|
||||
loadModels();
|
||||
updateUI();
|
||||
});
|
||||
|
||||
// ---------- 后端配置 ----------
|
||||
|
||||
async function loadConfig() {
|
||||
try {
|
||||
const resp = await fetch('/api/config');
|
||||
appConfig = await resp.json();
|
||||
} catch (e) {
|
||||
appConfig = {};
|
||||
}
|
||||
const backend = appConfig.backend || 'ollama';
|
||||
const label = document.getElementById('versionLabel');
|
||||
if (label) {
|
||||
label.textContent = `v2.1 | ${backend === 'openai' ? 'OpenAI 兼容' : 'Ollama'}`;
|
||||
}
|
||||
const welcome = document.getElementById('welcomeBackend');
|
||||
if (welcome) {
|
||||
welcome.textContent = backend === 'openai'
|
||||
? `(OpenAI 兼容:${appConfig.openai_base_url || ''})`
|
||||
: '(本地 Ollama 服务)';
|
||||
}
|
||||
}
|
||||
|
||||
// ---------- 用例列表 ----------
|
||||
|
||||
function renderTestCaseList() {
|
||||
const container = document.getElementById('testCaseList');
|
||||
const entries = Object.entries(backendCases);
|
||||
if (entries.length === 0) {
|
||||
container.innerHTML = '<p style="padding:10px;color:#64748b;font-size:13px;">无法加载用例列表</p>';
|
||||
return;
|
||||
}
|
||||
container.innerHTML = entries.map(([id, info]) => `
|
||||
<div class="test-case-item" onclick="toggleCase('${id}')">
|
||||
<input type="checkbox" id="cb_${id}" data-case-id="${id}" onclick="event.stopPropagation()">
|
||||
<div class="test-case-info">
|
||||
<div class="test-case-name">${info.name}</div>
|
||||
<div class="test-case-diff diff-${info.difficulty}">${info.difficulty} · 预计 ${info.estimated_seconds}s</div>
|
||||
</div>
|
||||
</div>
|
||||
`).join('');
|
||||
}
|
||||
|
||||
function toggleCase(caseId) {
|
||||
const cb = document.getElementById(`cb_${caseId}`);
|
||||
cb.checked = !cb.checked;
|
||||
updateUI();
|
||||
}
|
||||
|
||||
function selectAll() {
|
||||
Object.keys(backendCases).forEach(id => { document.getElementById(`cb_${id}`).checked = true; });
|
||||
updateUI();
|
||||
}
|
||||
|
||||
function clearAll() {
|
||||
Object.keys(backendCases).forEach(id => { document.getElementById(`cb_${id}`).checked = false; });
|
||||
updateUI();
|
||||
}
|
||||
|
||||
function selectRecommended() {
|
||||
Object.keys(backendCases).forEach(id => {
|
||||
document.getElementById(`cb_${id}`).checked = RECOMMENDED_CASES.includes(id);
|
||||
});
|
||||
updateUI();
|
||||
}
|
||||
|
||||
function getSelectedCases() {
|
||||
const selected = [];
|
||||
Object.keys(backendCases).forEach(id => {
|
||||
if (document.getElementById(`cb_${id}`).checked) selected.push(id);
|
||||
});
|
||||
return selected;
|
||||
}
|
||||
|
||||
function updateUI() {
|
||||
const startBtn = document.getElementById('startBtn');
|
||||
startBtn.disabled = getSelectedCases().length === 0;
|
||||
}
|
||||
|
||||
// ---------- 模型列表 ----------
|
||||
|
||||
async function loadModels() {
|
||||
const sel = document.getElementById('modelSelect');
|
||||
try {
|
||||
const resp = await fetch('/api/models');
|
||||
const data = await resp.json();
|
||||
if (data.models && data.models.length > 0) {
|
||||
sel.innerHTML = data.models.map(m => `<option value="${m}">${m}</option>`).join('');
|
||||
} else {
|
||||
sel.innerHTML = `<option value="">${data.error || '未找到已拉取的模型'}</option>`;
|
||||
}
|
||||
} catch (e) {
|
||||
sel.innerHTML = `<option value="">后端未启动(请运行 python run.py)</option>`;
|
||||
}
|
||||
}
|
||||
|
||||
// ---------- 用例列表(从后端加载) ----------
|
||||
|
||||
async function loadCases() {
|
||||
try {
|
||||
const resp = await fetch('/api/cases');
|
||||
backendCases = await resp.json();
|
||||
} catch (e) {
|
||||
console.warn('加载用例列表失败', e);
|
||||
backendCases = {};
|
||||
}
|
||||
renderTestCaseList();
|
||||
}
|
||||
|
||||
// ---------- 开始 / 停止 ----------
|
||||
|
||||
async function startTests() {
|
||||
const selectedCases = getSelectedCases();
|
||||
if (selectedCases.length === 0) {
|
||||
alert('请至少选择一个测试用例');
|
||||
return;
|
||||
}
|
||||
|
||||
const repeats = parseInt(document.getElementById('repeatsInput').value) || 3;
|
||||
const skipHeavy = document.getElementById('runMode').value === 'skip_heavy';
|
||||
const model = document.getElementById('modelSelect').value;
|
||||
if (!model) {
|
||||
const backend = appConfig.backend || 'ollama';
|
||||
alert(backend === 'openai'
|
||||
? '未选择模型。请检查 OpenAI 兼容服务地址与 API Key 配置,并刷新页面加载模型列表。'
|
||||
: '未选择模型。请确认 Ollama 已启动(ollama serve)并已拉取模型(ollama pull <模型名>)');
|
||||
return;
|
||||
}
|
||||
|
||||
// UI 状态
|
||||
document.getElementById('welcomeCard').style.display = 'none';
|
||||
document.getElementById('resultsArea').classList.remove('visible');
|
||||
document.getElementById('statusPanel').classList.add('active');
|
||||
document.getElementById('stepList').innerHTML = '';
|
||||
document.getElementById('currentCaseInfo').textContent = `模型: ${model} | 重复: ${repeats} 次`;
|
||||
document.getElementById('startBtn').disabled = true;
|
||||
document.getElementById('stopBtn').disabled = false;
|
||||
updateStatus('running', '正在启动测试...');
|
||||
updateProgress(0);
|
||||
|
||||
try {
|
||||
const resp = await fetch('/api/run', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
cases: selectedCases,
|
||||
repeats: repeats,
|
||||
model: model,
|
||||
skip_heavy: skipHeavy
|
||||
})
|
||||
});
|
||||
const data = await resp.json();
|
||||
if (data.error) {
|
||||
updateStatus('error', data.error);
|
||||
showError(data.error);
|
||||
resetButtons();
|
||||
return;
|
||||
}
|
||||
pollStatus();
|
||||
} catch (e) {
|
||||
updateStatus('error', '无法连接后端服务,请先运行 python run.py');
|
||||
showError('无法连接后端服务,请先运行 python run.py');
|
||||
resetButtons();
|
||||
}
|
||||
}
|
||||
|
||||
async function stopTests() {
|
||||
document.getElementById('stopBtn').disabled = true;
|
||||
updateStatus('running', '正在停止...');
|
||||
try {
|
||||
await fetch('/api/stop', { method: 'POST' });
|
||||
} catch (e) {
|
||||
// 忽略网络错误
|
||||
}
|
||||
}
|
||||
|
||||
// ---------- 轮询进度 ----------
|
||||
|
||||
async function pollStatus() {
|
||||
try {
|
||||
const resp = await fetch('/api/status');
|
||||
const st = await resp.json();
|
||||
renderRunStatus(st);
|
||||
|
||||
if (st.status === 'running') {
|
||||
pollTimer = setTimeout(pollStatus, 800);
|
||||
} else {
|
||||
finishRun(st);
|
||||
}
|
||||
} catch (e) {
|
||||
// 后端暂时不可达,继续重试
|
||||
pollTimer = setTimeout(pollStatus, 1500);
|
||||
}
|
||||
}
|
||||
|
||||
function renderRunStatus(st) {
|
||||
const total = st.total_cases || 0;
|
||||
const idx = st.current_case_index || 0;
|
||||
|
||||
updateStatus('running', st.message || '运行中...');
|
||||
updateProgress(total > 0 ? (idx / total) * 100 : 0);
|
||||
if (st.current_case_name) {
|
||||
document.getElementById('currentCaseInfo').textContent = `当前: ${st.current_case_name}`;
|
||||
}
|
||||
renderSteps(st);
|
||||
}
|
||||
|
||||
function renderSteps(st) {
|
||||
const cases = st.cases || [];
|
||||
const list = document.getElementById('stepList');
|
||||
const currentIdx = st.current_case_index || 0;
|
||||
list.innerHTML = cases.map((cid, i) => {
|
||||
const info = backendCases[cid] || { name: cid };
|
||||
let cls = '', icon = '○';
|
||||
if (i < currentIdx) { cls = 'completed'; icon = '✓'; }
|
||||
else if (i === currentIdx) { cls = 'active'; icon = '⟳'; }
|
||||
return `<li class="step-item ${cls}"><span class="step-icon">${icon}</span>${info.name}</li>`;
|
||||
}).join('');
|
||||
}
|
||||
|
||||
// ---------- 测试完成 ----------
|
||||
|
||||
async function finishRun(st) {
|
||||
resetButtons();
|
||||
|
||||
if (st.status === 'failed') {
|
||||
updateStatus('error', st.message || '测试失败');
|
||||
showError(st.message || '测试失败');
|
||||
return;
|
||||
}
|
||||
if (st.status === 'stopped') {
|
||||
updateStatus('error', '测试已停止');
|
||||
} else {
|
||||
updateStatus('complete', '测试完成!');
|
||||
updateProgress(100);
|
||||
}
|
||||
|
||||
try {
|
||||
const resp = await fetch('/api/results/latest.json');
|
||||
const report = await resp.json();
|
||||
if (report.error) throw new Error(report.error);
|
||||
currentTestData = report;
|
||||
displayResults(report);
|
||||
} catch (e) {
|
||||
console.warn('加载结果失败', e);
|
||||
showError('测试结束,但加载结果失败:' + e.message);
|
||||
}
|
||||
}
|
||||
|
||||
function resetButtons() {
|
||||
document.getElementById('stopBtn').disabled = true;
|
||||
document.getElementById('startBtn').disabled = false;
|
||||
}
|
||||
|
||||
// ---------- 结果显示 ----------
|
||||
|
||||
function showError(msg) {
|
||||
document.getElementById('resultsArea').classList.add('visible');
|
||||
document.getElementById('resultsArea').insertAdjacentHTML(
|
||||
'afterbegin',
|
||||
`<div class="error-banner">${msg}</div>`
|
||||
);
|
||||
}
|
||||
|
||||
function renderInferenceMetrics(data) {
|
||||
const results = data.results || [];
|
||||
const ttft = [], prefill = [], decode = [];
|
||||
|
||||
results.forEach(r => {
|
||||
if (r.raw_data && r.raw_data.results) {
|
||||
r.raw_data.results.forEach(item => {
|
||||
if (item.ttft_ms > 0) ttft.push(item.ttft_ms);
|
||||
if (item.prefill_ms > 0) prefill.push(item.prefill_ms);
|
||||
if (item.decode_speed_tok_s > 0) decode.push(item.decode_speed_tok_s);
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
const avg = arr => arr.length ? (arr.reduce((a, b) => a + b, 0) / arr.length).toFixed(2) : 'N/A';
|
||||
const grid = document.getElementById('inferenceMetricsGrid');
|
||||
grid.innerHTML = `
|
||||
<div class="metric-card" style="border-left-color:#6366f1;">
|
||||
<div class="label">TTFT (首Token延迟)</div>
|
||||
<div class="value" style="color:#6366f1;">${avg(ttft)} ms</div>
|
||||
<div class="metric-card-value-desc">均值(越低越好)</div>
|
||||
</div>
|
||||
<div class="metric-card" style="border-left-color:#8b5cf6;">
|
||||
<div class="label">Prefill Time</div>
|
||||
<div class="value" style="color:#8b5cf6;">${avg(prefill)} ms</div>
|
||||
<div class="metric-card-value-desc">Prompt 处理时间</div>
|
||||
</div>
|
||||
<div class="metric-card" style="border-left-color:#22c55e;">
|
||||
<div class="label">Decode Speed</div>
|
||||
<div class="value" style="color:#22c55e;">${avg(decode)} tok/s</div>
|
||||
<div class="metric-card-value-desc">生成速度(越高越好)</div>
|
||||
</div>
|
||||
`;
|
||||
document.getElementById('inferenceCard').style.display = 'block';
|
||||
}
|
||||
|
||||
function renderConcurrencySummary(data) {
|
||||
const card = document.getElementById('concurrencyCard');
|
||||
const grid = document.getElementById('concurrencyMetricsGrid');
|
||||
const concurrency = (data.results || []).find(r => r.case_id === 'case_08_concurrency');
|
||||
if (!concurrency || !concurrency.raw_data) {
|
||||
card.style.display = 'none';
|
||||
return;
|
||||
}
|
||||
const raw = concurrency.raw_data;
|
||||
const keep = raw.tps_keep_rate_pct;
|
||||
const keepColor = keep >= 60 ? '#22c55e' : (keep > 0 ? '#f59e0b' : '#ef4444');
|
||||
const errColor = raw.error_rate_pct === 0 ? '#22c55e' : '#ef4444';
|
||||
const metric = (label, val, color) => `
|
||||
<div class="metric-card" style="border-left-color:${color};">
|
||||
<div class="label">${label}</div>
|
||||
<div class="value" style="color:${color};">${val}</div>
|
||||
</div>`;
|
||||
grid.innerHTML =
|
||||
metric('并发数', raw.concurrency, '#8b5cf6') +
|
||||
metric('基准 TPS', `${raw.baseline_tps ?? 'N/A'} tok/s`, '#6366f1') +
|
||||
metric('并发 TPS', `${raw.concurrent_tps ?? 'N/A'} tok/s`, '#6366f1') +
|
||||
metric('TPS 保持率', `${raw.tps_keep_rate_pct ?? 'N/A'}%`, keepColor) +
|
||||
metric('P95 延迟', `${raw.e2e_p95_ms ?? 'N/A'} ms`, '#ef4444') +
|
||||
metric('P99 延迟', `${raw.e2e_p99_ms ?? 'N/A'} ms`, '#ef4444') +
|
||||
metric('错误率', `${raw.error_rate_pct ?? 'N/A'}%`, errColor) +
|
||||
metric('成功/失败', `${raw.success_count ?? 0}/${raw.failed_count ?? 0}`, '#64748b');
|
||||
card.style.display = 'block';
|
||||
}
|
||||
|
||||
function displayResults(data) {
|
||||
document.getElementById('resultsArea').classList.add('visible');
|
||||
document.querySelectorAll('#resultsArea .error-banner').forEach(el => el.remove());
|
||||
|
||||
renderInferenceMetrics(data);
|
||||
renderConcurrencySummary(data);
|
||||
|
||||
const summary = data.summary || {};
|
||||
document.getElementById('metricGrid').innerHTML = `
|
||||
<div class="metric-card"><div class="label">总用例数</div><div class="value">${summary.total_cases || 0}</div></div>
|
||||
<div class="metric-card success"><div class="label">通过</div><div class="value" style="color:#22c55e">${summary.passed || 0}</div></div>
|
||||
<div class="metric-card ${summary.failed > 0 ? 'danger' : 'success'}">
|
||||
<div class="label">失败</div>
|
||||
<div class="value" style="color:${summary.failed > 0 ? '#ef4444' : '#22c55e'}">${summary.failed || 0}</div>
|
||||
</div>
|
||||
<div class="metric-card warning"><div class="label">总耗时</div><div class="value">${(summary.total_elapsed_seconds || 0).toFixed(2)}s</div></div>
|
||||
`;
|
||||
document.getElementById('overviewCard').style.display = 'block';
|
||||
|
||||
buildCharts(data);
|
||||
document.getElementById('chartCard').style.display = 'block';
|
||||
|
||||
buildDataTable(data);
|
||||
document.getElementById('tableCard').style.display = 'block';
|
||||
|
||||
document.getElementById('actionButtons').style.display = 'flex';
|
||||
}
|
||||
|
||||
function buildCharts(data) {
|
||||
Object.values(charts).forEach(c => c.destroy());
|
||||
charts = {};
|
||||
|
||||
const results = data.results || [];
|
||||
const labels = [], means = [], p95s = [];
|
||||
results.forEach(r => {
|
||||
if (r.statistics && Object.keys(r.statistics).length > 0) {
|
||||
labels.push(r.case_name);
|
||||
means.push(r.statistics.mean_ms || 0);
|
||||
p95s.push(r.statistics.p95_ms || 0);
|
||||
}
|
||||
});
|
||||
|
||||
const barCtx = document.getElementById('barChart').getContext('2d');
|
||||
charts.bar = new Chart(barCtx, {
|
||||
type: 'bar',
|
||||
data: {
|
||||
labels, datasets: [{
|
||||
label: '平均延迟 (ms)', data: means,
|
||||
backgroundColor: 'rgba(99, 102, 241, 0.7)', borderColor: 'rgba(99, 102, 241, 1)', borderWidth: 1
|
||||
}]
|
||||
},
|
||||
options: { responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } }, scales: { y: { beginAtZero: true } } }
|
||||
});
|
||||
|
||||
const p95Ctx = document.getElementById('p95Chart').getContext('2d');
|
||||
charts.p95 = new Chart(p95Ctx, {
|
||||
type: 'bar',
|
||||
data: {
|
||||
labels, datasets: [{
|
||||
label: 'P95延迟 (ms)', data: p95s,
|
||||
backgroundColor: 'rgba(239, 68, 68, 0.7)', borderColor: 'rgba(239, 68, 68, 1)', borderWidth: 1
|
||||
}]
|
||||
},
|
||||
options: { responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } }, scales: { y: { beginAtZero: true } } }
|
||||
});
|
||||
}
|
||||
|
||||
function buildDataTable(data) {
|
||||
const tbody = document.getElementById('dataTableBody');
|
||||
const diffColors = { "简单": "#22c55e", "中等": "#f59e0b", "耗时": "#ef4444", "压力": "#8b5cf6" };
|
||||
|
||||
tbody.innerHTML = (data.results || []).map(r => {
|
||||
const stats = r.statistics || {};
|
||||
const statusClass = r.status === 'passed' ? 'status-passed' : 'status-failed';
|
||||
return `
|
||||
<tr>
|
||||
<td><strong>${r.case_name}</strong></td>
|
||||
<td><span class="diff-badge" style="background:${diffColors[r.difficulty] || '#6b7280'};padding:4px 10px;border-radius:12px;color:white;font-size:12px;">${r.difficulty}</span></td>
|
||||
<td class="${statusClass}" style="font-weight:bold;">${r.status}</td>
|
||||
<td class="mono">${stats.mean_ms != null ? stats.mean_ms : 'N/A'}</td>
|
||||
<td class="mono">${stats.std_ms != null ? stats.std_ms : 'N/A'}</td>
|
||||
<td class="mono">${stats.min_ms != null ? stats.min_ms : 'N/A'}</td>
|
||||
<td class="mono">${stats.max_ms != null ? stats.max_ms : 'N/A'}</td>
|
||||
<td class="mono">${stats.p95_ms != null ? stats.p95_ms : 'N/A'}</td>
|
||||
</tr>
|
||||
`;
|
||||
}).join('');
|
||||
}
|
||||
|
||||
// ---------- 操作按钮 ----------
|
||||
|
||||
function openReport() {
|
||||
window.open('/report.html', '_blank');
|
||||
}
|
||||
|
||||
function exportCSV() {
|
||||
if (!currentTestData) return;
|
||||
let csv = '用例ID,用例名称,难度,状态,平均延迟(ms),标准差(ms),最小延迟(ms),最大延迟(ms),P95延迟(ms)\n';
|
||||
currentTestData.results.forEach(r => {
|
||||
const stats = r.statistics || {};
|
||||
csv += `${r.case_id},${r.case_name},${r.difficulty},${r.status},${stats.mean_ms || ''},${stats.std_ms || ''},${stats.min_ms || ''},${stats.max_ms || ''},${stats.p95_ms || ''}\n`;
|
||||
});
|
||||
|
||||
const blob = new Blob(['' + csv], { type: 'text/csv;charset=utf-8;' });
|
||||
const link = document.createElement('a');
|
||||
link.href = URL.createObjectURL(blob);
|
||||
link.download = `speed_test_${new Date().toISOString().slice(0, 10)}.csv`;
|
||||
link.click();
|
||||
}
|
||||
|
||||
function exportJSON() {
|
||||
if (!currentTestData) return;
|
||||
const blob = new Blob([JSON.stringify(currentTestData, null, 2)], { type: 'application/json' });
|
||||
const link = document.createElement('a');
|
||||
link.href = URL.createObjectURL(blob);
|
||||
link.download = `speed_test_${new Date().toISOString().slice(0, 10)}.json`;
|
||||
link.click();
|
||||
}
|
||||
|
||||
function runNewTest() {
|
||||
if (pollTimer) clearTimeout(pollTimer);
|
||||
document.getElementById('resultsArea').classList.remove('visible');
|
||||
document.getElementById('statusPanel').classList.remove('active');
|
||||
document.getElementById('welcomeCard').style.display = 'block';
|
||||
document.getElementById('stepList').innerHTML = '';
|
||||
Object.values(charts).forEach(c => c.destroy());
|
||||
charts = {};
|
||||
currentTestData = null;
|
||||
updateUI();
|
||||
}
|
||||
|
||||
let historyData = null;
|
||||
let compareMode = false;
|
||||
|
||||
async function loadHistory() {
|
||||
try {
|
||||
const resp = await fetch('/api/results/list');
|
||||
const data = await resp.json();
|
||||
if (!data.results || data.results.length === 0) {
|
||||
alert('暂无历史结果');
|
||||
return;
|
||||
}
|
||||
|
||||
// 构建选择对话框
|
||||
const listHtml = data.results
|
||||
.map((r, i) => `<option value="${i}">${r.filename} — ${r.model} (${r.total_cases}用例, ${r.passed}/${r.total_cases}通过)</option>`)
|
||||
.join('\n');
|
||||
|
||||
const dialog = document.createElement('div');
|
||||
dialog.style.cssText = 'position:fixed;top:0;left:0;width:100%;height:100%;background:rgba(0,0,0,0.5);display:flex;align-items:center;justify-content:center;z-index:1000;';
|
||||
dialog.innerHTML = `
|
||||
<div style="background:white;border-radius:12px;padding:24px;max-width:600px;width:90%;max-height:80vh;overflow:auto;">
|
||||
<h3 style="margin-bottom:16px;">历史测试结果对比</h3>
|
||||
<p style="color:#64748b;font-size:13px;margin-bottom:12px;">选择两次运行结果进行对比(跳过耗时用例):</p>
|
||||
<div style="display:flex;gap:12px;margin-bottom:16px;">
|
||||
<div style="flex:1;">
|
||||
<label style="font-size:12px;color:#64748b;">第一次运行</label>
|
||||
<select id="historyA" style="width:100%;padding:8px;border:1px solid #e2e8f0;border-radius:8px;">${listHtml}</select>
|
||||
</div>
|
||||
<div style="flex:1;">
|
||||
<label style="font-size:12px;color:#64748b;">第二次运行</label>
|
||||
<select id="historyB" style="width:100%;padding:8px;border:1px solid #e2e8f0;border-radius:8px;">${listHtml}</select>
|
||||
</div>
|
||||
</div>
|
||||
<div style="display:flex;gap:8px;justify-content:flex-end;">
|
||||
<button onclick="this.closest('div[style]').parentElement.remove()" style="padding:8px 16px;border:1px solid #e2e8f0;border-radius:8px;cursor:pointer;">取消</button>
|
||||
<button id="compareBtn" style="padding:8px 16px;background:#6366f1;color:white;border:none;border-radius:8px;cursor:pointer;">对比</button>
|
||||
</div>
|
||||
</div>
|
||||
`;
|
||||
document.body.appendChild(dialog);
|
||||
|
||||
// 默认选择最后两个
|
||||
if (data.results.length >= 2) {
|
||||
dialog.querySelector('#historyA').value = data.results.length - 2;
|
||||
dialog.querySelector('#historyB').value = data.results.length - 1;
|
||||
}
|
||||
|
||||
dialog.querySelector('#compareBtn').onclick = async () => {
|
||||
const idxA = dialog.querySelector('#historyA').value;
|
||||
const idxB = dialog.querySelector('#historyB').value;
|
||||
dialog.remove();
|
||||
if (idxA === idxB) {
|
||||
alert('请选择不同的两次运行结果');
|
||||
return;
|
||||
}
|
||||
await showComparison(data.results[parseInt(idxA)], data.results[parseInt(idxB)]);
|
||||
};
|
||||
|
||||
// 点击背景关闭
|
||||
dialog.onclick = (e) => { if (e.target === dialog) dialog.remove(); };
|
||||
|
||||
} catch (e) {
|
||||
console.warn('加载历史记录失败', e);
|
||||
alert('无法加载历史记录');
|
||||
}
|
||||
}
|
||||
|
||||
async function showComparison(runA, runB) {
|
||||
if (runA.error || runB.error) {
|
||||
alert('无法读取选定的结果文件');
|
||||
return;
|
||||
}
|
||||
|
||||
// 加载完整数据
|
||||
const [dataA, dataB] = await Promise.all([
|
||||
fetch(`/api/results/${runA.filename}`).then(r => r.json()),
|
||||
fetch(`/api/results/${runB.filename}`).then(r => r.json()),
|
||||
]);
|
||||
|
||||
if (dataA.error || dataB.error) {
|
||||
alert('无法加载结果数据');
|
||||
return;
|
||||
}
|
||||
|
||||
// 显示对比
|
||||
document.getElementById('welcomeCard').style.display = 'none';
|
||||
document.getElementById('resultsArea').classList.add('visible');
|
||||
document.querySelectorAll('#resultsArea .error-banner').forEach(el => el.remove());
|
||||
|
||||
const summaryA = dataA.summary || {};
|
||||
const summaryB = dataB.summary || {};
|
||||
|
||||
// 对比卡片
|
||||
const card = document.createElement('div');
|
||||
card.className = 'report-card';
|
||||
card.innerHTML = `
|
||||
<h2>📊 运行对比:${runA.filename} vs ${runB.filename}</h2>
|
||||
<table class="data-table" style="margin-top:16px;">
|
||||
<thead>
|
||||
<tr><th>指标</th><th>${runA.timestamp ? new Date(runA.timestamp).toLocaleString('zh-CN') : runA.filename}</th><th>${runB.timestamp ? new Date(runB.timestamp).toLocaleString('zh-CN') : runB.filename}</th><th>变化</th></tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr><td>模型</td><td>${dataA.environment?.model || 'N/A'}</td><td>${dataB.environment?.model || 'N/A'}</td><td>${dataA.environment?.model !== dataB.environment?.model ? '⚠️ 不同' : '—'}</td></tr>
|
||||
<tr><td>总用例数</td><td>${summaryA.total_cases || 0}</td><td>${summaryB.total_cases || 0}</td><td>—</td></tr>
|
||||
<tr><td>通过</td><td>${summaryA.passed || 0}</td><td>${summaryB.passed || 0}</td><td>${diffArrow(summaryA.passed, summaryB.passed)}</td></tr>
|
||||
<tr><td>失败</td><td>${summaryA.failed || 0}</td><td>${summaryB.failed || 0}</td><td>${diffArrow(summaryA.failed, summaryB.failed)}</td></tr>
|
||||
<tr><td>总耗时</td><td>${(summaryA.total_elapsed_seconds || 0).toFixed(1)}s</td><td>${(summaryB.total_elapsed_seconds || 0).toFixed(1)}s</td><td>${diffArrow(summaryA.total_elapsed_seconds, summaryB.total_elapsed_seconds)}</td></tr>
|
||||
</tbody>
|
||||
</table>
|
||||
<div style="margin-top:16px;font-size:12px;color:#64748b;">
|
||||
<p>注:仅对比都通过的用例。跳过耗时用例(长上下文、推理、多轮对话)。</p>
|
||||
</div>
|
||||
`;
|
||||
|
||||
// 插入到 resultsArea 开头
|
||||
const resultsArea = document.getElementById('resultsArea');
|
||||
resultsArea.insertBefore(card, resultsArea.firstChild);
|
||||
|
||||
// 对比详细用例数据
|
||||
const compareCard = document.createElement('div');
|
||||
compareCard.className = 'report-card';
|
||||
compareCard.innerHTML = `
|
||||
<h2>📋 用例级对比(跳过耗时用例)</h2>
|
||||
<div style="overflow-x:auto;margin-top:16px;">
|
||||
<table class="data-table">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>用例名称</th>
|
||||
<th>运行A 平均延迟</th>
|
||||
<th>运行B 平均延迟</th>
|
||||
<th>变化</th>
|
||||
<th>运行A P95</th>
|
||||
<th>运行B P95</th>
|
||||
<th>变化</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody id="compareTableBody"></tbody>
|
||||
</table>
|
||||
</div>
|
||||
`;
|
||||
resultsArea.insertBefore(compareCard, card.nextSibling);
|
||||
|
||||
// 填充对比数据
|
||||
const tbody = document.getElementById('compareTableBody');
|
||||
const resultsA = (dataA.results || []).filter(r => r.difficulty !== '耗时' && r.status === 'passed');
|
||||
const resultsB = (dataB.results || []).filter(r => r.difficulty !== '耗时' && r.status === 'passed');
|
||||
const mapA = {};
|
||||
resultsA.forEach(r => mapA[r.case_id] = r);
|
||||
const mapB = {};
|
||||
resultsB.forEach(r => mapB[r.case_id] = r);
|
||||
|
||||
const allIds = [...new Set([...Object.keys(mapA), ...Object.keys(mapB)])];
|
||||
tbody.innerHTML = allIds.map(id => {
|
||||
const a = mapA[id], b = mapB[id];
|
||||
if (!a || !b) return `<tr><td>${id}</td><td colspan="6" style="color:#64748b;">仅在一次运行中出现</td></tr>`;
|
||||
const statsA = a.statistics || {};
|
||||
const statsB = b.statistics || {};
|
||||
const meanDiff = diffPercent(statsA.mean_ms, statsB.mean_ms);
|
||||
const p95Diff = diffPercent(statsA.p95_ms, statsB.p95_ms);
|
||||
return `
|
||||
<tr>
|
||||
<td><strong>${a.case_name}</strong></td>
|
||||
<td class="mono">${statsA.mean_ms || 'N/A'}</td>
|
||||
<td class="mono">${statsB.mean_ms || 'N/A'}</td>
|
||||
<td style="color:${meanDiff > 0 ? '#ef4444' : meanDiff < 0 ? '#22c55e' : '#64748b'}">${meanDiff === 0 ? '—' : (meanDiff > 0 ? '↑' : '↓') + Math.abs(meanDiff).toFixed(1) + '%'}</td>
|
||||
<td class="mono">${statsA.p95_ms || 'N/A'}</td>
|
||||
<td class="mono">${statsB.p95_ms || 'N/A'}</td>
|
||||
<td style="color:${p95Diff > 0 ? '#ef4444' : p95Diff < 0 ? '#22c55e' : '#64748b'}">${p95Diff === 0 ? '—' : (p95Diff > 0 ? '↑' : '↓') + Math.abs(p95Diff).toFixed(1) + '%'}</td>
|
||||
</tr>
|
||||
`;
|
||||
}).join('');
|
||||
|
||||
// 滚动到对比区域
|
||||
card.scrollIntoView({ behavior: 'smooth' });
|
||||
}
|
||||
|
||||
function diffArrow(current, previous) {
|
||||
if (!current || !previous) return '—';
|
||||
const diff = current - previous;
|
||||
if (diff === 0) return '—';
|
||||
return diff > 0 ? `↑+${diff}` : `↓${diff}`;
|
||||
}
|
||||
|
||||
function diffPercent(current, previous) {
|
||||
if (!current || !previous || previous === 0) return 0;
|
||||
return ((current - previous) / previous) * 100;
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
[build-system]
|
||||
requires = ["setuptools>=68.0"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "llm_speed_test_app"
|
||||
version = "2.0.0"
|
||||
description = "大模型速度测试助手 (Ollama)"
|
||||
requires-python = ">=3.9"
|
||||
dependencies = ["requests>=2.31"]
|
||||
|
||||
[project.scripts]
|
||||
llm-speed-test = "backend.server:main"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["backend*"]
|
||||
@@ -0,0 +1,149 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>大模型速度测试报告 - 20260826-004823</title>
|
||||
<script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.0/dist/chart.umd.min.js"></script>
|
||||
<style>
|
||||
* { margin: 0; padding: 0; box-sizing: border-box; }
|
||||
body {
|
||||
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', 'PingFang SC', 'Microsoft YaHei', sans-serif;
|
||||
background: #f8fafc; color: #1e293b; line-height: 1.6;
|
||||
}
|
||||
.container { max-width: 1400px; margin: 0 auto; padding: 20px; }
|
||||
.header { background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); color: white; padding: 30px; border-radius: 12px; margin-bottom: 20px; }
|
||||
.header h1 { font-size: 28px; margin-bottom: 10px; }
|
||||
.header .meta { font-size: 14px; opacity: 0.9; }
|
||||
.dashboard { display: grid; grid-template-columns: repeat(auto-fit, minmax(200px, 1fr)); gap: 15px; margin-bottom: 20px; }
|
||||
.stat-card { background: white; padding: 20px; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,0.1); }
|
||||
.stat-card .label { font-size: 12px; color: #64748b; text-transform: uppercase; }
|
||||
.stat-card .value { font-size: 28px; font-weight: bold; margin-top: 5px; }
|
||||
.section { background: white; border-radius: 10px; padding: 25px; margin-bottom: 20px; box-shadow: 0 2px 8px rgba(0,0,0,0.1); }
|
||||
.section h2 { font-size: 20px; margin-bottom: 15px; padding-bottom: 10px; border-bottom: 2px solid #e2e8f0; }
|
||||
table { width: 100%; border-collapse: collapse; }
|
||||
th, td { padding: 12px; text-align: left; border-bottom: 1px solid #e2e8f0; }
|
||||
th { background: #f1f5f9; font-weight: 600; font-size: 13px; text-transform: uppercase; }
|
||||
tr:hover { background: #f8fafc; }
|
||||
.mono { font-family: 'SF Mono', 'Fira Code', monospace; }
|
||||
.diff-badge { padding: 4px 10px; border-radius: 12px; color: white; font-size: 12px; font-weight: 600; }
|
||||
.status-passed { color: #22c55e; }
|
||||
.status-failed { color: #ef4444; }
|
||||
.chart-container { position: relative; height: 400px; margin: 20px 0; }
|
||||
.analysis-box { background: #f0fdf4; border-left: 4px solid #22c55e; padding: 15px; border-radius: 6px; }
|
||||
.analysis-box ul { padding-left: 20px; margin-top: 10px; }
|
||||
.analysis-box li { margin: 5px 0; }
|
||||
footer { text-align: center; padding: 20px; color: #64748b; font-size: 13px; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
<div class="header">
|
||||
<h1>大模型速度测试报告</h1>
|
||||
<div class="meta">
|
||||
Run ID: 20260826-004823 | 时间: 2026-08-26T00:48:23.163258 | 用例: 2 | 重复: 1次
|
||||
<br>模型: Q3.6-35B-A3B-Orig-Thi | 平台: win32
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="dashboard">
|
||||
<div class="stat-card"><div class="label">总用例数</div><div class="value">2</div></div>
|
||||
<div class="stat-card"><div class="label">通过</div><div class="value" style="color:#22c55e">2</div></div>
|
||||
<div class="stat-card"><div class="label">失败</div><div class="value" style="color:#ef4444">0</div></div>
|
||||
<div class="stat-card"><div class="label">总耗时</div><div class="value">60.01s</div></div>
|
||||
</div>
|
||||
|
||||
<div class="section"><h2>⚡ 并发压力测试摘要</h2><div class="dashboard"><div class="stat-card"><div class="label">并发数</div><div class="value" style="color:#8b5cf6">100</div></div><div class="stat-card"><div class="label">基准 TPS</div><div class="value" style="color:#6366f1">57.43</div></div><div class="stat-card"><div class="label">并发 TPS</div><div class="value" style="color:#6366f1">38.95</div></div><div class="stat-card"><div class="label">TPS 保持率</div><div class="value" style="color:#22c55e">67.82%</div></div><div class="stat-card"><div class="label">P95 延迟</div><div class="value" style="color:#ef4444">44577.95ms</div></div><div class="stat-card"><div class="label">P99 延迟</div><div class="value" style="color:#ef4444">45827.21ms</div></div><div class="stat-card"><div class="label">错误率</div><div class="value" style="color:#22c55e">0.0%</div></div></div></div>
|
||||
|
||||
<div class="analysis-box"><h3>性能分析</h3><ul><li><strong>最快用例:</strong> 纯文本生成基准 (5672.64ms)</li><li><strong>最慢用例:</strong> 并发压力测试 (25637.30ms)</li><li><strong>最快/最慢比:</strong> 4.5194653635696955x</li><li><strong>平均延迟:</strong> 15654.97ms</li></ul></div>
|
||||
|
||||
<div class="section">
|
||||
<h2>性能对比柱状图</h2>
|
||||
<div class="chart-container"><canvas id="barChart"></canvas></div>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<h2>P95延迟对比</h2>
|
||||
<div class="chart-container"><canvas id="p95Chart"></canvas></div>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<h2>详细数据</h2>
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>用例名称</th><th>难度</th><th>状态</th><th>平均延迟 (ms)</th>
|
||||
<th>标准差</th><th>最小</th><th>最大</th><th>P95</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr><td><strong>纯文本生成基准</strong></td><td><span class="diff-badge" style="background:#22c55e">简单</span></td><td class="status-passed" style="font-weight:bold">passed</td><td class="mono">5672.64</td><td class="mono">0.0</td><td class="mono">5672.64</td><td class="mono">5672.64</td><td class="mono">5672.64</td></tr>
|
||||
<tr><td><strong>并发压力测试</strong></td><td><span class="diff-badge" style="background:#8b5cf6">压力</span></td><td class="status-passed" style="font-weight:bold">passed</td><td class="mono">25637.3</td><td class="mono">11979.73</td><td class="mono">5264.59</td><td class="mono">45953.13</td><td class="mono">44577.96</td></tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<h2>环境信息</h2>
|
||||
<table>
|
||||
<tr><td>模型</td><td class="mono">Q3.6-35B-A3B-Orig-Thi</td></tr>
|
||||
<tr><td>运行平台</td><td class="mono">win32</td></tr>
|
||||
<tr><td>运行ID</td><td class="mono">20260826-004823</td></tr>
|
||||
<tr><td>测试时间</td><td class="mono">2026-08-26T00:48:23.163258</td></tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<footer>大模型速度测试助手 | 报告由 report.py 自动生成</footer>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
const barCtx = document.getElementById('barChart').getContext('2d');
|
||||
new Chart(barCtx, {
|
||||
type: 'bar',
|
||||
data: {
|
||||
labels: ["纯文本生成基准", "并发压力测试"],
|
||||
datasets: [{
|
||||
label: '平均延迟 (ms)',
|
||||
data: [5672.64, 25637.3],
|
||||
backgroundColor: 'rgba(102, 126, 234, 0.7)',
|
||||
borderColor: 'rgba(102, 126, 234, 1)',
|
||||
borderWidth: 1
|
||||
}]
|
||||
},
|
||||
options: {
|
||||
responsive: true,
|
||||
maintainAspectRatio: false,
|
||||
plugins: {
|
||||
legend: { display: false },
|
||||
title: { display: true, text: '各用例平均延迟对比 (ms)', font: { size: 16 } }
|
||||
},
|
||||
scales: { y: { beginAtZero: true, title: { display: true, text: '延迟 (ms)' } } }
|
||||
}
|
||||
});
|
||||
|
||||
const p95Ctx = document.getElementById('p95Chart').getContext('2d');
|
||||
new Chart(p95Ctx, {
|
||||
type: 'bar',
|
||||
data: {
|
||||
labels: ["纯文本生成基准", "并发压力测试"],
|
||||
datasets: [{
|
||||
label: 'P95延迟 (ms)',
|
||||
data: [5672.64, 44577.96],
|
||||
backgroundColor: 'rgba(239, 68, 68, 0.7)',
|
||||
borderColor: 'rgba(239, 68, 68, 1)',
|
||||
borderWidth: 1
|
||||
}]
|
||||
},
|
||||
options: {
|
||||
responsive: true,
|
||||
maintainAspectRatio: false,
|
||||
plugins: {
|
||||
legend: { display: false },
|
||||
title: { display: true, text: '各用例P95延迟 (ms)', font: { size: 16 } }
|
||||
},
|
||||
scales: { y: { beginAtZero: true, title: { display: true, text: '延迟 (ms)' } } }
|
||||
}
|
||||
});
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,3 @@
|
||||
# 大模型速度测试助手依赖
|
||||
# 安装:pip install -r requirements.txt
|
||||
requests>=2.31
|
||||
+1528
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"version": "1.0",
|
||||
"timestamp": "2026-08-25T16:39:51.611215",
|
||||
"run_id": "20260825-163951",
|
||||
"environment": {
|
||||
"python_version": "3.13.11",
|
||||
"platform": "linux",
|
||||
"model": "deepseek-r1:latest"
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": [
|
||||
"case_01_generation"
|
||||
],
|
||||
"repeats": 1,
|
||||
"skip_heavy": true,
|
||||
"total_cases": 1
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": 1,
|
||||
"passed": 1,
|
||||
"failed": 0,
|
||||
"stopped": false,
|
||||
"total_elapsed_seconds": 4.1
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"case_id": "case_01_generation",
|
||||
"case_name": "纯文本生成基准",
|
||||
"difficulty": "简单",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_01_generation",
|
||||
"name": "纯文本生成基准",
|
||||
"description": "基础生成速度基准测试",
|
||||
"difficulty": "简单",
|
||||
"estimated_seconds": 5,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 2752.43,
|
||||
"prefill_ms": 150.91,
|
||||
"decode_speed_tok_s": 78.35,
|
||||
"total_tokens": 325,
|
||||
"prompt_tokens": 25,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 4097.45,
|
||||
"elapsed_ms": 4097.45,
|
||||
"response_length": 167
|
||||
}
|
||||
],
|
||||
"prompt_used": "请用200字左右概述人工智能从诞生至今的关键发展阶段,每段不超过50字。"
|
||||
},
|
||||
"statistics": {
|
||||
"count": 1,
|
||||
"mean_ms": 4097.45,
|
||||
"std_ms": 0.0,
|
||||
"min_ms": 4097.45,
|
||||
"max_ms": 4097.45,
|
||||
"p50_ms": 4097.45,
|
||||
"p95_ms": 4097.45,
|
||||
"p99_ms": 4097.45
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 2752.43,
|
||||
"ttft_ms_p95": 2752.43,
|
||||
"prefill_ms_mean": 150.91,
|
||||
"prefill_ms_p95": 150.91,
|
||||
"decode_speed_tok_s_mean": 78.35,
|
||||
"decode_speed_tok_s_p95": 78.35,
|
||||
"e2e_ms_mean": 4097.45,
|
||||
"e2e_ms_p95": 4097.45,
|
||||
"total_tokens_mean": 325.0,
|
||||
"total_tokens_p95": 325.0
|
||||
},
|
||||
"elapsed_seconds": 4.1
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"version": "1.0",
|
||||
"timestamp": "2026-08-25T17:12:00.659272",
|
||||
"run_id": "20260825-171200",
|
||||
"environment": {
|
||||
"python_version": "3.13.11",
|
||||
"platform": "linux",
|
||||
"model": "qwen3.6:35b-a3b"
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": [
|
||||
"case_01_generation"
|
||||
],
|
||||
"repeats": 1,
|
||||
"skip_heavy": false,
|
||||
"total_cases": 1
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": 1,
|
||||
"passed": 1,
|
||||
"failed": 0,
|
||||
"stopped": false,
|
||||
"total_elapsed_seconds": 4.79
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"case_id": "case_01_generation",
|
||||
"case_name": "纯文本生成基准",
|
||||
"difficulty": "简单",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_01_generation",
|
||||
"name": "纯文本生成基准",
|
||||
"description": "基础生成速度基准测试",
|
||||
"difficulty": "简单",
|
||||
"estimated_seconds": 5,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 678.03,
|
||||
"prefill_ms": 528.43,
|
||||
"decode_speed_tok_s": 73.07,
|
||||
"total_tokens": 331,
|
||||
"prompt_tokens": 31,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 4787.47,
|
||||
"elapsed_ms": 4787.47,
|
||||
"response_length": 1082
|
||||
}
|
||||
],
|
||||
"prompt_used": "请用200字左右概述人工智能从诞生至今的关键发展阶段,每段不超过50字。"
|
||||
},
|
||||
"statistics": {
|
||||
"count": 1,
|
||||
"mean_ms": 4787.47,
|
||||
"std_ms": 0.0,
|
||||
"min_ms": 4787.47,
|
||||
"max_ms": 4787.47,
|
||||
"p50_ms": 4787.47,
|
||||
"p95_ms": 4787.47,
|
||||
"p99_ms": 4787.47
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 678.03,
|
||||
"ttft_ms_p95": 678.03,
|
||||
"prefill_ms_mean": 528.43,
|
||||
"prefill_ms_p95": 528.43,
|
||||
"decode_speed_tok_s_mean": 73.07,
|
||||
"decode_speed_tok_s_p95": 73.07,
|
||||
"e2e_ms_mean": 4787.47,
|
||||
"e2e_ms_p95": 4787.47,
|
||||
"total_tokens_mean": 331.0,
|
||||
"total_tokens_p95": 331.0
|
||||
},
|
||||
"elapsed_seconds": 4.79
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,562 @@
|
||||
{
|
||||
"version": "1.0",
|
||||
"timestamp": "2026-08-25T17:25:32.979324",
|
||||
"run_id": "20260825-172532",
|
||||
"environment": {
|
||||
"python_version": "3.13.11",
|
||||
"platform": "linux",
|
||||
"model": "qwen3.6:35b-a3b"
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": [
|
||||
"case_01_generation",
|
||||
"case_02_simple_tool",
|
||||
"case_03_file_analysis",
|
||||
"case_04_parallel_tool",
|
||||
"case_05_long_context",
|
||||
"case_06_reasoning",
|
||||
"case_07_multiturn"
|
||||
],
|
||||
"repeats": 2,
|
||||
"skip_heavy": false,
|
||||
"total_cases": 7
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": 7,
|
||||
"passed": 6,
|
||||
"failed": 1,
|
||||
"stopped": false,
|
||||
"total_elapsed_seconds": 357.49
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"case_id": "case_01_generation",
|
||||
"case_name": "纯文本生成基准",
|
||||
"difficulty": "简单",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_01_generation",
|
||||
"name": "纯文本生成基准",
|
||||
"description": "基础生成速度基准测试",
|
||||
"difficulty": "简单",
|
||||
"estimated_seconds": 5,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 41583.27,
|
||||
"prefill_ms": 319.34,
|
||||
"decode_speed_tok_s": 76.72,
|
||||
"total_tokens": 331,
|
||||
"prompt_tokens": 31,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 45496.03,
|
||||
"elapsed_ms": 45496.03,
|
||||
"response_length": 969
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 8929.4,
|
||||
"prefill_ms": 315.75,
|
||||
"decode_speed_tok_s": 28.7,
|
||||
"total_tokens": 331,
|
||||
"prompt_tokens": 31,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 19385.96,
|
||||
"elapsed_ms": 19385.96,
|
||||
"response_length": 961
|
||||
}
|
||||
],
|
||||
"prompt_used": "请用200字左右概述人工智能从诞生至今的关键发展阶段,每段不超过50字。"
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 32440.99,
|
||||
"std_ms": 18462.61,
|
||||
"min_ms": 19385.96,
|
||||
"max_ms": 45496.03,
|
||||
"p50_ms": 32440.99,
|
||||
"p95_ms": 44190.53,
|
||||
"p99_ms": 45234.93
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 25256.33,
|
||||
"ttft_ms_p95": 39950.58,
|
||||
"prefill_ms_mean": 317.54,
|
||||
"prefill_ms_p95": 319.16,
|
||||
"decode_speed_tok_s_mean": 52.71,
|
||||
"decode_speed_tok_s_p95": 74.32,
|
||||
"e2e_ms_mean": 32440.99,
|
||||
"e2e_ms_p95": 44190.53,
|
||||
"total_tokens_mean": 331.0,
|
||||
"total_tokens_p95": 331.0
|
||||
},
|
||||
"elapsed_seconds": 64.88
|
||||
},
|
||||
{
|
||||
"case_id": "case_02_simple_tool",
|
||||
"case_name": "简单工具调用延迟",
|
||||
"difficulty": "简单",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_02_simple_tool",
|
||||
"name": "简单工具调用延迟",
|
||||
"description": "测量简单工具调用的额外延迟",
|
||||
"difficulty": "简单",
|
||||
"estimated_seconds": 3,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 22079.81,
|
||||
"prefill_ms": 317.66,
|
||||
"decode_speed_tok_s": 77.75,
|
||||
"total_tokens": 121,
|
||||
"prompt_tokens": 61,
|
||||
"completion_tokens": 60,
|
||||
"e2e_ms": 22854.03,
|
||||
"elapsed_ms": 22854.03,
|
||||
"response_length": 117,
|
||||
"tool": "get_date",
|
||||
"matched": false
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 68555.6,
|
||||
"prefill_ms": 368.48,
|
||||
"decode_speed_tok_s": 82.77,
|
||||
"total_tokens": 122,
|
||||
"prompt_tokens": 62,
|
||||
"completion_tokens": 60,
|
||||
"e2e_ms": 69282.23,
|
||||
"elapsed_ms": 69282.23,
|
||||
"response_length": 118,
|
||||
"tool": "get_cpu_usage",
|
||||
"matched": false
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 46068.13,
|
||||
"std_ms": 32829.7,
|
||||
"min_ms": 22854.03,
|
||||
"max_ms": 69282.23,
|
||||
"p50_ms": 46068.13,
|
||||
"p95_ms": 66960.82,
|
||||
"p99_ms": 68817.95
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 45317.71,
|
||||
"ttft_ms_p95": 66231.81,
|
||||
"prefill_ms_mean": 343.07,
|
||||
"prefill_ms_p95": 365.94,
|
||||
"decode_speed_tok_s_mean": 80.26,
|
||||
"decode_speed_tok_s_p95": 82.52,
|
||||
"e2e_ms_mean": 46068.13,
|
||||
"e2e_ms_p95": 66960.82,
|
||||
"total_tokens_mean": 121.5,
|
||||
"total_tokens_p95": 121.95
|
||||
},
|
||||
"elapsed_seconds": 92.14
|
||||
},
|
||||
{
|
||||
"case_id": "case_03_file_analysis",
|
||||
"case_name": "文件读取+分析",
|
||||
"difficulty": "中等",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_03_file_analysis",
|
||||
"name": "文件读取+分析",
|
||||
"description": "文件读取与分析完整链路计时",
|
||||
"difficulty": "中等",
|
||||
"estimated_seconds": 2,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 72428.05,
|
||||
"prefill_ms": 643.53,
|
||||
"decode_speed_tok_s": 81.45,
|
||||
"total_tokens": 485,
|
||||
"prompt_tokens": 185,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 76116.09,
|
||||
"elapsed_ms": 76116.09,
|
||||
"response_length": 1083,
|
||||
"file_chars": 485
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 233.37,
|
||||
"prefill_ms": 103.1,
|
||||
"decode_speed_tok_s": 84.74,
|
||||
"total_tokens": 485,
|
||||
"prompt_tokens": 185,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 3778.0,
|
||||
"elapsed_ms": 3778.0,
|
||||
"response_length": 1111,
|
||||
"file_chars": 485
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 39947.04,
|
||||
"std_ms": 51150.75,
|
||||
"min_ms": 3778.0,
|
||||
"max_ms": 76116.09,
|
||||
"p50_ms": 39947.04,
|
||||
"p95_ms": 72499.19,
|
||||
"p99_ms": 75392.71
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 36330.71,
|
||||
"ttft_ms_p95": 68818.32,
|
||||
"prefill_ms_mean": 373.31,
|
||||
"prefill_ms_p95": 616.51,
|
||||
"decode_speed_tok_s_mean": 83.09,
|
||||
"decode_speed_tok_s_p95": 84.58,
|
||||
"e2e_ms_mean": 39947.04,
|
||||
"e2e_ms_p95": 72499.19,
|
||||
"total_tokens_mean": 485.0,
|
||||
"total_tokens_p95": 485.0
|
||||
},
|
||||
"elapsed_seconds": 79.9
|
||||
},
|
||||
{
|
||||
"case_id": "case_04_parallel_tool",
|
||||
"case_name": "并行工具调用",
|
||||
"difficulty": "中等",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_04_parallel_tool",
|
||||
"name": "并行工具调用",
|
||||
"description": "对比串行 vs 并行工具调用的延迟差异",
|
||||
"difficulty": "中等",
|
||||
"estimated_seconds": 5,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"parallel_tasks": 3,
|
||||
"serial_ms": 4045.91,
|
||||
"parallel_ms": 3855.57,
|
||||
"speedup_ratio": 1.05,
|
||||
"elapsed_ms": 3855.57,
|
||||
"avg_ttft_ms": 1662.77,
|
||||
"avg_decode_speed_tok_s": 84.11,
|
||||
"ttft_ms": 1662.77,
|
||||
"decode_speed_tok_s": 84.11
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"parallel_tasks": 3,
|
||||
"serial_ms": 4303.19,
|
||||
"parallel_ms": 3989.56,
|
||||
"speedup_ratio": 1.08,
|
||||
"elapsed_ms": 3989.56,
|
||||
"avg_ttft_ms": 1715.28,
|
||||
"avg_decode_speed_tok_s": 84.17,
|
||||
"ttft_ms": 1715.28,
|
||||
"decode_speed_tok_s": 84.17
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 3922.57,
|
||||
"std_ms": 94.75,
|
||||
"min_ms": 3855.57,
|
||||
"max_ms": 3989.56,
|
||||
"p50_ms": 3922.57,
|
||||
"p95_ms": 3982.86,
|
||||
"p99_ms": 3988.22
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 1689.03,
|
||||
"ttft_ms_p95": 1712.65,
|
||||
"decode_speed_tok_s_mean": 84.14,
|
||||
"decode_speed_tok_s_p95": 84.17
|
||||
},
|
||||
"elapsed_seconds": 16.19
|
||||
},
|
||||
{
|
||||
"case_id": "case_05_long_context",
|
||||
"case_name": "长上下文处理",
|
||||
"difficulty": "耗时",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_05_long_context",
|
||||
"name": "长上下文处理",
|
||||
"description": "不同上下文长度对延迟的影响",
|
||||
"difficulty": "耗时",
|
||||
"estimated_seconds": 10,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 820.58,
|
||||
"prefill_ms": 804.03,
|
||||
"decode_speed_tok_s": 77.15,
|
||||
"total_tokens": 728,
|
||||
"prompt_tokens": 648,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 1866.99,
|
||||
"elapsed_ms": 1866.99,
|
||||
"response_length": 319,
|
||||
"context_tokens_target": 500,
|
||||
"context_chars": 1254
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 70076.71,
|
||||
"prefill_ms": 97.84,
|
||||
"decode_speed_tok_s": 77.76,
|
||||
"total_tokens": 728,
|
||||
"prompt_tokens": 648,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 71112.83,
|
||||
"elapsed_ms": 71112.83,
|
||||
"response_length": 319,
|
||||
"context_tokens_target": 500,
|
||||
"context_chars": 1254
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 1050.97,
|
||||
"prefill_ms": 1030.17,
|
||||
"decode_speed_tok_s": 82.71,
|
||||
"total_tokens": 1355,
|
||||
"prompt_tokens": 1275,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 2030.18,
|
||||
"elapsed_ms": 2030.18,
|
||||
"response_length": 319,
|
||||
"context_tokens_target": 1000,
|
||||
"context_chars": 2508
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 121.85,
|
||||
"prefill_ms": 100.81,
|
||||
"decode_speed_tok_s": 81.98,
|
||||
"total_tokens": 1355,
|
||||
"prompt_tokens": 1275,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 1109.55,
|
||||
"elapsed_ms": 1109.55,
|
||||
"response_length": 319,
|
||||
"context_tokens_target": 1000,
|
||||
"context_chars": 2508
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 2106.57,
|
||||
"prefill_ms": 2077.92,
|
||||
"decode_speed_tok_s": 79.78,
|
||||
"total_tokens": 2628,
|
||||
"prompt_tokens": 2548,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 3132.13,
|
||||
"elapsed_ms": 3132.13,
|
||||
"response_length": 331,
|
||||
"context_tokens_target": 2000,
|
||||
"context_chars": 5054
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 133.36,
|
||||
"prefill_ms": 105.11,
|
||||
"decode_speed_tok_s": 81.07,
|
||||
"total_tokens": 2628,
|
||||
"prompt_tokens": 2548,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 1143.05,
|
||||
"elapsed_ms": 1143.05,
|
||||
"response_length": 331,
|
||||
"context_tokens_target": 2000,
|
||||
"context_chars": 5054
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 3200.13,
|
||||
"prefill_ms": 3153.14,
|
||||
"decode_speed_tok_s": 78.66,
|
||||
"total_tokens": 5155,
|
||||
"prompt_tokens": 5075,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 4259.59,
|
||||
"elapsed_ms": 4259.59,
|
||||
"response_length": 331,
|
||||
"context_tokens_target": 4000,
|
||||
"context_chars": 10108
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 165.67,
|
||||
"prefill_ms": 117.35,
|
||||
"decode_speed_tok_s": 79.5,
|
||||
"total_tokens": 5155,
|
||||
"prompt_tokens": 5075,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 1210.27,
|
||||
"elapsed_ms": 1210.27,
|
||||
"response_length": 326,
|
||||
"context_tokens_target": 4000,
|
||||
"context_chars": 10108
|
||||
}
|
||||
],
|
||||
"context_sizes_tested": [
|
||||
500,
|
||||
1000,
|
||||
2000,
|
||||
4000
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 8,
|
||||
"mean_ms": 10733.07,
|
||||
"std_ms": 24421.91,
|
||||
"min_ms": 1109.55,
|
||||
"max_ms": 71112.83,
|
||||
"p50_ms": 1948.59,
|
||||
"p95_ms": 47714.2,
|
||||
"p99_ms": 66433.1
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 9709.48,
|
||||
"ttft_ms_p95": 46669.91,
|
||||
"prefill_ms_mean": 935.8,
|
||||
"prefill_ms_p95": 2776.81,
|
||||
"decode_speed_tok_s_mean": 79.83,
|
||||
"decode_speed_tok_s_p95": 82.45,
|
||||
"e2e_ms_mean": 10733.07,
|
||||
"e2e_ms_p95": 47714.2,
|
||||
"total_tokens_mean": 2466.5,
|
||||
"total_tokens_p95": 5155.0
|
||||
},
|
||||
"elapsed_seconds": 85.87
|
||||
},
|
||||
{
|
||||
"case_id": "case_06_reasoning",
|
||||
"case_name": "复杂推理任务",
|
||||
"difficulty": "耗时",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_06_reasoning",
|
||||
"name": "复杂推理任务",
|
||||
"description": "不同推理复杂度下的延迟对比",
|
||||
"difficulty": "耗时",
|
||||
"estimated_seconds": 15,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 502.6,
|
||||
"prefill_ms": 245.68,
|
||||
"decode_speed_tok_s": 80.34,
|
||||
"total_tokens": 93,
|
||||
"prompt_tokens": 13,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 1500.12,
|
||||
"elapsed_ms": 1500.12,
|
||||
"response_length": 320,
|
||||
"reasoning_level": "简单问答"
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 220.67,
|
||||
"prefill_ms": 100.19,
|
||||
"decode_speed_tok_s": 83.2,
|
||||
"total_tokens": 93,
|
||||
"prompt_tokens": 13,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 1184.6,
|
||||
"elapsed_ms": 1184.6,
|
||||
"response_length": 317,
|
||||
"reasoning_level": "简单问答"
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 438.71,
|
||||
"prefill_ms": 268.76,
|
||||
"decode_speed_tok_s": 82.97,
|
||||
"total_tokens": 220,
|
||||
"prompt_tokens": 20,
|
||||
"completion_tokens": 200,
|
||||
"e2e_ms": 2852.11,
|
||||
"elapsed_ms": 2852.11,
|
||||
"response_length": 714,
|
||||
"reasoning_level": "中等推理"
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 232.27,
|
||||
"prefill_ms": 97.91,
|
||||
"decode_speed_tok_s": 84.27,
|
||||
"total_tokens": 220,
|
||||
"prompt_tokens": 20,
|
||||
"completion_tokens": 200,
|
||||
"e2e_ms": 2609.32,
|
||||
"elapsed_ms": 2609.32,
|
||||
"response_length": 861,
|
||||
"reasoning_level": "中等推理"
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 471.77,
|
||||
"prefill_ms": 297.05,
|
||||
"decode_speed_tok_s": 83.04,
|
||||
"total_tokens": 427,
|
||||
"prompt_tokens": 27,
|
||||
"completion_tokens": 400,
|
||||
"e2e_ms": 5293.99,
|
||||
"elapsed_ms": 5293.99,
|
||||
"response_length": 1768,
|
||||
"reasoning_level": "复杂推理"
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 234.98,
|
||||
"prefill_ms": 101.34,
|
||||
"decode_speed_tok_s": 82.71,
|
||||
"total_tokens": 427,
|
||||
"prompt_tokens": 27,
|
||||
"completion_tokens": 400,
|
||||
"e2e_ms": 5076.1,
|
||||
"elapsed_ms": 5076.1,
|
||||
"response_length": 1833,
|
||||
"reasoning_level": "复杂推理"
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 6,
|
||||
"mean_ms": 3086.04,
|
||||
"std_ms": 1746.31,
|
||||
"min_ms": 1184.6,
|
||||
"max_ms": 5293.99,
|
||||
"p50_ms": 2730.72,
|
||||
"p95_ms": 5239.52,
|
||||
"p99_ms": 5283.1
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 350.17,
|
||||
"ttft_ms_p95": 494.89,
|
||||
"prefill_ms_mean": 185.16,
|
||||
"prefill_ms_p95": 289.98,
|
||||
"decode_speed_tok_s_mean": 82.75,
|
||||
"decode_speed_tok_s_p95": 84.0,
|
||||
"e2e_ms_mean": 3086.04,
|
||||
"e2e_ms_p95": 5239.52,
|
||||
"total_tokens_mean": 246.67,
|
||||
"total_tokens_p95": 427.0
|
||||
},
|
||||
"elapsed_seconds": 18.52
|
||||
},
|
||||
{
|
||||
"case_id": "case_07_multiturn",
|
||||
"case_name": "多轮对话累积",
|
||||
"difficulty": "耗时",
|
||||
"status": "failed",
|
||||
"error": "run_test() got an unexpected keyword argument 'repeats'",
|
||||
"elapsed_seconds": 0.0
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
{
|
||||
"version": "1.0",
|
||||
"timestamp": "2026-08-25T17:27:37.663868",
|
||||
"run_id": "20260825-172737",
|
||||
"environment": {
|
||||
"python_version": "3.13.11",
|
||||
"platform": "linux",
|
||||
"model": "qwen3.6:35b-a3b"
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": [
|
||||
"case_07_multiturn"
|
||||
],
|
||||
"repeats": 1,
|
||||
"skip_heavy": false,
|
||||
"total_cases": 1
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": 1,
|
||||
"passed": 0,
|
||||
"failed": 1,
|
||||
"stopped": false,
|
||||
"total_elapsed_seconds": 0.0
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"case_id": "case_07_multiturn",
|
||||
"case_name": "多轮对话累积",
|
||||
"difficulty": "耗时",
|
||||
"status": "failed",
|
||||
"error": "run_test() got an unexpected keyword argument 'repeats'",
|
||||
"elapsed_seconds": 0.0
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
{
|
||||
"version": "1.0",
|
||||
"timestamp": "2026-08-25T17:31:46.368856",
|
||||
"run_id": "20260825-173146",
|
||||
"environment": {
|
||||
"python_version": "3.13.11",
|
||||
"platform": "linux",
|
||||
"model": "qwen3.6:35b-a3b"
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": [
|
||||
"case_07_multiturn"
|
||||
],
|
||||
"repeats": 1,
|
||||
"skip_heavy": false,
|
||||
"total_cases": 1
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": 1,
|
||||
"passed": 1,
|
||||
"failed": 0,
|
||||
"stopped": false,
|
||||
"total_elapsed_seconds": 129.17
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"case_id": "case_07_multiturn",
|
||||
"case_name": "多轮对话累积",
|
||||
"difficulty": "耗时",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_07_multiturn",
|
||||
"name": "多轮对话累积",
|
||||
"description": "追踪连续多轮对话中延迟随上下文累积的变化",
|
||||
"difficulty": "耗时",
|
||||
"estimated_seconds": 10,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 330.28,
|
||||
"prefill_ms": 330.28,
|
||||
"decode_speed_tok_s": 80.72,
|
||||
"total_tokens": 153,
|
||||
"prompt_tokens": 33,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 11799.24,
|
||||
"elapsed_ms": 11799.24,
|
||||
"response_length": 0,
|
||||
"round": 1,
|
||||
"repeat": 1,
|
||||
"scenario": "修复Python中的索引越界错误",
|
||||
"messages_in_context": 2
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 538.96,
|
||||
"prefill_ms": 538.96,
|
||||
"decode_speed_tok_s": 80.95,
|
||||
"total_tokens": 186,
|
||||
"prompt_tokens": 66,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 11741.17,
|
||||
"elapsed_ms": 11741.17,
|
||||
"response_length": 0,
|
||||
"round": 2,
|
||||
"repeat": 1,
|
||||
"scenario": "修复JavaScript闭包变量捕获问题",
|
||||
"messages_in_context": 4
|
||||
},
|
||||
{
|
||||
"iteration": 3,
|
||||
"ttft_ms": 447.1,
|
||||
"prefill_ms": 447.1,
|
||||
"decode_speed_tok_s": 84.25,
|
||||
"total_tokens": 216,
|
||||
"prompt_tokens": 96,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 32547.27,
|
||||
"elapsed_ms": 32547.27,
|
||||
"response_length": 0,
|
||||
"round": 3,
|
||||
"repeat": 1,
|
||||
"scenario": "修复SQL注入漏洞",
|
||||
"messages_in_context": 6
|
||||
},
|
||||
{
|
||||
"iteration": 4,
|
||||
"ttft_ms": 468.91,
|
||||
"prefill_ms": 468.91,
|
||||
"decode_speed_tok_s": 84.34,
|
||||
"total_tokens": 247,
|
||||
"prompt_tokens": 127,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 16912.8,
|
||||
"elapsed_ms": 16912.8,
|
||||
"response_length": 0,
|
||||
"round": 4,
|
||||
"repeat": 1,
|
||||
"scenario": "修复多线程竞态条件",
|
||||
"messages_in_context": 8
|
||||
},
|
||||
{
|
||||
"iteration": 5,
|
||||
"ttft_ms": 505.76,
|
||||
"prefill_ms": 505.76,
|
||||
"decode_speed_tok_s": 80.63,
|
||||
"total_tokens": 277,
|
||||
"prompt_tokens": 157,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 42758.37,
|
||||
"elapsed_ms": 42758.37,
|
||||
"response_length": 0,
|
||||
"round": 5,
|
||||
"repeat": 1,
|
||||
"scenario": "修复递归栈溢出",
|
||||
"messages_in_context": 10
|
||||
},
|
||||
{
|
||||
"iteration": 6,
|
||||
"ttft_ms": 467.09,
|
||||
"prefill_ms": 467.09,
|
||||
"decode_speed_tok_s": 80.18,
|
||||
"total_tokens": 308,
|
||||
"prompt_tokens": 188,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 13412.74,
|
||||
"elapsed_ms": 13412.74,
|
||||
"response_length": 0,
|
||||
"round": 6,
|
||||
"repeat": 1,
|
||||
"scenario": "修复正则表达式灾难回溯",
|
||||
"messages_in_context": 12
|
||||
}
|
||||
],
|
||||
"total_rounds": 6
|
||||
},
|
||||
"statistics": {
|
||||
"count": 6,
|
||||
"mean_ms": 21528.6,
|
||||
"std_ms": 13036.42,
|
||||
"min_ms": 11741.17,
|
||||
"max_ms": 42758.37,
|
||||
"p50_ms": 15162.77,
|
||||
"p95_ms": 40205.6,
|
||||
"p99_ms": 42247.82
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 459.68,
|
||||
"ttft_ms_p95": 530.66,
|
||||
"prefill_ms_mean": 459.68,
|
||||
"prefill_ms_p95": 530.66,
|
||||
"decode_speed_tok_s_mean": 81.84,
|
||||
"decode_speed_tok_s_p95": 84.32,
|
||||
"e2e_ms_mean": 21528.6,
|
||||
"e2e_ms_p95": 40205.6,
|
||||
"total_tokens_mean": 231.17,
|
||||
"total_tokens_p95": 300.25
|
||||
},
|
||||
"elapsed_seconds": 129.17
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,785 @@
|
||||
{
|
||||
"version": "1.0",
|
||||
"timestamp": "2026-08-25T17:44:25.622293",
|
||||
"run_id": "20260825-174425",
|
||||
"environment": {
|
||||
"python_version": "3.13.11",
|
||||
"platform": "linux",
|
||||
"model": "qwen3.6:35b-a3b"
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": [
|
||||
"case_01_generation",
|
||||
"case_02_simple_tool",
|
||||
"case_03_file_analysis",
|
||||
"case_04_parallel_tool",
|
||||
"case_05_long_context",
|
||||
"case_06_reasoning",
|
||||
"case_07_multiturn"
|
||||
],
|
||||
"repeats": 2,
|
||||
"skip_heavy": false,
|
||||
"total_cases": 7
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": 7,
|
||||
"passed": 7,
|
||||
"failed": 0,
|
||||
"stopped": false,
|
||||
"total_elapsed_seconds": 728.71
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"case_id": "case_01_generation",
|
||||
"case_name": "纯文本生成基准",
|
||||
"difficulty": "简单",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_01_generation",
|
||||
"name": "纯文本生成基准",
|
||||
"description": "基础生成速度基准测试",
|
||||
"difficulty": "简单",
|
||||
"estimated_seconds": 5,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 14715.57,
|
||||
"prefill_ms": 328.38,
|
||||
"decode_speed_tok_s": 80.65,
|
||||
"total_tokens": 331,
|
||||
"prompt_tokens": 31,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 18438.39,
|
||||
"elapsed_ms": 18438.39,
|
||||
"response_length": 912
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 13780.79,
|
||||
"prefill_ms": 338.35,
|
||||
"decode_speed_tok_s": 79.88,
|
||||
"total_tokens": 331,
|
||||
"prompt_tokens": 31,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 17539.77,
|
||||
"elapsed_ms": 17539.77,
|
||||
"response_length": 988
|
||||
}
|
||||
],
|
||||
"prompt_used": "请用200字左右概述人工智能从诞生至今的关键发展阶段,每段不超过50字。"
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 17989.08,
|
||||
"std_ms": 635.42,
|
||||
"min_ms": 17539.77,
|
||||
"max_ms": 18438.39,
|
||||
"p50_ms": 17989.08,
|
||||
"p95_ms": 18393.46,
|
||||
"p99_ms": 18429.4
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 14248.18,
|
||||
"ttft_ms_p95": 14668.83,
|
||||
"prefill_ms_mean": 333.37,
|
||||
"prefill_ms_p95": 337.85,
|
||||
"decode_speed_tok_s_mean": 80.27,
|
||||
"decode_speed_tok_s_p95": 80.61,
|
||||
"e2e_ms_mean": 17989.08,
|
||||
"e2e_ms_p95": 18393.46,
|
||||
"total_tokens_mean": 331.0,
|
||||
"total_tokens_p95": 331.0
|
||||
},
|
||||
"elapsed_seconds": 35.98
|
||||
},
|
||||
{
|
||||
"case_id": "case_02_simple_tool",
|
||||
"case_name": "简单工具调用延迟",
|
||||
"difficulty": "简单",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_02_simple_tool",
|
||||
"name": "简单工具调用延迟",
|
||||
"description": "测量简单工具调用的额外延迟",
|
||||
"difficulty": "简单",
|
||||
"estimated_seconds": 3,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 16196.0,
|
||||
"prefill_ms": 282.79,
|
||||
"decode_speed_tok_s": 84.08,
|
||||
"total_tokens": 83,
|
||||
"prompt_tokens": 61,
|
||||
"completion_tokens": 22,
|
||||
"e2e_ms": 16459.21,
|
||||
"elapsed_ms": 16459.21,
|
||||
"response_length": 60,
|
||||
"tool": "get_date",
|
||||
"matched": false
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 14408.31,
|
||||
"prefill_ms": 289.76,
|
||||
"decode_speed_tok_s": 81.91,
|
||||
"total_tokens": 122,
|
||||
"prompt_tokens": 62,
|
||||
"completion_tokens": 60,
|
||||
"e2e_ms": 15142.91,
|
||||
"elapsed_ms": 15142.91,
|
||||
"response_length": 117,
|
||||
"tool": "get_cpu_usage",
|
||||
"matched": false
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 15801.06,
|
||||
"std_ms": 930.76,
|
||||
"min_ms": 15142.91,
|
||||
"max_ms": 16459.21,
|
||||
"p50_ms": 15801.06,
|
||||
"p95_ms": 16393.4,
|
||||
"p99_ms": 16446.05
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 15302.15,
|
||||
"ttft_ms_p95": 16106.62,
|
||||
"prefill_ms_mean": 286.27,
|
||||
"prefill_ms_p95": 289.41,
|
||||
"decode_speed_tok_s_mean": 83.0,
|
||||
"decode_speed_tok_s_p95": 83.97,
|
||||
"e2e_ms_mean": 15801.06,
|
||||
"e2e_ms_p95": 16393.4,
|
||||
"total_tokens_mean": 102.5,
|
||||
"total_tokens_p95": 120.05
|
||||
},
|
||||
"elapsed_seconds": 31.6
|
||||
},
|
||||
{
|
||||
"case_id": "case_03_file_analysis",
|
||||
"case_name": "文件读取+分析",
|
||||
"difficulty": "中等",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_03_file_analysis",
|
||||
"name": "文件读取+分析",
|
||||
"description": "文件读取与分析完整链路计时",
|
||||
"difficulty": "中等",
|
||||
"estimated_seconds": 2,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 4497.2,
|
||||
"prefill_ms": 96.54,
|
||||
"decode_speed_tok_s": 81.86,
|
||||
"total_tokens": 485,
|
||||
"prompt_tokens": 185,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 8166.59,
|
||||
"elapsed_ms": 8166.59,
|
||||
"response_length": 1122,
|
||||
"file_chars": 485
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 30717.41,
|
||||
"prefill_ms": 103.84,
|
||||
"decode_speed_tok_s": 79.84,
|
||||
"total_tokens": 485,
|
||||
"prompt_tokens": 185,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 34478.78,
|
||||
"elapsed_ms": 34478.78,
|
||||
"response_length": 1075,
|
||||
"file_chars": 485
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 21322.68,
|
||||
"std_ms": 18605.53,
|
||||
"min_ms": 8166.59,
|
||||
"max_ms": 34478.78,
|
||||
"p50_ms": 21322.68,
|
||||
"p95_ms": 33163.17,
|
||||
"p99_ms": 34215.66
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 17607.31,
|
||||
"ttft_ms_p95": 29406.4,
|
||||
"prefill_ms_mean": 100.19,
|
||||
"prefill_ms_p95": 103.47,
|
||||
"decode_speed_tok_s_mean": 80.85,
|
||||
"decode_speed_tok_s_p95": 81.76,
|
||||
"e2e_ms_mean": 21322.68,
|
||||
"e2e_ms_p95": 33163.17,
|
||||
"total_tokens_mean": 485.0,
|
||||
"total_tokens_p95": 485.0
|
||||
},
|
||||
"elapsed_seconds": 42.65
|
||||
},
|
||||
{
|
||||
"case_id": "case_04_parallel_tool",
|
||||
"case_name": "并行工具调用",
|
||||
"difficulty": "中等",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_04_parallel_tool",
|
||||
"name": "并行工具调用",
|
||||
"description": "对比串行 vs 并行工具调用的延迟差异",
|
||||
"difficulty": "中等",
|
||||
"estimated_seconds": 5,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"parallel_tasks": 3,
|
||||
"serial_ms": 37309.93,
|
||||
"parallel_ms": 45353.04,
|
||||
"speedup_ratio": 0.82,
|
||||
"elapsed_ms": 45353.04,
|
||||
"avg_ttft_ms": 43165.98,
|
||||
"avg_decode_speed_tok_s": 83.01,
|
||||
"ttft_ms": 43165.98,
|
||||
"decode_speed_tok_s": 83.01
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"parallel_tasks": 3,
|
||||
"serial_ms": 30577.28,
|
||||
"parallel_ms": 10193.83,
|
||||
"speedup_ratio": 3.0,
|
||||
"elapsed_ms": 10193.83,
|
||||
"avg_ttft_ms": 7917.49,
|
||||
"avg_decode_speed_tok_s": 81.22,
|
||||
"ttft_ms": 7917.49,
|
||||
"decode_speed_tok_s": 81.22
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 2,
|
||||
"mean_ms": 27773.44,
|
||||
"std_ms": 24861.32,
|
||||
"min_ms": 10193.83,
|
||||
"max_ms": 45353.04,
|
||||
"p50_ms": 27773.44,
|
||||
"p95_ms": 43595.08,
|
||||
"p99_ms": 45001.45
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 25541.74,
|
||||
"ttft_ms_p95": 41403.56,
|
||||
"decode_speed_tok_s_mean": 82.12,
|
||||
"decode_speed_tok_s_p95": 82.92
|
||||
},
|
||||
"elapsed_seconds": 123.43
|
||||
},
|
||||
{
|
||||
"case_id": "case_05_long_context",
|
||||
"case_name": "长上下文处理",
|
||||
"difficulty": "耗时",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_05_long_context",
|
||||
"name": "长上下文处理",
|
||||
"description": "不同上下文长度对延迟的影响",
|
||||
"difficulty": "耗时",
|
||||
"estimated_seconds": 10,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 5375.1,
|
||||
"prefill_ms": 842.82,
|
||||
"decode_speed_tok_s": 80.26,
|
||||
"total_tokens": 728,
|
||||
"prompt_tokens": 648,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 6378.84,
|
||||
"elapsed_ms": 6378.84,
|
||||
"response_length": 319,
|
||||
"context_tokens_target": 500,
|
||||
"context_chars": 1254
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 57883.93,
|
||||
"prefill_ms": 95.78,
|
||||
"decode_speed_tok_s": 83.77,
|
||||
"total_tokens": 728,
|
||||
"prompt_tokens": 648,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 58844.25,
|
||||
"elapsed_ms": 58844.25,
|
||||
"response_length": 319,
|
||||
"context_tokens_target": 500,
|
||||
"context_chars": 1254
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 14493.28,
|
||||
"prefill_ms": 1041.01,
|
||||
"decode_speed_tok_s": 81.17,
|
||||
"total_tokens": 1355,
|
||||
"prompt_tokens": 1275,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 15491.6,
|
||||
"elapsed_ms": 15491.6,
|
||||
"response_length": 315,
|
||||
"context_tokens_target": 1000,
|
||||
"context_chars": 2508
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 88553.55,
|
||||
"prefill_ms": 102.99,
|
||||
"decode_speed_tok_s": 82.75,
|
||||
"total_tokens": 1355,
|
||||
"prompt_tokens": 1275,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 89532.55,
|
||||
"elapsed_ms": 89532.55,
|
||||
"response_length": 319,
|
||||
"context_tokens_target": 1000,
|
||||
"context_chars": 2508
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 19499.27,
|
||||
"prefill_ms": 1513.23,
|
||||
"decode_speed_tok_s": 80.77,
|
||||
"total_tokens": 2628,
|
||||
"prompt_tokens": 2548,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 20512.64,
|
||||
"elapsed_ms": 20512.64,
|
||||
"response_length": 331,
|
||||
"context_tokens_target": 2000,
|
||||
"context_chars": 5054
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 11929.32,
|
||||
"prefill_ms": 114.78,
|
||||
"decode_speed_tok_s": 77.52,
|
||||
"total_tokens": 2628,
|
||||
"prompt_tokens": 2548,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 12983.72,
|
||||
"elapsed_ms": 12983.72,
|
||||
"response_length": 329,
|
||||
"context_tokens_target": 2000,
|
||||
"context_chars": 5054
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 14999.33,
|
||||
"prefill_ms": 3185.49,
|
||||
"decode_speed_tok_s": 79.84,
|
||||
"total_tokens": 5155,
|
||||
"prompt_tokens": 5075,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 16042.1,
|
||||
"elapsed_ms": 16042.1,
|
||||
"response_length": 329,
|
||||
"context_tokens_target": 4000,
|
||||
"context_chars": 10108
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 9751.38,
|
||||
"prefill_ms": 116.54,
|
||||
"decode_speed_tok_s": 83.8,
|
||||
"total_tokens": 5155,
|
||||
"prompt_tokens": 5075,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 10748.62,
|
||||
"elapsed_ms": 10748.62,
|
||||
"response_length": 331,
|
||||
"context_tokens_target": 4000,
|
||||
"context_chars": 10108
|
||||
}
|
||||
],
|
||||
"context_sizes_tested": [
|
||||
500,
|
||||
1000,
|
||||
2000,
|
||||
4000
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 8,
|
||||
"mean_ms": 28816.79,
|
||||
"std_ms": 29467.59,
|
||||
"min_ms": 6378.84,
|
||||
"max_ms": 89532.55,
|
||||
"p50_ms": 15766.85,
|
||||
"p95_ms": 78791.64,
|
||||
"p99_ms": 87384.37
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 27810.65,
|
||||
"ttft_ms_p95": 77819.18,
|
||||
"prefill_ms_mean": 876.58,
|
||||
"prefill_ms_p95": 2600.2,
|
||||
"decode_speed_tok_s_mean": 81.23,
|
||||
"decode_speed_tok_s_p95": 83.79,
|
||||
"e2e_ms_mean": 28816.79,
|
||||
"e2e_ms_p95": 78791.64,
|
||||
"total_tokens_mean": 2466.5,
|
||||
"total_tokens_p95": 5155.0
|
||||
},
|
||||
"elapsed_seconds": 230.54
|
||||
},
|
||||
{
|
||||
"case_id": "case_06_reasoning",
|
||||
"case_name": "复杂推理任务",
|
||||
"difficulty": "耗时",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_06_reasoning",
|
||||
"name": "复杂推理任务",
|
||||
"description": "不同推理复杂度下的延迟对比",
|
||||
"difficulty": "耗时",
|
||||
"estimated_seconds": 15,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 17139.94,
|
||||
"prefill_ms": 284.82,
|
||||
"decode_speed_tok_s": 81.61,
|
||||
"total_tokens": 93,
|
||||
"prompt_tokens": 13,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 18122.0,
|
||||
"elapsed_ms": 18122.0,
|
||||
"response_length": 314,
|
||||
"reasoning_level": "简单问答"
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 47602.11,
|
||||
"prefill_ms": 285.31,
|
||||
"decode_speed_tok_s": 80.04,
|
||||
"total_tokens": 93,
|
||||
"prompt_tokens": 13,
|
||||
"completion_tokens": 80,
|
||||
"e2e_ms": 48603.85,
|
||||
"elapsed_ms": 48603.85,
|
||||
"response_length": 313,
|
||||
"reasoning_level": "简单问答"
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 10822.25,
|
||||
"prefill_ms": 315.46,
|
||||
"decode_speed_tok_s": 82.22,
|
||||
"total_tokens": 220,
|
||||
"prompt_tokens": 20,
|
||||
"completion_tokens": 200,
|
||||
"e2e_ms": 13257.81,
|
||||
"elapsed_ms": 13257.81,
|
||||
"response_length": 811,
|
||||
"reasoning_level": "中等推理"
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 20657.36,
|
||||
"prefill_ms": 314.18,
|
||||
"decode_speed_tok_s": 80.9,
|
||||
"total_tokens": 220,
|
||||
"prompt_tokens": 20,
|
||||
"completion_tokens": 200,
|
||||
"e2e_ms": 23132.11,
|
||||
"elapsed_ms": 23132.11,
|
||||
"response_length": 745,
|
||||
"reasoning_level": "中等推理"
|
||||
},
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 8117.27,
|
||||
"prefill_ms": 335.93,
|
||||
"decode_speed_tok_s": 82.11,
|
||||
"total_tokens": 427,
|
||||
"prompt_tokens": 27,
|
||||
"completion_tokens": 400,
|
||||
"e2e_ms": 12993.07,
|
||||
"elapsed_ms": 12993.07,
|
||||
"response_length": 1762,
|
||||
"reasoning_level": "复杂推理"
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 6073.91,
|
||||
"prefill_ms": 333.3,
|
||||
"decode_speed_tok_s": 81.79,
|
||||
"total_tokens": 427,
|
||||
"prompt_tokens": 27,
|
||||
"completion_tokens": 400,
|
||||
"e2e_ms": 10969.79,
|
||||
"elapsed_ms": 10969.79,
|
||||
"response_length": 1809,
|
||||
"reasoning_level": "复杂推理"
|
||||
}
|
||||
]
|
||||
},
|
||||
"statistics": {
|
||||
"count": 6,
|
||||
"mean_ms": 21179.77,
|
||||
"std_ms": 14136.94,
|
||||
"min_ms": 10969.79,
|
||||
"max_ms": 48603.85,
|
||||
"p50_ms": 15689.9,
|
||||
"p95_ms": 42235.91,
|
||||
"p99_ms": 47330.26
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 18402.14,
|
||||
"ttft_ms_p95": 40865.92,
|
||||
"prefill_ms_mean": 311.5,
|
||||
"prefill_ms_p95": 335.27,
|
||||
"decode_speed_tok_s_mean": 81.45,
|
||||
"decode_speed_tok_s_p95": 82.19,
|
||||
"e2e_ms_mean": 21179.77,
|
||||
"e2e_ms_p95": 42235.91,
|
||||
"total_tokens_mean": 246.67,
|
||||
"total_tokens_p95": 427.0
|
||||
},
|
||||
"elapsed_seconds": 127.08
|
||||
},
|
||||
{
|
||||
"case_id": "case_07_multiturn",
|
||||
"case_name": "多轮对话累积",
|
||||
"difficulty": "耗时",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_07_multiturn",
|
||||
"name": "多轮对话累积",
|
||||
"description": "追踪连续多轮对话中延迟随上下文累积的变化",
|
||||
"difficulty": "耗时",
|
||||
"estimated_seconds": 10,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 352.84,
|
||||
"prefill_ms": 352.84,
|
||||
"decode_speed_tok_s": 81.88,
|
||||
"total_tokens": 153,
|
||||
"prompt_tokens": 33,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 7458.81,
|
||||
"elapsed_ms": 7458.81,
|
||||
"response_length": 0,
|
||||
"round": 1,
|
||||
"repeat": 1,
|
||||
"scenario": "修复Python中的索引越界错误",
|
||||
"messages_in_context": 2
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 570.91,
|
||||
"prefill_ms": 570.91,
|
||||
"decode_speed_tok_s": 83.44,
|
||||
"total_tokens": 186,
|
||||
"prompt_tokens": 66,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 9924.72,
|
||||
"elapsed_ms": 9924.72,
|
||||
"response_length": 0,
|
||||
"round": 2,
|
||||
"repeat": 1,
|
||||
"scenario": "修复JavaScript闭包变量捕获问题",
|
||||
"messages_in_context": 4
|
||||
},
|
||||
{
|
||||
"iteration": 3,
|
||||
"ttft_ms": 592.78,
|
||||
"prefill_ms": 592.78,
|
||||
"decode_speed_tok_s": 80.86,
|
||||
"total_tokens": 216,
|
||||
"prompt_tokens": 96,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 8754.91,
|
||||
"elapsed_ms": 8754.91,
|
||||
"response_length": 0,
|
||||
"round": 3,
|
||||
"repeat": 1,
|
||||
"scenario": "修复SQL注入漏洞",
|
||||
"messages_in_context": 6
|
||||
},
|
||||
{
|
||||
"iteration": 4,
|
||||
"ttft_ms": 468.43,
|
||||
"prefill_ms": 468.43,
|
||||
"decode_speed_tok_s": 83.91,
|
||||
"total_tokens": 247,
|
||||
"prompt_tokens": 127,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 8867.78,
|
||||
"elapsed_ms": 8867.78,
|
||||
"response_length": 0,
|
||||
"round": 4,
|
||||
"repeat": 1,
|
||||
"scenario": "修复多线程竞态条件",
|
||||
"messages_in_context": 8
|
||||
},
|
||||
{
|
||||
"iteration": 5,
|
||||
"ttft_ms": 472.24,
|
||||
"prefill_ms": 472.24,
|
||||
"decode_speed_tok_s": 81.92,
|
||||
"total_tokens": 277,
|
||||
"prompt_tokens": 157,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 9069.2,
|
||||
"elapsed_ms": 9069.2,
|
||||
"response_length": 0,
|
||||
"round": 5,
|
||||
"repeat": 1,
|
||||
"scenario": "修复递归栈溢出",
|
||||
"messages_in_context": 10
|
||||
},
|
||||
{
|
||||
"iteration": 6,
|
||||
"ttft_ms": 496.68,
|
||||
"prefill_ms": 496.68,
|
||||
"decode_speed_tok_s": 83.06,
|
||||
"total_tokens": 308,
|
||||
"prompt_tokens": 188,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 9110.29,
|
||||
"elapsed_ms": 9110.29,
|
||||
"response_length": 0,
|
||||
"round": 6,
|
||||
"repeat": 1,
|
||||
"scenario": "修复正则表达式灾难回溯",
|
||||
"messages_in_context": 12
|
||||
},
|
||||
{
|
||||
"iteration": 7,
|
||||
"ttft_ms": 356.02,
|
||||
"prefill_ms": 356.02,
|
||||
"decode_speed_tok_s": 82.34,
|
||||
"total_tokens": 153,
|
||||
"prompt_tokens": 33,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 9302.89,
|
||||
"elapsed_ms": 9302.89,
|
||||
"response_length": 0,
|
||||
"round": 1,
|
||||
"repeat": 2,
|
||||
"scenario": "修复Python中的索引越界错误",
|
||||
"messages_in_context": 2
|
||||
},
|
||||
{
|
||||
"iteration": 8,
|
||||
"ttft_ms": 105.61,
|
||||
"prefill_ms": 105.61,
|
||||
"decode_speed_tok_s": 80.39,
|
||||
"total_tokens": 186,
|
||||
"prompt_tokens": 66,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 9569.96,
|
||||
"elapsed_ms": 9569.96,
|
||||
"response_length": 0,
|
||||
"round": 2,
|
||||
"repeat": 2,
|
||||
"scenario": "修复JavaScript闭包变量捕获问题",
|
||||
"messages_in_context": 4
|
||||
},
|
||||
{
|
||||
"iteration": 9,
|
||||
"ttft_ms": 596.24,
|
||||
"prefill_ms": 596.24,
|
||||
"decode_speed_tok_s": 78.49,
|
||||
"total_tokens": 216,
|
||||
"prompt_tokens": 96,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 9446.57,
|
||||
"elapsed_ms": 9446.57,
|
||||
"response_length": 0,
|
||||
"round": 3,
|
||||
"repeat": 2,
|
||||
"scenario": "修复SQL注入漏洞",
|
||||
"messages_in_context": 6
|
||||
},
|
||||
{
|
||||
"iteration": 10,
|
||||
"ttft_ms": 468.0,
|
||||
"prefill_ms": 468.0,
|
||||
"decode_speed_tok_s": 81.77,
|
||||
"total_tokens": 247,
|
||||
"prompt_tokens": 127,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 7437.1,
|
||||
"elapsed_ms": 7437.1,
|
||||
"response_length": 0,
|
||||
"round": 4,
|
||||
"repeat": 2,
|
||||
"scenario": "修复多线程竞态条件",
|
||||
"messages_in_context": 8
|
||||
},
|
||||
{
|
||||
"iteration": 11,
|
||||
"ttft_ms": 470.8,
|
||||
"prefill_ms": 470.8,
|
||||
"decode_speed_tok_s": 81.38,
|
||||
"total_tokens": 277,
|
||||
"prompt_tokens": 157,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 22114.83,
|
||||
"elapsed_ms": 22114.83,
|
||||
"response_length": 0,
|
||||
"round": 5,
|
||||
"repeat": 2,
|
||||
"scenario": "修复递归栈溢出",
|
||||
"messages_in_context": 10
|
||||
},
|
||||
{
|
||||
"iteration": 12,
|
||||
"ttft_ms": 503.16,
|
||||
"prefill_ms": 503.16,
|
||||
"decode_speed_tok_s": 81.86,
|
||||
"total_tokens": 308,
|
||||
"prompt_tokens": 188,
|
||||
"completion_tokens": 120,
|
||||
"e2e_ms": 26371.35,
|
||||
"elapsed_ms": 26371.35,
|
||||
"response_length": 0,
|
||||
"round": 6,
|
||||
"repeat": 2,
|
||||
"scenario": "修复正则表达式灾难回溯",
|
||||
"messages_in_context": 12
|
||||
}
|
||||
],
|
||||
"total_rounds": 12
|
||||
},
|
||||
"statistics": {
|
||||
"count": 12,
|
||||
"mean_ms": 11452.37,
|
||||
"std_ms": 6090.06,
|
||||
"min_ms": 7437.1,
|
||||
"max_ms": 26371.35,
|
||||
"p50_ms": 9206.59,
|
||||
"p95_ms": 24030.26,
|
||||
"p99_ms": 25903.13
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 454.48,
|
||||
"ttft_ms_p95": 594.34,
|
||||
"prefill_ms_mean": 454.48,
|
||||
"prefill_ms_p95": 594.34,
|
||||
"decode_speed_tok_s_mean": 81.77,
|
||||
"decode_speed_tok_s_p95": 83.65,
|
||||
"e2e_ms_mean": 11452.37,
|
||||
"e2e_ms_p95": 24030.26,
|
||||
"total_tokens_mean": 231.17,
|
||||
"total_tokens_p95": 308.0
|
||||
},
|
||||
"elapsed_seconds": 137.43
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,268 @@
|
||||
{
|
||||
"version": "1.0",
|
||||
"timestamp": "2026-08-26T00:43:28.377694",
|
||||
"run_id": "20260826-004328",
|
||||
"environment": {
|
||||
"python_version": "3.12.0",
|
||||
"platform": "win32",
|
||||
"model": "Q3.6-35B-A3B-Orig-Thi",
|
||||
"backend": "openai"
|
||||
},
|
||||
"config": {
|
||||
"cases_requested": [
|
||||
"case_01_generation",
|
||||
"case_08_concurrency"
|
||||
],
|
||||
"repeats": 1,
|
||||
"skip_heavy": false,
|
||||
"total_cases": 2
|
||||
},
|
||||
"summary": {
|
||||
"total_cases": 2,
|
||||
"passed": 2,
|
||||
"failed": 0,
|
||||
"stopped": false,
|
||||
"total_elapsed_seconds": 16.39
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"case_id": "case_01_generation",
|
||||
"case_name": "纯文本生成基准",
|
||||
"difficulty": "简单",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_01_generation",
|
||||
"name": "纯文本生成基准",
|
||||
"description": "基础生成速度基准测试",
|
||||
"difficulty": "简单",
|
||||
"estimated_seconds": 5,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 1469.4,
|
||||
"prefill_ms": 176.92,
|
||||
"decode_speed_tok_s": 56.05,
|
||||
"total_tokens": 331,
|
||||
"prompt_tokens": 31,
|
||||
"completion_tokens": 300,
|
||||
"e2e_ms": 5781.87,
|
||||
"elapsed_ms": 5781.87,
|
||||
"response_length": 931
|
||||
}
|
||||
],
|
||||
"prompt_used": "请用200字左右概述人工智能从诞生至今的关键发展阶段,每段不超过50字。"
|
||||
},
|
||||
"statistics": {
|
||||
"count": 1,
|
||||
"mean_ms": 5781.87,
|
||||
"std_ms": 0.0,
|
||||
"min_ms": 5781.87,
|
||||
"max_ms": 5781.87,
|
||||
"p50_ms": 5781.87,
|
||||
"p95_ms": 5781.87,
|
||||
"p99_ms": 5781.87
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 1469.4,
|
||||
"ttft_ms_p95": 1469.4,
|
||||
"prefill_ms_mean": 176.92,
|
||||
"prefill_ms_p95": 176.92,
|
||||
"e2e_ms_mean": 5781.87,
|
||||
"e2e_ms_p95": 5781.87,
|
||||
"decode_speed_tok_s_mean": 56.05,
|
||||
"decode_speed_tok_s_p95": 56.05,
|
||||
"total_tokens_mean": 331.0,
|
||||
"total_tokens_p95": 331
|
||||
},
|
||||
"elapsed_seconds": 5.78
|
||||
},
|
||||
{
|
||||
"case_id": "case_08_concurrency",
|
||||
"case_name": "并发压力测试",
|
||||
"difficulty": "压力",
|
||||
"status": "passed",
|
||||
"raw_data": {
|
||||
"case_id": "case_08_concurrency",
|
||||
"name": "并发压力测试",
|
||||
"description": "高并发下的 TPS 保持率、P95/P99 延迟、错误率",
|
||||
"difficulty": "压力",
|
||||
"estimated_seconds": 30,
|
||||
"results": [
|
||||
{
|
||||
"iteration": 1,
|
||||
"ttft_ms": 47.68,
|
||||
"prefill_ms": 47.68,
|
||||
"decode_speed_tok_s": 39.99,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 1737.06,
|
||||
"elapsed_ms": 1737.06,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 2,
|
||||
"ttft_ms": 68.83,
|
||||
"prefill_ms": 68.83,
|
||||
"decode_speed_tok_s": 36.31,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 1984.56,
|
||||
"elapsed_ms": 1984.56,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 3,
|
||||
"ttft_ms": 66.03,
|
||||
"prefill_ms": 66.03,
|
||||
"decode_speed_tok_s": 39.33,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 2687.98,
|
||||
"elapsed_ms": 2687.98,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 4,
|
||||
"ttft_ms": 67.81,
|
||||
"prefill_ms": 67.81,
|
||||
"decode_speed_tok_s": 39.4,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 2873.81,
|
||||
"elapsed_ms": 2873.81,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 5,
|
||||
"ttft_ms": 66.02,
|
||||
"prefill_ms": 66.02,
|
||||
"decode_speed_tok_s": 39.32,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 3567.66,
|
||||
"elapsed_ms": 3567.66,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 6,
|
||||
"ttft_ms": 67.78,
|
||||
"prefill_ms": 67.78,
|
||||
"decode_speed_tok_s": 39.37,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 3751.12,
|
||||
"elapsed_ms": 3751.12,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 7,
|
||||
"ttft_ms": 66.06,
|
||||
"prefill_ms": 66.06,
|
||||
"decode_speed_tok_s": 39.32,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 4435.97,
|
||||
"elapsed_ms": 4435.97,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 8,
|
||||
"ttft_ms": 67.75,
|
||||
"prefill_ms": 67.75,
|
||||
"decode_speed_tok_s": 39.41,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 4628.97,
|
||||
"elapsed_ms": 4628.97,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 9,
|
||||
"ttft_ms": 65.95,
|
||||
"prefill_ms": 65.95,
|
||||
"decode_speed_tok_s": 39.36,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 5330.84,
|
||||
"elapsed_ms": 5330.84,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
},
|
||||
{
|
||||
"iteration": 10,
|
||||
"ttft_ms": 67.72,
|
||||
"prefill_ms": 67.72,
|
||||
"decode_speed_tok_s": 43.93,
|
||||
"total_tokens": 36,
|
||||
"prompt_tokens": 4,
|
||||
"completion_tokens": 32,
|
||||
"e2e_ms": 5430.27,
|
||||
"elapsed_ms": 5430.27,
|
||||
"response_length": 91,
|
||||
"batch": "concurrent",
|
||||
"error": ""
|
||||
}
|
||||
],
|
||||
"concurrency": 10,
|
||||
"baseline_tps": 57.41,
|
||||
"concurrent_tps": 39.57,
|
||||
"tps_keep_rate_pct": 68.93,
|
||||
"e2e_p50_ms": 3659.39,
|
||||
"e2e_p95_ms": 5385.53,
|
||||
"e2e_p99_ms": 5421.32,
|
||||
"success_count": 10,
|
||||
"failed_count": 0,
|
||||
"error_rate_pct": 0.0
|
||||
},
|
||||
"statistics": {
|
||||
"count": 10,
|
||||
"mean_ms": 3642.82,
|
||||
"std_ms": 1314.37,
|
||||
"min_ms": 1737.06,
|
||||
"max_ms": 5430.27,
|
||||
"p50_ms": 3659.39,
|
||||
"p95_ms": 5385.53,
|
||||
"p99_ms": 5421.32
|
||||
},
|
||||
"inference_metrics": {
|
||||
"ttft_ms_mean": 65.16,
|
||||
"ttft_ms_p95": 68.37,
|
||||
"prefill_ms_mean": 65.16,
|
||||
"prefill_ms_p95": 68.37,
|
||||
"e2e_ms_mean": 3642.82,
|
||||
"e2e_ms_p95": 5385.53,
|
||||
"decode_speed_tok_s_mean": 39.57,
|
||||
"decode_speed_tok_s_p95": 42.16,
|
||||
"total_tokens_mean": 36.0,
|
||||
"total_tokens_p95": 36.0
|
||||
},
|
||||
"elapsed_seconds": 10.61
|
||||
}
|
||||
]
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,11 @@
|
||||
"""大模型速度测试助手 - 启动入口
|
||||
|
||||
用法:
|
||||
python run.py
|
||||
|
||||
启动后访问 http://localhost:8000
|
||||
"""
|
||||
from backend.server import main # noqa: E402
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,76 @@
|
||||
"""统计模块测试"""
|
||||
import math
|
||||
|
||||
from backend.stats import mean, std, percentile, percentiles, min_val, max_val
|
||||
|
||||
|
||||
class TestMean:
|
||||
def test_normal(self):
|
||||
assert mean([1, 2, 3, 4, 5]) == 3.0
|
||||
|
||||
def test_single(self):
|
||||
assert mean([42]) == 42.0
|
||||
|
||||
def test_empty(self):
|
||||
assert mean([]) == 0.0
|
||||
|
||||
def test_floats(self):
|
||||
assert abs(mean([1.5, 2.5, 3.5]) - 2.5) < 1e-9
|
||||
|
||||
|
||||
class TestStd:
|
||||
def test_known_std(self):
|
||||
data = [2, 4, 4, 4, 5, 5, 7, 9]
|
||||
assert abs(std(data) - 2.13809) < 1e-4
|
||||
|
||||
def test_single(self):
|
||||
assert std([42]) == 0.0
|
||||
|
||||
def test_two_values(self):
|
||||
assert std([0, 2]) == math.sqrt(2)
|
||||
|
||||
def test_empty(self):
|
||||
assert std([]) == 0.0
|
||||
|
||||
|
||||
class TestPercentile:
|
||||
def test_median(self):
|
||||
assert percentile([1, 2, 3, 4, 5], 50) == 3.0
|
||||
|
||||
def test_p0(self):
|
||||
assert percentile([1, 2, 3, 4, 5], 0) == 1.0
|
||||
|
||||
def test_p100(self):
|
||||
assert percentile([1, 2, 3, 4, 5], 100) == 5.0
|
||||
|
||||
def test_empty(self):
|
||||
assert percentile([], 50) == 0.0
|
||||
|
||||
def test_interpolation(self):
|
||||
assert percentile([1, 2], 50) == 1.5
|
||||
|
||||
|
||||
class TestPercentiles:
|
||||
def test_batch(self):
|
||||
result = percentiles([1, 2, 3, 4, 5, 6, 7, 8, 9, 10], [25, 50, 75])
|
||||
assert result[50] == 5.5
|
||||
assert result[25] > 0
|
||||
assert result[75] > result[50]
|
||||
|
||||
def test_empty(self):
|
||||
result = percentiles([], [50])
|
||||
assert result[50] == 0.0
|
||||
|
||||
|
||||
class TestMinMax:
|
||||
def test_min(self):
|
||||
assert min_val([3, 1, 2]) == 1
|
||||
|
||||
def test_max(self):
|
||||
assert max_val([3, 1, 2]) == 3
|
||||
|
||||
def test_min_empty(self):
|
||||
assert min_val([]) == 0.0
|
||||
|
||||
def test_max_empty(self):
|
||||
assert max_val([]) == 0.0
|
||||
Reference in New Issue
Block a user