diff --git a/.dockerignore b/.dockerignore index c9c34c2..392a2d8 100644 --- a/.dockerignore +++ b/.dockerignore @@ -4,3 +4,12 @@ .env *.log reports +test-results +threehost +threehost.tar.gz +Best_README_template-master +Best_README_template-master.zip +*.pyc +__pycache__ +frontend/node_modules +frontend/dist diff --git a/.env.example b/.env.example index 1e50f2c..16516db 100644 --- a/.env.example +++ b/.env.example @@ -15,6 +15,12 @@ MODEL_VELO_POSTGRES_CONNECT_TIMEOUT=5s MODEL_VELO_POSTGRES_MAX_CONN_LIFETIME=30m MODEL_VELO_POSTGRES_MAX_CONN_IDLE_TIME=5m MODEL_VELO_API_KEY_PEPPER=replace-with-at-least-32-random-bytes +MODEL_VELO_AUTH_CACHE_ENABLED=true +MODEL_VELO_AUTH_CACHE_L1_MAX_ENTRIES=10000 +MODEL_VELO_AUTH_CACHE_L1_TTL=15s +MODEL_VELO_AUTH_CACHE_L2_TTL=30s +MODEL_VELO_AUTH_CACHE_KEY_PREFIX=model-velo:development:auth:v1 +MODEL_VELO_AUTH_CACHE_INVALIDATION_CHANNEL=model-velo:development:auth:v1:invalidate MODEL_VELO_ADMIN_KEY_PEPPER=replace-with-a-different-32-byte-random-secret # Generate with: openssl rand -base64 32 MODEL_VELO_CONTROL_MASTER_KEY=replace-with-base64-encoded-32-random-bytes diff --git a/.gitignore b/.gitignore index 9353a82..e31fcd0 100644 --- a/.gitignore +++ b/.gitignore @@ -11,6 +11,8 @@ *.out coverage.* *.bench +__pycache__/ +*.pyc /test-results/ /test/threehost/client.env /test/threehost/benchmark.env diff --git a/1README.md b/1README.md new file mode 100644 index 0000000..210ba78 --- /dev/null +++ b/1README.md @@ -0,0 +1,179 @@ + + +# ProjectName + +ProjectName and Description + + + +[![Contributors][contributors-shield]][contributors-url] +[![Forks][forks-shield]][forks-url] +[![Stargazers][stars-shield]][stars-url] +[![Issues][issues-shield]][issues-url] +[![MIT License][license-shield]][license-url] +[![LinkedIn][linkedin-shield]][linkedin-url] + + +
+ +

+ + Logo + + +

"完美的"README模板

+

+ 一个"完美的"README模板去快速开始你的项目! +
+ 探索本项目的文档 » +
+
+ 查看Demo + · + 报告Bug + · + 提出新特性 +

+ +

+ + + 本篇README.md面向开发者 + +## 目录 + +- [上手指南](#上手指南) + - [开发前的配置要求](#开发前的配置要求) + - [安装步骤](#安装步骤) +- [文件目录说明](#文件目录说明) +- [开发的架构](#开发的架构) +- [部署](#部署) +- [使用到的框架](#使用到的框架) +- [贡献者](#贡献者) + - [如何参与开源项目](#如何参与开源项目) +- [版本控制](#版本控制) +- [作者](#作者) +- [鸣谢](#鸣谢) + +### 上手指南 + +请将所有链接中的“shaojintian/Best_README_template”改为“your_github_name/your_repository” + + + +###### 开发前的配置要求 + +1. xxxxx x.x.x +2. xxxxx x.x.x + +###### **安装步骤** + +1. Get a free API Key at [https://example.com](https://example.com) +2. Clone the repo + +```sh +git clone https://github.com/shaojintian/Best_README_template.git +``` + +### 文件目录说明 +eg: + +``` +filetree +├── ARCHITECTURE.md +├── LICENSE.txt +├── README.md +├── /account/ +├── /bbs/ +├── /docs/ +│ ├── /rules/ +│ │ ├── backend.txt +│ │ └── frontend.txt +├── manage.py +├── /oa/ +├── /static/ +├── /templates/ +├── useless.md +└── /util/ + +``` + + + + + +### 开发的架构 + +请阅读[ARCHITECTURE.md](https://github.com/shaojintian/Best_README_template/blob/master/ARCHITECTURE.md) 查阅为该项目的架构。 + +### 部署 + +暂无 + +### 使用到的框架 + +- [xxxxxxx](https://getbootstrap.com) +- [xxxxxxx](https://jquery.com) +- [xxxxxxx](https://laravel.com) + +### 贡献者 + +请阅读**CONTRIBUTING.md** 查阅为该项目做出贡献的开发者。 + +#### 如何参与开源项目 + +贡献使开源社区成为一个学习、激励和创造的绝佳场所。你所作的任何贡献都是**非常感谢**的。 + + +1. Fork the Project +2. Create your Feature Branch (`git checkout -b feature/AmazingFeature`) +3. Commit your Changes (`git commit -m 'Add some AmazingFeature'`) +4. Push to the Branch (`git push origin feature/AmazingFeature`) +5. Open a Pull Request + + + +### 版本控制 + +该项目使用Git进行版本管理。您可以在repository参看当前可用版本。 + +### 作者 + +xxx@xxxx + +知乎:xxxx   qq:xxxxxx + + *您也可以在贡献者名单中参看所有参与该项目的开发者。* + +### 版权说明 + +该项目签署了MIT 授权许可,详情请参阅 [LICENSE.txt](https://github.com/shaojintian/Best_README_template/blob/master/LICENSE.txt) + +### 鸣谢 + + +- [GitHub Emoji Cheat Sheet](https://www.webpagefx.com/tools/emoji-cheat-sheet) +- [Img Shields](https://shields.io) +- [Choose an Open Source License](https://choosealicense.com) +- [GitHub Pages](https://pages.github.com) +- [Animate.css](https://daneden.github.io/animate.css) +- [xxxxxxxxxxxxxx](https://connoratherton.com/loaders) + + +[your-project-path]:shaojintian/Best_README_template +[contributors-shield]: https://img.shields.io/github/contributors/shaojintian/Best_README_template.svg?style=flat-square +[contributors-url]: https://github.com/shaojintian/Best_README_template/graphs/contributors +[forks-shield]: https://img.shields.io/github/forks/shaojintian/Best_README_template.svg?style=flat-square +[forks-url]: https://github.com/shaojintian/Best_README_template/network/members +[stars-shield]: https://img.shields.io/github/stars/shaojintian/Best_README_template.svg?style=flat-square +[stars-url]: https://github.com/shaojintian/Best_README_template/stargazers +[issues-shield]: https://img.shields.io/github/issues/shaojintian/Best_README_template.svg?style=flat-square +[issues-url]: https://img.shields.io/github/issues/shaojintian/Best_README_template.svg +[license-shield]: https://img.shields.io/github/license/shaojintian/Best_README_template.svg?style=flat-square +[license-url]: https://github.com/shaojintian/Best_README_template/blob/master/LICENSE.txt +[linkedin-shield]: https://img.shields.io/badge/-LinkedIn-black.svg?style=flat-square&logo=linkedin&colorB=555 +[linkedin-url]: https://linkedin.com/in/shaojintian + + + + diff --git a/Best_README_template-master.zip b/Best_README_template-master.zip new file mode 100644 index 0000000..dea70fa Binary files /dev/null and b/Best_README_template-master.zip differ diff --git a/Best_README_template-master/Best_README_template-master/1README.md b/Best_README_template-master/Best_README_template-master/1README.md new file mode 100644 index 0000000..210ba78 --- /dev/null +++ b/Best_README_template-master/Best_README_template-master/1README.md @@ -0,0 +1,179 @@ + + +# ProjectName + +ProjectName and Description + + + +[![Contributors][contributors-shield]][contributors-url] +[![Forks][forks-shield]][forks-url] +[![Stargazers][stars-shield]][stars-url] +[![Issues][issues-shield]][issues-url] +[![MIT License][license-shield]][license-url] +[![LinkedIn][linkedin-shield]][linkedin-url] + + +
+ +

+ + Logo + + +

"完美的"README模板

+

+ 一个"完美的"README模板去快速开始你的项目! +
+ 探索本项目的文档 » +
+
+ 查看Demo + · + 报告Bug + · + 提出新特性 +

+ +

+ + + 本篇README.md面向开发者 + +## 目录 + +- [上手指南](#上手指南) + - [开发前的配置要求](#开发前的配置要求) + - [安装步骤](#安装步骤) +- [文件目录说明](#文件目录说明) +- [开发的架构](#开发的架构) +- [部署](#部署) +- [使用到的框架](#使用到的框架) +- [贡献者](#贡献者) + - [如何参与开源项目](#如何参与开源项目) +- [版本控制](#版本控制) +- [作者](#作者) +- [鸣谢](#鸣谢) + +### 上手指南 + +请将所有链接中的“shaojintian/Best_README_template”改为“your_github_name/your_repository” + + + +###### 开发前的配置要求 + +1. xxxxx x.x.x +2. xxxxx x.x.x + +###### **安装步骤** + +1. Get a free API Key at [https://example.com](https://example.com) +2. Clone the repo + +```sh +git clone https://github.com/shaojintian/Best_README_template.git +``` + +### 文件目录说明 +eg: + +``` +filetree +├── ARCHITECTURE.md +├── LICENSE.txt +├── README.md +├── /account/ +├── /bbs/ +├── /docs/ +│ ├── /rules/ +│ │ ├── backend.txt +│ │ └── frontend.txt +├── manage.py +├── /oa/ +├── /static/ +├── /templates/ +├── useless.md +└── /util/ + +``` + + + + + +### 开发的架构 + +请阅读[ARCHITECTURE.md](https://github.com/shaojintian/Best_README_template/blob/master/ARCHITECTURE.md) 查阅为该项目的架构。 + +### 部署 + +暂无 + +### 使用到的框架 + +- [xxxxxxx](https://getbootstrap.com) +- [xxxxxxx](https://jquery.com) +- [xxxxxxx](https://laravel.com) + +### 贡献者 + +请阅读**CONTRIBUTING.md** 查阅为该项目做出贡献的开发者。 + +#### 如何参与开源项目 + +贡献使开源社区成为一个学习、激励和创造的绝佳场所。你所作的任何贡献都是**非常感谢**的。 + + +1. Fork the Project +2. Create your Feature Branch (`git checkout -b feature/AmazingFeature`) +3. Commit your Changes (`git commit -m 'Add some AmazingFeature'`) +4. Push to the Branch (`git push origin feature/AmazingFeature`) +5. Open a Pull Request + + + +### 版本控制 + +该项目使用Git进行版本管理。您可以在repository参看当前可用版本。 + +### 作者 + +xxx@xxxx + +知乎:xxxx   qq:xxxxxx + + *您也可以在贡献者名单中参看所有参与该项目的开发者。* + +### 版权说明 + +该项目签署了MIT 授权许可,详情请参阅 [LICENSE.txt](https://github.com/shaojintian/Best_README_template/blob/master/LICENSE.txt) + +### 鸣谢 + + +- [GitHub Emoji Cheat Sheet](https://www.webpagefx.com/tools/emoji-cheat-sheet) +- [Img Shields](https://shields.io) +- [Choose an Open Source License](https://choosealicense.com) +- [GitHub Pages](https://pages.github.com) +- [Animate.css](https://daneden.github.io/animate.css) +- [xxxxxxxxxxxxxx](https://connoratherton.com/loaders) + + +[your-project-path]:shaojintian/Best_README_template +[contributors-shield]: https://img.shields.io/github/contributors/shaojintian/Best_README_template.svg?style=flat-square +[contributors-url]: https://github.com/shaojintian/Best_README_template/graphs/contributors +[forks-shield]: https://img.shields.io/github/forks/shaojintian/Best_README_template.svg?style=flat-square +[forks-url]: https://github.com/shaojintian/Best_README_template/network/members +[stars-shield]: https://img.shields.io/github/stars/shaojintian/Best_README_template.svg?style=flat-square +[stars-url]: https://github.com/shaojintian/Best_README_template/stargazers +[issues-shield]: https://img.shields.io/github/issues/shaojintian/Best_README_template.svg?style=flat-square +[issues-url]: https://img.shields.io/github/issues/shaojintian/Best_README_template.svg +[license-shield]: https://img.shields.io/github/license/shaojintian/Best_README_template.svg?style=flat-square +[license-url]: https://github.com/shaojintian/Best_README_template/blob/master/LICENSE.txt +[linkedin-shield]: https://img.shields.io/badge/-LinkedIn-black.svg?style=flat-square&logo=linkedin&colorB=555 +[linkedin-url]: https://linkedin.com/in/shaojintian + + + + diff --git a/Best_README_template-master/Best_README_template-master/LICENSE.txt b/Best_README_template-master/Best_README_template-master/LICENSE.txt new file mode 100644 index 0000000..f4ea8d9 --- /dev/null +++ b/Best_README_template-master/Best_README_template-master/LICENSE.txt @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2018 Othneil Drew + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/Best_README_template-master/Best_README_template-master/images/logo.png b/Best_README_template-master/Best_README_template-master/images/logo.png new file mode 100644 index 0000000..0f38ba9 Binary files /dev/null and b/Best_README_template-master/Best_README_template-master/images/logo.png differ diff --git a/README.md b/README.md index 4f8655c..84c90cd 100644 --- a/README.md +++ b/README.md @@ -1,604 +1,115 @@ # Model-Velo -Model-Velo 是一个用 Go 和 Gin 编写的多协议 LLM 网关。当前已实现 Chat Completions、Responses、Embeddings、Anthropic Messages 入站协议,多 Provider 原生转换与可靠性运行时,持久化 Usage/计费、在线控制平面、额度预算,以及阶段 6 的可观测性和工程门禁。 +Model-Velo 是一个使用 Go 构建的 LLM Gateway,为应用提供统一的模型访问入口、租户鉴权、可靠性治理和用量记录。 -## 当前已经实现 +## 核心能力 -- `GET /healthz` 健康检查; -- `GET /readyz` PostgreSQL/Redis 就绪检查,以及可选 Bearer 保护的 `GET /metrics`; -- `POST /v1/chat/completions` 非流式请求校验和转发; -- `POST /v1/responses`、`POST /v1/embeddings`、`POST /v1/messages` 和 `GET /v1/models`; -- 16 个内置厂商 Adapter:厂商身份与构造入口彼此独立;公开采用 OpenAI Chat 报文的厂商只复用协议编解码和 HTTP 边界; -- 一个厂商可配置多个 Provider 实例,每个实例可声明多个文本或视觉模型; -- request ID 生成、校验、响应回传和上游传播; -- 请求体与上游响应体大小限制; -- 上游超时、网络失败、HTTP 错误和非法响应的结构化错误; -- 操作系统退出信号和有界优雅关闭; -- 使用 `httptest.Server` 的本地测试,不调用真实付费 API; -- 固定版本的 PostgreSQL、Redis Compose 配置; -- 基于 GORM 的 PostgreSQL 连接、启动 Ping、连接池配置和退出关闭; -- 启动时通过 GORM `AutoMigrate` 同步租户/API Key、Usage/outbox、控制面/审计和额度账本; -- Model-Velo API Key 随机生成、摘要查找、HMAC 校验、过期判断、禁用和吊销; -- `model-velo-admin` 本地管理命令,可初始化租户、模型授权和首个 API Key; -- Gin Bearer 认证中间件、请求身份 Context 和租户模型授权检查; -- 官方 `go-redis/v9` Client、显式连接池、启动 Ping、可选启动降级和退出关闭; -- 基于 Redis Lua 的租户+模型固定窗口限流,以及可配置的 fail-open/fail-closed 运行时故障策略; -- 租户隔离的 Redis Exact Response Cache,支持规范化请求哈希、TTL、显式绕过和缓存故障降级; -- 精确模型/默认模型路由、有序候选去重、primary 选择和上游模型映射; -- 按模型声明 `text`、`image`、`audio`、`file`、`tools`、`structured` 能力,规划时先过滤协议或模型无法承载的候选; -- Provider Circuit Breaker 三态、指定故障计数、Open 快速拒绝和 HalfOpen 有界探测; -- 按 Provider 隔离的进程内有界 Queue,限制运行数和等待数,并传播请求取消; -- Provider 多 Key 安全身份与并发轮换:401 永久禁用错误 Key,403 只在当前请求内换 Key,429 按 `Retry-After` 临时冷却; -- 按 Provider ID 隔离的 Adapter、Circuit Breaker、Queue 和 Key Registry; -- 单候选 Attempt Executor:每个候选独立执行 Breaker、Queue、Key、有限 Retry 和上游调用; -- API Key Adapter 在装配时必须具备 Key Registry,错误装配会在启动边界失败而不是在请求期空指针崩溃; -- 有序 Fallback Orchestrator:成功立即停止,普通 400/取消停止,模型不可用等策略允许的失败进入下一候选;Fallback 成功响应不写 Exact Cache; -- Provider 执行总预算、单次调用超时和 Context-aware 退避取消; -- 每个 Provider 可独立覆盖 Breaker、Queue、Retry、Attempt Timeout 与 HTTP 连接池; -- Chat 文本、图片、音频、文件、Function Tool 与结构化输出:兼容协议保留原报文,原生协议只转换能够明确表达的字段并归一化响应; -- 流式预提交可靠性链:每次按 Breaker→Queue→Key→Adapter 建流并验证首事件,候选内支持有限 Retry,耗尽后按 Route Plan 有序 Fallback;最终 PreparedStream 持有成功流资源和完整安全 Trail,直到显式结束。 -- OpenAI-compatible 客户端 SSE:有效首事件后才提交 Header,逐事件同步 Write/Flush,正常转发 `[DONE]`,客户端断开会取消上游并释放 Queue。 -- Usage Event schema v2:记录 request、tenant、API Key ID、请求与实际模型、缓存、可靠性计数、详细 token、usage 来源、finish reason、TTFT、稳定终态和 UTC 延迟;原始 usage 子对象最多保留 64 KiB,不记录 Key Secret、提示词或完整上游响应; -- OpenAI-compatible 流请求默认合并 `stream_options.include_usage=true`;Provider 返回的缓存读写、音频、图像、推理和预测 token 会进入统一明细; -- 版本化价目表按 Provider、模型和事件时间生成不可变成本快照;Provider 明确上报的 USD 成本优先,缓存命中成本为已知零,无法定价时成本保持 `NULL` 而不是伪造零; -- API 在 Provider 前同步写 PostgreSQL pending 生命周期,终态先固化到 outbox,再以有界超时执行 Redis `XADD`;即时投递失败不会改写已经生成的模型响应; -- 独立 `model-velo-usage-worker` 使用 consumer group、`XREADGROUP`、`XAUTOCLAIM`、dead-letter 和 Context-aware 退避; -- PostgreSQL `usage_events.event_id` 主键与 `ON CONFLICT DO NOTHING` 提供幂等最终防线,数据库成功后才在 Redis 事务中执行 `XACK + XDEL`; -- 认证后的租户可查询 Usage 明细、汇总和时间序列;Worker 自动执行分批保留期清理,管理命令支持历史成本重算。 -- 独立管理员身份、owner/operator/billing/auditor RBAC、脱敏审计日志,以及 Provider/路由/Provider Key/价格、租户、业务 API Key 和额度策略的在线管理; -- PostgreSQL 强一致额度账本:分钟/小时/日/月请求、Token、USD 预算,支持 deny/allow/alert 超额策略、请求前预留、真实 Usage 结算和中断恢复; -- JSON 结构化日志、Prometheus、OpenTelemetry、非 root 容器、GitHub Actions race/集成门禁与可复现 benchmark。 +- OpenAI Chat Completions、Responses、Embeddings 与 Anthropic Messages 兼容接口 +- 非流式响应与 SSE 流式传输 +- 多 Provider 路由、Key 选择、重试、Fallback、熔断和有界队列 +- API Key 鉴权、模型授权、租户限流、配额与响应缓存 +- PostgreSQL 持久化与 Redis Stream 异步 Usage 链路 +- Prometheus 指标、结构化日志、健康检查与 OpenTelemetry Tracing -Usage 链路采用 **PostgreSQL outbox + Redis Stream at-least-once + 数据库幂等**,不宣称 exactly-once。API 在调用 Provider 前先写 pending 生命周期,结束时把完整 Event 固化为 ready;Redis 不可用不会丢失已固化事件,Worker 会从 outbox 重投。已标记 published 但尚未被 Usage 入库事务删除的记录也会周期性重发,覆盖 Redis 消息在消费前丢失或清理的窗口。数据库成功但 Redis ACK 响应丢失时,重投会命中 `usage_events.event_id` 唯一键。Worker 消失前未 ACK 的事件由 `XAUTOCLAIM` 恢复,坏版本/坏载荷达到阈值后进入有长度上限的 dead-letter Stream。进程在最终 Usage 形成前退出时,超时 pending 会转成明确的中断事件并保留“Usage 未知” caveat,而不是伪造 Token。 - -客户端始终收到 OpenAI SSE。Adapter 会按上游协议校验并转换 OpenAI-compatible SSE、Anthropic/Gemini/DashScope/Cohere SSE、Ollama NDJSON 或 Bedrock AWS EventStream;单行最大 1 MiB、单事件最大 2 MiB,Bedrock 二进制帧也在解码前限长。首事件前的 5xx、错误媒体类型、超时、EOF、坏 Chunk 和取消会沿用可靠性分类,可 Retry/Fallback,并在下一次尝试前释放资源;预提交总预算不会成为成功长流的上游 deadline。首事件提交后禁止切换 Provider,后续失败只安全记录并结束当前流。上游 heartbeat 当前不向客户端透传,但会重置事件空闲计时器;转换链使用同步背压,不创建无界 Chunk 队列。首事件验证成功后会清除当前 SSE 响应继承的 Server 总写截止时间,每个客户端帧仍有独立 15 秒 Write/Flush 截止时间,后续事件静默上限复用该 Provider 的 `attempt_timeout`。 - -阶段 3 的生产功能、合并故障矩阵、全量测试、vet 和独立性复查已经完成;Breaker、Queue、Key 并发用例也已通过普通执行,但 race detector 仍被本机 Go/race 工具链阻止,因此不能宣称 race 已通过。详细证据见 `STAGE3_GATE.md`。 - -阶段 5 的 Usage v2 生产链和真实 Redis/PostgreSQL 集成门禁已经通过;当前 PATH 没有 GCC,`go test -race` 在 `runtime/cgo` 编译前失败,因此阶段 5 race 与最终门禁仍保留,不能宣称 race 已通过。详细证据见 `STAGE5_GATE.md`。 - -当前请求会按 Route Plan 顺序执行候选;每个候选内部在策略允许时进行有限 Retry,耗尽后只有具备 Fallback 信号的失败才进入下一候选。合法请求所需的能力不被当前 Provider 支持时会直接尝试下一候选,不消耗 Retry,也不计入 Breaker;所有候选都不支持时返回 `400 unsupported_provider_capability`。`POST /v1/chat/completions` 必须携带有效的 Model-Velo API Key,请求模型必须存在于该租户的模型授权表,并依次通过租户限流和 Route Plan;缓存未命中后,每次 Provider 调用都会重新取得目标 Provider 的 Breaker Permit、Queue 槽位和可用 Provider Key。`GET /healthz` 保持公开。 - -## 当前 Chat 契约 - -请求必须使用 `Content-Type: application/json`,当前对以下字段做出保证: - -| 字段 | 当前行为 | -|---|---| -| `model` | 必填且不能是空白字符串;用于授权和 Route Plan,可通过 `upstream_model` 映射成厂商模型名。 | -| `messages` | 必填且至少包含一条消息。 | -| `messages[].role` | 接受 `system`、`developer`、`user`、`assistant`、`tool`;`tool`、assistant `tool_calls` 和 Tool 定义会要求目标模型声明 `tools`。 | -| `messages[].content` | 接受非空字符串或非空内容块数组;已建模 `text`/`input_text`、`image_url`、`input_audio` 和 `file`。未知块不会穿过原生转换器。 | -| `messages[].tool_calls` / `tool_call_id` | 校验 Function 名称、唯一调用 ID、JSON object 参数和历史引用;原生 Adapter 映射 Tool Use/Result,响应统一返回 OpenAI `tool_calls`。 | -| `tools` / `tool_choice` / `parallel_tool_calls` | 支持 Function Tool;厂商没有等价控制项时明确返回能力错误,不会删除字段继续请求。 | -| `response_format` | 支持 `text`、`json_object`、`json_schema`;需要模型声明 `structured`。DashScope 原生接口只接受 `json_object`,Cohere 不允许与 `tools` 组合。 | -| 生成参数 | 建模并校验 token 上限、`temperature`、`top_p`、`stop`、`seed`、penalty、`n`、logprobs 与 `reasoning_effort`;原生协议仅映射等价字段,其余明确拒绝。 | -| `stream` | 省略或为 `false` 时返回完整 JSON;`true` 时绕过 Exact Cache,并以 `text/event-stream` 逐事件返回兼容 Chunk 与 `[DONE]`。 | - -OpenAI、Mistral、DeepSeek、xAI、Zhipu、Groq、NVIDIA、Together 和 Cloudflare 均由各自的厂商装配入口设置端点与能力边界;它们公开采用 OpenAI Chat 报文,因此复用同一套 wire codec 和 HTTP 安全边界,只在路由配置要求时改写 `model`。Anthropic、Gemini、DashScope、Cohere、Ollama 和 Bedrock 使用各自原生消息、Tool、结构化输出、Usage 和流式事件格式。 - -请求 JSON 只解析一次。兼容协议会保留未知顶层字段和消息字段;原生协议在转换前检查这些字段,无法无损表达时返回能力不匹配并尝试下一候选,不会静默丢字段。 - -各协议在当前 Adapter 中可表达的上限如下;模型还必须在 `model_capabilities` 中显式声明对应能力: - -| 协议 | image | audio | file | tools | structured | stream | -|---|---:|---:|---:|---:|---:|---:| -| OpenAI / Azure / custom OpenAI-compatible | 是 | 是 | 是 | 是 | 是 | 是 | -| Anthropic Messages | 是 | 否 | 是 | 是 | 是 | 是 | -| Gemini generateContent | 是 | 是 | 是(内嵌数据) | 是 | 是 | 是 | -| DashScope Generation | 否 | 否 | 否 | 是 | 仅 `json_object` | 是 | -| Cohere v2 Chat | 是 | 否 | 否 | 是 | 是 | 是 | -| Ollama Chat | 是 | 否 | 否 | 是 | 是 | 是 | -| Bedrock Converse | 是 | 否 | 是(内嵌数据) | 是 | 是 | 是 | -| Cloudflare Workers AI Chat | 是 | 是 | 是 | 是 | 是 | 是 | - -这张表描述线协议转换能力,不承诺每个厂商模型都具备该能力。Anthropic 与 Cohere 可把远程图片 URL 交给上游;Gemini、Ollama 和 Bedrock 的当前转换器要求图片为 Base64 data URL。显式 `detail=low/high` 只在能保留该语义的 Cohere/OpenAI wire 上发送,其他原生协议会拒绝而不是忽略。Gemini、Bedrock 的 OpenAI `file_id` 没有安全等价物,当前只接受内嵌 `file_data`;Anthropic 同时接受 Files API `file_id` 和内嵌文件,使用 `file_id` 时 Adapter 会自动发送 Files API Beta Header。视频和厂商私有内容块不在 Chat 契约内。 - -成功时,兼容协议响应在通过 2xx、Content-Type、大小、错误信封和非空 Chat `choices[].message` 检查后原样返回;原生协议响应会转换为非流式 OpenAI Chat Completion。原生响应缺少 Usage 时省略 `usage`,不会伪造全零计费数据;出现当前无法表示的非文本输出时明确失败并按策略 Fallback。 - -## Usage 查询与成本 - -以下接口位于认证后的 `/v1` 路由组,并同时强制 tenant ID 与当前 API Key ID,普通模型 Key 不能读取同租户其他 Key 的账单或审计数据: - -- `GET /v1/usage/events`:游标分页明细,支持 `start`、`end`、`model`、`provider`、`api_key_id`、`request_id`、`status`、`cache_status`、`stream`、`limit` 和 `include_raw`; -- `GET /v1/usage/summary`:请求、成功/失败、缓存、token、已知/未知成本、延迟、TTFT 和重试/Fallback 汇总,`group_by` 支持 `model`、`provider`、`status`、`cache`、`api_key`; -- `GET /v1/usage/series`:按 `hour`、`day`、`week`、`month` 或 `year` 返回时间序列,并接受 IANA timezone。 - -`api_key_id` 省略时自动使用当前 Key,显式值也只能等于当前 Key。默认查询最近 30 天,单次范围最多 366 天,明细每页最多 200 条,分组最多返回 1000 组并显式标记截断。所有响应带 `Cache-Control: no-store`。成本以整数 nanoUSD 存储与聚合,接口同时返回精确十进制 USD 字符串;未知价格、缺失 usage 或无法覆盖早期失败 attempt 时会保留 caveat。 - -历史 schema v1 没有 API Key ID,Worker 仍能可靠消费和存储,但它不能被安全归属到某一把 Key,因此不会出现在普通 Key 的 HTTP 查询中;可通过 PostgreSQL 管理通道或带 tenant 条件的 `reprice-usage` 处理。系统不会为了补齐归属而猜测或把 v1 数据暴露给整个租户。 - -## 管理 API - -`/admin/v1` 只接受独立的 `mv_admin_...` Bearer Key,所有响应都带 `Cache-Control: no-store`。运行时、价格、管理员、租户/API Key 与额度变更和审计写入处于同一数据库事务;创建返回的管理员 Key 或业务 API Key 明文只出现一次。 - -- `GET/PUT /admin/v1/runtime`:版本化 Provider、路由、上游 Key 与可靠性参数;写入必须携带 `If-Match`; -- `GET/PUT /admin/v1/pricing`:版本化价目; -- `GET /admin/v1/principals`、`POST /admin/v1/principals`、`PATCH /admin/v1/principals/:id`:管理员与角色; -- `GET/POST/PUT /admin/v1/quotas`、`GET /admin/v1/quota-windows`:额度策略和当前已结算/预留窗口; -- `GET/POST /admin/v1/tenants`、`PUT /admin/v1/tenants/:id`、`GET/POST /admin/v1/tenants/:id/keys`、`PATCH /admin/v1/api-keys/:id`:租户、模型授权及业务 Key 生命周期; -- `GET /admin/v1/usage/events`、`GET /admin/v1/usage/summary`、`GET /admin/v1/usage/series`:需要 `usage:read` 的跨租户只读 Usage 查询,可按 `tenant_id` 与 `api_key_id` 下钻,汇总额外支持 `group_by=tenant`;明细不开放原始 Usage JSON; -- `GET /admin/v1/audit`:只读游标分页审计。 - -## 配置 - -| 环境变量 | 必填 | 默认值 | 用途 | -|---|---:|---|---| -| `MODEL_VELO_HTTP_ADDR` | 否 | `:8080` | Model-Velo HTTP 监听地址。 | -| `MODEL_VELO_ENVIRONMENT` | API 必填 | 无 | 1–32 位小写环境标识,用于隔离 Redis Key,例如 `development`、`staging`。 | -| `MODEL_VELO_PROVIDER_KEYS_JSON` | 条件必填 | 无 | 按需要鉴权的 Provider ID 配置一个或多个 Key;无鉴权的 Ollama 不配置 Key 集合。Secret 只用于上游鉴权,不进入快照、错误或日志。 | -| `MODEL_VELO_SHUTDOWN_TIMEOUT` | 否 | `10s` | 收到退出信号后等待活动请求结束的期限。 | -| `MODEL_VELO_POSTGRES_DB` | 否 | `model_velo` | Compose 创建的本地数据库名。 | -| `MODEL_VELO_POSTGRES_USER` | 否 | `model_velo` | Compose 创建的本地数据库用户。 | -| `MODEL_VELO_POSTGRES_PASSWORD` | Compose 必填 | 无 | 本地 PostgreSQL 密码,必须在 `.env` 中替换示例值。 | -| `MODEL_VELO_POSTGRES_PORT` | 否 | `5432` | PostgreSQL 映射到本机回环地址的端口。 | -| `MODEL_VELO_POSTGRES_DSN` | API 必填 | 无 | PostgreSQL URL;必须包含 `postgres/postgresql` scheme、用户、Host 和数据库名。 | -| `MODEL_VELO_POSTGRES_MAX_OPEN_CONNS` | 否 | `10` | `database/sql` 连接池最大打开连接数。 | -| `MODEL_VELO_POSTGRES_MAX_IDLE_CONNS` | 否 | `2` | `database/sql` 连接池最大空闲连接数,不能超过最大打开连接数。 | -| `MODEL_VELO_POSTGRES_CONNECT_TIMEOUT` | 否 | `5s` | PostgreSQL 启动连接检查期限。 | -| `MODEL_VELO_POSTGRES_MAX_CONN_LIFETIME` | 否 | `30m` | 单个 PostgreSQL 连接的最长寿命。 | -| `MODEL_VELO_POSTGRES_MAX_CONN_IDLE_TIME` | 否 | `5m` | PostgreSQL 连接的最大空闲时间。 | -| `MODEL_VELO_API_KEY_PEPPER` | API 与管理命令必填 | 无 | 至少 32 字节的服务端秘密,用于 HMAC 校验 Model-Velo API Key;更换后已有 Key 将全部失效。 | -| `MODEL_VELO_ADMIN_KEY_PEPPER` | API 与管理员初始化必填 | 无 | 与业务 Key 分离的至少 32 字节秘密。 | -| `MODEL_VELO_CONTROL_MASTER_KEY` | API 必填 | 无 | Base64 编码的 32 字节 AES-256-GCM 密钥,用于加密托管 Provider Key。 | -| `MODEL_VELO_CONTROL_REFRESH_INTERVAL` | 否 | `5s` | 多 API 实例同步活动运行时与价格版本的间隔。 | -| `MODEL_VELO_REDIS_ADDR` | API 必填 | 无 | Redis `host:port` 地址。 | -| `MODEL_VELO_REDIS_PASSWORD` | API 与 Compose 必填 | 无 | Redis 密码,错误信息和日志不得包含该值。 | -| `MODEL_VELO_REDIS_DB` | 否 | `0` | Go 应用使用的非负 Redis logical DB。 | -| `MODEL_VELO_REDIS_PORT` | 否 | `6379` | Redis 映射到本机回环地址的端口。 | -| `MODEL_VELO_REDIS_DIAL_TIMEOUT` | 否 | `5s` | Redis 建连期限。 | -| `MODEL_VELO_REDIS_READ_TIMEOUT` | 否 | `2s` | Redis 读取期限。 | -| `MODEL_VELO_REDIS_WRITE_TIMEOUT` | 否 | `2s` | Redis 写入期限。 | -| `MODEL_VELO_REDIS_POOL_SIZE` | 否 | `20` | Redis 连接池最大连接数。 | -| `MODEL_VELO_REDIS_MIN_IDLE_CONNS` | 否 | `2` | Redis 连接池预留的最小空闲连接数,不能超过池容量。 | -| `MODEL_VELO_REDIS_POOL_TIMEOUT` | 否 | `2s` | 连接池耗尽时等待可用连接的期限。 | -| `MODEL_VELO_REDIS_STARTUP_POLICY` | 否 | `required` | `required` 表示启动 Ping 失败终止;`optional` 表示记录警告后继续。它与限流运行时故障策略相互独立。 | -| `MODEL_VELO_RATE_LIMIT_REQUESTS` | 否 | `60` | 每个租户+模型在一个窗口内可接受的请求数,范围 1–1,000,000。 | -| `MODEL_VELO_RATE_LIMIT_WINDOW` | 否 | `1m` | 固定窗口时长,范围 `1s`–`24h`;窗口从该 Key 的首个请求开始。 | -| `MODEL_VELO_RATE_LIMIT_FAILURE_POLICY` | 否 | `fail-closed` | Redis 运行时失败时,`fail-closed` 返回 503;`fail-open` 标记绕过并继续 Provider。 | -| `MODEL_VELO_CACHE_TTL` | 否 | `5m` | Exact Cache 保存时间,范围 `1s`–`24h`;`0` 或 `off` 禁用缓存。 | -| `MODEL_VELO_CACHE_ROUTE_VERSION` | 否 | `routes-v1` | 环境变量启动路由的缓存命名空间;托管运行时切换会自动使用版本化命名空间。 | -| `MODEL_VELO_USAGE_EMIT_TIMEOUT` | 否 | `200ms` | API 在请求结束后投递 Usage Event 的独立短超时。 | -| `MODEL_VELO_USAGE_GROUP` | 否 | `model-velo-usage-workers` | Usage Worker consumer group。 | -| `MODEL_VELO_USAGE_CONSUMER` | 否 | `-` | 当前 Worker consumer 名;同组并发进程应不同。 | -| `MODEL_VELO_USAGE_BATCH_SIZE` | 否 | `50` | 每次读取或认领的最大消息数。 | -| `MODEL_VELO_USAGE_READ_BLOCK` | 否 | `2s` | 空 Stream 上 `XREADGROUP` 的阻塞时间。 | -| `MODEL_VELO_USAGE_CLAIM_IDLE` | 否 | `30s` | pending 消息允许被其他 Worker 认领前的空闲时间。 | -| `MODEL_VELO_USAGE_MAX_DELIVERIES` | 否 | `5` | 坏版本/坏载荷进入 dead-letter 前的最大投递次数。 | -| `MODEL_VELO_USAGE_RETRY_BACKOFF` | 否 | `500ms` | Redis 读取/认领失败后的 Context-aware 退避。 | -| `MODEL_VELO_USAGE_WORKER_TIMEOUT` | 否 | `10s` | 单批写库与关闭收尾共享的最大处理时间。 | -| `MODEL_VELO_USAGE_DEAD_LETTER_MAX_LEN` | 否 | `100000` | dead-letter Stream 的近似长度上限;旧 `MODEL_VELO_USAGE_STREAM_MAX_LEN` 仍兼容,但不能与新变量同时设置。 | -| `MODEL_VELO_USAGE_ENFORCE_STREAM` | 否 | `true` | 对 OpenAI-compatible 流请求强制合并 `stream_options.include_usage=true`;仅在确认自定义上游不兼容时关闭。 | -| `MODEL_VELO_USAGE_RETENTION_DAYS` | 否 | `90` | PostgreSQL Usage 保留天数,范围 0–3650;`0` 禁用自动清理。 | -| `MODEL_VELO_USAGE_MAINTENANCE_INTERVAL` | 否 | `1h` | Worker 执行保留期清理的间隔。 | -| `MODEL_VELO_USAGE_MAINTENANCE_BATCH_SIZE` | 否 | `1000` | 单次删除批量,Worker 会在维护超时内持续分批清理。 | -| `MODEL_VELO_USAGE_PRICING_JSON` | 否 | `[]` | 版本化 USD/百万 token 价目表;支持生效时间、缓存、音频、图像和推理 token 专属费率,最大 256 KiB。 | -| `MODEL_VELO_USAGE_PRICING_REFRESH_INTERVAL` | 否 | `30s` | 独立 Worker 刷新托管价格目录的间隔。 | -| `MODEL_VELO_USAGE_PENDING_TIMEOUT` | 否 | `15m` | 把未完成 outbox 生命周期恢复成中断事件前的保守等待时间,范围 5m–24h。 | -| `MODEL_VELO_QUOTA_RESERVATION_TTL` | 否 | `15m` | 崩溃后活动额度预留转为保守估算结算的时间。 | -| `MODEL_VELO_QUOTA_REAP_INTERVAL` | 否 | `1m` | 扫描过期额度预留的间隔。 | -| `MODEL_VELO_QUOTA_DEFAULT_MAX_OUTPUT_TOKENS` | 否 | `4096` | 请求未给输出上限时用于 Token/成本预留的默认值。 | -| `MODEL_VELO_LOG_FORMAT` | 否 | `json` | `json` 或本地开发用 `text`。 | -| `MODEL_VELO_LOG_LEVEL` | 否 | `info` | `debug`、`info`、`warn` 或 `error`。 | -| `MODEL_VELO_SERVICE_NAME` | 否 | `model-velo` | 日志与 OpenTelemetry service name。 | -| `MODEL_VELO_METRICS_TOKEN` | 否 | 无 | 设置后 API 与 Worker `/metrics` 要求至少 32 字节的 Bearer Token。 | -| `MODEL_VELO_WORKER_METRICS_ADDR` | 否 | `:9091` | Usage Worker 的 `/healthz`、`/readyz`、`/metrics` 监听地址。 | -| `MODEL_VELO_OTEL_EXPORTER_OTLP_ENDPOINT` | 否 | 无 | OTLP/HTTP Trace Collector 绝对 URL;空值禁用导出。 | -| `MODEL_VELO_OTEL_EXPORTER_OTLP_INSECURE` | 否 | `false` | 是否允许明文 OTLP/HTTP;`http` URL 也会启用。 | -| `MODEL_VELO_OTEL_SAMPLE_RATIO` | 否 | `0.1` | Parent-based Trace 采样比例,范围 0–1。 | -| `MODEL_VELO_READINESS_TIMEOUT` | 否 | `1s` | 单次依赖就绪检查的总期限。 | -| `MODEL_VELO_ROUTING_JSON` | API 必填 | 无 | 显式定义任意数量的 Provider、厂商预设、模型能力、Provider 级运行参数、精确/默认路由和有序候选;只配置一个 Provider 时也不能省略。 | -| `MODEL_VELO_BREAKER_FAILURE_THRESHOLD` | 否 | `5` | 连续可计数失败达到此值后 Open,范围 1–1000。 | -| `MODEL_VELO_BREAKER_OPEN_DURATION` | 否 | `30s` | Open 冷却时间,范围 `1s`–`10m`。 | -| `MODEL_VELO_BREAKER_HALF_OPEN_PROBES` | 否 | `1` | HalfOpen 同时允许的探测数,以及重新关闭所需成功数,范围 1–100。 | -| `MODEL_VELO_QUEUE_MAX_IN_FLIGHT` | 否 | `20` | 每个 Provider、每个网关进程允许的同时执行数,范围 1–10,000。 | -| `MODEL_VELO_QUEUE_MAX_WAITING` | 否 | `100` | 每个 Provider、每个网关进程允许的等待数;`0` 表示满载时立即拒绝,范围 0–100,000。 | -| `MODEL_VELO_QUEUE_WAIT_TIMEOUT` | 否 | `2s` | 等待 Provider 槽位的最长期限,范围 `10ms`–`1m`,同时受请求 Context 更早截止时间约束。 | -| `MODEL_VELO_RETRY_MAX_ATTEMPTS` | 否 | `3` | 每个候选最多调用次数,范围 1–10;包含第一次调用。 | -| `MODEL_VELO_RETRY_INITIAL_BACKOFF` | 否 | `100ms` | 首次可退避重试的基础等待,范围 `10ms`–`30s`。 | -| `MODEL_VELO_RETRY_MAX_BACKOFF` | 否 | `2s` | 指数退避上限,不得小于初始等待且不超过 `30s`。 | -| `MODEL_VELO_RETRY_BACKOFF_MULTIPLIER` | 否 | `2` | 退避倍数,范围 1–10。 | -| `MODEL_VELO_RETRY_JITTER_RATIO` | 否 | `0.2` | 退避随机抖动比例,范围 0–1。 | -| `MODEL_VELO_REQUEST_TIMEOUT` | 否 | `45s` | Cache miss 后 Retry 与全部 Fallback 候选共享的执行预算,范围 `1s`–`5m`。 | -| `MODEL_VELO_ATTEMPT_TIMEOUT` | 否 | `20s` | 非流式单次调用或流式建连+首事件等待期限,至少 `100ms` 且不得超过 Provider 执行总预算。 | - -价目使用十进制字符串,避免浮点配置误差;同一个 Provider/模型的生效时间窗口不能重叠。`provider` 或 `model` 可用 `*` 作为启动时已知的回退价格: - -```powershell -$env:MODEL_VELO_USAGE_PRICING_JSON = '[{"provider":"openai-main","model":"gpt-4o-mini","version":"openai-2026-07","effective_from":"2026-07-01T00:00:00Z","input_usd_per_million":"0.15","output_usd_per_million":"0.60","cached_read_usd_per_million":"0.075"}]' -``` - -上游 `http.Client` 不再维护第三套总超时。非流式调用服从 Attempt Context 和父级请求预算;流式调用在首事件前同时受 Attempt Timeout 与父 Context 约束,首事件验证通过后不再沿用短 Attempt deadline,但后续两个有效事件之间仍复用当前 Provider 的 Attempt Timeout 作为静默上限,上游 heartbeat 会重置计时器。HTTP Server 写超时固定覆盖托管运行时允许的最长 5 分钟请求预算并额外保留 15 秒收尾时间;每个非流式请求仍由自身快照中的较短总预算取消。流式 Handler 在首事件验证后清除总写截止时间,并为每个客户端帧设置独立 15 秒写截止时间。 - -### Provider 厂商预设 - -`vendor` 负责选择厂商目录和默认 API Base,`type` 必须显式声明协议;二者不匹配时启动失败。一个厂商可以配置多个 Provider ID,一个 Provider 的 `models` 可以列出多个文本、推理或 VLM 模型;`base_url` 可覆盖区域、私有部署或账号级端点。模型清单由配置显式声明,代码不硬编码容易过期的型号。 - -| `vendor` | 必须声明的 `type` | 默认 API Base | -|---|---|---| -| `openai` | `openai` | `https://api.openai.com/v1` | -| `anthropic` | `anthropic` | `https://api.anthropic.com` | -| `google` | `gemini` | `https://generativelanguage.googleapis.com/v1beta` | -| `azure` | `azure-openai` | 必须配置资源 Endpoint | -| `alibaba` | `dashscope` | `https://dashscope.aliyuncs.com/api/v1` | -| `cohere` | `cohere` | `https://api.cohere.com/v2` | -| `ollama` | `ollama` | `http://localhost:11434` | -| `bedrock` | `bedrock` | 必须配置区域 Bedrock Runtime Endpoint | -| `cloudflare` | `cloudflare` | 必须配置含 account ID 的 API Base | -| `mistral` | `mistral` | `https://api.mistral.ai/v1` | -| `xai` | `xai` | `https://api.x.ai/v1` | -| `deepseek` | `deepseek` | `https://api.deepseek.com` | -| `zhipu` | `zhipu` | `https://open.bigmodel.cn/api/paas/v4` | -| `groq` | `groq` | `https://api.groq.com/openai/v1` | -| `nvidia` | `nvidia` | `https://integrate.api.nvidia.com/v1` | -| `together` | `together` | `https://api.together.ai/v1` | - -当前 Azure Adapter 使用 v1 Endpoint 的 `api-key` 鉴权;Bedrock Adapter 使用 Bedrock Runtime Converse 与 Bearer API Key,尚未接 IAM Role/SigV4;Cloudflare Adapter 使用 Workers AI 官方 `/ai/v1/chat/completions`。配置其他鉴权方式或模型专属输入 schema 时会明确报不支持,不会静默伪装成兼容成功。 - -Mistral、DeepSeek、xAI、Zhipu、Groq、NVIDIA 和 Together 都有独立的 Adapter 类型与文件,后续厂商专属字段、错误解析或能力规则只进入对应 Adapter。它们没有复制七份相同的请求发送代码:官方相同的 Bearer 鉴权、OpenAI Chat JSON 和响应校验由 `compatible.go` 复用。DeepSeek Adapter 使用其 API Base 下的 `/chat/completions`,不会套用自定义兼容端点默认补 `/v1` 的规则。 - -另有 `custom`,必须显式配置 `type: "openai-compatible"` 和 `base_url`。已提供兼容接口的 Google、Alibaba 等厂商也可以显式把 `type` 设成 `openai-compatible`;这是主动选择兼容协议。 - -`MODEL_VELO_ROUTING_JSON` 没有隐式默认值。缺失或只有空白时进程会在连接 PostgreSQL、Redis 和监听端口前启动失败;即使只接一个自定义兼容上游,也必须显式写出 Provider 和 Route。配置中的 `model: "*"` 是默认路由;`upstream_model` 留空表示透传客户端模型,设置值则执行模型别名映射。`candidates` 的数组顺序就是稳定优先级;Router 先按请求所需能力过滤候选,首个实际执行的候选失败且错误策略允许 Fallback 时,Orchestrator 才会执行下一候选。 - -`model_capabilities` 按上游模型声明 `text`、`image`、`audio`、`file`、`tools`、`structured`;未声明的模型安全地默认为 `text`。能力不能超过当前 Adapter 协议真正能承载的范围,否则启动失败。Provider 的可选 `runtime` 可分别覆盖 `breaker`、`queue`、`retry` 和 `http`:请求总预算仍是全链路唯一值,Provider 只覆盖单次 `attempt_timeout`;HTTP 每 Host 连接数默认跟随该 Provider 的 `queue.max_in_flight`,避免 Queue 放行后又在默认连接池里形成第二条隐形队列。 - -`runtime` 支持的字段是:`breaker.failure_threshold/open_duration/half_open_max_probes`、`queue.max_in_flight/max_waiting/wait_timeout`、`retry.max_attempts/initial_backoff/max_backoff/backoff_multiplier/jitter_ratio/attempt_timeout`,以及 `http.max_idle_connections/max_idle_connections_per_host/max_connections_per_host`。未写的字段继承上表环境变量默认值,所有组合都在启动时校验。 - -以下示例让同一个 Gemini Provider 承载文本和 VLM,并准备 DeepSeek、OpenAI 作为其他路由或后续 fallback 候选: - -```json -{ - "providers": [ - {"id":"gemini-main","vendor":"google","type":"gemini","models":["gemini-2.5-flash","gemini-2.5-pro"],"model_capabilities":{"gemini-2.5-flash":["text"],"gemini-2.5-pro":["text","image"]},"runtime":{"queue":{"max_in_flight":40,"max_waiting":200,"wait_timeout":"2s"},"retry":{"max_attempts":2,"attempt_timeout":"15s"},"http":{"max_idle_connections":80}}}, - {"id":"deepseek-main","vendor":"deepseek","type":"deepseek","models":["deepseek-chat","deepseek-reasoner"],"model_capabilities":{"deepseek-chat":["text","tools"],"deepseek-reasoner":["text"]}}, - {"id":"openai-main","vendor":"openai","type":"openai","models":["gpt-4o-mini","gpt-4o"],"model_capabilities":{"gpt-4o-mini":["text","tools"],"gpt-4o":["text","image","tools"]}} - ], - "routes": [ - {"model":"fast-chat","candidates":[{"provider":"gemini-main","upstream_model":"gemini-2.5-flash"},{"provider":"deepseek-main","upstream_model":"deepseek-chat"}]}, - {"model":"vision","candidates":[{"provider":"gemini-main","upstream_model":"gemini-2.5-pro"},{"provider":"openai-main","upstream_model":"gpt-4o"}]} - ] -} -``` - -多 Key 配置与路由配置分离,避免 Route Plan 携带凭证。对应的多 Provider Key 配置: - -```json -{"providers":[{"provider_id":"gemini-main","keys":[{"id":"primary","secret":"replace-with-gemini-key"}]},{"provider_id":"deepseek-main","keys":[{"id":"primary","secret":"replace-with-deepseek-key"},{"id":"secondary","secret":"replace-with-second-deepseek-key"}]},{"provider_id":"openai-main","keys":[{"id":"primary","secret":"replace-with-openai-key"}]}]} -``` - -需要鉴权的 Provider ID 必须与 Key 配置一一对应;每个此类 Provider 至少一个 Key,Key ID 必须唯一,Secret 不能为空。Ollama 不出现在 Key 配置中。配置错误会在连接 PostgreSQL/Redis 前阻止启动。模型名只是路由声明,不表示网关替厂商保证该账号、区域或模型已开通。 - -可以复制 `.env.example` 的变量名用于本地配置,但程序当前不会自动读取 `.env` 文件,需要在启动进程前把变量注入环境。 - -> 预发布改名说明:项目已统一改为 `Model-Velo`。旧环境变量不再读取,旧 API Key 前缀不再接受,旧 Redis namespace 不再命中。新 Compose 默认数据库和用户是 `model_velo`;已有 PostgreSQL 数据不会被删除,需要时可继续通过新的 `MODEL_VELO_POSTGRES_DSN` 显式指向原数据库。 - -## 启动 PostgreSQL 和 Redis - -需要先启动 Docker Desktop。首次使用时复制配置并替换两个本地密码: - -```powershell -Copy-Item .env.example .env -``` - -`MODEL_VELO_POSTGRES_PASSWORD` 必须与 `MODEL_VELO_POSTGRES_DSN` 中的密码保持一致。然后解析配置并启动基础设施: - -```powershell -docker compose config --quiet -docker compose up -d postgres redis -docker compose ps -``` - -`docker compose ps` 中两个服务都应显示为 `healthy`。排查启动问题: - -```powershell -docker compose logs postgres -docker compose logs redis -``` - -停止容器但保留 PostgreSQL 和 Redis 数据: - -```powershell -docker compose down -``` - -显式删除容器、网络和两个命名数据卷,恢复全新环境: - -```powershell -docker compose down --volumes -``` - -不要在仍需保留本地开发数据时执行带 `--volumes` 的命令。 - -真实 PostgreSQL 综合用例只读取显式的 `MODEL_VELO_POSTGRES_TEST_DSN`,不会回退到 API 使用的 DSN。测试用户需要拥有 `CREATE SCHEMA` 权限;每次运行创建随机的 `model_velo_it_*` schema,在其中重复执行 GORM `AutoMigrate`、API Key 生命周期和认证授权验证,结束后只定向删除该随机 schema: - -```powershell -$env:MODEL_VELO_POSTGRES_TEST_DSN = $env:MODEL_VELO_POSTGRES_DSN -go test -count=1 -run '^TestPostgresAPIKeyLifecycle$' -v ./internal/apikey -``` - -未设置测试 DSN 时用例会跳过,跳过不等于成功。2026-07-18 已使用无持久卷的 PostgreSQL 17.10 一次性容器真实通过:随机 schema 中两次 AutoMigrate、约束/索引检查和完整认证生命周期均成功,随后 schema 与容器已清理。 - -## Redis Client - -Model-Velo 固定使用 `github.com/redis/go-redis/v9 v9.21.0`。API 启动时按照配置创建自带连接池的 Redis Client,并在 `MODEL_VELO_REDIS_DIAL_TIMEOUT` 期限内执行 `PING`: - -- `required`:网络、认证或 Redis 服务异常会关闭 Client 并阻止 HTTP Server 启动; -- `optional`:启动失败会记录不含密码的警告并继续,Client 保留,Redis 恢复后可供后续命令使用; -- 根 Context 已取消时,无论策略是什么都停止启动; -- 进程退出时通过幂等 `Close` 释放连接池。 - -Chat 请求会在认证和模型授权成功后访问 Redis 执行限流,再执行 Exact Cache 读写。`optional` 只决定启动 Ping 失败时能否继续监听;运行中的限流故障由 `MODEL_VELO_RATE_LIMIT_FAILURE_POLICY` 决定,缓存故障固定为 fail-open。 - -## Redis 租户限流 - -当前配额模型是“每个租户、每个网关模型、每个固定窗口最多 N 个已接受请求”。Redis Key 的结构为: - -```text -model-velo:rate-limit:v1::tenant::model: -``` - -Key 包含环境命名空间,并分别对 tenant ID 和规范化模型名做 SHA-256;它不包含 Model-Velo API Key、Provider Key 或原始模型名。多个 API Key 只要属于同一租户,就共享该租户对应模型的配额。 - -限流器通过一个 Lua 脚本完成读取计数、首次写入、递增、TTL、Redis 服务端时间和决策返回。Redis 把脚本作为单个命令执行,执行期间不会插入其他客户端命令,因此多个 Model-Velo 实例不会在应用侧 `GET` 与 `SET` 之间同时放过超额请求。拒绝请求不会递增计数或延长窗口。 - -请求通过限流时响应包含 `X-RateLimit-Limit`、`X-RateLimit-Remaining` 和 Unix 秒格式的 `X-RateLimit-Reset`。额度耗尽时返回结构化 `429 rate_limit_exceeded`,并增加整数秒 `Retry-After`。Redis 运行时故障时: - -- `fail-closed`(默认):返回结构化 `503 rate_limit_unavailable`,不调用 Provider; -- `fail-open`:继续调用 Provider,并返回 `X-RateLimit-Status: bypassed`,但不伪造额度与重置值; -- 请求 Context 已取消时:无论故障策略如何都立即停止,不把取消转换为放行。 - -该算法的边界是固定窗口交界处可能出现突发流量;当前阶段选择它是因为配额语义直接、单脚本可解释。Redis 8.8.0 容器测试已经验证窗口恢复、租户/模型隔离和双 Client 竞争不超发;race 门禁因本机缺少 CGO/C 编译器暂未完成,仍是阶段 2 的最后环境缺口。 - -真实 Redis 用例通过 `MODEL_VELO_REDIS_TEST_ADDR`、`MODEL_VELO_REDIS_TEST_PASSWORD` 和 `MODEL_VELO_REDIS_TEST_DB` 显式启用,使用随机 namespace 定向清理,不执行 `FLUSHDB`。未配置时测试会跳过,跳过不等于验证通过。 - -## Redis Exact Response Cache - -缓存只处理已经通过请求校验、认证、模型授权和限流的 `stream=false` 请求。请求携带 `Cache-Control: no-store` 时显式绕过。缓存命中仍会消耗一次租户配额,这是当前明确选择的调用顺序,而不是“命中免费”。 - -Redis Key 为: +## 架构 ```text -model-velo:response-cache:v1::tenant::model::route::request: +Client + │ + ▼ +Model-Velo Gateway ──► Provider APIs + │ + ├──► Redis ──► Usage Worker + │ + └──► PostgreSQL ``` -Key 不包含原始 API Key、tenant ID、模型名或提示词。规范化会递归排序 JSON 对象字段,并保留数组顺序、数字文本以及“字段缺失/显式给值”的区别;因此消息、工具和生成参数顺序不会被错误改写。包含重复对象字段的 JSON 直接绕过缓存,避免不同解析规则造成碰撞。 - -命中返回 Redis 中的完整合法 JSON;未命中只在首候选返回完整成功 JSON 后回填。Fallback 成功、上游错误、响应读取失败、客户端取消和显式绕过均不写缓存,避免临时降级结果覆盖正常路由语义。Redis 读写错误只记录 request ID 和安全错误,不记录 tenant、模型、Key 或提示词,并继续 Provider;响应通过 `X-Model-Velo-Cache: HIT|MISS|BYPASS` 表达本次状态,最终 Usage Event 也记录同一状态。 - -`MODEL_VELO_CACHE_ROUTE_VERSION` 只属于缓存命名空间,并以摘要进入缓存 Key。更换上游、模型映射或候选顺序时必须递增它,避免新路由命中旧响应。2026-07-18 真实 Redis 8.8.0 综合用例已验证跨请求命中、TTL、租户/参数隔离、错误不缓存和关闭 Client 后的故障降级;测试使用无持久卷的一次性容器并已自动清理。 - -当前调用顺序是 `认证 → 授权 → 限流 → Route Plan → Cache → Fallback Orchestrator → 当前候选 Attempt(Breaker → Queue → Key → Retry → Provider → 反馈/释放)→ 成功回填`。无路由返回 `503 route_unavailable`;Breaker Open 返回 `503 provider_circuit_open`;Queue 满或等待超时分别返回 `503 provider_queue_full`、`503 provider_queue_timeout`;没有可用 Key 返回 `503 provider_keys_exhausted`。本地拒绝不会调用当前 Provider;是否进入下一候选由统一 Fallback 策略决定。 - -## Provider Circuit Breaker - -Breaker 以稳定 Provider ID 隔离状态。Closed 累计网络错误、上游超时、500/502/503/504,以及上游返回 2xx 却不满足已声明响应协议的错误;成功会清零连续失败。401/403、429、普通 4xx、501 等非策略 5xx、网关主动限制的超大响应和客户端取消不会把整个 Provider 熔断。达到阈值后进入 Open,缓存未命中的新请求快速失败;冷却到期后进入 HalfOpen,只放行配置数量的探测,探测成功关闭,符合计数策略的失败重新 Open。 - -每次准入返回一次性 Permit,Provider 调用后必须反馈成功或分类后的 Failure;调用链额外使用 `defer Abandon()` 防止提前退出泄漏 HalfOpen 探测名额。401/403 与 429 不计 Provider 故障,而是反馈给当前 Key 的健康状态。 - -## Provider 有界 Queue - -Queue 是网关进程内的 Provider 容量保护,不是 Redis 租户配额。每个 Provider 拥有独立的带缓冲 channel:channel 容量是正在调用上游的硬上限;容量耗尽后,只有 `MAX_WAITING` 个请求可以等待,更多请求立即返回结构化 503。等待同时监听请求 Context 和 Queue Timer,因此客户端断开、请求总 deadline 或 Queue 等待超时都会停止占位。 - -成功取得槽位后会得到一次性 Lease,调用链用 `defer Release()` 配对归还;原子标记阻止重复释放。Queue Snapshot 只暴露 Provider ID、active、waiting、容量和拒绝/超时/取消计数,不包含 Provider Key、请求内容或错误 cause。Queue 拒绝不计入 Breaker 故障,因为它说明本实例容量饱和,不代表上游已经失败。 - -Queue Registry 按 Provider ID 隔离;primary 与每个 fallback 候选都会进入自己 Provider 的 Queue,不会复用上一个候选的 Lease。Queue 不是跨实例分布式信号:部署多个副本时,理论总并发上限约为“副本数 × `MAX_IN_FLIGHT`”,应结合上游账号配额设置每实例容量。 - -## Provider Key Selector +## 快速启动 -Key Registry 按 Provider ID 隔离。选择器只把稳定 Key ID 暴露给快照和失败元数据,Secret 只在构造上游 Authorization 时读取。每次选择通过原子游标轮换起点,再在读锁下跳过禁用或冷却中的 Key。401 表示凭证无效,会永久禁用当前 Key,直到进程使用修正配置重启;403 可能来自模型或账号权限,只把该 Key 排除在当前请求之外;429 优先按上游 `Retry-After` 冷却,缺失或非法时使用 30 秒,最长限制为 24 小时。成功反馈可以清除临时冷却,但不会恢复 401 禁用状态。 +### 环境要求 -401、403 或 429 反馈后,当前请求会释放 Queue/Breaker 资源并重新进入完整准入链,从其他可用 Key 继续。同一请求不会再次选择已返回 401/403 的 Key。所有可用 Key 都在冷却时,只有仍有 Retry 次数且最早恢复时间小于剩余请求预算才等待;等待期间可被客户端取消,且不占用 Queue 槽位或 Breaker Permit。等待不合算时立即 Fallback;没有候选可切换时返回 `503 provider_keys_exhausted`,并携带最早恢复时间对应的 `Retry-After`。指定 5xx、网络和上游超时则优先保持原 Key 并按指数退避重试。Key Selector 与 Retry 的完整并发/race 证据仍保留在阶段 3 门禁,不能据此宣称已经完成高并发验证。 +- Docker +- Docker Compose -## PostgreSQL 与 GORM +### 1. 配置环境变量 -Model-Velo 直接使用 GORM 连接 PostgreSQL。API 启动时先取得 GORM 底层的 `database/sql` 连接池,设置最大打开连接数、最大空闲连接数、连接寿命和空闲时间,再执行有界 `PingContext`。数据库不可达时 API 不会开始监听端口。官方 `gorm.io/driver/postgres` 内部依赖 pgx,因此模块图中仍会出现 `pgx // indirect`,但 Model-Velo 源码不直接导入或调用 pgx。 - -连接成功后,启动流程调用 GORM `AutoMigrate` 同步以下模型: - -1. `Tenant` → `tenants`; -2. `APIKey` → `api_keys`; -3. `TenantModelGrant` → `tenant_model_grants`。 - -当前阶段不再维护 `golang-migrate`、独立 Migration CLI 或手写 SQL 文件。`AutoMigrate` 用于创建缺失的表、列、索引、外键和检查约束,但不会负责删除废弃列,也没有 `down`/版本回滚命令。发生破坏性 schema 变更时,必须先备份数据并为该次变更单独设计显式升级方案,不能依赖启动时自动删改数据。 - -Schema 安全边界: - -- `tenants.slug` 唯一,租户状态只允许 `active/disabled`; -- `api_keys` 只保存 Key 前缀、查找摘要、不可逆哈希和哈希版本,不存在明文 Key 列; -- 租户存在 API Key 时禁止删除,避免凭证被级联误删; -- 删除没有 API Key 的租户时,其模型授权记录级联删除; -- 禁用租户不会删除 Key 或模型授权,后续认证查询必须同时检查租户状态; -- `tenant_model_grants` 使用 `(tenant_id, gateway_model)` 主键避免重复授权。 - -## Model-Velo API Key - -Key 格式固定为: - -```text -mvl_<16 字符公开定位符>_<43 字符随机密文> -``` - -定位符由 12 个随机字节编码得到,密文由 32 个随机字节编码得到,均使用无填充 Base64 URL 编码。数据库不会保存完整 Key: - -Base64URL 的合法字符本身包含 `_`,所以解析器按 16 字符 locator 和 43 字符 secret 的固定位置切分,不使用普通的下划线 `Split`;否则网关会偶发拒绝自己生成的合法 Key。 - -- `key_prefix` 保存 `mvl_<定位符>`,用于后台展示; -- `lookup_digest` 保存公开定位符前缀的 SHA-256,用于索引查询; -- `key_hash` 保存随机密文在服务端 Pepper 下的 HMAC-SHA-256; -- `hash_version` 当前为 `1`,为以后更换校验方案保留升级边界; -- 明文 Key 只在创建成功时输出一次,之后无法从数据库恢复。 - -先配置 PostgreSQL 和稳定的 Pepper。本地首次使用可以在 PowerShell 中生成随机 Pepper: - -```powershell -$pepperBytes = New-Object byte[] 48 -[System.Security.Cryptography.RandomNumberGenerator]::Fill($pepperBytes) -$env:MODEL_VELO_API_KEY_PEPPER = [Convert]::ToBase64String($pepperBytes) +```bash +cp .env.example .env ``` -同一个数据库必须长期使用同一个 Pepper。初始化租户、模型授权和首个 Key: - -```powershell -go run ./cmd/model-velo-admin bootstrap-tenant ` - --slug demo ` - --name "Demo Tenant" ` - --label "local development" ` - --models "gpt-4o-mini,gpt-4.1-mini" -``` +编辑 `.env`,替换示例密码和密钥,并配置: -命令成功后输出 `tenant_id`、`api_key_id`、公开前缀以及仅出现一次的 `api_key`。为已有租户创建第二个 Key: +- `MODEL_VELO_PROVIDER_KEYS_JSON`:上游 Provider Key +- `MODEL_VELO_ROUTING_JSON`:Provider、模型能力与路由规则 -```powershell -go run ./cmd/model-velo-admin create-key ` - --tenant-id "replace-with-tenant-uuid" ` - --label "second key" ` - --expires-in 720h -``` +### 2. 启动服务 -禁用或永久吊销 Key: - -```powershell -go run ./cmd/model-velo-admin disable-key --id "replace-with-api-key-uuid" -go run ./cmd/model-velo-admin revoke-key --id "replace-with-api-key-uuid" +```bash +docker compose up -d --build +docker compose ps ``` -禁用和吊销会立即影响后续认证。永久吊销的 Key 不能通过 `disable-key` 降级回普通禁用状态。 +网关默认监听 `http://localhost:8080`,Usage Worker 指标端口默认为 `9091`。 -## HTTP 认证与模型授权 +### 3. 创建租户和 API Key -`GET /healthz` 不需要 API Key。`POST /v1/chat/completions` 必须使用一个且只能使用一个 Authorization Header: - -```http -Authorization: Bearer mvl__ +```bash +docker compose --profile tools run --rm admin bootstrap-tenant \ + --slug demo \ + --name "Demo" \ + --models "*" ``` -网关按以下顺序处理请求: +命令只显示一次明文 `api_key`,请妥善保存。 -1. 严格解析 Bearer Header,拒绝缺失、重复、错误 scheme、空 Token 和额外空白; -2. 根据 locator 摘要查询 Key,并以 HMAC 常量时间比较验证 secret; -3. 检查 Key 状态、UTC 过期时间和租户状态; -4. 把 `tenant_id`、`api_key_id` 和公开 Key 前缀写入请求 Context; -5. 解析 Chat JSON 后,检查 `(tenant_id, model)` 是否存在于 `tenant_model_grants`; -6. 用 tenant ID 和规范化模型名执行 Redis 原子限流; -7. 生成有序 Route Plan,再以环境、租户、模型、路由版本和规范化请求查询 Exact Cache; -8. 命中直接返回;未命中或缓存故障时,在统一总预算内按候选顺序执行 Attempt; -9. 每个候选重新选择对应 Adapter、Breaker、Queue 和 Key,在候选内有限 Retry;允许 Fallback 的最终失败才进入下一候选; -10. 首个完整成功 JSON 回填缓存并返回客户端;普通 400、取消或候选耗尽返回结构化错误。 -11. 请求终态由 Collector finalize 一次,并在独立短超时内投递 Usage Event;流式请求在 `[DONE]`、取消或提交后中断时分别记录终态。 +### 4. 发送请求 -未知 Key、错误 secret、禁用、吊销、过期和禁用租户对客户端统一返回 `401 invalid_api_key`,避免暴露具体凭证状态。到期时刻本身已经算过期,创建 Key 时也只接受严格晚于当前时刻的 UTC 时间。缺少模型授权返回 `403 model_not_allowed`。认证数据库故障返回 `503 authentication_unavailable`,不会退化成匿名请求。 - -客户端的 Model-Velo Key 不会转发给外部 Provider。Provider 请求只使用 `MODEL_VELO_PROVIDER_KEYS_JSON` 中按目标 Provider ID 选出的 Secret。 - -## 运行 Model-Velo API - -需要 Go 工具链、一个已经健康的 PostgreSQL、Redis,以及至少一个显式配置的 Provider。启动时会自动同步当前 GORM 模型。下面用单个自定义 OpenAI-compatible Provider 演示;这仍然是一份显式路由,不是默认回退: +```bash +curl http://localhost:8080/v1/chat/completions \ + -H "Authorization: Bearer " \ + -H "Content-Type: application/json" \ + -d '{ + "model": "your-model", + "messages": [ + {"role": "user", "content": "Hello"} + ] + }' +``` + +## HTTP 接口 + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| `GET` | `/healthz` | 进程存活检查 | +| `GET` | `/readyz` | PostgreSQL 与 Redis 就绪检查 | +| `GET` | `/metrics` | Prometheus 指标 | +| `GET` | `/v1/models` | 模型列表 | +| `POST` | `/v1/chat/completions` | OpenAI Chat Completions | +| `POST` | `/v1/responses` | OpenAI Responses | +| `POST` | `/v1/embeddings` | OpenAI Embeddings | +| `POST` | `/v1/messages` | Anthropic Messages | +| `GET` | `/v1/usage/events` | Usage 明细 | +| `GET` | `/v1/usage/summary` | Usage 汇总 | +| `GET` | `/v1/usage/series` | Usage 时间序列 | + +`/v1/*` 接口使用网关 API Key。OpenAI 兼容接口通过 `Authorization: Bearer ` 认证;Anthropic Messages 接口通过 `x-api-key: ` 认证。 + +## 本地开发 -```powershell -$env:MODEL_VELO_ROUTING_JSON = '{"providers":[{"id":"upstream","vendor":"custom","type":"openai-compatible","base_url":"https://api.example.com","models":["*"],"model_capabilities":{"*":["text"]}}],"routes":[{"model":"*","candidates":[{"provider":"upstream"}]}]}' -$env:MODEL_VELO_PROVIDER_KEYS_JSON = '{"providers":[{"provider_id":"upstream","keys":[{"id":"primary","secret":"replace-with-provider-key"}]}]}' -$env:MODEL_VELO_ENVIRONMENT = "development" -$env:MODEL_VELO_POSTGRES_DSN = "postgres://model_velo:replace-with-local-postgres-password@localhost:5432/model_velo?sslmode=disable" -$env:MODEL_VELO_API_KEY_PEPPER = "use-the-same-stable-pepper-as-model-velo-admin" -$env:MODEL_VELO_REDIS_ADDR = "localhost:6379" -$env:MODEL_VELO_REDIS_PASSWORD = "replace-with-local-redis-password" -$env:MODEL_VELO_RATE_LIMIT_FAILURE_POLICY = "fail-closed" +```bash go run ./cmd/model-velo -``` - -Usage Worker 与 API 使用相同的 PostgreSQL、Redis 和 `MODEL_VELO_ENVIRONMENT`,但作为独立进程运行: - -```powershell go run ./cmd/model-velo-usage-worker ``` -Worker 启动会幂等创建 consumer group。数据库写入失败不 ACK;收到退出信号后停止新读取,并给当前批次最多 `MODEL_VELO_USAGE_WORKER_TIMEOUT` 完成。生产环境应为每个并发 Worker 设置不同的 `MODEL_VELO_USAGE_CONSUMER`,或使用默认的主机名与 PID。 - -价目新增或修正后,可显式重算缺少成本的历史记录。命令要求确认、限制时间范围和单批数量,不会把无法定价的记录写成零;输出非空 `next_cursor` 时,把它传给下一批即可越过当前批次中仍无法定价的行: - -```powershell -go run ./cmd/model-velo-admin reprice-usage ` - --start 2026-07-01T00:00:00Z ` - --end 2026-08-01T00:00:00Z ` - --missing-only=true ` - --limit 1000 ` - --confirm -``` - -请求示例: - -```powershell -$modelVeloKey = "replace-with-created-mvl-key" -$headers = @{ - Authorization = "Bearer $modelVeloKey" - "Content-Type" = "application/json" -} -$body = @{ - model = "gpt-4o-mini" - messages = @( - @{ role = "user"; content = "hello" } - ) -} | ConvertTo-Json -Depth 5 - -Invoke-RestMethod ` - -Method Post ` - -Uri "http://localhost:8080/v1/chat/completions" ` - -Headers $headers ` - -Body $body -``` - -同一接口也可直接使用 OpenAI Python SDK;`base_url` 必须指向网关的 `/v1`,Key 使用创建时仅返回一次的 Model-Velo 业务 Key: - -```python -import os -from openai import OpenAI - -client = OpenAI( - api_key=os.environ["MODEL_VELO_CLIENT_KEY"], - base_url="http://localhost:8080/v1", -) - -completion = client.chat.completions.create( - model="gpt-4o-mini", - messages=[{"role": "user", "content": "hello"}], -) -print(completion.choices[0].message.content) - -stream = client.chat.completions.create( - model="gpt-4o-mini", - messages=[{"role": "user", "content": "stream hello"}], - stream=True, -) -for chunk in stream: - if chunk.choices and chunk.choices[0].delta.content: - print(chunk.choices[0].delta.content, end="", flush=True) -``` - -Bash/curl 的最小重放请求: +代码检查: ```bash -curl --fail-with-body http://localhost:8080/v1/chat/completions \ - -H "Authorization: Bearer ${MODEL_VELO_CLIENT_KEY}" \ - -H "Content-Type: application/json" \ - -d '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"hello"}]}' -``` - -聊天、Responses 与 Anthropic Messages 请求体上限为 16 MiB,Embeddings 为 8 MiB;这允许常见内嵌多模态输入,同时仍在 JSON 解码前限制内存占用。更大的媒体应使用厂商支持的 URL/File ID,而不是继续扩大 Base64 请求。 - -不要使用示例地址调用真实服务,也不要把真实 Key 写入 `.env.example`、测试、日志或 Git。 - -日常开发检查命令: - -```powershell -$goFiles = rg --files -g '*.go' -gofmt -w $goFiles go test ./... go vet ./... ``` -项目当前采用快速交付模式:学习和讲解以生产代码、请求链和故障策略为主,不要求逐个学习测试文件。一个纵向功能通常最多维护一个合并测试文件;已有端到端证据能覆盖时,不再给 Handler、Service 和存储层重复补同类测试,也不追求覆盖率数字。 +停止 Docker 环境: -`go test ./...` 和 `go vet ./...` 在每次交付前统一运行一次。`go test -race ./...`、真实 Redis/PostgreSQL 故障矩阵和完整异常测试只在阶段门禁集中执行一次;环境不满足时记录缺口,不把纯测试任务继续当作开发进度。 - -阶段拆解和实时完成度见 `TODO.md`,完整规划边界见 `ARCHITECTURE.md`,开发约束见 `AGENTS.md`。 +```bash +docker compose down +``` diff --git "a/age)\357\200\272 stabilize and instrument benchmark delivery\357\200\242" "b/age)\357\200\272 stabilize and instrument benchmark delivery\357\200\242" new file mode 100644 index 0000000..c433bf4 --- /dev/null +++ "b/age)\357\200\272 stabilize and instrument benchmark delivery\357\200\242" @@ -0,0 +1,19 @@ +M cmd/model-velo-usage-worker/main.go +M cmd/model-velo/main.go +M internal/httpapi/auth.go +M internal/httpapi/chat.go +M internal/httpapi/chat_test.go +M internal/httpapi/router_options.go +M internal/httpapi/usage.go +A internal/observability/dependencies.go +M internal/observability/metrics.go +M internal/observability/observability_test.go +A internal/reliability/queue_observer.go +M internal/reliability/tracing.go +M internal/usage/collector.go +M internal/usage/emitter.go +M internal/usage/outbox.go +M internal/usage/store.go +M internal/usage/usage_test.go +M internal/usage/worker.go +M test/threehost/collect-run-evidence.sh diff --git a/cmd/model-velo-admin/main.go b/cmd/model-velo-admin/main.go index 2744004..24def49 100644 --- a/cmd/model-velo-admin/main.go +++ b/cmd/model-velo-admin/main.go @@ -14,6 +14,7 @@ import ( "model-velo/internal/apikey" "model-velo/internal/config" "model-velo/internal/postgres" + redisstore "model-velo/internal/redis" "model-velo/internal/usage" ) @@ -53,10 +54,25 @@ func run(ctx context.Context, arguments []string) error { if err != nil { return fmt.Errorf("configure API key security: %w", err) } - manager, err := apikey.NewManager(database.ORM(), security.Pepper) + var manager *apikey.Manager + closeCache := func() {} + if arguments[0] == "revoke-key" || + arguments[0] == "disable-key" { + manager, closeCache, err = newInvalidatingKeyManager( + ctx, + database, + security.Pepper, + ) + } else { + manager, err = apikey.NewManager( + database.ORM(), + security.Pepper, + ) + } if err != nil { return err } + defer closeCache() switch arguments[0] { case "bootstrap-tenant": @@ -72,6 +88,43 @@ func run(ctx context.Context, arguments []string) error { } } +func newInvalidatingKeyManager( + ctx context.Context, + database *postgres.Database, + pepper []byte, +) (*apikey.Manager, func(), error) { + settings, err := config.LoadAuthCache() + if err != nil { + return nil, nil, err + } + if !settings.Enabled { + manager, managerErr := apikey.NewManager(database.ORM(), pepper) + return manager, func() {}, managerErr + } + redisSettings, err := config.LoadRedis() + if err != nil { + return nil, nil, err + } + client, err := redisstore.Open(ctx, redisSettings) + if err != nil { + return nil, nil, err + } + manager, err := apikey.NewCachedManager( + database.ORM(), + pepper, + client.Native(), + settings, + nil, + ) + if err != nil { + _ = client.Close() + return nil, nil, err + } + return manager, func() { + _ = client.Close() + }, nil +} + func bootstrapAdmin( ctx context.Context, database *postgres.Database, diff --git a/cmd/model-velo-usage-worker/main.go b/cmd/model-velo-usage-worker/main.go index 81d0653..b9cbe00 100644 --- a/cmd/model-velo-usage-worker/main.go +++ b/cmd/model-velo-usage-worker/main.go @@ -96,7 +96,12 @@ func run() error { if err != nil { return fmt.Errorf("configure usage outbox emitter: %w", err) } - relay, err := usage.NewOutboxRelay(database.ORM(), redisEmitter, usageConfig.WorkerTimeout) + relay, err := usage.NewOutboxRelay( + database.ORM(), + redisEmitter, + usageConfig.Group, + usageConfig.BatchSize, + ) if err != nil { return fmt.Errorf("configure usage outbox relay: %w", err) } diff --git a/cmd/model-velo/main.go b/cmd/model-velo/main.go index ae7199d..2b7e426 100644 --- a/cmd/model-velo/main.go +++ b/cmd/model-velo/main.go @@ -84,11 +84,6 @@ func run() error { // 装配基础设施、启动 HTTP 服务并等待关闭。 return err // 表结构同步失败时终止启动。 } - access, err := apikey.NewManager(database.ORM(), startup.apiKeySecurity.Pepper) // 创建基于数据库的 API Key 认证授权服务。 - if err != nil { // API Key 管理器创建失败。 - return fmt.Errorf("configure API key manager: %w", err) // 返回认证组件配置错误。 - } - redisClient, err := redisstore.Open(ctx, startup.infrastructure.Redis) // 使用配置连接 Redis。 if err != nil { // Redis 客户端创建失败。 return fmt.Errorf("connect Redis: %w", err) // 终止启动并返回 Redis 错误。 @@ -97,6 +92,26 @@ func run() error { // 装配基础设施、启动 HTTP 服务并等待关闭。 if !redisClient.AvailableAtStartup() { // 启动时 Redis Ping 未成功。 slog.Warn("Redis startup ping failed; continuing with configured degradation policy") } + metrics := observability.NewMetrics() + var access *apikey.Manager + if startup.authCache.Enabled { + access, err = apikey.NewCachedManager( + database.ORM(), + startup.apiKeySecurity.Pepper, + redisClient.Native(), + startup.authCache, + metrics, + ) + } else { + access, err = apikey.NewManager( + database.ORM(), + startup.apiKeySecurity.Pepper, + ) + } + if err != nil { + return fmt.Errorf("configure API key manager: %w", err) + } + access.StartInvalidationListener(ctx) limiter, err := ratelimit.New(redisClient.Native(), startup.rateLimit) // 基于原生 Redis 客户端创建限流器。 if err != nil { // 限流器配置无效。 return fmt.Errorf("configure rate limiter: %w", err) // 返回限流组件配置错误。 @@ -110,22 +125,20 @@ func run() error { // 装配基础设施、启动 HTTP 服务并等待关闭。 if err != nil { // 响应缓存配置无效。 return fmt.Errorf("configure response cache: %w", err) // 返回缓存组件配置错误。 } - redisUsageEmitter, err := usage.NewRedisEmitter( - redisClient.Native(), - startup.usage.StreamKey, - startup.usage.EmitTimeout, - ) - if err != nil { - return fmt.Errorf("configure usage emitter: %w", err) - } usageEmitter, err := usage.NewDurableEmitter( database.ORM(), - redisUsageEmitter, startup.usage.EmitTimeout, ) if err != nil { return fmt.Errorf("configure durable usage emitter: %w", err) } + defer func() { + closeContext, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if err := usageEmitter.Close(closeContext); err != nil { + slog.Error("durable usage emitter shutdown failed", "error", err) + } + }() usagePricing, err := usage.NewPricingCatalog(startup.usage.Pricing) if err != nil { return fmt.Errorf("configure usage pricing: %w", err) @@ -177,12 +190,24 @@ func run() error { // 装配基础设施、启动 HTTP 服务并等待关闭。 if err != nil { return fmt.Errorf("configure quota manager: %w", err) } + if err := quotaManager.LoadPolicyIndex(ctx); err != nil { + return fmt.Errorf("load quota policy index: %w", err) + } + go quotaManager.RunPolicyIndexRefresh( + ctx, + startup.controlPlane.RefreshInterval, + ) go quotaManager.RunReaper(ctx) - metrics := observability.NewMetrics() if err := metrics.RegisterRuntime(runtimeManager); err != nil { return fmt.Errorf("register runtime metrics: %w", err) } + if err := metrics.RegisterDependencies( + database.SQL(), + redisClient.Native(), + ); err != nil { + return fmt.Errorf("register dependency metrics: %w", err) + } readiness := health.NewChecker( database.SQL(), redisClient.Native(), @@ -221,8 +246,9 @@ func run() error { // 装配基础设施、启动 HTTP 服务并等待关闭。 type startupConfig struct { // 保存启动后续步骤需要的全部配置和已创建组件。 infrastructure config.Infrastructure // PostgreSQL、Redis 等基础设施配置。 apiKeySecurity config.APIKeySecurity // API Key 哈希 Pepper 等安全配置。 - rateLimit config.RateLimit // 租户限流配置。 - responseCache config.ResponseCache // 响应缓存配置。 + authCache config.AuthCache + rateLimit config.RateLimit // 租户限流配置。 + responseCache config.ResponseCache // 响应缓存配置。 usage config.Usage observability config.Observability controlPlane config.ControlPlane @@ -247,6 +273,10 @@ func loadStartupConfig() (startupConfig, error) { // 加载所有启动配置并 if err != nil { // API Key 安全配置不合法。 return startupConfig{}, err } + authCache, err := config.LoadAuthCache() + if err != nil { + return startupConfig{}, err + } rateLimit, err := config.LoadRateLimit() // 读取租户限流配置。 if err != nil { // 限流配置不合法。 return startupConfig{}, err @@ -353,8 +383,9 @@ func loadStartupConfig() (startupConfig, error) { // 加载所有启动配置并 return startupConfig{ // 返回启动后续阶段所需的完整配置和组件。 infrastructure: infrastructure, // 保存基础设施配置。 apiKeySecurity: apiKeySecurity, // 保存 API Key 安全配置。 - rateLimit: rateLimit, // 保存限流配置。 - responseCache: responseCache, // 保存缓存配置。 + authCache: authCache, + rateLimit: rateLimit, // 保存限流配置。 + responseCache: responseCache, // 保存缓存配置。 usage: usageConfig, observability: observabilityConfig, controlPlane: controlPlaneConfig, diff --git a/cmd/model-velo/main_test.go b/cmd/model-velo/main_test.go index 98a7753..ea8a10f 100644 --- a/cmd/model-velo/main_test.go +++ b/cmd/model-velo/main_test.go @@ -235,6 +235,12 @@ func setValidInfrastructureEnv(t *testing.T) { t.Setenv("MODEL_VELO_REDIS_POOL_TIMEOUT", "") t.Setenv("MODEL_VELO_REDIS_STARTUP_POLICY", "") t.Setenv("MODEL_VELO_ENVIRONMENT", "test") + t.Setenv("MODEL_VELO_AUTH_CACHE_ENABLED", "") + t.Setenv("MODEL_VELO_AUTH_CACHE_L1_MAX_ENTRIES", "") + t.Setenv("MODEL_VELO_AUTH_CACHE_L1_TTL", "") + t.Setenv("MODEL_VELO_AUTH_CACHE_L2_TTL", "") + t.Setenv("MODEL_VELO_AUTH_CACHE_KEY_PREFIX", "") + t.Setenv("MODEL_VELO_AUTH_CACHE_INVALIDATION_CHANNEL", "") t.Setenv("MODEL_VELO_RATE_LIMIT_REQUESTS", "") t.Setenv("MODEL_VELO_RATE_LIMIT_WINDOW", "") t.Setenv("MODEL_VELO_RATE_LIMIT_FAILURE_POLICY", "") @@ -266,7 +272,11 @@ func (mainTestAccessController) Authenticate(context.Context, string) (apikey.Id return apikey.Identity{TenantID: "tenant-test-id"}, nil } -func (mainTestAccessController) AuthorizeModel(context.Context, string, string) error { +func (mainTestAccessController) AuthorizeModel( + context.Context, + apikey.Identity, + string, +) error { return nil } diff --git a/docs/detailed-reference.md b/docs/detailed-reference.md new file mode 100644 index 0000000..9a56fc6 --- /dev/null +++ b/docs/detailed-reference.md @@ -0,0 +1,606 @@ +# Model-Velo 详细参考 + +> 本文档保留完整环境变量、协议契约与实现细节。第一次了解或运行项目,请先阅读仓库根目录的 [README](../README.md)。 + +Model-Velo 是一个用 Go 和 Gin 编写的多协议 LLM 网关。当前已实现 Chat Completions、Responses、Embeddings、Anthropic Messages 入站协议,多 Provider 原生转换与可靠性运行时,持久化 Usage/计费、在线控制平面、额度预算,以及可观测性和工程化代码。阶段总门禁是否完成,仍以各阶段 Gate 文件中的实际证据为准。 + +## 当前已经实现 + +- `GET /healthz` 健康检查; +- `GET /readyz` PostgreSQL/Redis 就绪检查,以及可选 Bearer 保护的 `GET /metrics`; +- `POST /v1/chat/completions` 非流式请求校验和转发; +- `POST /v1/responses`、`POST /v1/embeddings`、`POST /v1/messages` 和 `GET /v1/models`; +- 16 个内置厂商 Adapter:厂商身份与构造入口彼此独立;公开采用 OpenAI Chat 报文的厂商只复用协议编解码和 HTTP 边界; +- 一个厂商可配置多个 Provider 实例,每个实例可声明多个文本或视觉模型; +- request ID 生成、校验、响应回传和上游传播; +- 请求体与上游响应体大小限制; +- 上游超时、网络失败、HTTP 错误和非法响应的结构化错误; +- 操作系统退出信号和有界优雅关闭; +- 使用 `httptest.Server` 的本地测试,不调用真实付费 API; +- 固定版本的 PostgreSQL、Redis Compose 配置; +- 基于 GORM 的 PostgreSQL 连接、启动 Ping、连接池配置和退出关闭; +- 启动时通过 GORM `AutoMigrate` 同步租户/API Key、Usage/outbox、控制面/审计和额度账本; +- Model-Velo API Key 随机生成、摘要查找、HMAC 校验、过期判断、禁用和吊销; +- `model-velo-admin` 本地管理命令,可初始化租户、模型授权和首个 API Key; +- Gin Bearer 认证中间件、请求身份 Context 和租户模型授权检查; +- 官方 `go-redis/v9` Client、显式连接池、启动 Ping、可选启动降级和退出关闭; +- 基于 Redis Lua 的租户+模型固定窗口限流,以及可配置的 fail-open/fail-closed 运行时故障策略; +- 租户隔离的 Redis Exact Response Cache,支持规范化请求哈希、TTL、显式绕过和缓存故障降级; +- 精确模型/默认模型路由、有序候选去重、primary 选择和上游模型映射; +- 按模型声明 `text`、`image`、`audio`、`file`、`tools`、`structured` 能力,规划时先过滤协议或模型无法承载的候选; +- Provider Circuit Breaker 三态、指定故障计数、Open 快速拒绝和 HalfOpen 有界探测; +- 按 Provider 隔离的进程内有界 Queue,限制运行数和等待数,并传播请求取消; +- Provider 多 Key 安全身份与并发轮换:401 永久禁用错误 Key,403 只在当前请求内换 Key,429 按 `Retry-After` 临时冷却; +- 按 Provider ID 隔离的 Adapter、Circuit Breaker、Queue 和 Key Registry; +- 单候选 Attempt Executor:每个候选独立执行 Breaker、Queue、Key、有限 Retry 和上游调用; +- API Key Adapter 在装配时必须具备 Key Registry,错误装配会在启动边界失败而不是在请求期空指针崩溃; +- 有序 Fallback Orchestrator:成功立即停止,普通 400/取消停止,模型不可用等策略允许的失败进入下一候选;Fallback 成功响应不写 Exact Cache; +- Provider 执行总预算、单次调用超时和 Context-aware 退避取消; +- 每个 Provider 可独立覆盖 Breaker、Queue、Retry、Attempt Timeout 与 HTTP 连接池; +- Chat 文本、图片、音频、文件、Function Tool 与结构化输出:兼容协议保留原报文,原生协议只转换能够明确表达的字段并归一化响应; +- 流式预提交可靠性链:每次按 Breaker→Queue→Key→Adapter 建流并验证首事件,候选内支持有限 Retry,耗尽后按 Route Plan 有序 Fallback;最终 PreparedStream 持有成功流资源和完整安全 Trail,直到显式结束。 +- OpenAI-compatible 客户端 SSE:有效首事件后才提交 Header,逐事件同步 Write/Flush,正常转发 `[DONE]`,客户端断开会取消上游并释放 Queue。 +- Usage Event schema v2:记录 request、tenant、API Key ID、请求与实际模型、缓存、可靠性计数、详细 token、usage 来源、finish reason、TTFT、稳定终态和 UTC 延迟;原始 usage 子对象最多保留 64 KiB,不记录 Key Secret、提示词或完整上游响应; +- OpenAI-compatible 流请求默认合并 `stream_options.include_usage=true`;Provider 返回的缓存读写、音频、图像、推理和预测 token 会进入统一明细; +- 版本化价目表按 Provider、模型和事件时间生成不可变成本快照;Provider 明确上报的 USD 成本优先,缓存命中成本为已知零,无法定价时成本保持 `NULL` 而不是伪造零; +- API 在 Provider 前同步写 PostgreSQL pending 生命周期,终态先固化到 outbox,再以有界超时执行 Redis `XADD`;即时投递失败不会改写已经生成的模型响应; +- 独立 `model-velo-usage-worker` 使用 consumer group、`XREADGROUP`、`XAUTOCLAIM`、dead-letter 和 Context-aware 退避; +- PostgreSQL `usage_events.event_id` 主键与 `ON CONFLICT DO NOTHING` 提供幂等最终防线,数据库成功后才在 Redis 事务中执行 `XACK + XDEL`; +- 认证后的租户可查询 Usage 明细、汇总和时间序列;Worker 自动执行分批保留期清理,管理命令支持历史成本重算。 +- 独立管理员身份、owner/operator/billing/auditor RBAC、脱敏审计日志,以及 Provider/路由/Provider Key/价格、租户、业务 API Key 和额度策略的在线管理; +- PostgreSQL 强一致额度账本:分钟/小时/日/月请求、Token、USD 预算,支持 deny/allow/alert 超额策略、请求前预留、真实 Usage 结算和中断恢复; +- JSON 结构化日志、Prometheus、OpenTelemetry、非 root 容器、GitHub Actions race/集成门禁与可复现 benchmark。 + +Usage 链路采用 **PostgreSQL outbox + Redis Stream at-least-once + 数据库幂等**,不宣称 exactly-once。API 在调用 Provider 前先写 pending 生命周期,结束时把完整 Event 固化为 ready;Redis 不可用不会丢失已固化事件,Worker 会从 outbox 重投。已标记 published 但尚未被 Usage 入库事务删除的记录也会周期性重发,覆盖 Redis 消息在消费前丢失或清理的窗口。数据库成功但 Redis ACK 响应丢失时,重投会命中 `usage_events.event_id` 唯一键。Worker 消失前未 ACK 的事件由 `XAUTOCLAIM` 恢复,坏版本/坏载荷达到阈值后进入有长度上限的 dead-letter Stream。进程在最终 Usage 形成前退出时,超时 pending 会转成明确的中断事件并保留“Usage 未知” caveat,而不是伪造 Token。 + +客户端始终收到 OpenAI SSE。Adapter 会按上游协议校验并转换 OpenAI-compatible SSE、Anthropic/Gemini/DashScope/Cohere SSE、Ollama NDJSON 或 Bedrock AWS EventStream;单行最大 1 MiB、单事件最大 2 MiB,Bedrock 二进制帧也在解码前限长。首事件前的 5xx、错误媒体类型、超时、EOF、坏 Chunk 和取消会沿用可靠性分类,可 Retry/Fallback,并在下一次尝试前释放资源;预提交总预算不会成为成功长流的上游 deadline。首事件提交后禁止切换 Provider,后续失败只安全记录并结束当前流。上游 heartbeat 当前不向客户端透传,但会重置事件空闲计时器;转换链使用同步背压,不创建无界 Chunk 队列。首事件验证成功后会清除当前 SSE 响应继承的 Server 总写截止时间,每个客户端帧仍有独立 15 秒 Write/Flush 截止时间,后续事件静默上限复用该 Provider 的 `attempt_timeout`。 + +阶段 3 的生产功能、合并故障矩阵、全量测试、vet 和独立性复查已经完成;Breaker、Queue、Key 并发用例也已通过普通执行,但 race detector 仍被本机 Go/race 工具链阻止,因此不能宣称 race 已通过。详细证据见 `STAGE3_GATE.md`。 + +阶段 5 的 Usage v2 生产链和真实 Redis/PostgreSQL 集成门禁已经通过;当前 PATH 没有 GCC,`go test -race` 在 `runtime/cgo` 编译前失败,因此阶段 5 race 与最终门禁仍保留,不能宣称 race 已通过。详细证据见 `STAGE5_GATE.md`。 + +当前请求会按 Route Plan 顺序执行候选;每个候选内部在策略允许时进行有限 Retry,耗尽后只有具备 Fallback 信号的失败才进入下一候选。合法请求所需的能力不被当前 Provider 支持时会直接尝试下一候选,不消耗 Retry,也不计入 Breaker;所有候选都不支持时返回 `400 unsupported_provider_capability`。`POST /v1/chat/completions` 必须携带有效的 Model-Velo API Key,请求模型必须存在于该租户的模型授权表,并依次通过租户限流和 Route Plan;缓存未命中后,每次 Provider 调用都会重新取得目标 Provider 的 Breaker Permit、Queue 槽位和可用 Provider Key。`GET /healthz` 保持公开。 + +## 当前 Chat 契约 + +请求必须使用 `Content-Type: application/json`,当前对以下字段做出保证: + +| 字段 | 当前行为 | +|---|---| +| `model` | 必填且不能是空白字符串;用于授权和 Route Plan,可通过 `upstream_model` 映射成厂商模型名。 | +| `messages` | 必填且至少包含一条消息。 | +| `messages[].role` | 接受 `system`、`developer`、`user`、`assistant`、`tool`;`tool`、assistant `tool_calls` 和 Tool 定义会要求目标模型声明 `tools`。 | +| `messages[].content` | 接受非空字符串或非空内容块数组;已建模 `text`/`input_text`、`image_url`、`input_audio` 和 `file`。未知块不会穿过原生转换器。 | +| `messages[].tool_calls` / `tool_call_id` | 校验 Function 名称、唯一调用 ID、JSON object 参数和历史引用;原生 Adapter 映射 Tool Use/Result,响应统一返回 OpenAI `tool_calls`。 | +| `tools` / `tool_choice` / `parallel_tool_calls` | 支持 Function Tool;厂商没有等价控制项时明确返回能力错误,不会删除字段继续请求。 | +| `response_format` | 支持 `text`、`json_object`、`json_schema`;需要模型声明 `structured`。DashScope 原生接口只接受 `json_object`,Cohere 不允许与 `tools` 组合。 | +| 生成参数 | 建模并校验 token 上限、`temperature`、`top_p`、`stop`、`seed`、penalty、`n`、logprobs 与 `reasoning_effort`;原生协议仅映射等价字段,其余明确拒绝。 | +| `stream` | 省略或为 `false` 时返回完整 JSON;`true` 时绕过 Exact Cache,并以 `text/event-stream` 逐事件返回兼容 Chunk 与 `[DONE]`。 | + +OpenAI、Mistral、DeepSeek、xAI、Zhipu、Groq、NVIDIA、Together 和 Cloudflare 均由各自的厂商装配入口设置端点与能力边界;它们公开采用 OpenAI Chat 报文,因此复用同一套 wire codec 和 HTTP 安全边界,只在路由配置要求时改写 `model`。Anthropic、Gemini、DashScope、Cohere、Ollama 和 Bedrock 使用各自原生消息、Tool、结构化输出、Usage 和流式事件格式。 + +请求 JSON 只解析一次。兼容协议会保留未知顶层字段和消息字段;原生协议在转换前检查这些字段,无法无损表达时返回能力不匹配并尝试下一候选,不会静默丢字段。 + +各协议在当前 Adapter 中可表达的上限如下;模型还必须在 `model_capabilities` 中显式声明对应能力: + +| 协议 | image | audio | file | tools | structured | stream | +|---|---:|---:|---:|---:|---:|---:| +| OpenAI / Azure / custom OpenAI-compatible | 是 | 是 | 是 | 是 | 是 | 是 | +| Anthropic Messages | 是 | 否 | 是 | 是 | 是 | 是 | +| Gemini generateContent | 是 | 是 | 是(内嵌数据) | 是 | 是 | 是 | +| DashScope Generation | 否 | 否 | 否 | 是 | 仅 `json_object` | 是 | +| Cohere v2 Chat | 是 | 否 | 否 | 是 | 是 | 是 | +| Ollama Chat | 是 | 否 | 否 | 是 | 是 | 是 | +| Bedrock Converse | 是 | 否 | 是(内嵌数据) | 是 | 是 | 是 | +| Cloudflare Workers AI Chat | 是 | 是 | 是 | 是 | 是 | 是 | + +这张表描述线协议转换能力,不承诺每个厂商模型都具备该能力。Anthropic 与 Cohere 可把远程图片 URL 交给上游;Gemini、Ollama 和 Bedrock 的当前转换器要求图片为 Base64 data URL。显式 `detail=low/high` 只在能保留该语义的 Cohere/OpenAI wire 上发送,其他原生协议会拒绝而不是忽略。Gemini、Bedrock 的 OpenAI `file_id` 没有安全等价物,当前只接受内嵌 `file_data`;Anthropic 同时接受 Files API `file_id` 和内嵌文件,使用 `file_id` 时 Adapter 会自动发送 Files API Beta Header。视频和厂商私有内容块不在 Chat 契约内。 + +成功时,兼容协议响应在通过 2xx、Content-Type、大小、错误信封和非空 Chat `choices[].message` 检查后原样返回;原生协议响应会转换为非流式 OpenAI Chat Completion。原生响应缺少 Usage 时省略 `usage`,不会伪造全零计费数据;出现当前无法表示的非文本输出时明确失败并按策略 Fallback。 + +## Usage 查询与成本 + +以下接口位于认证后的 `/v1` 路由组,并同时强制 tenant ID 与当前 API Key ID,普通模型 Key 不能读取同租户其他 Key 的账单或审计数据: + +- `GET /v1/usage/events`:游标分页明细,支持 `start`、`end`、`model`、`provider`、`api_key_id`、`request_id`、`status`、`cache_status`、`stream`、`limit` 和 `include_raw`; +- `GET /v1/usage/summary`:请求、成功/失败、缓存、token、已知/未知成本、延迟、TTFT 和重试/Fallback 汇总,`group_by` 支持 `model`、`provider`、`status`、`cache`、`api_key`; +- `GET /v1/usage/series`:按 `hour`、`day`、`week`、`month` 或 `year` 返回时间序列,并接受 IANA timezone。 + +`api_key_id` 省略时自动使用当前 Key,显式值也只能等于当前 Key。默认查询最近 30 天,单次范围最多 366 天,明细每页最多 200 条,分组最多返回 1000 组并显式标记截断。所有响应带 `Cache-Control: no-store`。成本以整数 nanoUSD 存储与聚合,接口同时返回精确十进制 USD 字符串;未知价格、缺失 usage 或无法覆盖早期失败 attempt 时会保留 caveat。 + +历史 schema v1 没有 API Key ID,Worker 仍能可靠消费和存储,但它不能被安全归属到某一把 Key,因此不会出现在普通 Key 的 HTTP 查询中;可通过 PostgreSQL 管理通道或带 tenant 条件的 `reprice-usage` 处理。系统不会为了补齐归属而猜测或把 v1 数据暴露给整个租户。 + +## 管理 API + +`/admin/v1` 只接受独立的 `mv_admin_...` Bearer Key,所有响应都带 `Cache-Control: no-store`。运行时、价格、管理员、租户/API Key 与额度变更和审计写入处于同一数据库事务;创建返回的管理员 Key 或业务 API Key 明文只出现一次。 + +- `GET/PUT /admin/v1/runtime`:版本化 Provider、路由、上游 Key 与可靠性参数;写入必须携带 `If-Match`; +- `GET/PUT /admin/v1/pricing`:版本化价目; +- `GET /admin/v1/principals`、`POST /admin/v1/principals`、`PATCH /admin/v1/principals/:id`:管理员与角色; +- `GET/POST/PUT /admin/v1/quotas`、`GET /admin/v1/quota-windows`:额度策略和当前已结算/预留窗口; +- `GET/POST /admin/v1/tenants`、`PUT /admin/v1/tenants/:id`、`GET/POST /admin/v1/tenants/:id/keys`、`PATCH /admin/v1/api-keys/:id`:租户、模型授权及业务 Key 生命周期; +- `GET /admin/v1/usage/events`、`GET /admin/v1/usage/summary`、`GET /admin/v1/usage/series`:需要 `usage:read` 的跨租户只读 Usage 查询,可按 `tenant_id` 与 `api_key_id` 下钻,汇总额外支持 `group_by=tenant`;明细不开放原始 Usage JSON; +- `GET /admin/v1/audit`:只读游标分页审计。 + +## 配置 + +| 环境变量 | 必填 | 默认值 | 用途 | +|---|---:|---|---| +| `MODEL_VELO_HTTP_ADDR` | 否 | `:8080` | Model-Velo HTTP 监听地址。 | +| `MODEL_VELO_ENVIRONMENT` | API 必填 | 无 | 1–32 位小写环境标识,用于隔离 Redis Key,例如 `development`、`staging`。 | +| `MODEL_VELO_PROVIDER_KEYS_JSON` | 条件必填 | 无 | 按需要鉴权的 Provider ID 配置一个或多个 Key;无鉴权的 Ollama 不配置 Key 集合。Secret 只用于上游鉴权,不进入快照、错误或日志。 | +| `MODEL_VELO_SHUTDOWN_TIMEOUT` | 否 | `10s` | 收到退出信号后等待活动请求结束的期限。 | +| `MODEL_VELO_POSTGRES_DB` | 否 | `model_velo` | Compose 创建的本地数据库名。 | +| `MODEL_VELO_POSTGRES_USER` | 否 | `model_velo` | Compose 创建的本地数据库用户。 | +| `MODEL_VELO_POSTGRES_PASSWORD` | Compose 必填 | 无 | 本地 PostgreSQL 密码,必须在 `.env` 中替换示例值。 | +| `MODEL_VELO_POSTGRES_PORT` | 否 | `5432` | PostgreSQL 映射到本机回环地址的端口。 | +| `MODEL_VELO_POSTGRES_DSN` | API 必填 | 无 | PostgreSQL URL;必须包含 `postgres/postgresql` scheme、用户、Host 和数据库名。 | +| `MODEL_VELO_POSTGRES_MAX_OPEN_CONNS` | 否 | `10` | `database/sql` 连接池最大打开连接数。 | +| `MODEL_VELO_POSTGRES_MAX_IDLE_CONNS` | 否 | `2` | `database/sql` 连接池最大空闲连接数,不能超过最大打开连接数。 | +| `MODEL_VELO_POSTGRES_CONNECT_TIMEOUT` | 否 | `5s` | PostgreSQL 启动连接检查期限。 | +| `MODEL_VELO_POSTGRES_MAX_CONN_LIFETIME` | 否 | `30m` | 单个 PostgreSQL 连接的最长寿命。 | +| `MODEL_VELO_POSTGRES_MAX_CONN_IDLE_TIME` | 否 | `5m` | PostgreSQL 连接的最大空闲时间。 | +| `MODEL_VELO_API_KEY_PEPPER` | API 与管理命令必填 | 无 | 至少 32 字节的服务端秘密,用于 HMAC 校验 Model-Velo API Key;更换后已有 Key 将全部失效。 | +| `MODEL_VELO_ADMIN_KEY_PEPPER` | API 与管理员初始化必填 | 无 | 与业务 Key 分离的至少 32 字节秘密。 | +| `MODEL_VELO_CONTROL_MASTER_KEY` | API 必填 | 无 | Base64 编码的 32 字节 AES-256-GCM 密钥,用于加密托管 Provider Key。 | +| `MODEL_VELO_CONTROL_REFRESH_INTERVAL` | 否 | `5s` | 多 API 实例同步活动运行时与价格版本的间隔。 | +| `MODEL_VELO_REDIS_ADDR` | API 必填 | 无 | Redis `host:port` 地址。 | +| `MODEL_VELO_REDIS_PASSWORD` | API 与 Compose 必填 | 无 | Redis 密码,错误信息和日志不得包含该值。 | +| `MODEL_VELO_REDIS_DB` | 否 | `0` | Go 应用使用的非负 Redis logical DB。 | +| `MODEL_VELO_REDIS_PORT` | 否 | `6379` | Redis 映射到本机回环地址的端口。 | +| `MODEL_VELO_REDIS_DIAL_TIMEOUT` | 否 | `5s` | Redis 建连期限。 | +| `MODEL_VELO_REDIS_READ_TIMEOUT` | 否 | `2s` | Redis 读取期限。 | +| `MODEL_VELO_REDIS_WRITE_TIMEOUT` | 否 | `2s` | Redis 写入期限。 | +| `MODEL_VELO_REDIS_POOL_SIZE` | 否 | `20` | Redis 连接池最大连接数。 | +| `MODEL_VELO_REDIS_MIN_IDLE_CONNS` | 否 | `2` | Redis 连接池预留的最小空闲连接数,不能超过池容量。 | +| `MODEL_VELO_REDIS_POOL_TIMEOUT` | 否 | `2s` | 连接池耗尽时等待可用连接的期限。 | +| `MODEL_VELO_REDIS_STARTUP_POLICY` | 否 | `required` | `required` 表示启动 Ping 失败终止;`optional` 表示记录警告后继续。它与限流运行时故障策略相互独立。 | +| `MODEL_VELO_RATE_LIMIT_REQUESTS` | 否 | `60` | 每个租户+模型在一个窗口内可接受的请求数,范围 1–1,000,000。 | +| `MODEL_VELO_RATE_LIMIT_WINDOW` | 否 | `1m` | 固定窗口时长,范围 `1s`–`24h`;窗口从该 Key 的首个请求开始。 | +| `MODEL_VELO_RATE_LIMIT_FAILURE_POLICY` | 否 | `fail-closed` | Redis 运行时失败时,`fail-closed` 返回 503;`fail-open` 标记绕过并继续 Provider。 | +| `MODEL_VELO_CACHE_TTL` | 否 | `5m` | Exact Cache 保存时间,范围 `1s`–`24h`;`0` 或 `off` 禁用缓存。 | +| `MODEL_VELO_CACHE_ROUTE_VERSION` | 否 | `routes-v1` | 环境变量启动路由的缓存命名空间;托管运行时切换会自动使用版本化命名空间。 | +| `MODEL_VELO_USAGE_EMIT_TIMEOUT` | 否 | `200ms` | API 在请求结束后投递 Usage Event 的独立短超时。 | +| `MODEL_VELO_USAGE_GROUP` | 否 | `model-velo-usage-workers` | Usage Worker consumer group。 | +| `MODEL_VELO_USAGE_CONSUMER` | 否 | `-` | 当前 Worker consumer 名;同组并发进程应不同。 | +| `MODEL_VELO_USAGE_BATCH_SIZE` | 否 | `50` | 每次读取或认领的最大消息数。 | +| `MODEL_VELO_USAGE_READ_BLOCK` | 否 | `2s` | 空 Stream 上 `XREADGROUP` 的阻塞时间。 | +| `MODEL_VELO_USAGE_CLAIM_IDLE` | 否 | `30s` | pending 消息允许被其他 Worker 认领前的空闲时间。 | +| `MODEL_VELO_USAGE_MAX_DELIVERIES` | 否 | `5` | 坏版本/坏载荷进入 dead-letter 前的最大投递次数。 | +| `MODEL_VELO_USAGE_RETRY_BACKOFF` | 否 | `500ms` | Redis 读取/认领失败后的 Context-aware 退避。 | +| `MODEL_VELO_USAGE_WORKER_TIMEOUT` | 否 | `10s` | 单批写库与关闭收尾共享的最大处理时间。 | +| `MODEL_VELO_USAGE_DEAD_LETTER_MAX_LEN` | 否 | `100000` | dead-letter Stream 的近似长度上限;旧 `MODEL_VELO_USAGE_STREAM_MAX_LEN` 仍兼容,但不能与新变量同时设置。 | +| `MODEL_VELO_USAGE_ENFORCE_STREAM` | 否 | `true` | 对 OpenAI-compatible 流请求强制合并 `stream_options.include_usage=true`;仅在确认自定义上游不兼容时关闭。 | +| `MODEL_VELO_USAGE_RETENTION_DAYS` | 否 | `90` | PostgreSQL Usage 保留天数,范围 0–3650;`0` 禁用自动清理。 | +| `MODEL_VELO_USAGE_MAINTENANCE_INTERVAL` | 否 | `1h` | Worker 执行保留期清理的间隔。 | +| `MODEL_VELO_USAGE_MAINTENANCE_BATCH_SIZE` | 否 | `1000` | 单次删除批量,Worker 会在维护超时内持续分批清理。 | +| `MODEL_VELO_USAGE_PRICING_JSON` | 否 | `[]` | 版本化 USD/百万 token 价目表;支持生效时间、缓存、音频、图像和推理 token 专属费率,最大 256 KiB。 | +| `MODEL_VELO_USAGE_PRICING_REFRESH_INTERVAL` | 否 | `30s` | 独立 Worker 刷新托管价格目录的间隔。 | +| `MODEL_VELO_USAGE_PENDING_TIMEOUT` | 否 | `15m` | 把未完成 outbox 生命周期恢复成中断事件前的保守等待时间,范围 5m–24h。 | +| `MODEL_VELO_QUOTA_RESERVATION_TTL` | 否 | `15m` | 崩溃后活动额度预留转为保守估算结算的时间。 | +| `MODEL_VELO_QUOTA_REAP_INTERVAL` | 否 | `1m` | 扫描过期额度预留的间隔。 | +| `MODEL_VELO_QUOTA_DEFAULT_MAX_OUTPUT_TOKENS` | 否 | `4096` | 请求未给输出上限时用于 Token/成本预留的默认值。 | +| `MODEL_VELO_LOG_FORMAT` | 否 | `json` | `json` 或本地开发用 `text`。 | +| `MODEL_VELO_LOG_LEVEL` | 否 | `info` | `debug`、`info`、`warn` 或 `error`。 | +| `MODEL_VELO_SERVICE_NAME` | 否 | `model-velo` | 日志与 OpenTelemetry service name。 | +| `MODEL_VELO_METRICS_TOKEN` | 否 | 无 | 设置后 API 与 Worker `/metrics` 要求至少 32 字节的 Bearer Token。 | +| `MODEL_VELO_WORKER_METRICS_ADDR` | 否 | `:9091` | Usage Worker 的 `/healthz`、`/readyz`、`/metrics` 监听地址。 | +| `MODEL_VELO_OTEL_EXPORTER_OTLP_ENDPOINT` | 否 | 无 | OTLP/HTTP Trace Collector 绝对 URL;空值禁用导出。 | +| `MODEL_VELO_OTEL_EXPORTER_OTLP_INSECURE` | 否 | `false` | 是否允许明文 OTLP/HTTP;`http` URL 也会启用。 | +| `MODEL_VELO_OTEL_SAMPLE_RATIO` | 否 | `0.1` | Parent-based Trace 采样比例,范围 0–1。 | +| `MODEL_VELO_READINESS_TIMEOUT` | 否 | `1s` | 单次依赖就绪检查的总期限。 | +| `MODEL_VELO_ROUTING_JSON` | API 必填 | 无 | 显式定义任意数量的 Provider、厂商预设、模型能力、Provider 级运行参数、精确/默认路由和有序候选;只配置一个 Provider 时也不能省略。 | +| `MODEL_VELO_BREAKER_FAILURE_THRESHOLD` | 否 | `5` | 连续可计数失败达到此值后 Open,范围 1–1000。 | +| `MODEL_VELO_BREAKER_OPEN_DURATION` | 否 | `30s` | Open 冷却时间,范围 `1s`–`10m`。 | +| `MODEL_VELO_BREAKER_HALF_OPEN_PROBES` | 否 | `1` | HalfOpen 同时允许的探测数,以及重新关闭所需成功数,范围 1–100。 | +| `MODEL_VELO_QUEUE_MAX_IN_FLIGHT` | 否 | `20` | 每个 Provider、每个网关进程允许的同时执行数,范围 1–10,000。 | +| `MODEL_VELO_QUEUE_MAX_WAITING` | 否 | `100` | 每个 Provider、每个网关进程允许的等待数;`0` 表示满载时立即拒绝,范围 0–100,000。 | +| `MODEL_VELO_QUEUE_WAIT_TIMEOUT` | 否 | `2s` | 等待 Provider 槽位的最长期限,范围 `10ms`–`1m`,同时受请求 Context 更早截止时间约束。 | +| `MODEL_VELO_RETRY_MAX_ATTEMPTS` | 否 | `3` | 每个候选最多调用次数,范围 1–10;包含第一次调用。 | +| `MODEL_VELO_RETRY_INITIAL_BACKOFF` | 否 | `100ms` | 首次可退避重试的基础等待,范围 `10ms`–`30s`。 | +| `MODEL_VELO_RETRY_MAX_BACKOFF` | 否 | `2s` | 指数退避上限,不得小于初始等待且不超过 `30s`。 | +| `MODEL_VELO_RETRY_BACKOFF_MULTIPLIER` | 否 | `2` | 退避倍数,范围 1–10。 | +| `MODEL_VELO_RETRY_JITTER_RATIO` | 否 | `0.2` | 退避随机抖动比例,范围 0–1。 | +| `MODEL_VELO_REQUEST_TIMEOUT` | 否 | `45s` | Cache miss 后 Retry 与全部 Fallback 候选共享的执行预算,范围 `1s`–`5m`。 | +| `MODEL_VELO_ATTEMPT_TIMEOUT` | 否 | `20s` | 非流式单次调用或流式建连+首事件等待期限,至少 `100ms` 且不得超过 Provider 执行总预算。 | + +价目使用十进制字符串,避免浮点配置误差;同一个 Provider/模型的生效时间窗口不能重叠。`provider` 或 `model` 可用 `*` 作为启动时已知的回退价格: + +```powershell +$env:MODEL_VELO_USAGE_PRICING_JSON = '[{"provider":"openai-main","model":"gpt-4o-mini","version":"openai-2026-07","effective_from":"2026-07-01T00:00:00Z","input_usd_per_million":"0.15","output_usd_per_million":"0.60","cached_read_usd_per_million":"0.075"}]' +``` + +上游 `http.Client` 不再维护第三套总超时。非流式调用服从 Attempt Context 和父级请求预算;流式调用在首事件前同时受 Attempt Timeout 与父 Context 约束,首事件验证通过后不再沿用短 Attempt deadline,但后续两个有效事件之间仍复用当前 Provider 的 Attempt Timeout 作为静默上限,上游 heartbeat 会重置计时器。HTTP Server 写超时固定覆盖托管运行时允许的最长 5 分钟请求预算并额外保留 15 秒收尾时间;每个非流式请求仍由自身快照中的较短总预算取消。流式 Handler 在首事件验证后清除总写截止时间,并为每个客户端帧设置独立 15 秒写截止时间。 + +### Provider 厂商预设 + +`vendor` 负责选择厂商目录和默认 API Base,`type` 必须显式声明协议;二者不匹配时启动失败。一个厂商可以配置多个 Provider ID,一个 Provider 的 `models` 可以列出多个文本、推理或 VLM 模型;`base_url` 可覆盖区域、私有部署或账号级端点。模型清单由配置显式声明,代码不硬编码容易过期的型号。 + +| `vendor` | 必须声明的 `type` | 默认 API Base | +|---|---|---| +| `openai` | `openai` | `https://api.openai.com/v1` | +| `anthropic` | `anthropic` | `https://api.anthropic.com` | +| `google` | `gemini` | `https://generativelanguage.googleapis.com/v1beta` | +| `azure` | `azure-openai` | 必须配置资源 Endpoint | +| `alibaba` | `dashscope` | `https://dashscope.aliyuncs.com/api/v1` | +| `cohere` | `cohere` | `https://api.cohere.com/v2` | +| `ollama` | `ollama` | `http://localhost:11434` | +| `bedrock` | `bedrock` | 必须配置区域 Bedrock Runtime Endpoint | +| `cloudflare` | `cloudflare` | 必须配置含 account ID 的 API Base | +| `mistral` | `mistral` | `https://api.mistral.ai/v1` | +| `xai` | `xai` | `https://api.x.ai/v1` | +| `deepseek` | `deepseek` | `https://api.deepseek.com` | +| `zhipu` | `zhipu` | `https://open.bigmodel.cn/api/paas/v4` | +| `groq` | `groq` | `https://api.groq.com/openai/v1` | +| `nvidia` | `nvidia` | `https://integrate.api.nvidia.com/v1` | +| `together` | `together` | `https://api.together.ai/v1` | + +当前 Azure Adapter 使用 v1 Endpoint 的 `api-key` 鉴权;Bedrock Adapter 使用 Bedrock Runtime Converse 与 Bearer API Key,尚未接 IAM Role/SigV4;Cloudflare Adapter 使用 Workers AI 官方 `/ai/v1/chat/completions`。配置其他鉴权方式或模型专属输入 schema 时会明确报不支持,不会静默伪装成兼容成功。 + +Mistral、DeepSeek、xAI、Zhipu、Groq、NVIDIA 和 Together 都有独立的 Adapter 类型与文件,后续厂商专属字段、错误解析或能力规则只进入对应 Adapter。它们没有复制七份相同的请求发送代码:官方相同的 Bearer 鉴权、OpenAI Chat JSON 和响应校验由 `compatible.go` 复用。DeepSeek Adapter 使用其 API Base 下的 `/chat/completions`,不会套用自定义兼容端点默认补 `/v1` 的规则。 + +另有 `custom`,必须显式配置 `type: "openai-compatible"` 和 `base_url`。已提供兼容接口的 Google、Alibaba 等厂商也可以显式把 `type` 设成 `openai-compatible`;这是主动选择兼容协议。 + +`MODEL_VELO_ROUTING_JSON` 没有隐式默认值。缺失或只有空白时进程会在连接 PostgreSQL、Redis 和监听端口前启动失败;即使只接一个自定义兼容上游,也必须显式写出 Provider 和 Route。配置中的 `model: "*"` 是默认路由;`upstream_model` 留空表示透传客户端模型,设置值则执行模型别名映射。`candidates` 的数组顺序就是稳定优先级;Router 先按请求所需能力过滤候选,首个实际执行的候选失败且错误策略允许 Fallback 时,Orchestrator 才会执行下一候选。 + +`model_capabilities` 按上游模型声明 `text`、`image`、`audio`、`file`、`tools`、`structured`;未声明的模型安全地默认为 `text`。能力不能超过当前 Adapter 协议真正能承载的范围,否则启动失败。Provider 的可选 `runtime` 可分别覆盖 `breaker`、`queue`、`retry` 和 `http`:请求总预算仍是全链路唯一值,Provider 只覆盖单次 `attempt_timeout`;HTTP 每 Host 连接数默认跟随该 Provider 的 `queue.max_in_flight`,避免 Queue 放行后又在默认连接池里形成第二条隐形队列。 + +`runtime` 支持的字段是:`breaker.failure_threshold/open_duration/half_open_max_probes`、`queue.max_in_flight/max_waiting/wait_timeout`、`retry.max_attempts/initial_backoff/max_backoff/backoff_multiplier/jitter_ratio/attempt_timeout`,以及 `http.max_idle_connections/max_idle_connections_per_host/max_connections_per_host`。未写的字段继承上表环境变量默认值,所有组合都在启动时校验。 + +以下示例让同一个 Gemini Provider 承载文本和 VLM,并准备 DeepSeek、OpenAI 作为其他路由或后续 fallback 候选: + +```json +{ + "providers": [ + {"id":"gemini-main","vendor":"google","type":"gemini","models":["gemini-2.5-flash","gemini-2.5-pro"],"model_capabilities":{"gemini-2.5-flash":["text"],"gemini-2.5-pro":["text","image"]},"runtime":{"queue":{"max_in_flight":40,"max_waiting":200,"wait_timeout":"2s"},"retry":{"max_attempts":2,"attempt_timeout":"15s"},"http":{"max_idle_connections":80}}}, + {"id":"deepseek-main","vendor":"deepseek","type":"deepseek","models":["deepseek-chat","deepseek-reasoner"],"model_capabilities":{"deepseek-chat":["text","tools"],"deepseek-reasoner":["text"]}}, + {"id":"openai-main","vendor":"openai","type":"openai","models":["gpt-4o-mini","gpt-4o"],"model_capabilities":{"gpt-4o-mini":["text","tools"],"gpt-4o":["text","image","tools"]}} + ], + "routes": [ + {"model":"fast-chat","candidates":[{"provider":"gemini-main","upstream_model":"gemini-2.5-flash"},{"provider":"deepseek-main","upstream_model":"deepseek-chat"}]}, + {"model":"vision","candidates":[{"provider":"gemini-main","upstream_model":"gemini-2.5-pro"},{"provider":"openai-main","upstream_model":"gpt-4o"}]} + ] +} +``` + +多 Key 配置与路由配置分离,避免 Route Plan 携带凭证。对应的多 Provider Key 配置: + +```json +{"providers":[{"provider_id":"gemini-main","keys":[{"id":"primary","secret":"replace-with-gemini-key"}]},{"provider_id":"deepseek-main","keys":[{"id":"primary","secret":"replace-with-deepseek-key"},{"id":"secondary","secret":"replace-with-second-deepseek-key"}]},{"provider_id":"openai-main","keys":[{"id":"primary","secret":"replace-with-openai-key"}]}]} +``` + +需要鉴权的 Provider ID 必须与 Key 配置一一对应;每个此类 Provider 至少一个 Key,Key ID 必须唯一,Secret 不能为空。Ollama 不出现在 Key 配置中。配置错误会在连接 PostgreSQL/Redis 前阻止启动。模型名只是路由声明,不表示网关替厂商保证该账号、区域或模型已开通。 + +可以复制 `.env.example` 的变量名用于本地配置,但程序当前不会自动读取 `.env` 文件,需要在启动进程前把变量注入环境。 + +> 预发布改名说明:项目已统一改为 `Model-Velo`。旧环境变量不再读取,旧 API Key 前缀不再接受,旧 Redis namespace 不再命中。新 Compose 默认数据库和用户是 `model_velo`;已有 PostgreSQL 数据不会被删除,需要时可继续通过新的 `MODEL_VELO_POSTGRES_DSN` 显式指向原数据库。 + +## 启动 PostgreSQL 和 Redis + +需要先启动 Docker Desktop。首次使用时复制配置并替换两个本地密码: + +```powershell +Copy-Item .env.example .env +``` + +`MODEL_VELO_POSTGRES_PASSWORD` 必须与 `MODEL_VELO_POSTGRES_DSN` 中的密码保持一致。然后解析配置并启动基础设施: + +```powershell +docker compose config --quiet +docker compose up -d postgres redis +docker compose ps +``` + +`docker compose ps` 中两个服务都应显示为 `healthy`。排查启动问题: + +```powershell +docker compose logs postgres +docker compose logs redis +``` + +停止容器但保留 PostgreSQL 和 Redis 数据: + +```powershell +docker compose down +``` + +显式删除容器、网络和两个命名数据卷,恢复全新环境: + +```powershell +docker compose down --volumes +``` + +不要在仍需保留本地开发数据时执行带 `--volumes` 的命令。 + +真实 PostgreSQL 综合用例只读取显式的 `MODEL_VELO_POSTGRES_TEST_DSN`,不会回退到 API 使用的 DSN。测试用户需要拥有 `CREATE SCHEMA` 权限;每次运行创建随机的 `model_velo_it_*` schema,在其中重复执行 GORM `AutoMigrate`、API Key 生命周期和认证授权验证,结束后只定向删除该随机 schema: + +```powershell +$env:MODEL_VELO_POSTGRES_TEST_DSN = $env:MODEL_VELO_POSTGRES_DSN +go test -count=1 -run '^TestPostgresAPIKeyLifecycle$' -v ./internal/apikey +``` + +未设置测试 DSN 时用例会跳过,跳过不等于成功。2026-07-18 已使用无持久卷的 PostgreSQL 17.10 一次性容器真实通过:随机 schema 中两次 AutoMigrate、约束/索引检查和完整认证生命周期均成功,随后 schema 与容器已清理。 + +## Redis Client + +Model-Velo 固定使用 `github.com/redis/go-redis/v9 v9.21.0`。API 启动时按照配置创建自带连接池的 Redis Client,并在 `MODEL_VELO_REDIS_DIAL_TIMEOUT` 期限内执行 `PING`: + +- `required`:网络、认证或 Redis 服务异常会关闭 Client 并阻止 HTTP Server 启动; +- `optional`:启动失败会记录不含密码的警告并继续,Client 保留,Redis 恢复后可供后续命令使用; +- 根 Context 已取消时,无论策略是什么都停止启动; +- 进程退出时通过幂等 `Close` 释放连接池。 + +Chat 请求会在认证和模型授权成功后访问 Redis 执行限流,再执行 Exact Cache 读写。`optional` 只决定启动 Ping 失败时能否继续监听;运行中的限流故障由 `MODEL_VELO_RATE_LIMIT_FAILURE_POLICY` 决定,缓存故障固定为 fail-open。 + +## Redis 租户限流 + +当前配额模型是“每个租户、每个网关模型、每个固定窗口最多 N 个已接受请求”。Redis Key 的结构为: + +```text +model-velo:rate-limit:v1::tenant::model: +``` + +Key 包含环境命名空间,并分别对 tenant ID 和规范化模型名做 SHA-256;它不包含 Model-Velo API Key、Provider Key 或原始模型名。多个 API Key 只要属于同一租户,就共享该租户对应模型的配额。 + +限流器通过一个 Lua 脚本完成读取计数、首次写入、递增、TTL、Redis 服务端时间和决策返回。Redis 把脚本作为单个命令执行,执行期间不会插入其他客户端命令,因此多个 Model-Velo 实例不会在应用侧 `GET` 与 `SET` 之间同时放过超额请求。拒绝请求不会递增计数或延长窗口。 + +请求通过限流时响应包含 `X-RateLimit-Limit`、`X-RateLimit-Remaining` 和 Unix 秒格式的 `X-RateLimit-Reset`。额度耗尽时返回结构化 `429 rate_limit_exceeded`,并增加整数秒 `Retry-After`。Redis 运行时故障时: + +- `fail-closed`(默认):返回结构化 `503 rate_limit_unavailable`,不调用 Provider; +- `fail-open`:继续调用 Provider,并返回 `X-RateLimit-Status: bypassed`,但不伪造额度与重置值; +- 请求 Context 已取消时:无论故障策略如何都立即停止,不把取消转换为放行。 + +该算法的边界是固定窗口交界处可能出现突发流量;当前阶段选择它是因为配额语义直接、单脚本可解释。Redis 8.8.0 容器测试已经验证窗口恢复、租户/模型隔离和双 Client 竞争不超发;race 门禁因本机缺少 CGO/C 编译器暂未完成,仍是阶段 2 的最后环境缺口。 + +真实 Redis 用例通过 `MODEL_VELO_REDIS_TEST_ADDR`、`MODEL_VELO_REDIS_TEST_PASSWORD` 和 `MODEL_VELO_REDIS_TEST_DB` 显式启用,使用随机 namespace 定向清理,不执行 `FLUSHDB`。未配置时测试会跳过,跳过不等于验证通过。 + +## Redis Exact Response Cache + +缓存只处理已经通过请求校验、认证、模型授权和限流的 `stream=false` 请求。请求携带 `Cache-Control: no-store` 时显式绕过。缓存命中仍会消耗一次租户配额,这是当前明确选择的调用顺序,而不是“命中免费”。 + +Redis Key 为: + +```text +model-velo:response-cache:v1::tenant::model::route::request: +``` + +Key 不包含原始 API Key、tenant ID、模型名或提示词。规范化会递归排序 JSON 对象字段,并保留数组顺序、数字文本以及“字段缺失/显式给值”的区别;因此消息、工具和生成参数顺序不会被错误改写。包含重复对象字段的 JSON 直接绕过缓存,避免不同解析规则造成碰撞。 + +命中返回 Redis 中的完整合法 JSON;未命中只在首候选返回完整成功 JSON 后回填。Fallback 成功、上游错误、响应读取失败、客户端取消和显式绕过均不写缓存,避免临时降级结果覆盖正常路由语义。Redis 读写错误只记录 request ID 和安全错误,不记录 tenant、模型、Key 或提示词,并继续 Provider;响应通过 `X-Model-Velo-Cache: HIT|MISS|BYPASS` 表达本次状态,最终 Usage Event 也记录同一状态。 + +`MODEL_VELO_CACHE_ROUTE_VERSION` 只属于缓存命名空间,并以摘要进入缓存 Key。更换上游、模型映射或候选顺序时必须递增它,避免新路由命中旧响应。2026-07-18 真实 Redis 8.8.0 综合用例已验证跨请求命中、TTL、租户/参数隔离、错误不缓存和关闭 Client 后的故障降级;测试使用无持久卷的一次性容器并已自动清理。 + +当前调用顺序是 `认证 → 授权 → 限流 → Route Plan → Cache → Fallback Orchestrator → 当前候选 Attempt(Breaker → Queue → Key → Retry → Provider → 反馈/释放)→ 成功回填`。无路由返回 `503 route_unavailable`;Breaker Open 返回 `503 provider_circuit_open`;Queue 满或等待超时分别返回 `503 provider_queue_full`、`503 provider_queue_timeout`;没有可用 Key 返回 `503 provider_keys_exhausted`。本地拒绝不会调用当前 Provider;是否进入下一候选由统一 Fallback 策略决定。 + +## Provider Circuit Breaker + +Breaker 以稳定 Provider ID 隔离状态。Closed 累计网络错误、上游超时、500/502/503/504,以及上游返回 2xx 却不满足已声明响应协议的错误;成功会清零连续失败。401/403、429、普通 4xx、501 等非策略 5xx、网关主动限制的超大响应和客户端取消不会把整个 Provider 熔断。达到阈值后进入 Open,缓存未命中的新请求快速失败;冷却到期后进入 HalfOpen,只放行配置数量的探测,探测成功关闭,符合计数策略的失败重新 Open。 + +每次准入返回一次性 Permit,Provider 调用后必须反馈成功或分类后的 Failure;调用链额外使用 `defer Abandon()` 防止提前退出泄漏 HalfOpen 探测名额。401/403 与 429 不计 Provider 故障,而是反馈给当前 Key 的健康状态。 + +## Provider 有界 Queue + +Queue 是网关进程内的 Provider 容量保护,不是 Redis 租户配额。每个 Provider 拥有独立的带缓冲 channel:channel 容量是正在调用上游的硬上限;容量耗尽后,只有 `MAX_WAITING` 个请求可以等待,更多请求立即返回结构化 503。等待同时监听请求 Context 和 Queue Timer,因此客户端断开、请求总 deadline 或 Queue 等待超时都会停止占位。 + +成功取得槽位后会得到一次性 Lease,调用链用 `defer Release()` 配对归还;原子标记阻止重复释放。Queue Snapshot 只暴露 Provider ID、active、waiting、容量和拒绝/超时/取消计数,不包含 Provider Key、请求内容或错误 cause。Queue 拒绝不计入 Breaker 故障,因为它说明本实例容量饱和,不代表上游已经失败。 + +Queue Registry 按 Provider ID 隔离;primary 与每个 fallback 候选都会进入自己 Provider 的 Queue,不会复用上一个候选的 Lease。Queue 不是跨实例分布式信号:部署多个副本时,理论总并发上限约为“副本数 × `MAX_IN_FLIGHT`”,应结合上游账号配额设置每实例容量。 + +## Provider Key Selector + +Key Registry 按 Provider ID 隔离。选择器只把稳定 Key ID 暴露给快照和失败元数据,Secret 只在构造上游 Authorization 时读取。每次选择通过原子游标轮换起点,再在读锁下跳过禁用或冷却中的 Key。401 表示凭证无效,会永久禁用当前 Key,直到进程使用修正配置重启;403 可能来自模型或账号权限,只把该 Key 排除在当前请求之外;429 优先按上游 `Retry-After` 冷却,缺失或非法时使用 30 秒,最长限制为 24 小时。成功反馈可以清除临时冷却,但不会恢复 401 禁用状态。 + +401、403 或 429 反馈后,当前请求会释放 Queue/Breaker 资源并重新进入完整准入链,从其他可用 Key 继续。同一请求不会再次选择已返回 401/403 的 Key。所有可用 Key 都在冷却时,只有仍有 Retry 次数且最早恢复时间小于剩余请求预算才等待;等待期间可被客户端取消,且不占用 Queue 槽位或 Breaker Permit。等待不合算时立即 Fallback;没有候选可切换时返回 `503 provider_keys_exhausted`,并携带最早恢复时间对应的 `Retry-After`。指定 5xx、网络和上游超时则优先保持原 Key 并按指数退避重试。Key Selector 与 Retry 的完整并发/race 证据仍保留在阶段 3 门禁,不能据此宣称已经完成高并发验证。 + +## PostgreSQL 与 GORM + +Model-Velo 直接使用 GORM 连接 PostgreSQL。API 启动时先取得 GORM 底层的 `database/sql` 连接池,设置最大打开连接数、最大空闲连接数、连接寿命和空闲时间,再执行有界 `PingContext`。数据库不可达时 API 不会开始监听端口。官方 `gorm.io/driver/postgres` 内部依赖 pgx,因此模块图中仍会出现 `pgx // indirect`,但 Model-Velo 源码不直接导入或调用 pgx。 + +连接成功后,启动流程调用 GORM `AutoMigrate` 同步以下模型: + +1. `Tenant` → `tenants`; +2. `APIKey` → `api_keys`; +3. `TenantModelGrant` → `tenant_model_grants`。 + +当前阶段不再维护 `golang-migrate`、独立 Migration CLI 或手写 SQL 文件。`AutoMigrate` 用于创建缺失的表、列、索引、外键和检查约束,但不会负责删除废弃列,也没有 `down`/版本回滚命令。发生破坏性 schema 变更时,必须先备份数据并为该次变更单独设计显式升级方案,不能依赖启动时自动删改数据。 + +Schema 安全边界: + +- `tenants.slug` 唯一,租户状态只允许 `active/disabled`; +- `api_keys` 只保存 Key 前缀、查找摘要、不可逆哈希和哈希版本,不存在明文 Key 列; +- 租户存在 API Key 时禁止删除,避免凭证被级联误删; +- 删除没有 API Key 的租户时,其模型授权记录级联删除; +- 禁用租户不会删除 Key 或模型授权,后续认证查询必须同时检查租户状态; +- `tenant_model_grants` 使用 `(tenant_id, gateway_model)` 主键避免重复授权。 + +## Model-Velo API Key + +Key 格式固定为: + +```text +mvl_<16 字符公开定位符>_<43 字符随机密文> +``` + +定位符由 12 个随机字节编码得到,密文由 32 个随机字节编码得到,均使用无填充 Base64 URL 编码。数据库不会保存完整 Key: + +Base64URL 的合法字符本身包含 `_`,所以解析器按 16 字符 locator 和 43 字符 secret 的固定位置切分,不使用普通的下划线 `Split`;否则网关会偶发拒绝自己生成的合法 Key。 + +- `key_prefix` 保存 `mvl_<定位符>`,用于后台展示; +- `lookup_digest` 保存公开定位符前缀的 SHA-256,用于索引查询; +- `key_hash` 保存随机密文在服务端 Pepper 下的 HMAC-SHA-256; +- `hash_version` 当前为 `1`,为以后更换校验方案保留升级边界; +- 明文 Key 只在创建成功时输出一次,之后无法从数据库恢复。 + +先配置 PostgreSQL 和稳定的 Pepper。本地首次使用可以在 PowerShell 中生成随机 Pepper: + +```powershell +$pepperBytes = New-Object byte[] 48 +[System.Security.Cryptography.RandomNumberGenerator]::Fill($pepperBytes) +$env:MODEL_VELO_API_KEY_PEPPER = [Convert]::ToBase64String($pepperBytes) +``` + +同一个数据库必须长期使用同一个 Pepper。初始化租户、模型授权和首个 Key: + +```powershell +go run ./cmd/model-velo-admin bootstrap-tenant ` + --slug demo ` + --name "Demo Tenant" ` + --label "local development" ` + --models "gpt-4o-mini,gpt-4.1-mini" +``` + +命令成功后输出 `tenant_id`、`api_key_id`、公开前缀以及仅出现一次的 `api_key`。为已有租户创建第二个 Key: + +```powershell +go run ./cmd/model-velo-admin create-key ` + --tenant-id "replace-with-tenant-uuid" ` + --label "second key" ` + --expires-in 720h +``` + +禁用或永久吊销 Key: + +```powershell +go run ./cmd/model-velo-admin disable-key --id "replace-with-api-key-uuid" +go run ./cmd/model-velo-admin revoke-key --id "replace-with-api-key-uuid" +``` + +禁用和吊销会立即影响后续认证。永久吊销的 Key 不能通过 `disable-key` 降级回普通禁用状态。 + +## HTTP 认证与模型授权 + +`GET /healthz` 不需要 API Key。`POST /v1/chat/completions` 必须使用一个且只能使用一个 Authorization Header: + +```http +Authorization: Bearer mvl__ +``` + +网关按以下顺序处理请求: + +1. 严格解析 Bearer Header,拒绝缺失、重复、错误 scheme、空 Token 和额外空白; +2. 根据 locator 摘要查询 Key,并以 HMAC 常量时间比较验证 secret; +3. 检查 Key 状态、UTC 过期时间和租户状态; +4. 把 `tenant_id`、`api_key_id` 和公开 Key 前缀写入请求 Context; +5. 解析 Chat JSON 后,检查 `(tenant_id, model)` 是否存在于 `tenant_model_grants`; +6. 用 tenant ID 和规范化模型名执行 Redis 原子限流; +7. 生成有序 Route Plan,再以环境、租户、模型、路由版本和规范化请求查询 Exact Cache; +8. 命中直接返回;未命中或缓存故障时,在统一总预算内按候选顺序执行 Attempt; +9. 每个候选重新选择对应 Adapter、Breaker、Queue 和 Key,在候选内有限 Retry;允许 Fallback 的最终失败才进入下一候选; +10. 首个完整成功 JSON 回填缓存并返回客户端;普通 400、取消或候选耗尽返回结构化错误。 +11. 请求终态由 Collector finalize 一次,并在独立短超时内投递 Usage Event;流式请求在 `[DONE]`、取消或提交后中断时分别记录终态。 + +未知 Key、错误 secret、禁用、吊销、过期和禁用租户对客户端统一返回 `401 invalid_api_key`,避免暴露具体凭证状态。到期时刻本身已经算过期,创建 Key 时也只接受严格晚于当前时刻的 UTC 时间。缺少模型授权返回 `403 model_not_allowed`。认证数据库故障返回 `503 authentication_unavailable`,不会退化成匿名请求。 + +客户端的 Model-Velo Key 不会转发给外部 Provider。Provider 请求只使用 `MODEL_VELO_PROVIDER_KEYS_JSON` 中按目标 Provider ID 选出的 Secret。 + +## 运行 Model-Velo API + +需要 Go 工具链、一个已经健康的 PostgreSQL、Redis,以及至少一个显式配置的 Provider。启动时会自动同步当前 GORM 模型。下面用单个自定义 OpenAI-compatible Provider 演示;这仍然是一份显式路由,不是默认回退: + +```powershell +$env:MODEL_VELO_ROUTING_JSON = '{"providers":[{"id":"upstream","vendor":"custom","type":"openai-compatible","base_url":"https://api.example.com","models":["*"],"model_capabilities":{"*":["text"]}}],"routes":[{"model":"*","candidates":[{"provider":"upstream"}]}]}' +$env:MODEL_VELO_PROVIDER_KEYS_JSON = '{"providers":[{"provider_id":"upstream","keys":[{"id":"primary","secret":"replace-with-provider-key"}]}]}' +$env:MODEL_VELO_ENVIRONMENT = "development" +$env:MODEL_VELO_POSTGRES_DSN = "postgres://model_velo:replace-with-local-postgres-password@localhost:5432/model_velo?sslmode=disable" +$env:MODEL_VELO_API_KEY_PEPPER = "use-the-same-stable-pepper-as-model-velo-admin" +$env:MODEL_VELO_REDIS_ADDR = "localhost:6379" +$env:MODEL_VELO_REDIS_PASSWORD = "replace-with-local-redis-password" +$env:MODEL_VELO_RATE_LIMIT_FAILURE_POLICY = "fail-closed" +go run ./cmd/model-velo +``` + +Usage Worker 与 API 使用相同的 PostgreSQL、Redis 和 `MODEL_VELO_ENVIRONMENT`,但作为独立进程运行: + +```powershell +go run ./cmd/model-velo-usage-worker +``` + +Worker 启动会幂等创建 consumer group。数据库写入失败不 ACK;收到退出信号后停止新读取,并给当前批次最多 `MODEL_VELO_USAGE_WORKER_TIMEOUT` 完成。生产环境应为每个并发 Worker 设置不同的 `MODEL_VELO_USAGE_CONSUMER`,或使用默认的主机名与 PID。 + +价目新增或修正后,可显式重算缺少成本的历史记录。命令要求确认、限制时间范围和单批数量,不会把无法定价的记录写成零;输出非空 `next_cursor` 时,把它传给下一批即可越过当前批次中仍无法定价的行: + +```powershell +go run ./cmd/model-velo-admin reprice-usage ` + --start 2026-07-01T00:00:00Z ` + --end 2026-08-01T00:00:00Z ` + --missing-only=true ` + --limit 1000 ` + --confirm +``` + +请求示例: + +```powershell +$modelVeloKey = "replace-with-created-mvl-key" +$headers = @{ + Authorization = "Bearer $modelVeloKey" + "Content-Type" = "application/json" +} +$body = @{ + model = "gpt-4o-mini" + messages = @( + @{ role = "user"; content = "hello" } + ) +} | ConvertTo-Json -Depth 5 + +Invoke-RestMethod ` + -Method Post ` + -Uri "http://localhost:8080/v1/chat/completions" ` + -Headers $headers ` + -Body $body +``` + +同一接口也可直接使用 OpenAI Python SDK;`base_url` 必须指向网关的 `/v1`,Key 使用创建时仅返回一次的 Model-Velo 业务 Key: + +```python +import os +from openai import OpenAI + +client = OpenAI( + api_key=os.environ["MODEL_VELO_CLIENT_KEY"], + base_url="http://localhost:8080/v1", +) + +completion = client.chat.completions.create( + model="gpt-4o-mini", + messages=[{"role": "user", "content": "hello"}], +) +print(completion.choices[0].message.content) + +stream = client.chat.completions.create( + model="gpt-4o-mini", + messages=[{"role": "user", "content": "stream hello"}], + stream=True, +) +for chunk in stream: + if chunk.choices and chunk.choices[0].delta.content: + print(chunk.choices[0].delta.content, end="", flush=True) +``` + +Bash/curl 的最小重放请求: + +```bash +curl --fail-with-body http://localhost:8080/v1/chat/completions \ + -H "Authorization: Bearer ${MODEL_VELO_CLIENT_KEY}" \ + -H "Content-Type: application/json" \ + -d '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"hello"}]}' +``` + +聊天、Responses 与 Anthropic Messages 请求体上限为 16 MiB,Embeddings 为 8 MiB;这允许常见内嵌多模态输入,同时仍在 JSON 解码前限制内存占用。更大的媒体应使用厂商支持的 URL/File ID,而不是继续扩大 Base64 请求。 + +不要使用示例地址调用真实服务,也不要把真实 Key 写入 `.env.example`、测试、日志或 Git。 + +日常开发检查命令: + +```powershell +$goFiles = rg --files -g '*.go' +gofmt -w $goFiles +go test ./... +go vet ./... +``` + +项目当前采用快速交付模式:学习和讲解以生产代码、请求链和故障策略为主,不要求逐个学习测试文件。一个纵向功能通常最多维护一个合并测试文件;已有端到端证据能覆盖时,不再给 Handler、Service 和存储层重复补同类测试,也不追求覆盖率数字。 + +`go test ./...` 和 `go vet ./...` 在每次交付前统一运行一次。`go test -race ./...`、真实 Redis/PostgreSQL 故障矩阵和完整异常测试只在阶段门禁集中执行一次;环境不满足时记录缺口,不把纯测试任务继续当作开发进度。 + +阶段拆解和实时完成度见 `TODO.md`,完整规划边界见 `ARCHITECTURE.md`,开发约束见 `AGENTS.md`。 diff --git a/go.mod b/go.mod index 86e242f..2d93599 100644 --- a/go.mod +++ b/go.mod @@ -12,6 +12,7 @@ require ( go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.44.0 go.opentelemetry.io/otel/sdk v1.44.0 go.opentelemetry.io/otel/trace v1.44.0 + golang.org/x/sync v0.21.0 gorm.io/driver/postgres v1.6.0 gorm.io/gorm v1.31.2 ) @@ -67,7 +68,6 @@ require ( golang.org/x/arch v0.22.0 // indirect golang.org/x/crypto v0.53.0 // indirect golang.org/x/net v0.56.0 // indirect - golang.org/x/sync v0.21.0 // indirect golang.org/x/sys v0.47.0 // indirect golang.org/x/text v0.38.0 // indirect google.golang.org/genproto/googleapis/api v0.0.0-20260526163538-3dc84a4a5aaa // indirect diff --git a/internal/apikey/admin.go b/internal/apikey/admin.go index a5d3c87..adb63fd 100644 --- a/internal/apikey/admin.go +++ b/internal/apikey/admin.go @@ -190,6 +190,7 @@ func (manager *Manager) UpdateTenantAudited( if err != nil { return TenantView{}, err } + manager.invalidateTenant(ctx, tenantID) return tenantView(updated, models), nil } @@ -305,7 +306,11 @@ func (manager *Manager) UpdateKeyStatusAudited( keyView(current), keyView(updated), ) }) - return keyView(updated), err + if err != nil { + return KeyView{}, err + } + manager.invalidateKey(ctx, keyID) + return keyView(updated), nil } func replaceModelGrants( diff --git a/internal/apikey/cache.go b/internal/apikey/cache.go new file mode 100644 index 0000000..10c51a3 --- /dev/null +++ b/internal/apikey/cache.go @@ -0,0 +1,486 @@ +package apikey + +import ( + "context" + "encoding/hex" + "encoding/json" + "errors" + "sync" + "sync/atomic" + "time" + + goredis "github.com/redis/go-redis/v9" + "golang.org/x/sync/singleflight" + + "model-velo/internal/config" +) + +const ( + authCacheLoadTimeout = 5 * time.Second + authCacheWriteTimeout = 2 * time.Second + invalidationRetryPeriod = time.Second + invalidationVersion = 1 +) + +type CacheObserver interface { + AuthCacheLookup(layer, result string, duration time.Duration) + AuthCacheEvent(event, result string) + AuthPostgresFallback(result string) + AuthDatabaseQueries(count int) + ModelAuthorization(result string) +} + +type credentialStore struct { + enabled bool + settings config.AuthCache + l1 *l1Cache + shared sharedAuthCache + source snapshotSource + observer CacheObserver + loads singleflight.Group + now func() time.Time + start sync.Once + + invalidationMu sync.RWMutex + generation atomic.Uint64 +} + +type cacheLoad struct { + snapshot authSnapshot + source string + queries int + err error + generation uint64 + promoteOnce sync.Once + touchOnce sync.Once +} + +type invalidationEvent struct { + Version int `json:"version"` + EventID string `json:"event_id"` + Scope string `json:"scope"` + ID string `json:"id"` +} + +func newCredentialStore( + source snapshotSource, + settings config.AuthCache, + shared sharedAuthCache, + observer CacheObserver, +) *credentialStore { + store := &credentialStore{ + enabled: settings.Enabled, settings: settings, + shared: shared, source: source, observer: observer, now: time.Now, + } + if settings.Enabled { + store.l1 = newL1Cache(settings.L1MaxEntries, settings.L1TTL) + } + return store +} + +func newRedisCredentialStore( + source snapshotSource, + client *goredis.Client, + settings config.AuthCache, + observer CacheObserver, +) (*credentialStore, error) { + if client == nil { + return nil, errors.New("authentication cache requires Redis") + } + if !settings.Enabled || settings.L1MaxEntries <= 0 || + settings.L1TTL <= 0 || settings.L2TTL < settings.L1TTL || + settings.KeyPrefix == "" || settings.InvalidationChannel == "" { + return nil, errors.New("authentication cache settings are invalid") + } + return newCredentialStore( + source, + settings, + redisAuthCache{client: client, prefix: settings.KeyPrefix}, + observer, + ), nil +} + +func (store *credentialStore) Lookup( + ctx context.Context, + lookupDigest []byte, +) (*cacheLoad, bool) { + cacheKey := store.cacheKey(lookupDigest) + if !store.enabled { + snapshot, queries, err := store.source.Load( + ctx, lookupDigest, store.now().UTC(), + ) + result := "success" + if errors.Is(err, errSnapshotNotFound) { + result = "not_found" + } else if err != nil { + result = "error" + } + store.observePostgresFallback(result) + return &cacheLoad{ + snapshot: snapshot, source: "postgres", + queries: queries, err: err, + }, true + } + + startedAt := time.Now() + if snapshot, result := store.l1.Get(cacheKey, store.now().UTC()); result == "hit" { + store.observeLookup("l1", result, time.Since(startedAt)) + return &cacheLoad{snapshot: snapshot, source: "l1"}, false + } else { + store.observeLookup("l1", result, time.Since(startedAt)) + } + + var leader atomic.Bool + resultChannel := store.loads.DoChan(cacheKey, func() (any, error) { + leader.Store(true) + loadContext, cancel := context.WithTimeout( + context.WithoutCancel(ctx), + authCacheLoadTimeout, + ) + defer cancel() + return store.load(loadContext, cacheKey, lookupDigest), nil + }) + select { + case result := <-resultChannel: + return result.Val.(*cacheLoad), leader.Load() + case <-ctx.Done(): + return &cacheLoad{err: ctx.Err()}, false + } +} + +func (store *credentialStore) load( + ctx context.Context, + cacheKey string, + lookupDigest []byte, +) *cacheLoad { + generation := store.generation.Load() + if snapshot, result := store.l1.Get(cacheKey, store.now().UTC()); result == "hit" { + return &cacheLoad{ + snapshot: snapshot, source: "l1", generation: generation, + } + } + + startedAt := time.Now() + payload, err := store.shared.Load(ctx, cacheKey) + switch { + case err == nil: + var snapshot authSnapshot + decodeErr := json.Unmarshal(payload, &snapshot) + if decodeErr == nil { + decodeErr = snapshot.validate( + lookupDigest, + store.now().UTC(), + true, + ) + } + if decodeErr == nil { + store.observeLookup("l2", "hit", time.Since(startedAt)) + store.invalidationMu.RLock() + if store.generation.Load() == generation { + store.setL1(cacheKey, snapshot) + } + store.invalidationMu.RUnlock() + return &cacheLoad{ + snapshot: snapshot, source: "l2", generation: generation, + } + } + store.observeLookup("l2", "error", time.Since(startedAt)) + store.observeEvent("invalid_payload", "deleted") + deleteContext, cancel := context.WithTimeout( + context.WithoutCancel(ctx), + authCacheWriteTimeout, + ) + if deleteErr := store.shared.DeleteValue( + deleteContext, + cacheKey, + ); deleteErr != nil { + store.observeEvent("delete", "error") + } + cancel() + case errors.Is(err, errSharedCacheMiss): + store.observeLookup("l2", "miss", time.Since(startedAt)) + default: + store.observeLookup("l2", "error", time.Since(startedAt)) + } + + snapshot, queries, sourceErr := store.source.Load( + ctx, + lookupDigest, + store.now().UTC(), + ) + fallbackResult := "success" + if errors.Is(sourceErr, errSnapshotNotFound) { + fallbackResult = "not_found" + } else if sourceErr != nil { + fallbackResult = "error" + } + store.observePostgresFallback(fallbackResult) + return &cacheLoad{ + snapshot: snapshot, source: "postgres", + queries: queries, err: sourceErr, generation: generation, + } +} + +func (store *credentialStore) Promote( + ctx context.Context, + cacheKey string, + load *cacheLoad, +) { + if !store.enabled || load == nil || load.err != nil || + load.source != "postgres" { + return + } + store.invalidationMu.RLock() + defer store.invalidationMu.RUnlock() + if store.generation.Load() != load.generation { + return + } + load.promoteOnce.Do(func() { + now := store.now().UTC() + snapshot := load.snapshot + snapshot.CachedUntil = now.Add(store.settings.L2TTL) + payload, err := json.Marshal(snapshot) + if err == nil { + writeContext, cancel := context.WithTimeout( + context.WithoutCancel(ctx), + authCacheWriteTimeout, + ) + err = store.shared.Store( + writeContext, + cacheKey, + snapshot.APIKeyID, + snapshot.TenantID, + payload, + store.settings.L2TTL, + ) + cancel() + } + if err != nil { + store.observeEvent("write", "error") + } else { + store.observeEvent("write", "success") + } + store.setL1(cacheKey, snapshot) + }) +} + +func (store *credentialStore) Touch( + ctx context.Context, + load *cacheLoad, + now time.Time, +) int { + if load == nil || load.source != "postgres" || + load.snapshot.APIKeyID == "" { + return 0 + } + performed := false + queries := 0 + load.touchOnce.Do(func() { + performed = true + touchContext, cancel := context.WithTimeout( + context.WithoutCancel(ctx), + authCacheWriteTimeout, + ) + defer cancel() + queries, _ = store.source.Touch( + touchContext, + load.snapshot.APIKeyID, + now, + ) + }) + if !performed { + return 0 + } + return queries +} + +func (store *credentialStore) Start(ctx context.Context) { + if !store.enabled || store.shared == nil { + return + } + store.start.Do(func() { + go store.listen(ctx) + }) +} + +func (store *credentialStore) InvalidateKey( + ctx context.Context, + keyID string, +) { + if !store.enabled { + return + } + store.invalidationMu.Lock() + defer store.invalidationMu.Unlock() + store.generation.Add(1) + store.l1.DeleteKey(keyID) + store.invalidate(ctx, "key", keyID) +} + +func (store *credentialStore) InvalidateTenant( + ctx context.Context, + tenantID string, +) { + if !store.enabled { + return + } + store.invalidationMu.Lock() + defer store.invalidationMu.Unlock() + store.generation.Add(1) + store.l1.DeleteTenant(tenantID) + store.invalidate(ctx, "tenant", tenantID) +} + +func (store *credentialStore) invalidate( + ctx context.Context, + scope string, + id string, +) { + writeContext, cancel := context.WithTimeout( + context.WithoutCancel(ctx), + authCacheWriteTimeout, + ) + defer cancel() + var err error + switch scope { + case "key": + err = store.shared.InvalidateKey(writeContext, id) + case "tenant": + err = store.shared.InvalidateTenant(writeContext, id) + } + if err != nil { + store.observeEvent("invalidation", "error") + } else { + store.observeEvent("invalidation", "success") + } + eventID, eventErr := randomUUID() + if eventErr != nil { + store.observeEvent("publish", "error") + return + } + payload, eventErr := json.Marshal(invalidationEvent{ + Version: invalidationVersion, + EventID: eventID, + Scope: scope, + ID: id, + }) + if eventErr != nil || + store.shared.Publish( + writeContext, + store.settings.InvalidationChannel, + payload, + ) != nil { + store.observeEvent("publish", "error") + return + } + store.observeEvent("publish", "success") +} + +func (store *credentialStore) listen(ctx context.Context) { + for ctx.Err() == nil { + err := store.shared.Listen( + ctx, + store.settings.InvalidationChannel, + store.handleInvalidation, + ) + if ctx.Err() != nil { + return + } + if err != nil { + store.observeEvent("subscription", "error") + } + timer := time.NewTimer(invalidationRetryPeriod) + select { + case <-ctx.Done(): + timer.Stop() + return + case <-timer.C: + } + } +} + +func (store *credentialStore) handleInvalidation(payload []byte) { + var event invalidationEvent + if json.Unmarshal(payload, &event) != nil || + event.Version != invalidationVersion || + event.EventID == "" || event.ID == "" { + store.observeEvent("invalidation_message", "error") + return + } + switch event.Scope { + case "key": + store.invalidationMu.Lock() + store.generation.Add(1) + store.l1.DeleteKey(event.ID) + store.deleteSharedAfterEvent("key", event.ID) + store.invalidationMu.Unlock() + case "tenant": + store.invalidationMu.Lock() + store.generation.Add(1) + store.l1.DeleteTenant(event.ID) + store.deleteSharedAfterEvent("tenant", event.ID) + store.invalidationMu.Unlock() + default: + store.observeEvent("invalidation_message", "error") + return + } + store.observeEvent("invalidation_message", "success") +} + +func (store *credentialStore) deleteSharedAfterEvent( + scope string, + id string, +) { + ctx, cancel := context.WithTimeout( + context.Background(), + authCacheWriteTimeout, + ) + defer cancel() + var err error + if scope == "key" { + err = store.shared.InvalidateKey(ctx, id) + } else { + err = store.shared.InvalidateTenant(ctx, id) + } + if err != nil { + store.observeEvent("invalidation_recheck", "error") + } else { + store.observeEvent("invalidation_recheck", "success") + } +} + +func (store *credentialStore) setL1( + cacheKey string, + snapshot authSnapshot, +) { + if store.l1.Set(cacheKey, snapshot, store.now().UTC()) { + store.observeEvent("eviction", "success") + } +} + +func (store *credentialStore) cacheKey(lookupDigest []byte) string { + return store.settings.KeyPrefix + ":snapshot:" + + hex.EncodeToString(lookupDigest) +} + +func (store *credentialStore) observeLookup( + layer string, + result string, + duration time.Duration, +) { + if store.observer != nil { + store.observer.AuthCacheLookup(layer, result, duration) + } +} + +func (store *credentialStore) observeEvent(event, result string) { + if store.observer != nil { + store.observer.AuthCacheEvent(event, result) + } +} + +func (store *credentialStore) observePostgresFallback(result string) { + if store.observer != nil { + store.observer.AuthPostgresFallback(result) + } +} diff --git a/internal/apikey/cache_l1.go b/internal/apikey/cache_l1.go new file mode 100644 index 0000000..68172eb --- /dev/null +++ b/internal/apikey/cache_l1.go @@ -0,0 +1,146 @@ +package apikey + +import ( + "container/list" + "sync" + "time" +) + +type l1Cache struct { + mu sync.Mutex + maxEntries int + ttl time.Duration + entries map[string]*list.Element + keys map[string]map[string]struct{} + tenants map[string]map[string]struct{} + order *list.List +} + +type l1Entry struct { + cacheKey string + snapshot authSnapshot + expiresAt time.Time +} + +func newL1Cache(maxEntries int, ttl time.Duration) *l1Cache { + return &l1Cache{ + maxEntries: maxEntries, + ttl: ttl, + entries: make(map[string]*list.Element, maxEntries), + keys: make(map[string]map[string]struct{}), + tenants: make(map[string]map[string]struct{}), + order: list.New(), + } +} + +func (cache *l1Cache) Get( + cacheKey string, + now time.Time, +) (authSnapshot, string) { + cache.mu.Lock() + defer cache.mu.Unlock() + + element, ok := cache.entries[cacheKey] + if !ok { + return authSnapshot{}, "miss" + } + entry := element.Value.(*l1Entry) + if !entry.expiresAt.After(now) { + cache.remove(element) + return authSnapshot{}, "expired" + } + cache.order.MoveToFront(element) + return entry.snapshot, "hit" +} + +func (cache *l1Cache) Set( + cacheKey string, + snapshot authSnapshot, + now time.Time, +) bool { + cache.mu.Lock() + defer cache.mu.Unlock() + + expiresAt := now.Add(cache.ttl) + if !snapshot.CachedUntil.IsZero() && + snapshot.CachedUntil.Before(expiresAt) { + expiresAt = snapshot.CachedUntil + } + if element, ok := cache.entries[cacheKey]; ok { + cache.remove(element) + } + entry := &l1Entry{ + cacheKey: cacheKey, snapshot: snapshot, expiresAt: expiresAt, + } + element := cache.order.PushFront(entry) + cache.entries[cacheKey] = element + addCacheIndex(cache.keys, snapshot.APIKeyID, cacheKey) + addCacheIndex(cache.tenants, snapshot.TenantID, cacheKey) + + if cache.order.Len() <= cache.maxEntries { + return false + } + cache.remove(cache.order.Back()) + return true +} + +func (cache *l1Cache) DeleteKey(keyID string) int { + cache.mu.Lock() + defer cache.mu.Unlock() + return cache.deleteIndexed(cache.keys[keyID]) +} + +func (cache *l1Cache) DeleteTenant(tenantID string) int { + cache.mu.Lock() + defer cache.mu.Unlock() + return cache.deleteIndexed(cache.tenants[tenantID]) +} + +func (cache *l1Cache) deleteIndexed(cacheKeys map[string]struct{}) int { + removed := 0 + for cacheKey := range cacheKeys { + element, ok := cache.entries[cacheKey] + if !ok { + continue + } + cache.remove(element) + removed++ + } + return removed +} + +func (cache *l1Cache) remove(element *list.Element) { + if element == nil { + return + } + entry := element.Value.(*l1Entry) + delete(cache.entries, entry.cacheKey) + deleteCacheIndex(cache.keys, entry.snapshot.APIKeyID, entry.cacheKey) + deleteCacheIndex(cache.tenants, entry.snapshot.TenantID, entry.cacheKey) + cache.order.Remove(element) +} + +func addCacheIndex( + index map[string]map[string]struct{}, + id string, + cacheKey string, +) { + cacheKeys := index[id] + if cacheKeys == nil { + cacheKeys = make(map[string]struct{}) + index[id] = cacheKeys + } + cacheKeys[cacheKey] = struct{}{} +} + +func deleteCacheIndex( + index map[string]map[string]struct{}, + id string, + cacheKey string, +) { + cacheKeys := index[id] + delete(cacheKeys, cacheKey) + if len(cacheKeys) == 0 { + delete(index, id) + } +} diff --git a/internal/apikey/cache_redis.go b/internal/apikey/cache_redis.go new file mode 100644 index 0000000..5aacaa6 --- /dev/null +++ b/internal/apikey/cache_redis.go @@ -0,0 +1,134 @@ +package apikey + +import ( + "context" + "errors" + "time" + + goredis "github.com/redis/go-redis/v9" +) + +var errSharedCacheMiss = errors.New("shared authentication cache miss") + +type sharedAuthCache interface { + Load(context.Context, string) ([]byte, error) + Store( + context.Context, + string, + string, + string, + []byte, + time.Duration, + ) error + DeleteValue(context.Context, string) error + InvalidateKey(context.Context, string) error + InvalidateTenant(context.Context, string) error + Publish(context.Context, string, []byte) error + Listen(context.Context, string, func([]byte)) error +} + +type redisAuthCache struct { + client *goredis.Client + prefix string +} + +func (cache redisAuthCache) Load( + ctx context.Context, + cacheKey string, +) ([]byte, error) { + payload, err := cache.client.Get(ctx, cacheKey).Bytes() + if errors.Is(err, goredis.Nil) { + return nil, errSharedCacheMiss + } + return payload, err +} + +func (cache redisAuthCache) Store( + ctx context.Context, + cacheKey string, + keyID string, + tenantID string, + payload []byte, + ttl time.Duration, +) error { + keyIndex := cache.keyIndex(keyID) + tenantIndex := cache.tenantIndex(tenantID) + pipeline := cache.client.TxPipeline() + pipeline.Set(ctx, cacheKey, payload, ttl) + pipeline.Set(ctx, keyIndex, cacheKey, ttl) + pipeline.SAdd(ctx, tenantIndex, cacheKey, keyIndex) + pipeline.Expire(ctx, tenantIndex, ttl) + _, err := pipeline.Exec(ctx) + return err +} + +func (cache redisAuthCache) DeleteValue( + ctx context.Context, + cacheKey string, +) error { + return cache.client.Del(ctx, cacheKey).Err() +} + +func (cache redisAuthCache) InvalidateKey( + ctx context.Context, + keyID string, +) error { + keyIndex := cache.keyIndex(keyID) + cacheKey, err := cache.client.Get(ctx, keyIndex).Result() + if err != nil && !errors.Is(err, goredis.Nil) { + return err + } + keys := []string{keyIndex} + if cacheKey != "" { + keys = append(keys, cacheKey) + } + return cache.client.Del(ctx, keys...).Err() +} + +func (cache redisAuthCache) InvalidateTenant( + ctx context.Context, + tenantID string, +) error { + tenantIndex := cache.tenantIndex(tenantID) + keys, err := cache.client.SMembers(ctx, tenantIndex).Result() + if err != nil && !errors.Is(err, goredis.Nil) { + return err + } + keys = append(keys, tenantIndex) + return cache.client.Del(ctx, keys...).Err() +} + +func (cache redisAuthCache) Publish( + ctx context.Context, + channel string, + payload []byte, +) error { + return cache.client.Publish(ctx, channel, payload).Err() +} + +func (cache redisAuthCache) Listen( + ctx context.Context, + channel string, + handle func([]byte), +) error { + subscription := cache.client.Subscribe(ctx, channel) + defer subscription.Close() + if _, err := subscription.Receive(ctx); err != nil { + return err + } + for { + message, err := subscription.ReceiveMessage(ctx) + if err != nil { + return err + } + handle([]byte(message.Payload)) + } +} + +func (cache redisAuthCache) keyIndex(keyID string) string { + return cache.prefix + ":key:" + keyID +} + +func (cache redisAuthCache) tenantIndex(tenantID string) string { + return cache.prefix + ":tenant:" + tenantID +} diff --git a/internal/apikey/cache_test.go b/internal/apikey/cache_test.go new file mode 100644 index 0000000..aeb1f6f --- /dev/null +++ b/internal/apikey/cache_test.go @@ -0,0 +1,864 @@ +package apikey + +import ( + "bytes" + "context" + "encoding/hex" + "errors" + "sync" + "testing" + "time" + + "model-velo/internal/config" + "model-velo/internal/postgres" +) + +func TestAuthenticationCacheLayersAndFailureSafety(t *testing.T) { + pepper := bytes.Repeat([]byte("p"), 32) + token := mustTestToken(t, pepper) + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + now := time.Date(2026, 7, 28, 12, 0, 0, 0, time.UTC) + manager := newTestCachedManager(source, shared, pepper, &now) + + identity, err := manager.Authenticate(context.Background(), token.plaintext) + if err != nil { + t.Fatalf("Authenticate(database fallback) error = %v", err) + } + if source.Loads() != 1 || shared.Loads() != 1 || shared.Stores() != 1 { + t.Fatalf( + "initial loads source/Redis/store = %d/%d/%d, want 1/1/1", + source.Loads(), + shared.Loads(), + shared.Stores(), + ) + } + source.ResetCounts() + shared.ResetCounts() + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(L1 hit) error = %v", err) + } + if source.Loads() != 0 || shared.Loads() != 0 { + t.Fatalf( + "L1 hit source/Redis loads = %d/%d, want 0/0", + source.Loads(), + shared.Loads(), + ) + } + + second := newTestCachedManager(source, shared, pepper, &now) + if _, err := second.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(L2 hit) error = %v", err) + } + if source.Loads() != 0 || shared.Loads() != 1 { + t.Fatalf( + "L2 hit source/Redis loads = %d/%d, want 0/1", + source.Loads(), + shared.Loads(), + ) + } + if err := second.AuthorizeModel( + context.Background(), + identity, + "model-a", + ); err != nil { + t.Fatalf("AuthorizeModel(cached grant) error = %v", err) + } + if !errors.Is( + second.AuthorizeModel( + context.Background(), + identity, + "model-denied", + ), + ErrModelNotAllowed, + ) { + t.Fatal("AuthorizeModel() allowed an uncached model grant") + } + + t.Run("concurrent miss is collapsed", func(t *testing.T) { + source := newFakeSnapshotSource(testSnapshot(token)) + source.delay = 20 * time.Millisecond + shared := newFakeSharedAuthCache() + manager := newTestCachedManager(source, shared, pepper, &now) + const callers = 64 + var wait sync.WaitGroup + errorsFound := make(chan error, callers) + for range callers { + wait.Add(1) + go func() { + defer wait.Done() + _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ) + errorsFound <- err + }() + } + wait.Wait() + close(errorsFound) + for err := range errorsFound { + if err != nil { + t.Fatalf("Authenticate(concurrent miss) error = %v", err) + } + } + if source.Loads() != 1 || shared.Loads() != 1 { + t.Fatalf( + "concurrent source/Redis loads = %d/%d, want 1/1", + source.Loads(), + shared.Loads(), + ) + } + }) + + t.Run("Redis error falls back and database error fails closed", func(t *testing.T) { + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + shared.readErr = errors.New("Redis unavailable") + manager := newTestCachedManager(source, shared, pepper, &now) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(Redis unavailable) error = %v", err) + } + source.err = errors.New("PostgreSQL unavailable") + manager.credentials.l1.DeleteKey(token.prefix) + manager = newTestCachedManager( + source, + newFakeSharedAuthCache(), + pepper, + &now, + ) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err == nil { + t.Fatal("Authenticate(PostgreSQL unavailable) error = nil") + } + + source = newFakeSnapshotSource(testSnapshot(token)) + shared = newFakeSharedAuthCache() + shared.writeErr = errors.New("Redis write unavailable") + manager = newTestCachedManager(source, shared, pepper, &now) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(Redis write unavailable) error = %v", err) + } + source.ResetCounts() + shared.ResetCounts() + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(L1 after Redis write error) error = %v", err) + } + if source.Loads() != 0 || shared.Loads() != 0 { + t.Fatal("Redis write error prevented safe L1 degradation") + } + }) + + t.Run("bad payload and both TTLs reload the next layer", func(t *testing.T) { + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + manager := newTestCachedManager(source, shared, pepper, &now) + cacheKey := manager.credentials.cacheKey(token.lookupDigest) + shared.SetRaw(cacheKey, []byte(`{"version":999}`)) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(bad L2) error = %v", err) + } + if shared.Deletes() == 0 || source.Loads() != 1 { + t.Fatalf( + "bad L2 deletes/source loads = %d/%d, want >0/1", + shared.Deletes(), + source.Loads(), + ) + } + now = now.Add(16 * time.Second) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(expired L1) error = %v", err) + } + if source.Loads() != 1 || shared.Loads() != 2 { + t.Fatalf( + "expired L1 source/Redis loads = %d/%d, want 1/2", + source.Loads(), + shared.Loads(), + ) + } + now = now.Add(15 * time.Second) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatalf("Authenticate(expired L2) error = %v", err) + } + if source.Loads() != 2 { + t.Fatalf("expired L2 source loads = %d, want 2", source.Loads()) + } + }) + + t.Run("payload contains no plaintext secret or pepper", func(t *testing.T) { + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + manager := newTestCachedManager(source, shared, pepper, &now) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + t.Fatal(err) + } + payload := shared.LastPayload() + parsed, _ := parseToken(token.plaintext) + if bytes.Contains(payload, []byte(token.plaintext)) || + bytes.Contains(payload, []byte(parsed.secret)) || + bytes.Contains(payload, pepper) { + t.Fatal("shared cache payload contains plaintext API key or pepper") + } + }) +} + +func TestAuthenticationInvalidationAcrossInstances(t *testing.T) { + pepper := bytes.Repeat([]byte("q"), 32) + token := mustTestToken(t, pepper) + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + now := time.Date(2026, 7, 28, 13, 0, 0, 0, time.UTC) + instanceA := newTestCachedManager(source, shared, pepper, &now) + instanceB := newTestCachedManager(source, shared, pepper, &now) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + instanceA.StartInvalidationListener(ctx) + instanceB.StartInvalidationListener(ctx) + shared.WaitForListeners(t, 2) + + identity, err := instanceA.Authenticate(ctx, token.plaintext) + if err != nil { + t.Fatal(err) + } + revoked := testSnapshot(token) + revoked.KeyStatus = postgres.APIKeyRevoked + source.Set(revoked) + instanceB.credentials.InvalidateKey(ctx, revoked.APIKeyID) + waitForCondition(t, func() bool { + _, err := instanceA.Authenticate(ctx, token.plaintext) + return errors.Is(err, ErrKeyRevoked) + }) + + activeWithoutGrants := revoked + activeWithoutGrants.KeyStatus = postgres.APIKeyActive + activeWithoutGrants.AllowedModels = nil + source.Set(activeWithoutGrants) + instanceB.credentials.InvalidateTenant(ctx, revoked.TenantID) + var refreshed Identity + waitForCondition(t, func() bool { + var authenticateErr error + refreshed, authenticateErr = instanceA.Authenticate( + ctx, + token.plaintext, + ) + return authenticateErr == nil + }) + if !errors.Is( + instanceA.AuthorizeModel(ctx, refreshed, "model-a"), + ErrModelNotAllowed, + ) { + t.Fatal("tenant grant invalidation did not reject the removed model") + } + + disabledTenant := activeWithoutGrants + disabledTenant.TenantStatus = postgres.TenantDisabled + source.Set(disabledTenant) + instanceB.credentials.InvalidateTenant(ctx, disabledTenant.TenantID) + waitForCondition(t, func() bool { + _, err := instanceA.Authenticate(ctx, token.plaintext) + return errors.Is(err, ErrTenantInactive) + }) + if identity.APIKeyID == "" { + t.Fatal("initial identity was incomplete") + } +} + +func TestAuthenticationL1IsBounded(t *testing.T) { + pepper := bytes.Repeat([]byte("r"), 32) + first := mustTestToken(t, pepper) + second := mustTestToken(t, pepper) + source := newFakeSnapshotSource( + testSnapshot(first), + testSnapshot(second), + ) + shared := newFakeSharedAuthCache() + now := time.Date(2026, 7, 28, 14, 0, 0, 0, time.UTC) + settings := testCacheSettings() + settings.L1MaxEntries = 1 + manager := newTestManagerWithSettings( + source, + shared, + pepper, + &now, + settings, + ) + for _, plaintext := range []string{ + first.plaintext, + second.plaintext, + first.plaintext, + } { + if _, err := manager.Authenticate( + context.Background(), + plaintext, + ); err != nil { + t.Fatal(err) + } + } + if shared.Loads() != 3 { + t.Fatalf("L1 max=1 Redis loads = %d, want 3", shared.Loads()) + } +} + +func TestAuthenticationInvalidationBlocksStalePromotion(t *testing.T) { + pepper := bytes.Repeat([]byte("s"), 32) + token := mustTestToken(t, pepper) + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + now := time.Date(2026, 7, 28, 14, 30, 0, 0, time.UTC) + manager := newTestCachedManager(source, shared, pepper, &now) + load := &cacheLoad{ + snapshot: testSnapshot(token), + source: "postgres", + generation: manager.credentials.generation.Load(), + } + manager.credentials.InvalidateKey( + context.Background(), + load.snapshot.APIKeyID, + ) + shared.ResetCounts() + manager.credentials.Promote( + context.Background(), + manager.credentials.cacheKey(token.lookupDigest), + load, + ) + if shared.Stores() != 0 { + t.Fatal("an invalidated in-flight snapshot was written to Redis") + } + if _, result := manager.credentials.l1.Get( + manager.credentials.cacheKey(token.lookupDigest), + now, + ); result != "miss" { + t.Fatalf("stale L1 promotion result = %q, want miss", result) + } +} + +// BenchmarkAuthenticationCachePaths isolates gateway cache and HMAC overhead. +// The fake Redis and snapshot source make command/query counts deterministic; +// their ns/op values do not represent networked Redis or PostgreSQL latency. +func BenchmarkAuthenticationCachePaths(b *testing.B) { + pepper := bytes.Repeat([]byte("b"), 32) + token, err := generateToken(pepper) + if err != nil { + b.Fatal(err) + } + now := time.Date(2026, 7, 28, 15, 0, 0, 0, time.UTC) + + b.Run("l1_hit", func(b *testing.B) { + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + manager := newTestCachedManager(source, shared, pepper, &now) + _, _ = manager.Authenticate(context.Background(), token.plaintext) + source.ResetCounts() + shared.ResetCounts() + b.ReportAllocs() + b.ResetTimer() + for range b.N { + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + reportBenchmarkCounts(b, source, shared, 100) + }) + + b.Run("l2_hit", func(b *testing.B) { + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + manager := newTestCachedManager(source, shared, pepper, &now) + _, _ = manager.Authenticate(context.Background(), token.plaintext) + source.ResetCounts() + shared.ResetCounts() + b.ReportAllocs() + b.ResetTimer() + for range b.N { + manager.credentials.l1.DeleteKey(testSnapshot(token).APIKeyID) + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + reportBenchmarkCounts(b, source, shared, 100) + }) + + b.Run("postgres_fallback", func(b *testing.B) { + source := newFakeSnapshotSource(testSnapshot(token)) + shared := newFakeSharedAuthCache() + manager := newTestCachedManager(source, shared, pepper, &now) + b.ReportAllocs() + b.ResetTimer() + for range b.N { + manager.credentials.l1.DeleteKey(testSnapshot(token).APIKeyID) + shared.Clear() + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + reportBenchmarkCounts(b, source, shared, 0) + }) + + b.Run("cache_disabled", func(b *testing.B) { + source := newFakeSnapshotSource(testSnapshot(token)) + manager := &Manager{ + pepper: append([]byte(nil), pepper...), + now: func() time.Time { return now }, + } + manager.credentials = newCredentialStore( + source, + config.AuthCache{}, + nil, + nil, + ) + manager.credentials.now = manager.now + b.ReportAllocs() + b.ResetTimer() + for range b.N { + if _, err := manager.Authenticate( + context.Background(), + token.plaintext, + ); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + reportBenchmarkCounts( + b, + source, + newFakeSharedAuthCache(), + 0, + ) + }) +} + +func reportBenchmarkCounts( + b *testing.B, + source *fakeSnapshotSource, + shared *fakeSharedAuthCache, + hitRate float64, +) { + denominator := float64(b.N) + b.ReportMetric(float64(source.Queries())/denominator, "pg_queries/op") + b.ReportMetric(float64(shared.Commands())/denominator, "redis_cmds/op") + b.ReportMetric(hitRate, "cache_hit_pct") +} + +func newTestCachedManager( + source *fakeSnapshotSource, + shared *fakeSharedAuthCache, + pepper []byte, + now *time.Time, +) *Manager { + return newTestManagerWithSettings( + source, + shared, + pepper, + now, + testCacheSettings(), + ) +} + +func newTestManagerWithSettings( + source snapshotSource, + shared sharedAuthCache, + pepper []byte, + now *time.Time, + settings config.AuthCache, +) *Manager { + manager := &Manager{ + pepper: append([]byte(nil), pepper...), + now: func() time.Time { return *now }, + } + manager.credentials = newCredentialStore( + source, + settings, + shared, + nil, + ) + manager.credentials.now = manager.now + return manager +} + +func testCacheSettings() config.AuthCache { + return config.AuthCache{ + Enabled: true, + L1MaxEntries: 100, + L1TTL: 15 * time.Second, + L2TTL: 30 * time.Second, + KeyPrefix: "model-velo:test:auth:v1", + InvalidationChannel: "model-velo:test:auth:v1:invalidate", + } +} + +func testSnapshot(token generatedToken) authSnapshot { + return authSnapshot{ + Version: authSnapshotVersion, + APIKeyID: "00000000-0000-4000-8000-000000000011", + TenantID: "00000000-0000-4000-8000-000000000022", + KeyPrefix: token.prefix, + LookupDigest: append([]byte(nil), token.lookupDigest...), + KeyHash: append([]byte(nil), token.keyHash...), + HashVersion: token.hashVersion, + KeyStatus: postgres.APIKeyActive, + TenantStatus: postgres.TenantActive, + TenantVersion: 1, + AllowedModels: []string{"model-a"}, + } +} + +func mustTestToken(t *testing.T, pepper []byte) generatedToken { + t.Helper() + token, err := generateToken(pepper) + if err != nil { + t.Fatal(err) + } + return token +} + +type fakeSnapshotSource struct { + mu sync.Mutex + snapshots map[string]authSnapshot + loads int + touches int + err error + delay time.Duration +} + +func newFakeSnapshotSource( + snapshots ...authSnapshot, +) *fakeSnapshotSource { + source := &fakeSnapshotSource{ + snapshots: make(map[string]authSnapshot, len(snapshots)), + } + for _, snapshot := range snapshots { + source.Set(snapshot) + } + return source +} + +func (source *fakeSnapshotSource) Load( + ctx context.Context, + digest []byte, + now time.Time, +) (authSnapshot, int, error) { + if source.delay > 0 { + timer := time.NewTimer(source.delay) + select { + case <-ctx.Done(): + timer.Stop() + return authSnapshot{}, 0, ctx.Err() + case <-timer.C: + } + } + source.mu.Lock() + defer source.mu.Unlock() + source.loads++ + if source.err != nil { + return authSnapshot{}, 1, source.err + } + snapshot, ok := source.snapshots[hex.EncodeToString(digest)] + if !ok { + return authSnapshot{}, 1, errSnapshotNotFound + } + snapshot.GeneratedAt = now + snapshot.CachedUntil = time.Time{} + return snapshot, 2, nil +} + +func (source *fakeSnapshotSource) Touch( + _ context.Context, + keyID string, + now time.Time, +) (int, error) { + source.mu.Lock() + source.touches++ + for digest, snapshot := range source.snapshots { + if snapshot.APIKeyID != keyID { + continue + } + snapshot.LastUsedAt = cloneTime(&now) + source.snapshots[digest] = snapshot + } + source.mu.Unlock() + return 1, nil +} + +func (source *fakeSnapshotSource) Set(snapshot authSnapshot) { + source.mu.Lock() + if source.snapshots == nil { + source.snapshots = make(map[string]authSnapshot) + } + source.snapshots[hex.EncodeToString(snapshot.LookupDigest)] = snapshot + source.mu.Unlock() +} + +func (source *fakeSnapshotSource) Loads() int { + source.mu.Lock() + defer source.mu.Unlock() + return source.loads +} + +func (source *fakeSnapshotSource) Queries() int { + source.mu.Lock() + defer source.mu.Unlock() + return source.loads*2 + source.touches +} + +func (source *fakeSnapshotSource) ResetCounts() { + source.mu.Lock() + source.loads = 0 + source.touches = 0 + source.mu.Unlock() +} + +type fakeSharedAuthCache struct { + mu sync.Mutex + values map[string][]byte + keyIndex map[string]string + tenantIndex map[string]map[string]struct{} + listeners []chan []byte + loads int + stores int + deletes int + publishes int + readErr error + writeErr error + lastPayload []byte +} + +func newFakeSharedAuthCache() *fakeSharedAuthCache { + return &fakeSharedAuthCache{ + values: make(map[string][]byte), + keyIndex: make(map[string]string), + tenantIndex: make(map[string]map[string]struct{}), + } +} + +func (cache *fakeSharedAuthCache) Load( + _ context.Context, + key string, +) ([]byte, error) { + cache.mu.Lock() + defer cache.mu.Unlock() + cache.loads++ + if cache.readErr != nil { + return nil, cache.readErr + } + value, ok := cache.values[key] + if !ok { + return nil, errSharedCacheMiss + } + return append([]byte(nil), value...), nil +} + +func (cache *fakeSharedAuthCache) Store( + _ context.Context, + cacheKey string, + keyID string, + tenantID string, + payload []byte, + _ time.Duration, +) error { + cache.mu.Lock() + defer cache.mu.Unlock() + cache.stores++ + if cache.writeErr != nil { + return cache.writeErr + } + cache.values[cacheKey] = append([]byte(nil), payload...) + cache.keyIndex[keyID] = cacheKey + keys := cache.tenantIndex[tenantID] + if keys == nil { + keys = make(map[string]struct{}) + cache.tenantIndex[tenantID] = keys + } + keys[cacheKey] = struct{}{} + cache.lastPayload = append([]byte(nil), payload...) + return nil +} + +func (cache *fakeSharedAuthCache) DeleteValue( + _ context.Context, + key string, +) error { + cache.mu.Lock() + delete(cache.values, key) + cache.deletes++ + cache.mu.Unlock() + return nil +} + +func (cache *fakeSharedAuthCache) InvalidateKey( + _ context.Context, + keyID string, +) error { + cache.mu.Lock() + if key := cache.keyIndex[keyID]; key != "" { + delete(cache.values, key) + } + delete(cache.keyIndex, keyID) + cache.deletes++ + cache.mu.Unlock() + return nil +} + +func (cache *fakeSharedAuthCache) InvalidateTenant( + _ context.Context, + tenantID string, +) error { + cache.mu.Lock() + for key := range cache.tenantIndex[tenantID] { + delete(cache.values, key) + } + delete(cache.tenantIndex, tenantID) + cache.deletes++ + cache.mu.Unlock() + return nil +} + +func (cache *fakeSharedAuthCache) Publish( + _ context.Context, + _ string, + payload []byte, +) error { + cache.mu.Lock() + listeners := append([]chan []byte(nil), cache.listeners...) + cache.publishes++ + cache.mu.Unlock() + for _, listener := range listeners { + listener <- append([]byte(nil), payload...) + } + return nil +} + +func (cache *fakeSharedAuthCache) Listen( + ctx context.Context, + _ string, + handle func([]byte), +) error { + events := make(chan []byte, 16) + cache.mu.Lock() + cache.listeners = append(cache.listeners, events) + cache.mu.Unlock() + for { + select { + case <-ctx.Done(): + return ctx.Err() + case event := <-events: + handle(event) + } + } +} + +func (cache *fakeSharedAuthCache) SetRaw(key string, payload []byte) { + cache.mu.Lock() + cache.values[key] = append([]byte(nil), payload...) + cache.mu.Unlock() +} + +func (cache *fakeSharedAuthCache) Clear() { + cache.mu.Lock() + cache.values = make(map[string][]byte) + cache.keyIndex = make(map[string]string) + cache.tenantIndex = make(map[string]map[string]struct{}) + cache.mu.Unlock() +} + +func (cache *fakeSharedAuthCache) WaitForListeners( + t *testing.T, + want int, +) { + t.Helper() + waitForCondition(t, func() bool { + cache.mu.Lock() + defer cache.mu.Unlock() + return len(cache.listeners) >= want + }) +} + +func (cache *fakeSharedAuthCache) Loads() int { + cache.mu.Lock() + defer cache.mu.Unlock() + return cache.loads +} + +func (cache *fakeSharedAuthCache) Stores() int { + cache.mu.Lock() + defer cache.mu.Unlock() + return cache.stores +} + +func (cache *fakeSharedAuthCache) Deletes() int { + cache.mu.Lock() + defer cache.mu.Unlock() + return cache.deletes +} + +func (cache *fakeSharedAuthCache) Commands() int { + cache.mu.Lock() + defer cache.mu.Unlock() + return cache.loads + cache.stores*4 +} + +func (cache *fakeSharedAuthCache) LastPayload() []byte { + cache.mu.Lock() + defer cache.mu.Unlock() + return append([]byte(nil), cache.lastPayload...) +} + +func (cache *fakeSharedAuthCache) ResetCounts() { + cache.mu.Lock() + cache.loads = 0 + cache.stores = 0 + cache.deletes = 0 + cache.publishes = 0 + cache.mu.Unlock() +} + +func waitForCondition(t *testing.T, condition func() bool) { + t.Helper() + deadline := time.Now().Add(2 * time.Second) + for time.Now().Before(deadline) { + if condition() { + return + } + time.Sleep(time.Millisecond) + } + t.Fatal("condition was not satisfied before timeout") +} diff --git a/internal/apikey/manager.go b/internal/apikey/manager.go index f1686e5..40af015 100644 --- a/internal/apikey/manager.go +++ b/internal/apikey/manager.go @@ -33,8 +33,10 @@ import ( "time" "unicode/utf8" + goredis "github.com/redis/go-redis/v9" "gorm.io/gorm" + "model-velo/internal/config" "model-velo/internal/postgres" ) @@ -54,9 +56,11 @@ var ( var tenantSlugPattern = regexp.MustCompile(`^[a-z0-9][a-z0-9_-]{1,78}[a-z0-9]$`) type Manager struct { - database *gorm.DB - pepper []byte - now func() time.Time + database *gorm.DB + pepper []byte + now func() time.Time + credentials *credentialStore + observer CacheObserver } // BootstrapTenantInput 表示“创建一个新租户”需要提供的参数。 @@ -100,6 +104,12 @@ type Identity struct { // // 它不会暴露完整 secret。 KeyPrefix string + + authorization *authorizationSnapshot +} + +type authorizationSnapshot struct { + models []string } func NewManager(database *gorm.DB, pepper []byte) (*Manager, error) { @@ -110,11 +120,55 @@ func NewManager(database *gorm.DB, pepper []byte) (*Manager, error) { return nil, errors.New("API key manager requires at least 32 pepper bytes") } - return &Manager{ + manager := &Manager{ database: database, pepper: append([]byte(nil), pepper...), now: time.Now, - }, nil + } + manager.credentials = newCredentialStore( + postgresSnapshotSource{database: database}, + config.AuthCache{}, + nil, + nil, + ) + manager.credentials.now = func() time.Time { + return manager.now() + } + return manager, nil +} + +func NewCachedManager( + database *gorm.DB, + pepper []byte, + client *goredis.Client, + settings config.AuthCache, + observer CacheObserver, +) (*Manager, error) { + manager, err := NewManager(database, pepper) + if err != nil { + return nil, err + } + credentials, err := newRedisCredentialStore( + postgresSnapshotSource{database: database}, + client, + settings, + observer, + ) + if err != nil { + return nil, err + } + credentials.now = func() time.Time { + return manager.now() + } + manager.credentials = credentials + manager.observer = observer + return manager, nil +} + +func (manager *Manager) StartInvalidationListener(ctx context.Context) { + if manager != nil && manager.credentials != nil { + manager.credentials.Start(ctx) + } } func (manager *Manager) BootstrapTenant(ctx context.Context, input BootstrapTenantInput) (IssuedKey, error) { @@ -204,98 +258,107 @@ func (manager *Manager) Authenticate(ctx context.Context, plaintext string) (Ide token, err := parseToken(plaintext) if err != nil { manager.consumeDummyHash() + manager.observeDatabaseQueries(0) return Identity{}, ErrInvalidCredential } - var key postgres.APIKey - err = manager.database.WithContext(ctx). - Preload("Tenant"). - Where("lookup_digest = ?", digestPrefix(token.prefix)). - First(&key).Error - if err != nil { + lookupDigest := digestPrefix(token.prefix) + load, leader := manager.credentials.Lookup(ctx, lookupDigest) + queries := 0 + if leader { + queries = load.queries + } + defer func() { + manager.observeDatabaseQueries(queries) + }() + if load.err != nil { manager.consumeDummyHash() - if errors.Is(err, gorm.ErrRecordNotFound) { + if errors.Is(load.err, errSnapshotNotFound) { return Identity{}, ErrInvalidCredential } + if ctx.Err() != nil { + return Identity{}, ctx.Err() + } return Identity{}, errors.New("read API key") } - if !verifyToken(token, key.KeyHash, key.HashVersion, manager.pepper) { + snapshot := load.snapshot + if !verifyToken( + token, + snapshot.KeyHash, + snapshot.HashVersion, + manager.pepper, + ) || snapshot.KeyPrefix != token.prefix { return Identity{}, ErrInvalidCredential } - if key.Status == postgres.APIKeyRevoked { + manager.credentials.Promote( + ctx, + manager.credentials.cacheKey(lookupDigest), + load, + ) + if snapshot.KeyStatus == postgres.APIKeyRevoked { return Identity{}, ErrKeyRevoked } - if key.Status != postgres.APIKeyActive { + if snapshot.KeyStatus != postgres.APIKeyActive { return Identity{}, ErrKeyInactive } - if key.Tenant.Status != postgres.TenantActive { + if snapshot.TenantStatus != postgres.TenantActive { return Identity{}, ErrTenantInactive } - if expirationReached(key.ExpiresAt, manager.now().UTC()) { + now := manager.now().UTC() + if expirationReached(snapshot.KeyExpiresAt, now) { return Identity{}, ErrKeyExpired } - now := manager.now().UTC() - if key.LastUsedAt == nil || now.Sub(*key.LastUsedAt) >= 5*time.Minute { - _ = manager.database.WithContext(ctx). - Model(&postgres.APIKey{}). - Where("id = ?", key.ID). - Update("last_used_at", now).Error + if snapshot.LastUsedAt == nil || + now.Sub(*snapshot.LastUsedAt) >= 5*time.Minute { + queries += manager.credentials.Touch(ctx, load, now) } - return Identity{ - TenantID: key.TenantID, - APIKeyID: key.ID, - KeyPrefix: key.KeyPrefix, - }, nil + return snapshot.identity(), nil } -func (manager *Manager) AuthorizeModel(ctx context.Context, tenantID, model string) error { - tenantID = strings.TrimSpace(tenantID) +func (manager *Manager) AuthorizeModel( + ctx context.Context, + identity Identity, + model string, +) error { + tenantID := strings.TrimSpace(identity.TenantID) model = strings.TrimSpace(model) if tenantID == "" || model == "" { + manager.observeAuthorization("denied") return ErrModelNotAllowed } - - var count int64 - err := manager.database.WithContext(ctx). - Model(&postgres.TenantModelGrant{}). - Where( - "tenant_id = ? AND gateway_model IN ?", - tenantID, []string{model, "*"}, - ). - Count(&count).Error - if err != nil { - if ctx.Err() != nil { - return ctx.Err() - } - return errors.New("read tenant model grant") + if err := ctx.Err(); err != nil { + manager.observeAuthorization("canceled") + return err } - if count == 0 { + if identity.authorization == nil { + manager.observeAuthorization("denied") return ErrModelNotAllowed } - return nil + for _, allowed := range identity.authorization.models { + if allowed == "*" || allowed == model { + manager.observeAuthorization("allowed") + return nil + } + } + manager.observeAuthorization("denied") + return ErrModelNotAllowed } func (manager *Manager) AuthorizedModels( ctx context.Context, - tenantID string, + identity Identity, ) ([]string, error) { - tenantID = strings.TrimSpace(tenantID) - if tenantID == "" { + if strings.TrimSpace(identity.TenantID) == "" { return nil, ErrModelNotAllowed } - var models []string - if err := manager.database.WithContext(ctx). - Model(&postgres.TenantModelGrant{}). - Where("tenant_id = ?", tenantID). - Order("gateway_model ASC"). - Pluck("gateway_model", &models).Error; err != nil { - if ctx.Err() != nil { - return nil, ctx.Err() - } - return nil, errors.New("read tenant model grants") + if err := ctx.Err(); err != nil { + return nil, err } - return models, nil + if identity.authorization == nil { + return nil, nil + } + return append([]string(nil), identity.authorization.models...), nil } func (manager *Manager) Revoke(ctx context.Context, keyID string) error { @@ -325,10 +388,12 @@ func (manager *Manager) Revoke(ctx context.Context, keyID string) error { return errors.New("check API key status") } if count > 0 { + manager.invalidateKey(ctx, keyID) return ErrKeyRevoked } return ErrKeyNotFound } + manager.invalidateKey(ctx, keyID) return nil } @@ -358,13 +423,42 @@ func (manager *Manager) Disable(ctx context.Context, keyID string) error { return errors.New("check API key status") } if count > 0 { + manager.invalidateKey(ctx, keyID) return ErrKeyRevoked } return ErrKeyNotFound } + manager.invalidateKey(ctx, keyID) return nil } +func (manager *Manager) invalidateKey(ctx context.Context, keyID string) { + if manager.credentials != nil { + manager.credentials.InvalidateKey(ctx, keyID) + } +} + +func (manager *Manager) invalidateTenant( + ctx context.Context, + tenantID string, +) { + if manager.credentials != nil { + manager.credentials.InvalidateTenant(ctx, tenantID) + } +} + +func (manager *Manager) observeDatabaseQueries(count int) { + if manager.observer != nil { + manager.observer.AuthDatabaseQueries(count) + } +} + +func (manager *Manager) observeAuthorization(result string) { + if manager.observer != nil { + manager.observer.ModelAuthorization(result) + } +} + func (manager *Manager) createKey(transaction *gorm.DB, input CreateKeyInput) (IssuedKey, error) { token, err := generateToken(manager.pepper) if err != nil { diff --git a/internal/apikey/postgres_integration_test.go b/internal/apikey/postgres_integration_test.go index dad9e42..cda965d 100644 --- a/internal/apikey/postgres_integration_test.go +++ b/internal/apikey/postgres_integration_test.go @@ -98,10 +98,14 @@ func TestPostgresAPIKeyLifecycle(t *testing.T) { if identity.TenantID != issued.TenantID || identity.APIKeyID != issued.ID || identity.KeyPrefix != issued.Prefix { t.Errorf("Authenticate(valid key) identity = %+v, want tenant %q key %q prefix %q", identity, issued.TenantID, issued.ID, issued.Prefix) } - if err := manager.AuthorizeModel(ctx, issued.TenantID, "model-a"); err != nil { + if err := manager.AuthorizeModel(ctx, identity, "model-a"); err != nil { t.Fatalf("AuthorizeModel(allowed) error = %v", err) } - requireErrorIs(t, manager.AuthorizeModel(ctx, issued.TenantID, "model-denied"), ErrModelNotAllowed) + requireErrorIs( + t, + manager.AuthorizeModel(ctx, identity, "model-denied"), + ErrModelNotAllowed, + ) unknown, err := generateToken(pepper) if err != nil { diff --git a/internal/apikey/snapshot.go b/internal/apikey/snapshot.go new file mode 100644 index 0000000..76d5e83 --- /dev/null +++ b/internal/apikey/snapshot.go @@ -0,0 +1,194 @@ +package apikey + +import ( + "bytes" + "context" + "errors" + "strings" + "time" + + "gorm.io/gorm" + + "model-velo/internal/postgres" +) + +const authSnapshotVersion = 1 + +var errSnapshotNotFound = errors.New("authentication snapshot not found") + +type authSnapshot struct { + Version int `json:"version"` + APIKeyID string `json:"api_key_id"` + TenantID string `json:"tenant_id"` + KeyPrefix string `json:"key_prefix"` + LookupDigest []byte `json:"lookup_digest"` + KeyHash []byte `json:"key_hash"` + HashVersion int16 `json:"hash_version"` + KeyStatus postgres.APIKeyStatus `json:"key_status"` + KeyExpiresAt *time.Time `json:"key_expires_at,omitempty"` + LastUsedAt *time.Time `json:"last_used_at,omitempty"` + TenantStatus postgres.TenantStatus `json:"tenant_status"` + TenantVersion uint64 `json:"tenant_version"` + AllowedModels []string `json:"allowed_models"` + GeneratedAt time.Time `json:"generated_at"` + CachedUntil time.Time `json:"cached_until"` +} + +func (snapshot authSnapshot) identity() Identity { + return Identity{ + TenantID: snapshot.TenantID, + APIKeyID: snapshot.APIKeyID, + KeyPrefix: snapshot.KeyPrefix, + authorization: &authorizationSnapshot{ + models: append([]string(nil), snapshot.AllowedModels...), + }, + } +} + +func (snapshot *authSnapshot) validate( + expectedDigest []byte, + now time.Time, + requireCacheLifetime bool, +) error { + if snapshot.Version != authSnapshotVersion || + strings.TrimSpace(snapshot.APIKeyID) == "" || + strings.TrimSpace(snapshot.TenantID) == "" || + strings.TrimSpace(snapshot.KeyPrefix) == "" || + len(snapshot.KeyHash) != 32 || + snapshot.HashVersion != hashVersion || + snapshot.TenantVersion == 0 || + snapshot.GeneratedAt.IsZero() || + !bytes.Equal(snapshot.LookupDigest, expectedDigest) || + !bytes.Equal(digestPrefix(snapshot.KeyPrefix), expectedDigest) { + return errors.New("authentication snapshot is invalid") + } + switch snapshot.KeyStatus { + case postgres.APIKeyActive, postgres.APIKeyDisabled, postgres.APIKeyRevoked: + default: + return errors.New("authentication snapshot key status is invalid") + } + switch snapshot.TenantStatus { + case postgres.TenantActive, postgres.TenantDisabled: + default: + return errors.New("authentication snapshot tenant status is invalid") + } + models, err := normalizeModels(snapshot.AllowedModels) + if err != nil { + return errors.New("authentication snapshot model grants are invalid") + } + snapshot.AllowedModels = models + if requireCacheLifetime && (snapshot.CachedUntil.IsZero() || + !snapshot.CachedUntil.After(now)) { + return errors.New("authentication snapshot is expired") + } + return nil +} + +type snapshotSource interface { + Load( + context.Context, + []byte, + time.Time, + ) (authSnapshot, int, error) + Touch(context.Context, string, time.Time) (int, error) +} + +type postgresSnapshotSource struct { + database *gorm.DB +} + +func (source postgresSnapshotSource) Load( + ctx context.Context, + lookupDigest []byte, + now time.Time, +) (authSnapshot, int, error) { + var row struct { + APIKeyID string + TenantID string + KeyPrefix string + LookupDigest []byte + KeyHash []byte + HashVersion int16 + KeyStatus postgres.APIKeyStatus + KeyExpiresAt *time.Time + LastUsedAt *time.Time + TenantStatus postgres.TenantStatus + TenantVersion uint64 + } + err := source.database.WithContext(ctx). + Table("api_keys AS keys"). + Select( + "keys.id AS api_key_id, keys.tenant_id, keys.key_prefix, "+ + "keys.lookup_digest, keys.key_hash, keys.hash_version, "+ + "keys.status AS key_status, keys.expires_at AS key_expires_at, "+ + "keys.last_used_at, tenants.status AS tenant_status, "+ + "tenants.version AS tenant_version", + ). + Joins("JOIN tenants ON tenants.id = keys.tenant_id"). + Where("keys.lookup_digest = ?", lookupDigest). + Take(&row).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return authSnapshot{}, 1, errSnapshotNotFound + } + if err != nil { + return authSnapshot{}, 1, errors.New("read API key authentication snapshot") + } + + var models []string + err = source.database.WithContext(ctx). + Model(&postgres.TenantModelGrant{}). + Where("tenant_id = ?", row.TenantID). + Order("gateway_model ASC"). + Pluck("gateway_model", &models).Error + if err != nil { + return authSnapshot{}, 2, errors.New("read API key model grants") + } + snapshot := authSnapshot{ + Version: authSnapshotVersion, + APIKeyID: row.APIKeyID, + TenantID: row.TenantID, + KeyPrefix: row.KeyPrefix, + LookupDigest: append([]byte(nil), row.LookupDigest...), + KeyHash: append([]byte(nil), row.KeyHash...), + HashVersion: row.HashVersion, + KeyStatus: row.KeyStatus, + KeyExpiresAt: cloneTime(row.KeyExpiresAt), + LastUsedAt: cloneTime(row.LastUsedAt), + TenantStatus: row.TenantStatus, + TenantVersion: row.TenantVersion, + AllowedModels: models, + GeneratedAt: now.UTC(), + } + if err := snapshot.validate(lookupDigest, now, false); err != nil { + return authSnapshot{}, 2, err + } + return snapshot, 2, nil +} + +func (source postgresSnapshotSource) Touch( + ctx context.Context, + keyID string, + now time.Time, +) (int, error) { + err := source.database.WithContext(ctx). + Session(&gorm.Session{SkipDefaultTransaction: true}). + Model(&postgres.APIKey{}). + Where( + "id = ? AND (last_used_at IS NULL OR last_used_at < ?)", + keyID, + now.Add(-5*time.Minute), + ). + Update("last_used_at", now.UTC()).Error + if err != nil { + return 1, errors.New("update API key last used time") + } + return 1, nil +} + +func cloneTime(value *time.Time) *time.Time { + if value == nil { + return nil + } + cloned := value.UTC() + return &cloned +} diff --git a/internal/config/authcache.go b/internal/config/authcache.go new file mode 100644 index 0000000..2b64835 --- /dev/null +++ b/internal/config/authcache.go @@ -0,0 +1,123 @@ +package config + +import ( + "fmt" + "os" + "regexp" + "strconv" + "strings" + "time" +) + +const ( + authCacheEnabledEnv = "MODEL_VELO_AUTH_CACHE_ENABLED" + authCacheL1MaxEntriesEnv = "MODEL_VELO_AUTH_CACHE_L1_MAX_ENTRIES" + authCacheL1TTLEnv = "MODEL_VELO_AUTH_CACHE_L1_TTL" + authCacheL2TTLEnv = "MODEL_VELO_AUTH_CACHE_L2_TTL" + authCacheKeyPrefixEnv = "MODEL_VELO_AUTH_CACHE_KEY_PREFIX" + authCacheInvalidationChannelEnv = "MODEL_VELO_AUTH_CACHE_INVALIDATION_CHANNEL" + + defaultAuthCacheL1MaxEntries = 10_000 + defaultAuthCacheL1TTL = 15 * time.Second + defaultAuthCacheL2TTL = 30 * time.Second +) + +var authCacheNamespacePattern = regexp.MustCompile(`^[a-zA-Z0-9][a-zA-Z0-9:_.-]{0,127}$`) + +type AuthCache struct { + Enabled bool + L1MaxEntries int + L1TTL time.Duration + L2TTL time.Duration + KeyPrefix string + InvalidationChannel string +} + +func LoadAuthCache() (AuthCache, error) { + enabled, err := loadOptionalBool(authCacheEnabledEnv, true) + if err != nil { + return AuthCache{}, err + } + maxEntries, err := loadInt( + authCacheL1MaxEntriesEnv, + defaultAuthCacheL1MaxEntries, + false, + ) + if err != nil || maxEntries > 1_000_000 { + return AuthCache{}, fmt.Errorf( + "%s must be between 1 and 1000000", + authCacheL1MaxEntriesEnv, + ) + } + l1TTL, err := loadPositiveDuration(authCacheL1TTLEnv, defaultAuthCacheL1TTL) + if err != nil || l1TTL < time.Second || l1TTL > 5*time.Minute { + return AuthCache{}, fmt.Errorf( + "%s must be between 1s and 5m", + authCacheL1TTLEnv, + ) + } + l2TTL, err := loadPositiveDuration(authCacheL2TTLEnv, defaultAuthCacheL2TTL) + if err != nil || l2TTL < l1TTL || l2TTL > 10*time.Minute { + return AuthCache{}, fmt.Errorf( + "%s must be between %s and 10m", + authCacheL2TTLEnv, + l1TTL, + ) + } + + environment := strings.ToLower(strings.TrimSpace(os.Getenv(environmentEnv))) + if !environmentPattern.MatchString(environment) { + return AuthCache{}, fmt.Errorf( + "%s must contain 1 to 32 lowercase letters, digits, underscores, or hyphens", + environmentEnv, + ) + } + keyPrefix := strings.TrimSpace(os.Getenv(authCacheKeyPrefixEnv)) + if keyPrefix == "" { + keyPrefix = "model-velo:" + environment + ":auth:v1" + } + if !authCacheNamespacePattern.MatchString(keyPrefix) { + return AuthCache{}, fmt.Errorf( + "%s must contain 1 to 128 safe namespace characters", + authCacheKeyPrefixEnv, + ) + } + channel := strings.TrimSpace(os.Getenv(authCacheInvalidationChannelEnv)) + if channel == "" { + channel = keyPrefix + ":invalidate" + } + if !authCacheNamespacePattern.MatchString(channel) { + return AuthCache{}, fmt.Errorf( + "%s must contain 1 to 128 safe namespace characters", + authCacheInvalidationChannelEnv, + ) + } + if channel == keyPrefix { + return AuthCache{}, fmt.Errorf( + "%s must differ from %s", + authCacheInvalidationChannelEnv, + authCacheKeyPrefixEnv, + ) + } + + return AuthCache{ + Enabled: enabled, + L1MaxEntries: maxEntries, + L1TTL: l1TTL, + L2TTL: l2TTL, + KeyPrefix: keyPrefix, + InvalidationChannel: channel, + }, nil +} + +func loadOptionalBool(name string, fallback bool) (bool, error) { + raw := strings.TrimSpace(os.Getenv(name)) + if raw == "" { + return fallback, nil + } + value, err := strconv.ParseBool(raw) + if err != nil { + return false, fmt.Errorf("%s must be true or false", name) + } + return value, nil +} diff --git a/internal/config/authcache_test.go b/internal/config/authcache_test.go new file mode 100644 index 0000000..a49b57e --- /dev/null +++ b/internal/config/authcache_test.go @@ -0,0 +1,64 @@ +package config + +import ( + "testing" + "time" +) + +func TestLoadAuthCacheDefaultsAndValidation(t *testing.T) { + t.Setenv(environmentEnv, "test") + for _, name := range []string{ + authCacheEnabledEnv, + authCacheL1MaxEntriesEnv, + authCacheL1TTLEnv, + authCacheL2TTLEnv, + authCacheKeyPrefixEnv, + authCacheInvalidationChannelEnv, + } { + t.Setenv(name, "") + } + settings, err := LoadAuthCache() + if err != nil { + t.Fatalf("LoadAuthCache() error = %v", err) + } + if !settings.Enabled || + settings.L1MaxEntries != defaultAuthCacheL1MaxEntries || + settings.L1TTL != 15*time.Second || + settings.L2TTL != 30*time.Second || + settings.KeyPrefix != "model-velo:test:auth:v1" || + settings.InvalidationChannel != "model-velo:test:auth:v1:invalidate" { + t.Fatalf("LoadAuthCache() = %#v", settings) + } + + tests := []struct { + name string + env string + value string + }{ + {name: "bad enabled", env: authCacheEnabledEnv, value: "sometimes"}, + {name: "zero entries", env: authCacheL1MaxEntriesEnv, value: "0"}, + {name: "too many entries", env: authCacheL1MaxEntriesEnv, value: "1000001"}, + {name: "short L1", env: authCacheL1TTLEnv, value: "500ms"}, + {name: "L2 shorter than L1", env: authCacheL2TTLEnv, value: "5s"}, + {name: "unsafe prefix", env: authCacheKeyPrefixEnv, value: "auth cache"}, + {name: "same channel", env: authCacheInvalidationChannelEnv, value: "custom"}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + t.Setenv(environmentEnv, "test") + t.Setenv(authCacheEnabledEnv, "") + t.Setenv(authCacheL1MaxEntriesEnv, "") + t.Setenv(authCacheL1TTLEnv, "") + t.Setenv(authCacheL2TTLEnv, "") + t.Setenv(authCacheKeyPrefixEnv, "") + t.Setenv(authCacheInvalidationChannelEnv, "") + if test.name == "same channel" { + t.Setenv(authCacheKeyPrefixEnv, "custom") + } + t.Setenv(test.env, test.value) + if _, err := LoadAuthCache(); err == nil { + t.Fatal("LoadAuthCache() error = nil, want error") + } + }) + } +} diff --git a/internal/httpapi/auth.go b/internal/httpapi/auth.go index de24926..db760f0 100644 --- a/internal/httpapi/auth.go +++ b/internal/httpapi/auth.go @@ -5,6 +5,7 @@ import ( "errors" // 判断具体认证错误类型。 "net/http" // 读取请求头并使用 HTTP 状态码。 "strings" // 解析 Authorization 请求头。 + "time" "github.com/gin-gonic/gin" // Gin 中间件与请求上下文。 @@ -17,8 +18,8 @@ const identityGinKey = "model-velo.identity" // 身份在 Gin Context 中的存 type identityContextKey struct{} // 身份在标准 context.Context 中使用的专用键类型。 type AccessController interface { // 认证与模型授权服务接口。 - Authenticate(ctx context.Context, plaintext string) (apikey.Identity, error) // 验证网关 API Key 并返回租户身份。 - AuthorizeModel(ctx context.Context, tenantID, model string) error // 检查租户是否允许使用指定模型。 + Authenticate(ctx context.Context, plaintext string) (apikey.Identity, error) // 验证网关 API Key 并返回租户身份。 + AuthorizeModel(ctx context.Context, identity apikey.Identity, model string) error // 使用认证快照检查模型权限。 } type RateLimiter interface { // 租户限流服务接口。 @@ -44,8 +45,25 @@ func authenticationMiddleware(access AccessController) gin.HandlerFunc { // 创 return // 不再进入后续路由。 } + startedAt := time.Now() identity, authenticateErr := access.Authenticate(c.Request.Context(), plaintext) // 验证 API Key 并取得租户身份。 - if authenticateErr != nil { // API Key 验证失败。 + result := "accepted" + switch { + case c.Request.Context().Err() != nil: + result = "canceled" + case errors.Is(authenticateErr, apikey.ErrInvalidCredential), + errors.Is(authenticateErr, apikey.ErrKeyInactive), + errors.Is(authenticateErr, apikey.ErrKeyRevoked), + errors.Is(authenticateErr, apikey.ErrKeyExpired), + errors.Is(authenticateErr, apikey.ErrTenantInactive): + result = "rejected" + case authenticateErr != nil: + result = "error" + } + routerMetrics(c).RequestStage( + "authentication", result, "", time.Since(startedAt), + ) + if authenticateErr != nil { // API Key 验证失败。 if c.Request.Context().Err() != nil { // 请求已经取消时不再写响应。 return } diff --git a/internal/httpapi/auth_test.go b/internal/httpapi/auth_test.go index 0fdbfba..7dfe298 100644 --- a/internal/httpapi/auth_test.go +++ b/internal/httpapi/auth_test.go @@ -150,7 +150,11 @@ func (access *middlewareAccessController) Authenticate(context.Context, string) return access.identity, access.authenticateErr } -func (*middlewareAccessController) AuthorizeModel(context.Context, string, string) error { +func (*middlewareAccessController) AuthorizeModel( + context.Context, + apikey.Identity, + string, +) error { return nil } diff --git a/internal/httpapi/chat.go b/internal/httpapi/chat.go index f0835d3..cb49c79 100644 --- a/internal/httpapi/chat.go +++ b/internal/httpapi/chat.go @@ -163,15 +163,29 @@ func (h chatHandler) complete(c *gin.Context) { return } - if err := h.access.AuthorizeModel( // 检查该租户是否允许使用当前模型。 + startedAt := time.Now() + authorizationErr := h.access.AuthorizeModel( // 检查该租户是否允许使用当前模型。 c.Request.Context(), - identity.TenantID, + identity, model, - ); err != nil { + ) + authorizationResult := "allowed" + switch { + case c.Request.Context().Err() != nil: + authorizationResult = "canceled" + case errors.Is(authorizationErr, apikey.ErrModelNotAllowed): + authorizationResult = "denied" + case authorizationErr != nil: + authorizationResult = "error" + } + h.metrics.RequestStage( + "authorization", authorizationResult, "", time.Since(startedAt), + ) + if authorizationErr != nil { if c.Request.Context().Err() != nil { // 请求已经取消时不再写响应。 return } - if errors.Is(err, apikey.ErrModelNotAllowed) { // API Key 没有模型权限。 + if errors.Is(authorizationErr, apikey.ErrModelNotAllowed) { // API Key 没有模型权限。 writeAPIError( c, http.StatusForbidden, // 返回 403。 @@ -194,7 +208,22 @@ func (h chatHandler) complete(c *gin.Context) { return } + startedAt = time.Now() limitDecision, err := h.limiter.Allow(c.Request.Context(), identity.TenantID, model) // 查询租户和模型的限流额度。 + limitResult := "allowed" + switch { + case c.Request.Context().Err() != nil: + limitResult = "canceled" + case err != nil: + limitResult = "error" + case !limitDecision.Allowed: + limitResult = "rejected" + case limitDecision.Bypassed: + limitResult = "bypassed" + } + h.metrics.RequestStage( + "rate_limit", limitResult, "", time.Since(startedAt), + ) if err != nil { if c.Request.Context().Err() != nil { // 请求已经取消。 return @@ -257,7 +286,15 @@ func (h chatHandler) complete(c *gin.Context) { c.Request.Context(), active.CacheNamespace, )) + startedAt = time.Now() routePlan, err := active.Routes.Plan(model, requiredCapabilities) // 生成符合模型和能力要求的候选 Provider。 + routeResult := "planned" + if err != nil { + routeResult = "error" + } + h.metrics.RequestStage( + "route_plan", routeResult, "", time.Since(startedAt), + ) if err != nil { if errors.Is(err, routing.ErrCapabilityUnavailable) { // 没有 Provider 支持请求能力。 writeAPIError( @@ -306,7 +343,15 @@ func (h chatHandler) complete(c *gin.Context) { cacheResult := responsecache.Result{Status: responsecache.StatusBypass} // 默认标记为绕过缓存。 if !requestBypassesResponseCache(c.Request) { // 请求头没有 no-store 时才查询缓存。 + startedAt = time.Now() cacheResult, err = h.cache.Lookup(c.Request.Context(), identity.TenantID, model, requestBody) // 用租户、模型和请求体查缓存。 + cacheLookupResult := string(cacheResult.Status) + if err != nil { + cacheLookupResult = "error" + } + h.metrics.RequestStage( + "cache_lookup", cacheLookupResult, "", time.Since(startedAt), + ) if err != nil { if c.Request.Context().Err() != nil { // 请求取消时停止处理。 return @@ -349,11 +394,19 @@ func (h chatHandler) complete(c *gin.Context) { h.metrics.Cache("lookup", string(cacheResult.Status)) usageSession.setCacheStatus(string(cacheResult.Status)) + startedAt = time.Now() execution, failure := active.Chat.Execute(c.Request.Context(), reliability.ExecutionInput{ // 执行 Provider 调用和回退。 RequestID: requestIDFromContext(c.Request.Context()), // 当前网关请求 ID。 Request: request, // 解析后的聊天请求。 Plan: routePlan, // 路由候选计划。 }) + executionResult := "success" + if failure != nil { + executionResult = string(failure.Category) + } + h.metrics.RequestStage( + "reliability", executionResult, "", time.Since(startedAt), + ) if failure != nil { h.observeFailure(requestIDFromContext(c.Request.Context()), failure) usageSession.recordFailure(failure) @@ -379,13 +432,22 @@ func (h chatHandler) complete(c *gin.Context) { usageSession.recordExecution(execution) usageSession.observe(responseBody) if cacheResult.Status == responsecache.StatusMiss && execution.Fallbacks == 0 { // 只有缓存未命中且未切换 Provider 时才缓存。 - if err := h.cache.Store( + startedAt = time.Now() + err := h.cache.Store( c.Request.Context(), identity.TenantID, model, requestBody, responseBody, - ); err != nil { + ) + cacheStoreResult := "stored" + if err != nil { + cacheStoreResult = "error" + } + h.metrics.RequestStage( + "cache_store", cacheStoreResult, "", time.Since(startedAt), + ) + if err != nil { if c.Request.Context().Err() != nil { // 请求取消时停止。 return } @@ -433,6 +495,11 @@ func (h chatHandler) reserveQuota( if h.quota == nil { return true } + if !h.quota.HasPolicy(tenantID, model) { + h.metrics.RequestStage("quota_reserve", "no_policy", "", 0) + h.metrics.Quota("no_policy") + return true + } output := h.quota.DefaultMaxOutputTokens() if request.MaxCompletionTokens != nil { output = int64(*request.MaxCompletionTokens) @@ -442,6 +509,7 @@ func (h chatHandler) reserveQuota( if request.N != nil && *request.N > 1 { output *= int64(*request.N) } + startedAt := time.Now() decision, err := h.quota.Reserve(c.Request.Context(), quota.ReserveInput{ GroupID: session.eventID, TenantID: tenantID, @@ -450,6 +518,24 @@ func (h chatHandler) reserveQuota( EstimatedOutputTokens: output, Plan: plan, }) + result := "allowed" + switch { + case c.Request.Context().Err() != nil: + result = "canceled" + case errors.Is(err, quota.ErrExceeded): + result = "denied" + case err != nil: + result = "error" + case decision.Exceeded: + result = "allowed_overage" + case len(decision.Alerts) > 0: + result = "allowed_alert" + case decision.AppliedPolicies == 0: + result = "no_policy" + } + h.metrics.RequestStage( + "quota_reserve", result, "", time.Since(startedAt), + ) if err != nil { if errors.Is(err, quota.ErrExceeded) { h.metrics.Quota("denied") @@ -598,11 +684,19 @@ func (h chatHandler) stream( return } + startedAt := time.Now() prepared, failure := orchestrator.OpenStream(c.Request.Context(), reliability.ExecutionInput{ // 打开上游流并取得首个事件。 RequestID: requestIDFromContext(c.Request.Context()), Request: request, Plan: plan, }) + result := "success" + if failure != nil { + result = string(failure.Category) + } + h.metrics.RequestStage( + "reliability", result, "", time.Since(startedAt), + ) if failure != nil { h.observeFailure(requestIDFromContext(c.Request.Context()), failure) usageSession.recordFailure(failure) diff --git a/internal/httpapi/chat_test.go b/internal/httpapi/chat_test.go index 698d8bf..248ccba 100644 --- a/internal/httpapi/chat_test.go +++ b/internal/httpapi/chat_test.go @@ -158,8 +158,11 @@ func TestChatCompletionUsageLifecycle(t *testing.T) { } event := singleUsageEvent(t, emitter) + if err := event.Validate(); err != nil { + t.Fatalf("cache usage event validation error = %v", err) + } if event.Status != usage.StatusCacheHit || - event.CacheStatus != string(responsecache.StatusHit) || + event.CacheStatus != "hit" || event.UsageSource != usage.UsageSourceCacheReplay || event.ProviderID != "" || event.Attempts != 0 || @@ -236,11 +239,14 @@ func TestChatCompletionUsageLifecycle(t *testing.T) { } event := singleUsageEvent(t, emitter) + if err := event.Validate(); err != nil { + t.Fatalf("stream usage event validation error = %v", err) + } if event.Status != usage.StatusStreamCompleted || !event.Stream || event.FirstTokenMS == nil || event.UsageSource != usage.UsageSourceProvider || - event.CacheStatus != string(responsecache.StatusBypass) || + event.CacheStatus != "bypass" || event.Usage == nil || event.Usage.Total != 7 { t.Fatalf("stream usage event = %#v", event) diff --git a/internal/httpapi/embeddings.go b/internal/httpapi/embeddings.go index cb9e9ec..ecd21af 100644 --- a/internal/httpapi/embeddings.go +++ b/internal/httpapi/embeddings.go @@ -95,7 +95,7 @@ func (h embeddingHandler) create(c *gin.Context) { } defer session.finish(c) - if err := h.access.AuthorizeModel(c.Request.Context(), identity.TenantID, model); err != nil { + if err := h.access.AuthorizeModel(c.Request.Context(), identity, model); err != nil { if errors.Is(err, apikey.ErrModelNotAllowed) { writeAPIError( c, http.StatusForbidden, @@ -246,6 +246,11 @@ func (h embeddingHandler) reserveQuota( if h.quota == nil { return true } + if !h.quota.HasPolicy(tenantID, model) { + h.metrics.RequestStage("quota_reserve", "no_policy", "", 0) + h.metrics.Quota("no_policy") + return true + } decision, err := h.quota.Reserve(c.Request.Context(), quota.ReserveInput{ GroupID: session.eventID, TenantID: tenantID, GatewayModel: model, EstimatedInputTokens: estimateEmbeddingTokens(request.Input), diff --git a/internal/httpapi/models.go b/internal/httpapi/models.go index 856acb1..14f3686 100644 --- a/internal/httpapi/models.go +++ b/internal/httpapi/models.go @@ -19,7 +19,7 @@ type modelHandler struct { } type authorizedModelLister interface { - AuthorizedModels(context.Context, string) ([]string, error) + AuthorizedModels(context.Context, apikey.Identity) ([]string, error) } func (handler modelHandler) list(c *gin.Context) { @@ -37,7 +37,7 @@ func (handler modelHandler) list(c *gin.Context) { writeIdentityUnavailable(c) return } - allowed, err := handler.allowedModels(c.Request.Context(), identity.TenantID, models) + allowed, err := handler.allowedModels(c.Request.Context(), identity, models) if err != nil { writeAPIError( c, http.StatusServiceUnavailable, "model catalog authorization is unavailable", @@ -61,7 +61,7 @@ func (handler modelHandler) get(c *gin.Context) { return } if err := handler.access.AuthorizeModel( - c.Request.Context(), identity.TenantID, model, + c.Request.Context(), identity, model, ); err != nil { if errors.Is(err, apikey.ErrModelNotAllowed) { writeAPIError( @@ -104,11 +104,11 @@ func (handler modelHandler) get(c *gin.Context) { func (handler modelHandler) allowedModels( ctx context.Context, - tenantID string, + identity apikey.Identity, configured []string, ) ([]string, error) { if lister, ok := handler.access.(authorizedModelLister); ok { - granted, err := lister.AuthorizedModels(ctx, tenantID) + granted, err := lister.AuthorizedModels(ctx, identity) if err != nil { return nil, err } @@ -129,7 +129,7 @@ func (handler modelHandler) allowedModels( } result := make([]string, 0, len(configured)) for _, model := range configured { - if err := handler.access.AuthorizeModel(ctx, tenantID, model); err == nil { + if err := handler.access.AuthorizeModel(ctx, identity, model); err == nil { result = append(result, model) } else if !errors.Is(err, apikey.ErrModelNotAllowed) { return nil, err diff --git a/internal/httpapi/router_options.go b/internal/httpapi/router_options.go index aeccafc..13eb3be 100644 --- a/internal/httpapi/router_options.go +++ b/internal/httpapi/router_options.go @@ -20,6 +20,7 @@ import ( "model-velo/internal/gateway" "model-velo/internal/observability" "model-velo/internal/quota" + "model-velo/internal/reliability" ) type ReadinessChecker interface { @@ -104,6 +105,9 @@ func requestSummaryMiddleware( c.Header("request-id", requestIDFromContext(c.Request.Context())) } c.Set("model-velo.metrics", metrics) + c.Request = c.Request.WithContext( + reliability.WithQueueObserver(c.Request.Context(), metrics), + ) finishMetrics := metrics.BeginRequest() c.Next() @@ -116,6 +120,7 @@ func requestSummaryMiddleware( status := c.Writer.Status() duration := time.Since(startedAt) finishMetrics(route, c.Request.Method, status, isStream, duration) + metrics.HTTPError(route, status, apiErrorCode(c)) attributes := []any{ "request_id", requestIDFromContext(c.Request.Context()), diff --git a/internal/httpapi/router_test.go b/internal/httpapi/router_test.go index 14a39fc..c0c49ff 100644 --- a/internal/httpapi/router_test.go +++ b/internal/httpapi/router_test.go @@ -383,9 +383,13 @@ func (access testAccessController) Authenticate(context.Context, string) (apikey return apikey.Identity{TenantID: "tenant-test-id", APIKeyID: "api-key-test-id", KeyPrefix: "mvl_test"}, nil } -func (access testAccessController) AuthorizeModel(_ context.Context, tenantID, model string) error { +func (access testAccessController) AuthorizeModel( + _ context.Context, + identity apikey.Identity, + model string, +) error { if access.onAuthorize != nil { - access.onAuthorize(tenantID, model) + access.onAuthorize(identity.TenantID, model) } return access.authorizeErr } diff --git a/internal/httpapi/usage.go b/internal/httpapi/usage.go index 38672a6..e73cc55 100644 --- a/internal/httpapi/usage.go +++ b/internal/httpapi/usage.go @@ -36,6 +36,14 @@ func newUsageSession( stream bool, metrics *observability.Metrics, ) (*usageSession, error) { + startedAt := time.Now() + stageResult := "error" + defer func() { + metrics.RequestStage( + "usage_begin", stageResult, "", time.Since(startedAt), + ) + }() + collector, err := usage.NewCollector(usage.NewEventInput{ RequestID: requestIDFromContext(ctx), TenantID: tenantID, @@ -57,6 +65,7 @@ func newUsageSession( } metrics.UsageDelivery("lifecycle_begin", "success") } + stageResult = "success" return &usageSession{ collector: collector, emitter: emitter, eventID: pending.EventID, metrics: metrics, @@ -94,9 +103,13 @@ func (session *usageSession) finish(c *gin.Context) { settleContext, cancel := context.WithTimeout( context.WithoutCancel(c.Request.Context()), 2*time.Second, ) - if err := session.quota.Settle( + startedAt := time.Now() + err := session.quota.Settle( settleContext, session.reservationID, session.event, - ); err != nil { + ) + result := "success" + if err != nil { + result = "error" slog.Error( "quota settlement failed", "request_id", session.event.RequestID, @@ -104,19 +117,34 @@ func (session *usageSession) finish(c *gin.Context) { "error", err, ) } + session.metrics.RequestStage( + "quota_settle", result, "", time.Since(startedAt), + ) cancel() } - if _, err := session.emitter.Emit(c.Request.Context(), session.event); err != nil { + startedAt := time.Now() + entryID, err := session.emitter.Emit(c.Request.Context(), session.event) + result := "queued" + if entryID != "" { + result = "published" + } + if err != nil { + result = "deferred" + } + session.metrics.RequestStage( + "usage_finalize", result, "", time.Since(startedAt), + ) + if err != nil { session.metrics.UsageDelivery("finalize", "deferred") slog.Warn( - "usage immediate publish deferred", + "usage finalization deferred", "request_id", session.event.RequestID, "event_id", session.event.EventID, "error", err, ) return } - session.metrics.UsageDelivery("finalize", "published") + session.metrics.UsageDelivery("finalize", result) } func (session *usageSession) attachQuota( diff --git a/internal/observability/dependencies.go b/internal/observability/dependencies.go new file mode 100644 index 0000000..30f1a43 --- /dev/null +++ b/internal/observability/dependencies.go @@ -0,0 +1,145 @@ +package observability + +import ( + "database/sql" + + "github.com/prometheus/client_golang/prometheus" + goredis "github.com/redis/go-redis/v9" +) + +type dependencyCollector struct { + database *sql.DB + redis *goredis.Client + + postgresConnections *prometheus.Desc + postgresWaits *prometheus.Desc + postgresWaitTime *prometheus.Desc + redisConnections *prometheus.Desc + redisPoolEvents *prometheus.Desc + redisWaitTime *prometheus.Desc +} + +func (metrics *Metrics) RegisterDependencies( + database *sql.DB, + redis *goredis.Client, +) error { + if metrics == nil || (database == nil && redis == nil) { + return nil + } + return metrics.registry.Register(newDependencyCollector(database, redis)) +} + +func newDependencyCollector( + database *sql.DB, + redis *goredis.Client, +) *dependencyCollector { + return &dependencyCollector{ + database: database, + redis: redis, + postgresConnections: prometheus.NewDesc( + "model_velo_postgres_connections", + "Current PostgreSQL connection-pool state.", + []string{"state"}, nil, + ), + postgresWaits: prometheus.NewDesc( + "model_velo_postgres_waits_total", + "PostgreSQL connection-pool waits.", + nil, nil, + ), + postgresWaitTime: prometheus.NewDesc( + "model_velo_postgres_wait_duration_seconds_total", + "Total time waiting for a PostgreSQL connection.", + nil, nil, + ), + redisConnections: prometheus.NewDesc( + "model_velo_redis_pool_connections", + "Current Redis connection-pool state.", + []string{"state"}, nil, + ), + redisPoolEvents: prometheus.NewDesc( + "model_velo_redis_pool_events_total", + "Redis connection-pool cumulative outcomes.", + []string{"event"}, nil, + ), + redisWaitTime: prometheus.NewDesc( + "model_velo_redis_pool_wait_duration_seconds_total", + "Total time waiting for a Redis connection.", + nil, nil, + ), + } +} + +func (collector *dependencyCollector) Describe(output chan<- *prometheus.Desc) { + output <- collector.postgresConnections + output <- collector.postgresWaits + output <- collector.postgresWaitTime + output <- collector.redisConnections + output <- collector.redisPoolEvents + output <- collector.redisWaitTime +} + +func (collector *dependencyCollector) Collect(output chan<- prometheus.Metric) { + if collector.database != nil { + stats := collector.database.Stats() + for state, value := range map[string]int{ + "open": stats.OpenConnections, + "in_use": stats.InUse, + "idle": stats.Idle, + "max_open": stats.MaxOpenConnections, + } { + output <- prometheus.MustNewConstMetric( + collector.postgresConnections, + prometheus.GaugeValue, + float64(value), + state, + ) + } + output <- prometheus.MustNewConstMetric( + collector.postgresWaits, + prometheus.CounterValue, + float64(stats.WaitCount), + ) + output <- prometheus.MustNewConstMetric( + collector.postgresWaitTime, + prometheus.CounterValue, + stats.WaitDuration.Seconds(), + ) + } + + if collector.redis == nil { + return + } + stats := collector.redis.PoolStats() + for state, value := range map[string]uint32{ + "total": stats.TotalConns, + "idle": stats.IdleConns, + "pending": stats.PendingRequests, + } { + output <- prometheus.MustNewConstMetric( + collector.redisConnections, + prometheus.GaugeValue, + float64(value), + state, + ) + } + for event, value := range map[string]uint32{ + "hit": stats.Hits, + "miss": stats.Misses, + "timeout": stats.Timeouts, + "wait": stats.WaitCount, + "unusable": stats.Unusable, + "stale": stats.StaleConns, + } { + output <- prometheus.MustNewConstMetric( + collector.redisPoolEvents, + prometheus.CounterValue, + float64(value), + event, + ) + } + output <- prometheus.MustNewConstMetric( + collector.redisWaitTime, + prometheus.CounterValue, + float64(stats.WaitDurationNs)/1e9, + ) +} diff --git a/internal/observability/metrics.go b/internal/observability/metrics.go index 847f7a6..b3d4417 100644 --- a/internal/observability/metrics.go +++ b/internal/observability/metrics.go @@ -16,19 +16,27 @@ import ( ) type Metrics struct { - registry *prometheus.Registry - requests *prometheus.CounterVec - requestDuration *prometheus.HistogramVec - inFlight prometheus.Gauge - providerAttempts *prometheus.CounterVec - providerDuration *prometheus.HistogramVec - retries *prometheus.CounterVec - fallbacks *prometheus.CounterVec - cache *prometheus.CounterVec - rateLimit *prometheus.CounterVec - auth *prometheus.CounterVec - usageDelivery *prometheus.CounterVec - quota *prometheus.CounterVec + registry *prometheus.Registry + requests *prometheus.CounterVec + requestDuration *prometheus.HistogramVec + requestErrors *prometheus.CounterVec + stageDuration *prometheus.HistogramVec + inFlight prometheus.Gauge + providerAttempts *prometheus.CounterVec + providerDuration *prometheus.HistogramVec + retries *prometheus.CounterVec + fallbacks *prometheus.CounterVec + cache *prometheus.CounterVec + rateLimit *prometheus.CounterVec + auth *prometheus.CounterVec + authCache *prometheus.CounterVec + authCacheDuration *prometheus.HistogramVec + authCacheEvents *prometheus.CounterVec + authFallback *prometheus.CounterVec + authDBQueries prometheus.Histogram + authorization *prometheus.CounterVec + usageDelivery *prometheus.CounterVec + quota *prometheus.CounterVec } func NewMetrics() *Metrics { @@ -43,6 +51,18 @@ func NewMetrics() *Metrics { Help: "End-to-end HTTP request duration.", Buckets: []float64{0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 30, 60, 120, 300}, }, []string{"route", "method", "status", "stream"}), + requestErrors: prometheus.NewCounterVec(prometheus.CounterOpts{ + Name: "model_velo_http_errors_total", + Help: "Completed HTTP errors by stable gateway error code.", + }, []string{"route", "status", "code"}), + stageDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{ + Name: "model_velo_request_stage_duration_seconds", + Help: "Duration of bounded gateway request stages.", + Buckets: []float64{ + 0.0001, 0.00025, 0.0005, 0.001, 0.0025, 0.005, + 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, + }, + }, []string{"stage", "result", "provider"}), inFlight: prometheus.NewGauge(prometheus.GaugeOpts{ Name: "model_velo_http_in_flight", Help: "Current in-flight HTTP requests.", @@ -76,6 +96,35 @@ func NewMetrics() *Metrics { Name: "model_velo_authentication_total", Help: "Gateway authentication outcomes.", }, []string{"result"}), + authCache: prometheus.NewCounterVec(prometheus.CounterOpts{ + Name: "model_velo_auth_cache_lookups_total", + Help: "Authentication cache lookups by bounded layer and outcome.", + }, []string{"layer", "result"}), + authCacheDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{ + Name: "model_velo_auth_cache_lookup_duration_seconds", + Help: "Authentication cache lookup duration by bounded layer and outcome.", + Buckets: []float64{ + 0.00001, 0.000025, 0.00005, 0.0001, 0.00025, + 0.0005, 0.001, 0.0025, 0.005, 0.01, 0.025, + }, + }, []string{"layer", "result"}), + authCacheEvents: prometheus.NewCounterVec(prometheus.CounterOpts{ + Name: "model_velo_auth_cache_events_total", + Help: "Authentication cache writes, invalidations, evictions, and subscription outcomes.", + }, []string{"event", "result"}), + authFallback: prometheus.NewCounterVec(prometheus.CounterOpts{ + Name: "model_velo_auth_postgres_fallback_total", + Help: "Authentication snapshot PostgreSQL fallback outcomes.", + }, []string{"result"}), + authDBQueries: prometheus.NewHistogram(prometheus.HistogramOpts{ + Name: "model_velo_auth_postgres_queries", + Help: "Physical PostgreSQL queries performed by one authentication request.", + Buckets: []float64{0, 1, 2, 3, 4, 5}, + }), + authorization: prometheus.NewCounterVec(prometheus.CounterOpts{ + Name: "model_velo_model_authorization_total", + Help: "Model authorization decisions made from authentication snapshots.", + }, []string{"result"}), usageDelivery: prometheus.NewCounterVec(prometheus.CounterOpts{ Name: "model_velo_usage_delivery_total", Help: "Usage delivery outcomes.", @@ -88,6 +137,8 @@ func NewMetrics() *Metrics { metrics.registry.MustRegister( metrics.requests, metrics.requestDuration, + metrics.requestErrors, + metrics.stageDuration, metrics.inFlight, metrics.providerAttempts, metrics.providerDuration, @@ -96,8 +147,16 @@ func NewMetrics() *Metrics { metrics.cache, metrics.rateLimit, metrics.auth, + metrics.authCache, + metrics.authCacheDuration, + metrics.authCacheEvents, + metrics.authFallback, + metrics.authDBQueries, + metrics.authorization, metrics.usageDelivery, metrics.quota, + prometheus.NewGoCollector(), + prometheus.NewProcessCollector(prometheus.ProcessCollectorOpts{}), ) return metrics } @@ -163,6 +222,70 @@ func (metrics *Metrics) Authentication(result string) { } } +func (metrics *Metrics) AuthCacheLookup( + layer string, + result string, + duration time.Duration, +) { + if metrics == nil { + return + } + metrics.authCache.WithLabelValues(layer, result).Inc() + metrics.authCacheDuration.WithLabelValues(layer, result). + Observe(duration.Seconds()) +} + +func (metrics *Metrics) AuthCacheEvent(event, result string) { + if metrics != nil { + metrics.authCacheEvents.WithLabelValues(event, result).Inc() + } +} + +func (metrics *Metrics) AuthPostgresFallback(result string) { + if metrics != nil { + metrics.authFallback.WithLabelValues(result).Inc() + } +} + +func (metrics *Metrics) AuthDatabaseQueries(count int) { + if metrics != nil { + metrics.authDBQueries.Observe(float64(count)) + } +} + +func (metrics *Metrics) ModelAuthorization(result string) { + if metrics != nil { + metrics.authorization.WithLabelValues(result).Inc() + } +} + +func (metrics *Metrics) HTTPError(route string, status int, code string) { + if metrics == nil || status < http.StatusBadRequest { + return + } + if code == "" { + code = "unclassified" + } + metrics.requestErrors.WithLabelValues(route, strconv.Itoa(status), code).Inc() +} + +func (metrics *Metrics) RequestStage( + stage, result, provider string, + duration time.Duration, +) { + if metrics != nil { + metrics.stageDuration.WithLabelValues(stage, result, provider). + Observe(duration.Seconds()) + } +} + +func (metrics *Metrics) ObserveQueueWait( + provider, result string, + duration time.Duration, +) { + metrics.RequestStage("provider_queue", result, provider, duration) +} + func (metrics *Metrics) RateLimit(result string) { if metrics != nil { metrics.rateLimit.WithLabelValues(result).Inc() @@ -185,6 +308,7 @@ func (metrics *Metrics) ProviderAttempt( } metrics.providerAttempts.WithLabelValues(provider, result, category).Inc() metrics.providerDuration.WithLabelValues(provider, result).Observe(duration.Seconds()) + metrics.RequestStage("provider_call", result, provider, duration) if retry { metrics.retries.WithLabelValues(provider, category).Inc() } diff --git a/internal/observability/observability_test.go b/internal/observability/observability_test.go index 147bdf3..0acbd3a 100644 --- a/internal/observability/observability_test.go +++ b/internal/observability/observability_test.go @@ -2,12 +2,15 @@ package observability import ( "context" + "database/sql" "net/http" "net/http/httptest" "strings" "testing" "time" + goredis "github.com/redis/go-redis/v9" + "model-velo/internal/usage" ) @@ -16,9 +19,24 @@ func TestMetricsAreBoundedAndProtected(t *testing.T) { if err := metrics.RegisterUsageWorker(fakeUsageWorker{}); err != nil { t.Fatal(err) } + redisClient := goredis.NewClient(&goredis.Options{Addr: "127.0.0.1:1"}) + t.Cleanup(func() { + _ = redisClient.Close() + }) + if err := metrics.RegisterDependencies(&sql.DB{}, redisClient); err != nil { + t.Fatal(err) + } finish := metrics.BeginRequest() finish("/v1/test", http.MethodPost, http.StatusOK, false, 20*time.Millisecond) + metrics.HTTPError("/v1/test", http.StatusServiceUnavailable, "gateway_overloaded") + metrics.RequestStage("authentication", "accepted", "", time.Millisecond) metrics.Authentication("accepted") + metrics.AuthCacheLookup("l1", "hit", time.Microsecond) + metrics.AuthCacheLookup("l2", "miss", time.Millisecond) + metrics.AuthCacheEvent("write", "success") + metrics.AuthPostgresFallback("success") + metrics.AuthDatabaseQueries(2) + metrics.ModelAuthorization("allowed") metrics.RateLimit("allowed") metrics.Cache("lookup", "hit") metrics.ProviderAttempt("provider-a", "success", "", time.Millisecond, false) @@ -53,11 +71,22 @@ func TestMetricsAreBoundedAndProtected(t *testing.T) { payload := response.Body.String() for _, metricName := range []string{ "model_velo_http_requests_total", + "model_velo_http_errors_total", + "model_velo_request_stage_duration_seconds", "model_velo_provider_attempts_total", "model_velo_authentication_total", + "model_velo_auth_cache_lookups_total", + "model_velo_auth_cache_lookup_duration_seconds", + "model_velo_auth_cache_events_total", + "model_velo_auth_postgres_fallback_total", + "model_velo_auth_postgres_queries", + "model_velo_model_authorization_total", "model_velo_usage_delivery_total", "model_velo_usage_worker_pending", "model_velo_quota_decisions_total", + "model_velo_postgres_connections", + "model_velo_redis_pool_connections", + "go_goroutines", } { if !strings.Contains(payload, metricName) { t.Errorf("scrape does not contain %s", metricName) diff --git a/internal/quota/manager.go b/internal/quota/manager.go index 9636f64..ef6b27c 100644 --- a/internal/quota/manager.go +++ b/internal/quota/manager.go @@ -100,6 +100,7 @@ type Manager struct { pricing Quoter settings config.Quota now func() time.Time + policies *policyIndex } func NewManager( @@ -115,7 +116,8 @@ func NewManager( return nil, errors.New("quota manager settings are invalid") } return &Manager{ - database: database, pricing: pricing, settings: settings, now: time.Now, + database: database, pricing: pricing, settings: settings, + now: time.Now, policies: newPolicyIndex(), }, nil } @@ -181,6 +183,7 @@ func (manager *Manager) createPolicy( if err != nil { return PolicyView{}, err } + manager.policies.Put(policy) return view, nil } @@ -252,7 +255,11 @@ func (manager *Manager) updatePolicy( } return nil }) - return policyView(updated), err + if err != nil { + return PolicyView{}, err + } + manager.policies.Put(updated) + return policyView(updated), nil } func ensureTenant(transaction *gorm.DB, tenantID string) error { @@ -364,6 +371,9 @@ func (manager *Manager) Reserve( input.EstimatedInputTokens < 0 || input.EstimatedOutputTokens < 0 { return Decision{}, errors.New("quota reservation input is invalid") } + if !manager.HasPolicy(input.TenantID, input.GatewayModel) { + return Decision{}, nil + } totalTokens, ok := add(input.EstimatedInputTokens, input.EstimatedOutputTokens) if !ok { return Decision{}, errors.New("quota token estimate overflow") diff --git a/internal/quota/manager_test.go b/internal/quota/manager_test.go index 7f72d9a..6eea307 100644 --- a/internal/quota/manager_test.go +++ b/internal/quota/manager_test.go @@ -1,6 +1,7 @@ package quota import ( + "context" "errors" "testing" "time" @@ -8,6 +9,36 @@ import ( "model-velo/internal/postgres" ) +func TestReserveSkipsDatabaseWithoutEnabledPolicy(t *testing.T) { + manager := &Manager{policies: newPolicyIndex()} + decision, err := manager.Reserve(context.Background(), ReserveInput{ + GroupID: "request-1", TenantID: "tenant-1", + GatewayModel: "model-a", + }) + if err != nil { + t.Fatalf("Reserve(no policy) error = %v", err) + } + if decision.ReservationID != "" || + decision.AppliedPolicies != 0 { + t.Fatalf("Reserve(no policy) decision = %#v", decision) + } + + manager.policies.Put(postgres.TenantQuotaPolicy{ + ID: "policy-disabled", TenantID: "tenant-1", + GatewayModel: "*", Enabled: false, + }) + if manager.HasPolicy("tenant-1", "model-a") { + t.Fatal("disabled quota policy was added to the hot-path index") + } + manager.policies.Put(postgres.TenantQuotaPolicy{ + ID: "policy-enabled", TenantID: "tenant-1", + GatewayModel: "*", Enabled: true, + }) + if !manager.HasPolicy("tenant-1", "model-a") { + t.Fatal("enabled wildcard quota policy was absent from the index") + } +} + func TestPolicyWindowsAndLimits(t *testing.T) { requestLimit := int64(2) tokenLimit := int64(100) diff --git a/internal/quota/policy_index.go b/internal/quota/policy_index.go new file mode 100644 index 0000000..ad6e7c1 --- /dev/null +++ b/internal/quota/policy_index.go @@ -0,0 +1,123 @@ +package quota + +import ( + "context" + "errors" + "strings" + "sync" + "time" + + "model-velo/internal/postgres" +) + +type policyIndex struct { + mu sync.RWMutex + policies map[string]postgres.TenantQuotaPolicy + matches map[string]map[string]int +} + +func newPolicyIndex() *policyIndex { + return &policyIndex{ + policies: make(map[string]postgres.TenantQuotaPolicy), + matches: make(map[string]map[string]int), + } +} + +func (index *policyIndex) Replace( + policies []postgres.TenantQuotaPolicy, +) { + next := newPolicyIndex() + for _, policy := range policies { + next.put(policy) + } + index.mu.Lock() + index.policies = next.policies + index.matches = next.matches + index.mu.Unlock() +} + +func (index *policyIndex) Put(policy postgres.TenantQuotaPolicy) { + index.mu.Lock() + defer index.mu.Unlock() + index.remove(policy.ID) + index.put(policy) +} + +func (index *policyIndex) Has(tenantID, model string) bool { + index.mu.RLock() + defer index.mu.RUnlock() + models := index.matches[tenantID] + return models[model] > 0 || models["*"] > 0 +} + +func (index *policyIndex) put(policy postgres.TenantQuotaPolicy) { + index.policies[policy.ID] = policy + if !policy.Enabled { + return + } + models := index.matches[policy.TenantID] + if models == nil { + models = make(map[string]int) + index.matches[policy.TenantID] = models + } + models[policy.GatewayModel]++ +} + +func (index *policyIndex) remove(policyID string) { + current, ok := index.policies[policyID] + if !ok { + return + } + delete(index.policies, policyID) + if !current.Enabled { + return + } + models := index.matches[current.TenantID] + models[current.GatewayModel]-- + if models[current.GatewayModel] <= 0 { + delete(models, current.GatewayModel) + } + if len(models) == 0 { + delete(index.matches, current.TenantID) + } +} + +func (manager *Manager) LoadPolicyIndex(ctx context.Context) error { + var policies []postgres.TenantQuotaPolicy + if err := manager.database.WithContext(ctx). + Where("enabled = ?", true). + Find(&policies).Error; err != nil { + return errors.New("load quota policy index") + } + manager.policies.Replace(policies) + return nil +} + +func (manager *Manager) RunPolicyIndexRefresh( + ctx context.Context, + interval time.Duration, +) { + if interval <= 0 { + return + } + ticker := time.NewTicker(interval) + defer ticker.Stop() + for { + select { + case <-ctx.Done(): + return + case <-ticker.C: + _ = manager.LoadPolicyIndex(ctx) + } + } +} + +func (manager *Manager) HasPolicy(tenantID, model string) bool { + if manager == nil || manager.policies == nil { + return false + } + return manager.policies.Has( + strings.TrimSpace(tenantID), + strings.TrimSpace(model), + ) +} diff --git a/internal/reliability/queue_observer.go b/internal/reliability/queue_observer.go new file mode 100644 index 0000000..75eec4c --- /dev/null +++ b/internal/reliability/queue_observer.go @@ -0,0 +1,36 @@ +package reliability + +import ( + "context" + "time" +) + +type QueueObserver interface { + ObserveQueueWait(provider, result string, duration time.Duration) +} + +type queueObserverContextKey struct{} + +func WithQueueObserver(ctx context.Context, observer QueueObserver) context.Context { + if ctx == nil { + ctx = context.Background() + } + if observer == nil { + return ctx + } + return context.WithValue(ctx, queueObserverContextKey{}, observer) +} + +func observeQueueWait( + ctx context.Context, + provider, result string, + duration time.Duration, +) { + if ctx == nil { + return + } + observer, _ := ctx.Value(queueObserverContextKey{}).(QueueObserver) + if observer != nil { + observer.ObserveQueueWait(provider, result, duration) + } +} diff --git a/internal/reliability/tracing.go b/internal/reliability/tracing.go index a944885..66d1e86 100644 --- a/internal/reliability/tracing.go +++ b/internal/reliability/tracing.go @@ -87,17 +87,25 @@ func traceQueueAcquire( attribute.String("gateway.provider.id", providerID), ), ) + startedAt := time.Now() lease, failure := queues.Acquire(ctx, providerID) + duration := time.Since(startedAt) + result := "acquired" if failure == nil { span.SetAttributes(attribute.String("gateway.queue.result", "acquired")) span.SetStatus(codes.Ok, "") } else { + result = string(failure.Queue) + if result == "" { + result = string(failure.Category) + } span.SetAttributes( attribute.String("gateway.queue.result", string(failure.Queue)), attribute.String("gateway.failure.category", string(failure.Category)), ) span.SetStatus(codes.Error, string(failure.Category)) } + observeQueueWait(ctx, providerID, result, duration) span.End() return lease, failure } diff --git a/internal/usage/collector.go b/internal/usage/collector.go index 1158f97..8bb283d 100644 --- a/internal/usage/collector.go +++ b/internal/usage/collector.go @@ -1,6 +1,7 @@ package usage import ( + "strings" "sync" "time" ) @@ -40,6 +41,7 @@ func (collector *Collector) SetCacheStatus(status string) { if collector == nil { return } + status = strings.ToLower(strings.TrimSpace(status)) collector.mu.Lock() defer collector.mu.Unlock() if !collector.finalized && status != "" { diff --git a/internal/usage/emitter.go b/internal/usage/emitter.go index 1824276..fffa25c 100644 --- a/internal/usage/emitter.go +++ b/internal/usage/emitter.go @@ -9,6 +9,8 @@ import ( ) type Emitter interface { + // Emit returns a Redis entry ID for immediate delivery. A durable emitter + // may return an empty ID after the event is safely queued in its outbox. Emit(ctx context.Context, event Event) (string, error) } diff --git a/internal/usage/outbox.go b/internal/usage/outbox.go index 73f07f8..a20698c 100644 --- a/internal/usage/outbox.go +++ b/internal/usage/outbox.go @@ -4,6 +4,7 @@ import ( "context" "errors" "fmt" + "sync" "time" "gorm.io/gorm" @@ -12,9 +13,12 @@ import ( "model-velo/internal/postgres" ) +const defaultOutboxRepublishPeriod = 30 * time.Second + const ( - defaultOutboxBatchSize = 100 - defaultOutboxRepublishPeriod = 30 * time.Second + durableEmitterBatchSize = 100 + durableEmitterBatchWait = 5 * time.Millisecond + durableEmitterQueueSize = 4_096 ) // PendingEvent is the durable minimum recorded before an authenticated request @@ -36,29 +40,45 @@ type LifecycleEmitter interface { Begin(context.Context, PendingEvent) error } -// DurableEmitter stores request lifecycle state in PostgreSQL before making a -// best-effort immediate delivery to Redis. The worker relays anything left in -// the outbox, so Redis downtime cannot silently discard a finalized event. +// DurableEmitter stores request lifecycle state in PostgreSQL. Concurrent +// requests share short, bounded batches so durability does not require two +// individual SQL round trips per request. type DurableEmitter struct { database *gorm.DB - redis *RedisEmitter timeout time.Duration + writes chan durableWrite + stop chan struct{} + done chan struct{} + stateMu sync.RWMutex + closed bool + stopOnce sync.Once +} + +type durableWrite struct { + record postgres.UsageOutbox + ready bool + result chan error } func NewDurableEmitter( database *gorm.DB, - redis *RedisEmitter, timeout time.Duration, ) (*DurableEmitter, error) { switch { case database == nil: return nil, errors.New("durable usage emitter requires PostgreSQL") - case redis == nil: - return nil, errors.New("durable usage emitter requires Redis") case timeout <= 0: return nil, errors.New("durable usage emitter timeout must be positive") default: - return &DurableEmitter{database: database, redis: redis, timeout: timeout}, nil + emitter := &DurableEmitter{ + database: database, + timeout: timeout, + writes: make(chan durableWrite, durableEmitterQueueSize), + stop: make(chan struct{}), + done: make(chan struct{}), + } + go emitter.run() + return emitter, nil } } @@ -66,8 +86,6 @@ func (emitter *DurableEmitter) Begin(ctx context.Context, pending PendingEvent) if err := validatePendingEvent(pending); err != nil { return err } - writeContext, cancel := detachedTimeout(ctx, emitter.timeout) - defer cancel() record := postgres.UsageOutbox{ EventID: pending.EventID, RequestID: pending.RequestID, @@ -78,11 +96,8 @@ func (emitter *DurableEmitter) Begin(ctx context.Context, pending PendingEvent) State: postgres.UsageOutboxPending, StartedAt: pending.StartedAt.UTC(), } - result := emitter.database.WithContext(writeContext). - Clauses(clause.OnConflict{Columns: []clause.Column{{Name: "event_id"}}, DoNothing: true}). - Create(&record) - if result.Error != nil { - return fmt.Errorf("record usage request lifecycle: %w", result.Error) + if err := emitter.enqueue(ctx, durableWrite{record: record}); err != nil { + return fmt.Errorf("record usage request lifecycle: %w", err) } return nil } @@ -92,42 +107,195 @@ func (emitter *DurableEmitter) Emit(ctx context.Context, event Event) (string, e if err != nil { return "", err } + payloadText := string(payload) + record := postgres.UsageOutbox{ + EventID: event.EventID, + RequestID: event.RequestID, + TenantID: event.TenantID, + APIKeyID: event.APIKeyID, + RequestedModel: event.RequestedModel, + Stream: event.Stream, + State: postgres.UsageOutboxReady, + Payload: &payloadText, + StartedAt: event.StartedAt.UTC(), + } + if err := emitter.enqueue(ctx, durableWrite{record: record, ready: true}); err != nil { + return "", fmt.Errorf("finalize usage outbox event: %w", err) + } + return "", nil +} + +func (emitter *DurableEmitter) Close(ctx context.Context) error { + emitter.stopOnce.Do(func() { + emitter.stateMu.Lock() + emitter.closed = true + close(emitter.stop) + emitter.stateMu.Unlock() + }) + if ctx == nil { + ctx = context.Background() + } + select { + case <-emitter.done: + return nil + case <-ctx.Done(): + return fmt.Errorf("close durable usage emitter: %w", ctx.Err()) + } +} + +func (emitter *DurableEmitter) enqueue( + ctx context.Context, + write durableWrite, +) error { writeContext, cancel := detachedTimeout(ctx, emitter.timeout) defer cancel() - result := emitter.database.WithContext(writeContext). - Model(&postgres.UsageOutbox{}). - Where("event_id = ?", event.EventID). - Updates(map[string]any{ - "payload": string(payload), - "state": postgres.UsageOutboxReady, - "published_at": nil, - }) - if result.Error != nil { - return "", fmt.Errorf("finalize usage outbox event: %w", result.Error) + write.result = make(chan error, 1) + + emitter.stateMu.RLock() + if emitter.closed { + emitter.stateMu.RUnlock() + return errors.New("durable usage emitter is closed") } - if result.RowsAffected != 1 { - return "", errors.New("usage outbox lifecycle record is missing") + select { + case emitter.writes <- write: + emitter.stateMu.RUnlock() + case <-emitter.done: + emitter.stateMu.RUnlock() + return errors.New("durable usage emitter is closed") + case <-writeContext.Done(): + emitter.stateMu.RUnlock() + return writeContext.Err() } - entryID, publishErr := emitter.redis.Emit(ctx, event) - if publishErr != nil { - return "", publishErr + select { + case err := <-write.result: + return err + case <-emitter.done: + return errors.New("durable usage emitter closed before write completed") + case <-writeContext.Done(): + return writeContext.Err() } - emitter.markPublished(event.EventID) - return entryID, nil } -func (emitter *DurableEmitter) markPublished(eventID string) { +func (emitter *DurableEmitter) run() { + defer close(emitter.done) + for { + select { + case first := <-emitter.writes: + emitter.writeBatch(emitter.collect(first, false)) + case <-emitter.stop: + emitter.drain() + return + } + } +} + +func (emitter *DurableEmitter) collect( + first durableWrite, + stopping bool, +) []durableWrite { + batch := make([]durableWrite, 0, durableEmitterBatchSize) + batch = append(batch, first) + if stopping { + for len(batch) < durableEmitterBatchSize { + select { + case write := <-emitter.writes: + batch = append(batch, write) + default: + return batch + } + } + return batch + } + + timer := time.NewTimer(durableEmitterBatchWait) + defer timer.Stop() + for len(batch) < durableEmitterBatchSize { + select { + case write := <-emitter.writes: + batch = append(batch, write) + case <-timer.C: + return batch + case <-emitter.stop: + return batch + } + } + return batch +} + +func (emitter *DurableEmitter) drain() { + for { + select { + case first := <-emitter.writes: + emitter.writeBatch(emitter.collect(first, true)) + default: + return + } + } +} + +func (emitter *DurableEmitter) writeBatch(batch []durableWrite) { + pending := make([]postgres.UsageOutbox, 0, len(batch)) + ready := make([]postgres.UsageOutbox, 0, len(batch)) + for _, write := range batch { + if write.ready { + ready = append(ready, write.record) + continue + } + pending = append(pending, write.record) + } + + pendingErr := emitter.writePending(pending) + readyErr := emitter.writeReady(ready) + for _, write := range batch { + if write.ready { + write.result <- readyErr + continue + } + write.result <- pendingErr + } +} + +func (emitter *DurableEmitter) writePending( + records []postgres.UsageOutbox, +) error { + if len(records) == 0 { + return nil + } ctx, cancel := context.WithTimeout(context.Background(), emitter.timeout) defer cancel() - now := time.Now().UTC() - _ = emitter.database.WithContext(ctx). - Model(&postgres.UsageOutbox{}). - Where("event_id = ? AND state = ?", eventID, postgres.UsageOutboxReady). - Updates(map[string]any{ - "state": postgres.UsageOutboxPublished, - "published_at": now, - }).Error + if err := emitter.database.WithContext(ctx). + Session(&gorm.Session{SkipDefaultTransaction: true}). + Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "event_id"}}, + DoNothing: true, + }). + CreateInBatches(&records, len(records)).Error; err != nil { + return err + } + return nil +} + +func (emitter *DurableEmitter) writeReady( + records []postgres.UsageOutbox, +) error { + if len(records) == 0 { + return nil + } + ctx, cancel := context.WithTimeout(context.Background(), emitter.timeout) + defer cancel() + if err := emitter.database.WithContext(ctx). + Session(&gorm.Session{SkipDefaultTransaction: true}). + Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "event_id"}}, + DoUpdates: clause.AssignmentColumns([]string{ + "payload", "state", "published_at", "updated_at", + }), + }). + CreateInBatches(&records, len(records)).Error; err != nil { + return err + } + return nil } // OutboxRelay republishes ready records and safely republishes published @@ -135,8 +303,8 @@ func (emitter *DurableEmitter) markPublished(eventID string) { type OutboxRelay struct { database *gorm.DB emitter *RedisEmitter + consumerGroup string batchSize int - timeout time.Duration pendingTimeout time.Duration republishAfter time.Duration } @@ -144,21 +312,24 @@ type OutboxRelay struct { func NewOutboxRelay( database *gorm.DB, emitter *RedisEmitter, - timeout time.Duration, + consumerGroup string, + batchSize int64, ) (*OutboxRelay, error) { switch { case database == nil: return nil, errors.New("usage outbox relay requires PostgreSQL") case emitter == nil: return nil, errors.New("usage outbox relay requires Redis") - case timeout <= 0: - return nil, errors.New("usage outbox relay timeout must be positive") + case consumerGroup == "": + return nil, errors.New("usage outbox relay requires a consumer group") + case batchSize <= 0 || batchSize > 1_000: + return nil, errors.New("usage outbox relay batch size is invalid") default: return &OutboxRelay{ database: database, emitter: emitter, - batchSize: defaultOutboxBatchSize, - timeout: timeout, + consumerGroup: consumerGroup, + batchSize: int(batchSize), pendingTimeout: 15 * time.Minute, republishAfter: defaultOutboxRepublishPeriod, }, nil @@ -177,46 +348,110 @@ func (relay *OutboxRelay) Publish(ctx context.Context) (int, error) { if _, err := relay.recoverPending(ctx); err != nil { return 0, err } + + records, err := relay.readyRecords(ctx) + if err != nil { + return 0, err + } + if len(records) > 0 { + return relay.publishRecords(ctx, records) + } + + caughtUp, err := relay.consumerCaughtUp(ctx) + if err != nil { + return 0, err + } + if !caughtUp { + return 0, nil + } + + records, err = relay.stalePublishedRecords(ctx) + if err != nil { + return 0, err + } + return relay.publishRecords(ctx, records) +} + +func (relay *OutboxRelay) readyRecords( + ctx context.Context, +) ([]postgres.UsageOutbox, error) { + var records []postgres.UsageOutbox + if err := relay.database.WithContext(ctx). + Where("state = ?", postgres.UsageOutboxReady). + Order("updated_at ASC"). + Limit(relay.batchSize). + Find(&records).Error; err != nil { + return nil, fmt.Errorf("read ready usage outbox: %w", err) + } + return records, nil +} + +func (relay *OutboxRelay) stalePublishedRecords( + ctx context.Context, +) ([]postgres.UsageOutbox, error) { republishBefore := time.Now().UTC().Add(-relay.republishAfter) var records []postgres.UsageOutbox if err := relay.database.WithContext(ctx). Where( - "state = ? OR (state = ? AND (published_at IS NULL OR published_at <= ?))", - postgres.UsageOutboxReady, + "state = ? AND (published_at IS NULL OR published_at <= ?)", postgres.UsageOutboxPublished, republishBefore, ). - Order("updated_at ASC"). + Order("published_at ASC"). Limit(relay.batchSize). Find(&records).Error; err != nil { - return 0, fmt.Errorf("read usage outbox: %w", err) + return nil, fmt.Errorf("read published usage outbox: %w", err) } + return records, nil +} - published := 0 +func (relay *OutboxRelay) consumerCaughtUp(ctx context.Context) (bool, error) { + groups, err := relay.emitter.client.XInfoGroups(ctx, relay.emitter.stream).Result() + if err != nil { + return false, fmt.Errorf("read usage consumer group: %w", err) + } + for _, group := range groups { + if group.Name == relay.consumerGroup { + return group.Pending == 0 && group.Lag == 0, nil + } + } + return false, fmt.Errorf("usage consumer group %q was not found", relay.consumerGroup) +} + +func (relay *OutboxRelay) publishRecords( + ctx context.Context, + records []postgres.UsageOutbox, +) (int, error) { + eventIDs := make([]string, 0, len(records)) for _, record := range records { if record.Payload == nil { - return published, fmt.Errorf("usage outbox event %s has no payload", record.EventID) + return 0, fmt.Errorf("usage outbox event %s has no payload", record.EventID) } event, err := Decode([]byte(*record.Payload)) if err != nil { - return published, fmt.Errorf("decode usage outbox event %s: %w", record.EventID, err) + return 0, fmt.Errorf("decode usage outbox event %s: %w", record.EventID, err) } if _, err := relay.emitter.Emit(ctx, event); err != nil { - return published, err - } - now := time.Now().UTC() - if err := relay.database.WithContext(ctx). - Model(&postgres.UsageOutbox{}). - Where("event_id = ?", record.EventID). - Updates(map[string]any{ - "state": postgres.UsageOutboxPublished, - "published_at": now, - }).Error; err != nil { - return published, fmt.Errorf("mark usage outbox published: %w", err) + return 0, err } - published++ + eventIDs = append(eventIDs, record.EventID) + } + if len(eventIDs) == 0 { + return 0, nil + } + + now := time.Now().UTC() + if err := relay.database.WithContext(ctx). + Session(&gorm.Session{SkipDefaultTransaction: true}). + Model(&postgres.UsageOutbox{}). + Where("event_id IN ?", eventIDs). + Updates(map[string]any{ + "state": postgres.UsageOutboxPublished, + "published_at": now, + }).Error; err != nil { + return 0, fmt.Errorf("mark usage outbox published: %w", err) } - return published, nil + return len(eventIDs), nil } func (relay *OutboxRelay) recoverPending(ctx context.Context) (int, error) { diff --git a/internal/usage/store.go b/internal/usage/store.go index 2eab8ce..7aabcca 100644 --- a/internal/usage/store.go +++ b/internal/usage/store.go @@ -20,6 +20,11 @@ type Store struct { now func() time.Time } +type storeEntry struct { + entryID string + event Event +} + func NewStore(database *gorm.DB, pricing *PricingCatalog) (*Store, error) { if database == nil { return nil, errors.New("usage store requires PostgreSQL") @@ -78,33 +83,61 @@ func (store *Store) ReloadManagedPricing(ctx context.Context) (bool, error) { } func (store *Store) Put(ctx context.Context, entryID string, event Event) (bool, error) { - if err := event.Validate(); err != nil { + stored, duplicates, err := store.putBatch( + ctx, + []storeEntry{{entryID: entryID, event: event}}, + ) + if err != nil { return false, err } - record := store.usageRecord(entryID, event, store.now().UTC()) - duplicate := false + return stored == 0 && duplicates == 1, nil +} + +func (store *Store) putBatch( + ctx context.Context, + entries []storeEntry, +) (int64, int64, error) { + if len(entries) == 0 { + return 0, 0, nil + } + processedAt := store.now().UTC() + records := make([]postgres.UsageEvent, 0, len(entries)) + eventIDs := make([]string, 0, len(entries)) + for _, entry := range entries { + if err := entry.event.Validate(); err != nil { + return 0, 0, err + } + records = append(records, store.usageRecord( + entry.entryID, + entry.event, + processedAt, + )) + eventIDs = append(eventIDs, entry.event.EventID) + } + + var stored int64 err := store.database.WithContext(ctx).Transaction(func(transaction *gorm.DB) error { result := transaction. Clauses(clause.OnConflict{ Columns: []clause.Column{{Name: "event_id"}}, DoNothing: true, }). - Create(&record) + Create(&records) if result.Error != nil { return result.Error } - duplicate = result.RowsAffected == 0 + stored = result.RowsAffected if err := transaction. - Where("event_id = ?", event.EventID). + Where("event_id IN ?", eventIDs). Delete(&postgres.UsageOutbox{}).Error; err != nil { return err } return nil }) if err != nil { - return false, err + return 0, 0, err } - return duplicate, nil + return stored, int64(len(entries)) - stored, nil } func (store *Store) usageRecord(entryID string, event Event, processedAt time.Time) postgres.UsageEvent { diff --git a/internal/usage/usage_test.go b/internal/usage/usage_test.go index 52987a9..e626eeb 100644 --- a/internal/usage/usage_test.go +++ b/internal/usage/usage_test.go @@ -34,7 +34,7 @@ func TestCollectorFinalizesOnce(t *testing.T) { if err != nil { t.Fatalf("NewCollector() error = %v", err) } - collector.SetCacheStatus("bypass") + collector.SetCacheStatus("BYPASS") collector.SetRoute("primary", "upstream-model", 3, 1, 1) collector.ObserveResponse([]byte( `{"usage":{"prompt_tokens":11,"completion_tokens":4,"total_tokens":15}}`, @@ -51,6 +51,7 @@ func TestCollectorFinalizesOnce(t *testing.T) { t.Fatal("second Finalize() = true") } if event.LatencyMS != 1500 || + event.CacheStatus != "bypass" || event.Attempts != 3 || event.Retries != 1 || event.Fallbacks != 1 || @@ -513,10 +514,84 @@ func TestUsageRedisPostgresPipeline(t *testing.T) { t.Fatalf("first Worker.Run() error = %v", err) } - durable, err := NewDurableEmitter(database.ORM(), emitter, time.Second) + durable, err := NewDurableEmitter(database.ORM(), time.Second) if err != nil { t.Fatalf("NewDurableEmitter() error = %v", err) } + defer func() { + closeContext, closeCancel := context.WithTimeout(context.Background(), 2*time.Second) + defer closeCancel() + if err := durable.Close(closeContext); err != nil { + t.Errorf("DurableEmitter.Close() error = %v", err) + } + }() + + const concurrentLifecycleCount = 200 + concurrentEvents := make([]Event, 0, concurrentLifecycleCount) + concurrentEventIDs := make([]string, 0, concurrentLifecycleCount) + for index := 0; index < concurrentLifecycleCount; index++ { + event := integrationEvent(t, "request-batched-"+strconv.Itoa(index)) + concurrentEvents = append(concurrentEvents, event) + concurrentEventIDs = append(concurrentEventIDs, event.EventID) + } + var lifecycleWait sync.WaitGroup + lifecycleErrors := make(chan error, concurrentLifecycleCount) + for _, event := range concurrentEvents { + lifecycleWait.Add(1) + go func(event Event) { + defer lifecycleWait.Done() + if err := durable.Begin(ctx, PendingEvent{ + EventID: event.EventID, RequestID: event.RequestID, + TenantID: event.TenantID, APIKeyID: event.APIKeyID, + RequestedModel: event.RequestedModel, Stream: event.Stream, + StartedAt: event.StartedAt, + }); err != nil { + lifecycleErrors <- err + return + } + if _, err := durable.Emit(ctx, event); err != nil { + lifecycleErrors <- err + } + }(event) + } + lifecycleWait.Wait() + close(lifecycleErrors) + for err := range lifecycleErrors { + t.Errorf("batched durable lifecycle error = %v", err) + } + var batchedReady int64 + if err := database.ORM().Model(&postgres.UsageOutbox{}). + Where("event_id IN ? AND state = ?", concurrentEventIDs, postgres.UsageOutboxReady). + Count(&batchedReady).Error; err != nil || batchedReady != concurrentLifecycleCount { + t.Fatalf( + "batched ready lifecycles = %d, want %d, error = %v", + batchedReady, concurrentLifecycleCount, err, + ) + } + if err := database.ORM(). + Where("event_id IN ?", concurrentEventIDs). + Delete(&postgres.UsageOutbox{}).Error; err != nil { + t.Fatalf("delete batched usage lifecycles: %v", err) + } + + finalOnlyEvent := integrationEvent(t, "request-final-without-begin") + if _, err := durable.Emit(ctx, finalOnlyEvent); err != nil { + t.Fatalf("DurableEmitter.Emit(without Begin) error = %v", err) + } + var finalOnly postgres.UsageOutbox + if err := database.ORM(). + Where("event_id = ?", finalOnlyEvent.EventID). + First(&finalOnly).Error; err != nil || + finalOnly.State != postgres.UsageOutboxReady || + finalOnly.Payload == nil { + t.Fatalf("final-only outbox row = %#v, error = %v", finalOnly, err) + } + if err := database.ORM(). + Where("event_id = ?", finalOnlyEvent.EventID). + Delete(&postgres.UsageOutbox{}).Error; err != nil { + t.Fatalf("delete final-only usage lifecycle: %v", err) + } + relayEvent := integrationEvent(t, "request-outbox-republish") if err := durable.Begin(ctx, PendingEvent{ EventID: relayEvent.EventID, RequestID: relayEvent.RequestID, @@ -530,10 +605,15 @@ func TestUsageRedisPostgresPipeline(t *testing.T) { if err != nil { t.Fatalf("DurableEmitter.Emit() error = %v", err) } - if err := client.XDel(ctx, settings.StreamKey, relayEntryID).Err(); err != nil { - t.Fatalf("delete published event before worker storage: %v", err) + if relayEntryID != "" { + t.Fatalf("DurableEmitter.Emit() entry ID = %q, want asynchronous relay", relayEntryID) } - relay, err := NewOutboxRelay(database.ORM(), emitter, time.Second) + relay, err := NewOutboxRelay( + database.ORM(), + emitter, + settings.Group, + settings.BatchSize, + ) if err != nil { t.Fatalf("NewOutboxRelay() error = %v", err) } @@ -542,7 +622,30 @@ func TestUsageRedisPostgresPipeline(t *testing.T) { t.Fatalf("OutboxRelay.Publish() published=%d error=%v", published, err) } if length, err := client.XLen(ctx, settings.StreamKey).Result(); err != nil || length != 1 { - t.Fatalf("republished stream length=%d error=%v", length, err) + t.Fatalf("relayed stream length=%d error=%v", length, err) + } + if published, err := relay.Publish(ctx); err != nil || published != 0 { + t.Fatalf("OutboxRelay.Publish(backlogged) published=%d error=%v", published, err) + } + messages, err := client.XReadGroup(ctx, &goredis.XReadGroupArgs{ + Group: settings.Group, + Consumer: settings.Consumer, + Streams: []string{settings.StreamKey, ">"}, + Count: 1, + }).Result() + if err != nil || len(messages) != 1 || len(messages[0].Messages) != 1 { + t.Fatalf("XReadGroup(relayed) streams=%d error=%v", len(messages), err) + } + lostEntryID := messages[0].Messages[0].ID + if _, err := client.TxPipelined(ctx, func(pipe goredis.Pipeliner) error { + pipe.XAck(ctx, settings.StreamKey, settings.Group, lostEntryID) + pipe.XDel(ctx, settings.StreamKey, lostEntryID) + return nil + }); err != nil { + t.Fatalf("remove relayed event before storage: %v", err) + } + if published, err := relay.Publish(ctx); err != nil || published != 1 { + t.Fatalf("OutboxRelay.Publish(lost) published=%d error=%v", published, err) } if duplicate, err := store.Put(ctx, "direct-relay", relayEvent); err != nil || duplicate { t.Fatalf("Put(relayed event) duplicate=%t error=%v", duplicate, err) diff --git a/internal/usage/worker.go b/internal/usage/worker.go index 463ff7b..835f918 100644 --- a/internal/usage/worker.go +++ b/internal/usage/worker.go @@ -260,49 +260,65 @@ func (worker *Worker) processBatch(ctx context.Context, messages []goredis.XMess ) defer cancel() + entries := make([]storeEntry, 0, len(messages)) + entryIDs := make([]string, 0, len(messages)) for _, message := range messages { - if err := worker.processMessage(batchContext, message); err != nil { - worker.stats.failed.Add(1) - slog.Error( - "usage worker entry failed", - "entry_id", message.ID, - "error", err, + payload, ok := streamString(message.Values["payload"]) + if !ok { + worker.recordEntryError( + message.ID, + worker.handlePoison(batchContext, message, "missing_payload"), ) + continue } + event, err := Decode([]byte(payload)) + if err != nil { + worker.recordEntryError( + message.ID, + worker.handlePoison(batchContext, message, "invalid_event"), + ) + continue + } + entries = append(entries, storeEntry{entryID: message.ID, event: event}) + entryIDs = append(entryIDs, message.ID) if batchContext.Err() != nil { return } } -} - -func (worker *Worker) processMessage(ctx context.Context, message goredis.XMessage) error { - payload, ok := streamString(message.Values["payload"]) - if !ok { - return worker.handlePoison(ctx, message, "missing_payload") + if len(entries) == 0 { + return } - event, err := Decode([]byte(payload)) + + stored, duplicates, err := worker.store.putBatch(batchContext, entries) if err != nil { - return worker.handlePoison(ctx, message, "invalid_event") + worker.stats.failed.Add(1) + slog.Error("usage worker batch failed", "entries", len(entries), "error", err) + return } + worker.stats.stored.Add(stored) + worker.stats.duplicates.Add(duplicates) - duplicate, err := worker.store.Put(ctx, message.ID, event) - if err != nil { - return err + if err := worker.acknowledge(batchContext, entryIDs); err != nil { + worker.stats.failed.Add(1) + slog.Error("usage worker acknowledgement failed", "entries", len(entryIDs), "error", err) } - if duplicate { - worker.stats.duplicates.Add(1) - } else { - worker.stats.stored.Add(1) +} + +func (worker *Worker) recordEntryError(entryID string, err error) { + if err == nil { + return } - _, err = worker.client.TxPipelined(ctx, func(pipe goredis.Pipeliner) error { - pipe.XAck(ctx, worker.config.StreamKey, worker.config.Group, message.ID) - pipe.XDel(ctx, worker.config.StreamKey, message.ID) + worker.stats.failed.Add(1) + slog.Error("usage worker entry failed", "entry_id", entryID, "error", err) +} + +func (worker *Worker) acknowledge(ctx context.Context, entryIDs []string) error { + _, err := worker.client.TxPipelined(ctx, func(pipe goredis.Pipeliner) error { + pipe.XAck(ctx, worker.config.StreamKey, worker.config.Group, entryIDs...) + pipe.XDel(ctx, worker.config.StreamKey, entryIDs...) return nil }) - if err != nil { - return err - } - return nil + return err } func (worker *Worker) handlePoison( diff --git a/llm_gateway_performance_report_2026-07-27.html b/llm_gateway_performance_report_2026-07-27.html new file mode 100644 index 0000000..c443c0d --- /dev/null +++ b/llm_gateway_performance_report_2026-07-27.html @@ -0,0 +1,1100 @@ + + + + + + +LLM 网关性能公开证据报告|2026-07-27 + + + +
+
+
公开数据审阅 · 截止 2026-07-27 JST
+

LLM 网关性能
公开证据报告

+

+ 覆盖 GoModel、Bifrost、LiteLLM、Portkey、Kong AI Gateway、Ferro Labs AI Gateway 与 agentgateway。 + 本报告保留测试硬件、并发方式、上游模型、功能开关、版本与作者关系;不同测试套件的数据不合并成一张“总排名”。 +

+
+
7网关产品
+
8运行时路径(LiteLLM Python / Rust 分开)
+
7公开基准组
+
16核心原始来源与复现实物
+
+
+
+ + + +
+
+
+

范围与阅读规则

+

先看测试条件,再看数字。报告正文在结尾之前不做产品推荐,也不把不同硬件和不同功能开关下的结果混算。

+
+ +
+
+
+ 同一测试组内可横向比较 + 同一主机、同一 mock 上游、同一负载生成器、相同功能开关时,网关之间的差异才有直接意义。 +
+
+ 跨测试组只看量级与一致性 + 2 vCPU、8 vCPU、Kubernetes 集群、18 worker 本地容器的数据不能直接按 RPS 排队。 +
+
+ “纯转发快”不等于“生产配置快” + 多数公开测试关闭了鉴权、数据库写入、预算、缓存、重试、Guardrail、详细日志或持久化。 +
+
+
+ 公开基准几乎都由某个被测产品的项目方、供应商或相关贡献者发布。报告因此把“作者关系”和“可复现材料”放在数字旁边;没有把供应商自测包装成独立实验室结论。 +
+
+ +
+
RPS / QPS每秒完成请求数。必须同时看错误率、并发数、响应体大小和上游延迟。
+
p50 / p95 / p9950%、95%、99% 请求不超过该延迟。p99 更能暴露排队和长尾。
+
Added latency网关路径延迟减去直连 mock 的延迟,试图隔离网关自身开销。
+
TTFT流式响应收到首个事件/首字节的时间;与总响应时间不是同一指标。
+
Peak RSS进程峰值常驻内存。多 worker / 多实例时通常近似随进程数放大。
+
Cold start容器或进程启动至可服务的时间。受镜像、依赖、迁移和健康检查影响。
+
Mock upstream固定、快速的假 LLM。适合隔离代理开销,但不覆盖真实模型波动与网络。
+
Feature parity相同端点并不保证相同鉴权、记账、重试、缓存、路由或协议转换工作量。
+
+
+ +
+
+

公开证据覆盖矩阵

+

“出现”表示该产品在该套件中有可识别数字;不代表套件作者独立,也不代表启用了完整生产功能。

+
+ +
+
+ + + + + + + + + + + + + + + + + + + + + + + +
网关 × 测试套件覆盖(版本写在对应章节)
产品 / 路径GoModel AWS
统一测试
AIGatewayBench
统一测试
Ferro
统一测试
Kong EKS
统一测试
项目方单项
性能测试
公开流式
数据
GoModel●———同一套件●
Bifrost●●●—●部分
LiteLLM Python●●●●●●
LiteLLM Rust beta—●——●仍在补齐
Portkey OSS●●●●未找到同级公开表部分
Kong AI Gateway——●●同一套件未见统一表
Ferro Labs——●—同一套件套件声称覆盖
agentgateway————●该文未测
+
+
+ +
+

测试套件的证据元数据

+
+ + + + + + + + + + + + + +
套件发布方与关系脚本原始结果硬件 / 版本主要限制
GoModel AWS 2026-06-25GoModel 项目方;被测产品作者有有披露;镜像摘要留档仅 2 vCPU;多数生产功能关闭;GoModel 尚无外部复测
AIGatewayBench 2026-07LiteLLM 项目方;Rust 路径作者有CSV 已提交版本披露;单主机无重复试验误差条;只测转发路径;Rust 为 early beta
Ferro 多网关套件Ferro Labs;被测产品作者有有报告披露Portkey 以 Docker 运行,其余原生;Bifrost 默认池配置成为瓶颈
Kong EKS 2025-07Kong;被测产品作者有图表 / 基础设施材料披露版本较旧;仅代理,无鉴权、缓存等策略;精确结果主要在图片
Bifrost 官方Maxim / Bifrost;被测产品作者有独立 benchmark 文档汇总表披露500 RPS 对比与 5k 调优压力测试不是同一场景
LiteLLM 官方LiteLLM;被测产品作者部分公开汇总表 / CSV披露稳定 Python 与 Rust beta 必须分开看
agentgateway 2026-06项目贡献者;agentgateway 相关方有文章粘贴原始输出镜像版本披露;发布机器规格未披露18 个 LiteLLM worker;3 秒跑满测试很短;资源结果强依赖进程数
+
+
+
+ +
+
+

GoModel AWS 统一测试

+

四个网关在同一台 2 vCPU / 4 GiB AWS 实例上连接同一即时 mock;这一节中的数字可在本节内比较。

+
+ +
+
+
+

2026-06-25 · AWS c7i.large

+

+ 每个协议变体 8,000 请求、并发 10、两次随机顺序试验;容量扫描并发为 1 / 16 / 128。 + 全部关闭重试,GoModel 关闭熔断,LiteLLM Python 使用 2 个 worker(每个 CPU 核一个)。 +

+
+
+ 作者:GoModel 项目方 + 原始 JSON / CSV / 脚本 + S1S2 +
+
+ +
+
主机AWS c7i.large
2 vCPU / 4 GiB
+
上游共享即时 mock
隔离代理开销
+
协议Chat / Responses / Messages
流式与非流式
+
延迟样本8,000 × 2 次 / 变体
取两次中位结果
+
功能开关重试关闭
无完整生产插件栈
+
+ +
+
+

Chat 非流式 p99

越短越好 · 同套件
+
GoModel
+
Bifrost
+
Portkey
+
LiteLLM Python
+
+ +
+

Chat 非流式峰值持续吞吐

越长越高 · 同套件
+
GoModel
+
Bifrost
+
Portkey
+
LiteLLM Python
+
+ +
+

负载下峰值 RSS

对数长度 · 数字为原值
+
GoModel
+
Portkey
+
Bifrost
+
LiteLLM Python
+
+ +
+

Chat 流式 TTFT p50

越短越好 · 同套件
+
GoModel
+
Bifrost
+
Portkey
+
LiteLLM Python
+
+
+ +
+ 完整延迟、吞吐和资源表 +
+
+ + + + + + + + + + + +
非流式延迟(ms;两次试验结果的中位数)
工作负载 / 指标直连 baselineGoModelBifrostPortkeyLiteLLM Python
Chat p500.231.812.519.7030.56
Chat p992.776.8818.2730.5439.26
Responses p500.262.012.739.0739.12
Responses p992.337.2816.5526.9248.60
Anthropic Messages p500.261.762.65此设置不支持61.06
Anthropic Messages p992.236.5919.08此设置不支持98.12
+
+
+
+ + + + + + + + + + + +
流式延迟 p50(ms)
工作负载 / 指标GoModelBifrostPortkeyLiteLLM Python
Chat TTFT4.719.0227.97151.94
Chat total4.9511.8927.98151.95
Responses TTFT4.6912.8727.9047.53
Responses total5.0014.9427.9347.55
Messages TTFT7.50无终止事件,10s 内 0 完成此设置不支持48.86
Messages total8.38同上此设置不支持48.89
+
+
+
+ + + + + + + + + + +
Chat 非流式容量扫描(req/s)
网关并发 1并发 16并发 128峰值
直连 baseline15,51029,70130,01530,015
GoModel2,7454,9284,5674,928
Bifrost1,8853,0882,9043,088
Portkey636946900946
LiteLLM Python227324254324
+
+
+
+ + + + + + + + + + + + +
镜像、启动与资源
指标GoModelPortkeyBifrostLiteLLM Python
压缩镜像(MB)165977372
磁盘镜像(MB)47.2177.4230.71,159.9
Cold start(s)0.561.057.0725.49
峰值 RSS(MB)37.0112.0143.02,272.3
平均 CPU(%)92.6116.9117.6101.1
资源窗口持续 RPS4,8249602,977261
RPS / CPU%52.18.225.32.6
+
+
+
+ +
+ 范围限制:GoModel、Bifrost、Portkey 使用当时镜像,LiteLLM 为 main-stable 且 2 worker;测试记录了镜像摘要。 + Portkey 的 Messages 缺失是该单 Provider → OpenAI mock 配置限制,不应外推为产品绝对不支持 Anthropic Messages。 +
+
+
+ +
+
+

AIGatewayBench

+

LiteLLM 项目方在 2026 年 7 月发布的可复现套件,比较 Rust early beta、Python v1、Bifrost v1.6.4 与当前 Portkey OSS。

+
+ +
+
+
+

单主机确定性 Rust mock · Anthropic Messages 转发路径

+

+ 通过“网关路径 p99 − 直连 mock p99”计算 added latency;关闭日志 callback、支出追踪和持久化。 + 原始 CSV、负载驱动器与每个网关的启动说明均已公开。 +

+
+
+ 作者:LiteLLM 项目方 + CSV 已提交 + Rust:early beta + S3S4S5S6 +
+
+ +
+
+

p99 Added latency

对数长度 · 数字为原值
+
LiteLLM Rust
+
LiteLLM Python
+
Bifrost
+
Portkey
+
+ +
+

峰值 RSS

越短越低 · 同套件
+
LiteLLM Rust
+
LiteLLM Python
+
Bifrost
+
Portkey
+
+
+ +

并发 64 的同一测点

+
+ + + + + + + + +
路径直连 p99网关 p99Added p99网关 RPS错误
LiteLLM Rust beta24.015 ms64.299 ms40.284 ms2,722.170
LiteLLM Python v124.015 ms570.058 ms546.043 ms179.750
Bifrost v1.6.424.015 ms52.699 ms28.683 ms2,719.080
Portkey OSS24.015 ms132.434 ms108.419 ms1,360.260
+
+ +
+ 查看并发 1 / 4 / 16 / 64 全表 +
+ + + + + + + + + + + + + + + + + + + + +
路径并发网关 p99(ms)Added p99(ms)网关 RPS错误
LiteLLM Rust122.8690.88745.250
LiteLLM Python131.7679.78534.380
Bifrost122.7300.74944.820
Portkey124.7632.78143.160
LiteLLM Rust423.0340.610181.240
LiteLLM Python449.37426.95092.710
Bifrost423.0530.629180.230
Portkey425.2002.776174.040
LiteLLM Rust1623.0010.327721.790
LiteLLM Python16274.018251.345158.390
Bifrost1623.5640.891722.670
Portkey1630.9128.238673.600
LiteLLM Rust6464.29940.2842,722.170
LiteLLM Python64570.058546.043179.750
Bifrost6452.69928.6832,719.080
Portkey64132.434108.4191,360.260
+
+
+ +

套件给出的资源成本模型(不是云账单)

+
+ + + + + + + + +
路径保留测点持续 RPS估算小时资源成本RPS / 美元·小时测点 p99
LiteLLM Rust beta2,814.35$0.00992283,83326.945 ms
LiteLLM Python177.48$0.041314,296557.356 ms
Bifrost2,744.12$0.0433963,24632.212 ms
Portkey1,095.53$0.0609817,964299.744 ms
+
+
+ 成本模型使用固定单价(vCPU 与 GB·小时)把本地 CPU、RSS 和吞吐换算成美元,不含模型 token 费用、数据库、Redis、日志、网络、控制平面与冗余副本。它适合比较该套件内的资源量级,不是采购报价。 +
+
+
+ +
+
+

Ferro 多网关统一测试

+

GCP n2-standard-8(8 vCPU / 32 GB)、Debian 12、固定 60ms mock。该套件把 Ferro、Kong、Bifrost、LiteLLM 与 Portkey 放在同一负载配置下。

+
+ +
+
+
+

150 → 1,000 并发用户的吞吐与资源

+

Ferro、Kong、Bifrost、LiteLLM 以原生进程运行;Portkey 以 Docker host network 运行。版本与设置由套件记录。

+
+
+ 作者:Ferro Labs + 脚本 / 配置 / 报告 + S7 +
+
+ +
+ + + + + + + + + + +
吞吐(RPS)与内存;“失败/未测”不按 0 吞吐参与比较
网关 / 版本150 VU300 VU500 VU1,000 VU报告内存
Ferro Labs v1.0.02,4474,8908,01413,92532–135 MB
Kong OSS 3.9.12,4434,8858,13315,89143 MB(报告称近似平坦)
Bifrost v1.0.02,441连接池饥饿连接池饥饿连接池饥饿107–333 MB
LiteLLM 1.82.6175未继续测未继续测未继续测335–1,124 MB
Portkey latest85184385589167 MB
+
+ +
+
+

150 VU:所有五个网关均有测点

同套件
+
Ferro
+
Kong
+
Bifrost
+
LiteLLM
+
Portkey
+
+
+

1,000 VU:仅三个网关有有效测点

缺失值不绘制
+
Ferro
+
Kong
+
Portkey
+
+
+ +

Ferro 自身在同一套件中的延迟曲线

+
+ + + + + + + + + +
VURPSp50p99内存相对 60ms mock 的 p50 开销
5081361.3 ms64.1 ms36 MB约 1.3 ms
1502,44761.2 ms63.4 ms47 MB约 1.2 ms
3004,89061.2 ms64.4 ms72 MB约 1.2 ms
5008,01461.5 ms72.9 ms89 MB1.5 ms
1,00013,92568.1 ms111.9 ms135 MB8.1 ms
+
+ +
+ 该结果同时说明了“默认配置”对结论的影响:Bifrost 在 150 VU 有 2,441 RPS,但套件报告其在 ≥300 VU 因连接池饥饿出现大量失败; + Bifrost 自家 5,000 RPS 测试则显式把 buffer 与 initial pool 调得很大。两组结果应并列阅读,而不是任选一组外推。 +
+
+
+ +
+
+

其他官方 / 项目相关方测试

+

这些测试补充了扩容、固定吞吐、Kubernetes 上限和调优后压力场景,但不能与前面的统一测试直接合并。

+
+ +
+
+
+

agentgateway v1.3.1 vs LiteLLM main-latest

+

Fortio + 本地固定响应 mock。发布文先做 3 秒跑满测试,再做 30 秒固定目标 3,000 QPS 测试;LiteLLM 使用 18 个 worker。

+
+
+ 项目相关方发布 + 脚本公开 + 发布主机规格未写明 + S13S14S15 +
+
+
+ + + + + + + +
固定目标 3,000 QPS;32 连接、1 KB 请求、30 秒
网关实际 QPSp50p90p99平均 CPU峰值内存
agentgateway2,998.940.227 ms0.249 ms0.436 ms13.4%17.07 MiB
LiteLLM Python(18 worker)2,465.8912.318 ms19.739 ms30.626 ms345.5%11.69 GiB
+
+
+
+ + + + + + + +
各自跑满;32 连接、1 KB 请求、3 秒
网关QPSp50p99平均 CPU峰值内存
agentgateway36,933.620.831 ms1.970 ms104.8%28.79 MiB
LiteLLM Python(18 worker)3,198.487.076 ms32.192 ms330.8%11.81 GiB
+
+
+ 11.7 GiB 是 18 个 Python worker 共同产生的容器结果,不应当描述成“一个 LiteLLM 进程需要 11.7 GiB”。 + 同时,文章未列出发布机器的 CPU / RAM,3 秒跑满窗口也较短,因此这组数据最适合说明该特定配置下的代理开销,不适合做跨硬件容量规划。 +
+
+ +
+
+
+

Kong EKS:Kong 3.10、Portkey 1.9.19、LiteLLM 1.63.7

+

每个网关上限 12 CPU,400 VU、每轮 3 分钟、1,000 prompt token;WireMock 与网关分置于 c5.4xlarge 节点。

+
+
+ 作者:Kong + 基础设施仓库公开 + 2025 版本;仅 proxy + S8S9 +
+
+
+
29,005.51 RPS
WireMock 直连 baseline;p95 24.07ms,p99 30.35ms
+
+228%
Kong 文章给出的 Konnect Data Plane 相对 Portkey 吞吐差异
+
+859%
Kong 文章给出的 Konnect Data Plane 相对 LiteLLM 吞吐差异
+
−65% / −86%
文章给出的 Kong 相对 Portkey / LiteLLM 延迟差异
+
+
+ Kong 的公开正文把精确网关 RPS 和延迟放在图片中,文本提供的是相对百分比;测试明确关闭缓存、API Key 鉴权等策略。 + 因此该组数字证明的是“纯代理、旧版本、固定 EKS 配额”下的相对关系,不是完整 AI policy 栈的开销。 +
+
+ +
+
+
+
+

Bifrost 官方:500 RPS 对比

+

AWS t3.medium(2 vCPU / 4 GB)、60 秒、500 VU;真实 OpenAI Tier 5 场景与单独 60ms mock overhead 场景。

+
+
作者:BifrostS10
+
+
+ + + + + + + + + + + +
指标BifrostLiteLLM Python
成功率100%88.78%
p50804 ms38.65 s
p991.68 s90.72 s
实际吞吐424 req/s44.84 req/s
峰值内存120 MB372 MB
60ms mock 中位总延迟60.99 ms100 ms
由此计算的 overhead0.99 ms40 ms
+
+
+ +
+
+
+

Bifrost 官方:5,000 RPS 调优压力测试

+

仅 Bifrost;两个 AWS 规格均使用 mock。显式扩大 buffer 与 initial pool。

+
+
作者:BifrostS11
+
+
+ + + + + + + + + + + + +
指标t3.medium
2 vCPU / 4GB
t3.xlarge
4 vCPU / 16GB
成功率 @ 5k RPS100%100%
报告的 Bifrost overhead59 µs11 µs
平均总延迟2.12 s1.61 s
队列等待47.13 µs1.67 µs
响应解析11.30 ms2.11 ms
峰值内存1,312.79 MB3,340.44 MB
Buffer / initial pool15,000 / 10,00020,000 / 15,000
响应体备注约 1 KB约 10 KB
+
+
+
+ +
+
+
+
+

LiteLLM Python 官方扩容表

+

每台 4 CPU / 8 GB,PostgreSQL,未使用 Redis,对 fake OpenAI endpoint。

+
+
作者:LiteLLMS12
+
+
+ + + + + + + + + + +
配置 / 指标中位p95p99Current RPS
2 实例:/chat/completions200 ms630 ms1,200 ms1,035.7
2 实例:LiteLLM overhead12 ms29 ms43 ms1,035.7
2 实例:官方 Aggregated100 ms430 ms930 ms2,071.4
4 实例:/chat/completions100 ms150 ms240 ms1,170
4 实例:LiteLLM overhead2 ms8 ms13 ms1,170
4 实例:官方 Aggregated77 ms130 ms180 ms2,340
+
+
这是横向扩容后的整体测试,不应与单容器 2 vCPU 测试的 324 RPS直接比较;官方建议 worker 数等于 CPU 核数。
+
+ +
+
+
+

LiteLLM Rust 迁移微基准

+

同一 mock;10 并发测 overhead,50 并发测吞吐与内存。2026-07-22 官方仍称 early beta,完整流式与功能面仍在补齐。

+
+
作者:LiteLLMearly betaS6S16
+
+
+ + + + + + +
路径每请求 overhead持续吞吐峰值内存
Rust forwarding path约 0.05 ms6,782 req/s31.7 MB
当前 Python path约 7.5 ms453 req/s358.9 MB
+
+
这是转发 hot path 微基准,不是带完整鉴权、数据库、预算、日志与路由策略的生产 workload。
+
+
+
+ +
+
+

七个网关的公开事实卡

+

只列数字出现在哪些公开测试、测试间有哪些差异、目前缺什么证据;解释和选择结论留到报告末尾。

+
+ +
+
+

GoModel

Go
+
    +
  • GoModel AWS 2 vCPU:Chat p50 1.81ms、p99 6.88ms;容量峰值 4,928 req/s;峰值 RSS 37MB;cold start 0.56s。
  • +
  • 同一套件覆盖 Chat、Responses、Anthropic Messages 的流式和非流式 6 个变体。
  • +
  • 目前找到的核心性能数字来自 GoModel 自己发布的统一套件;本报告未找到 Ferro、BerriAI、Kong 等外部套件对 GoModel 的复测。
  • +
+
+ +
+

Bifrost

Go
+
    +
  • GoModel 套件:3,088 req/s、Chat p99 18.27ms、143MB RSS。
  • +
  • AIGatewayBench:保留测点 2,744 req/s、199.1MB;并发 64 的 p99 为 52.70ms。
  • +
  • Ferro 套件:150 VU 为 2,441 RPS,但 ≥300 VU 报告连接池饥饿;Bifrost 官方调大 pool/buffer 后报告 5,000 RPS、100% 成功。
  • +
  • 不同测试之间最明显的变量是连接池、buffer、响应体大小与内存换吞吐。
  • +
+
+ +
+

LiteLLM Python

稳定主路径
+
    +
  • GoModel 2 worker / 2 vCPU:324 req/s、2.27GB RSS、25.49s cold start。
  • +
  • AIGatewayBench:约 177–180 req/s、329MB RSS;Ferro:约 175 RPS、335–1,124MB。
  • +
  • agentgateway 对比使用 18 worker:固定目标下 2,466 QPS、11.69GiB 峰值内存。
  • +
  • LiteLLM 官方多实例表:4 实例配置给出 Aggregated 2,340 RPS,p95 overhead 8ms。
  • +
+
+ +
+

LiteLLM Rust

2026-07:early beta
+
    +
  • AIGatewayBench:p99 added latency 0.66ms、21.85MB;保留测点约 2,814 req/s。
  • +
  • LiteLLM forwarding-path 微基准:约 0.05ms overhead、6,782 req/s、31.7MB。
  • +
  • 官方在 2026-07-22 仍明确表示 streaming 与完整功能面尚在落地,不应与稳定 Python 网关按“同功能产品”直接替换。
  • +
+
+ +
+

Portkey OSS

Node / JS 路径
+
    +
  • GoModel 2 vCPU:峰值 946 req/s、Chat p99 30.54ms、112MB RSS、1.05s cold start。
  • +
  • AIGatewayBench:保留测点约 1,096 req/s、90.44MB;并发 64 的网关 RPS 为 1,360。
  • +
  • Ferro 8 vCPU:150–1,000 VU 约 843–891 RPS,报告称出现 event-loop 拥塞与高并发错误。
  • +
  • Kong 2025 EKS 测试包含 Portkey 1.9.19,但正文主要给相对百分比。
  • +
+
+ +
+

Kong AI Gateway

API Gateway 路径
+
    +
  • Ferro 8 vCPU 套件:150 / 300 / 500 / 1,000 VU 分别 2,443 / 4,885 / 8,133 / 15,891 RPS;报告内存约 43MB。
  • +
  • Kong 自己的 EKS 测试声称 Konnect Data Plane 相对 Portkey +228%、相对 LiteLLM +859% 吞吐。
  • +
  • 两组测试均以纯代理或极轻策略为主;公开材料没有给出同一压力下开启完整 AI 鉴权、token quota、缓存、日志后的精确增量。
  • +
+
+ +
+

Ferro Labs AI Gateway

v1.0.0 测点
+
    +
  • 自家 8 vCPU 统一套件:150–1,000 VU 为 2,447 → 13,925 RPS;内存 32–135MB。
  • +
  • 60ms mock 下,1,000 VU 时 p50 68.1ms、p99 111.9ms。
  • +
  • 公开性能证据主要由 Ferro 自己的套件提供;本报告未找到另一套主流多网关基准对 Ferro 的复测。
  • +
+
+ +
+

agentgateway

Rust
+
    +
  • 固定 3,000 QPS 测试:实际 2,998.94 QPS、p99 0.436ms、17.07MiB 峰值内存。
  • +
  • 3 秒跑满测试:36,933.62 QPS、p99 1.970ms、28.79MiB。
  • +
  • 镜像固定为 v1.3.1,但文章未写发布主机规格;对手 LiteLLM 为 main-latest、18 worker,因此资源差异高度依赖该配置。
  • +
+
+
+
+ +
+
+

公开数据仍未回答的问题

+

这些缺口决定了“公开最高 RPS”不能直接变成生产容量承诺。

+
+
+

完整功能开销

缺少一套同时开启租户鉴权、RBAC、Redis 限流、数据库记账、预算、审计、可观测性、缓存、Guardrail 与动态路由的跨产品测试。

+

长流与背压

公开结果多用立即完成的短响应。真实 SSE 可能持续几十秒,需要测慢客户端、断连、缓冲、首事件和尾部资源释放。

+

失败路径

大多数吞吐表没有统一测 429、5xx、超时、重试、熔断、跨供应商 fallback、Key 冷却和排队上限。

+

混合协议与大请求体

Chat、Responses、Messages、embeddings、audio、批处理与 100k context 的解析成本不同;多数测试只取一个端点。

+

水平扩展效率

很少有相同 Kubernetes 配额下从 1 → 2 → 4 → 8 副本的线性度、负载均衡、共享数据库与 Redis 瓶颈数据。

+

稳定性窗口

3 秒、30 秒、3 分钟和短时容量扫描无法替代 6–24 小时 soak test,也难以暴露内存增长、连接泄漏和长尾漂移。

+

版本漂移

LiteLLM 正在迁移 Rust,其他项目也快速发布。使用 latest 或 main-latest 的结果必须保留镜像摘要才能复现。

+

SaaS 边缘网络

自托管 OSS 进程与供应商托管边缘服务的 TLS、跨区网络、WAF、控制平面和多租户隔离不是同一个性能对象。

+

独立复核

本轮检索未发现一个与所有七个产品无利益关系、同时公开脚本和原始数据的统一实验室测试。

+
+
+ +
+
+

来源与数据追溯

+

均为官方文档、官方或项目相关 GitHub 仓库、原始 CSV。检索日期:2026-07-27(JST)。

+
+
+
S1

GoModel Benchmarks and AI Gateway Performance Results测试条件与摘要;AWS c7i.large、8,000 请求、两次试验、功能开关。

+
S2

GoModel AWS Benchmark — RESULTS.md完整非流式、流式、容量、镜像、冷启动、RSS 与 CPU 表。

+
S3

BerriAI / AIGatewayBench场景、负载驱动、网关启动说明、复现步骤与限制。

+
S4

AIGatewayBench — overhead_comparison.csvp99 added latency 与 peak RSS 原始汇总。

+
S5

AIGatewayBench — latency_vs_concurrency.csv并发 1 / 4 / 16 / 64 的 direct p99、gateway p99、added p99、RPS、错误。

+
S6

LiteLLM Rust AI Gateway Benchmark(2026-07-22)版本、测试边界、early beta 状态、成本模型与供应商自测声明。

+
S7

Ferro Labs AI Gateway Performance BenchmarksGCP 统一套件、版本、吞吐、内存、Ferro 延迟曲线与复现说明。

+
S8

Kong AI Gateway vs Portkey vs LiteLLMEKS 架构、版本、baseline、相对吞吐与延迟。

+
S9

Kong benchmark repositoryEKS 1.32、WireMock、400 VU、3 分钟、资源上限与部署材料。

+
S10

Bifrost vs LiteLLM Benchmarks500 RPS 对比、60ms mock overhead、成功率、延迟、实际吞吐和内存。

+
S11

Bifrost 5,000 RPS Benchmarking Guide两种 EC2 规格、buffer / pool、队列、解析、内存和响应体说明。

+
S12

LiteLLM Gateway Benchmarks2 / 4 实例 Python 网关、机器规格、PostgreSQL、overhead 与 RPS。

+
S13

agentgateway vs LiteLLM — maximum throughput3 秒跑满、32 连接、1KB、QPS、延迟、CPU、内存与原始输出。

+
S14

agentgateway vs LiteLLM — fixed 3,000 QPS30 秒固定目标测试与原始输出。

+
S15

litellm-agw-perf repository复现脚本;agentgateway v1.3.1、LiteLLM main-latest 与 worker 配置。

+
S16

Migrating LiteLLM to RustRust forwarding path 微基准、迁移阶段、目标与“非完整生产 workload”限制。

+
+
+ +
+
+

最后总结

+
+
+ GoModel:公开数字很强,但证据来源集中 + 在自己的 2 vCPU 统一套件中,GoModel同时给出较低延迟、4,928 req/s、37MB RSS和0.56s冷启动;脚本与原始结果透明。当前缺口是没有进入其他主流多网关套件,外部复核不足。 +
+
+ Bifrost:跨多个套件都处于高吞吐组,但对配置敏感 + GoModel与AIGatewayBench均给出约2.7k–3.1k req/s量级;Ferro默认测试出现连接池饥饿,而Bifrost调大池与buffer后报告5k RPS。它的生产容量不能脱离连接池、buffer和内存预算谈。 +
+
+ LiteLLM:必须把 Python 稳定路径与 Rust beta 分开 + Python路径在多个统一开销测试中普遍表现为更高内存、更低单实例吞吐和更明显长尾,但官方多实例扩容可达到约2.3k aggregate RPS。Rust beta把开销与内存降到另一量级,不过2026-07仍未达到完整功能与流式面等价。 +
+
+ Portkey:多个统一套件中的结果相对一致 + GoModel、AIGatewayBench和Ferro给出的吞吐大致在0.85k–1.36k req/s范围、内存约67–112MB;高并发下的event-loop拥塞与平台配置需要自行验证。 +
+
+ Kong:纯代理容量上限高,完整 AI 策略开销未量化 + Ferro套件中Kong在8 vCPU和1,000 VU达到15,891 RPS;Kong自家EKS测试也给出明显相对优势。但公开对比关闭了鉴权、缓存和其他策略,不能直接代表完整LLM治理路径。 +
+
+ Ferro 与 agentgateway:自家结果突出,跨来源验证不足 + Ferro在自家8 vCPU套件达到13,925 RPS;agentgateway在相关方测试中固定3k QPS时p99低于0.5ms。两者目前都更需要无利益关系的复测,agentgateway报告还缺发布机器规格。 +
+
+ 公开数据能支持的判断 + 编译型 / Rust / Go / 高性能代理路径在“快速 mock + 纯转发”条件下通常能显著降低每请求开销、内存和冷启动;Python多进程可以横向换吞吐,但资源会随worker和实例放大。 +
+
+ 公开数据不能支持的判断 + 现有证据不足以选出一个对所有业务都最快的网关。真正采购或项目定位前,应在相同硬件上开启你需要的鉴权、限流、记账、缓存、重试、流式和观测,用真实请求体与失败比例复跑至少30分钟并做长时间稳定性测试。 +
+
+
+
+
+ +
+
+ 本报告是公开证据整理,不是对任何供应商的背书,也没有代替用户在同一环境中重新执行基准。数值保留来源上下文;跨套件图表未做归一化排名。 +
+
+ + + + diff --git a/test/fakeupstream/main.go b/test/fakeupstream/main.go index 7f6e029..bc5b416 100644 --- a/test/fakeupstream/main.go +++ b/test/fakeupstream/main.go @@ -95,7 +95,7 @@ func run(config commandConfig) error { "name", upstream.providerName, "scenario", - upstream.scenarioOverride, + upstream.forcedScenario(), ) select { diff --git a/test/fakeupstream/server.go b/test/fakeupstream/server.go index 21330c3..400e0b9 100644 --- a/test/fakeupstream/server.go +++ b/test/fakeupstream/server.go @@ -144,8 +144,9 @@ var scenarioCatalog = map[string]scenario{ } type upstreamServer struct { - providerName string - scenarioOverride string + providerName string + scenarioMu sync.RWMutex + scenario string attemptsMu sync.Mutex attempts map[string]attemptState @@ -169,9 +170,11 @@ type attemptState struct { } type scenarioStats struct { - Requests int64 `json:"requests"` - Errors int64 `json:"errors"` - Streams int64 `json:"streams"` + Requests int64 + Errors int64 + Streams int64 + FirstRequestAt time.Time + LastRequestAt time.Time } func newUpstreamServer(providerName, scenarioOverride string) (*upstreamServer, error) { @@ -189,10 +192,10 @@ func newUpstreamServer(providerName, scenarioOverride string) (*upstreamServer, } } server := &upstreamServer{ - providerName: providerName, - scenarioOverride: scenarioOverride, - attempts: map[string]attemptState{}, - scenarioStats: map[string]scenarioStats{}, + providerName: providerName, + scenario: scenarioOverride, + attempts: map[string]attemptState{}, + scenarioStats: map[string]scenarioStats{}, } server.statsStartedAt.Store(time.Now().UnixNano()) return server, nil @@ -204,6 +207,7 @@ func (s *upstreamServer) handler() http.Handler { mux.HandleFunc("GET /__admin/scenarios", s.handleScenarios) mux.HandleFunc("GET /__admin/stats", s.handleStats) mux.HandleFunc("POST /__admin/reset", s.handleReset) + mux.HandleFunc("POST /__admin/scenario", s.handleScenario) mux.HandleFunc("POST /v1/chat/completions", s.handleChatCompletions) mux.HandleFunc("POST /chat/completions", s.handleChatCompletions) return mux @@ -213,7 +217,7 @@ func (s *upstreamServer) handleHealth(w http.ResponseWriter, _ *http.Request) { writeJSON(w, http.StatusOK, healthResponse{ Status: "ok", Provider: s.providerName, - ScenarioOverride: s.scenarioOverride, + ScenarioOverride: s.forcedScenario(), }) } @@ -234,7 +238,7 @@ func (s *upstreamServer) handleScenarios(w http.ResponseWriter, _ *http.Request) } writeJSON(w, http.StatusOK, scenariosResponse{ Default: defaultScenarioName, - Override: s.scenarioOverride, + Override: s.forcedScenario(), Scenarios: available, }) } @@ -250,10 +254,12 @@ func (s *upstreamServer) handleStats(w http.ResponseWriter, _ *http.Request) { for _, name := range names { stats := s.scenarioStats[name] scenarios = append(scenarios, scenarioStatsEntry{ - Name: name, - Requests: stats.Requests, - Errors: stats.Errors, - Streams: stats.Streams, + Name: name, + Requests: stats.Requests, + Errors: stats.Errors, + Streams: stats.Streams, + FirstRequestAt: stats.FirstRequestAt, + LastRequestAt: stats.LastRequestAt, }) } s.statsMu.Unlock() @@ -289,6 +295,31 @@ func (s *upstreamServer) handleReset(w http.ResponseWriter, _ *http.Request) { w.WriteHeader(http.StatusNoContent) } +func (s *upstreamServer) handleScenario(w http.ResponseWriter, r *http.Request) { + decoder := json.NewDecoder(http.MaxBytesReader(w, r.Body, 1<<10)) + decoder.DisallowUnknownFields() + var request scenarioOverrideRequest + if err := decoder.Decode(&request); err != nil { + writeUpstreamError(w, http.StatusBadRequest, "invalid scenario override") + return + } + if err := decoder.Decode(&struct{}{}); !errors.Is(err, io.EOF) { + writeUpstreamError(w, http.StatusBadRequest, "invalid scenario override") + return + } + request.Scenario = strings.TrimSpace(request.Scenario) + if request.Scenario != "" { + if _, exists := scenarioCatalog[request.Scenario]; !exists { + writeUpstreamError(w, http.StatusBadRequest, "unknown scenario override") + return + } + } + s.scenarioMu.Lock() + s.scenario = request.Scenario + s.scenarioMu.Unlock() + w.WriteHeader(http.StatusNoContent) +} + func (s *upstreamServer) handleChatCompletions(w http.ResponseWriter, r *http.Request) { s.beginRequest() defer s.finishRequest() @@ -388,6 +419,11 @@ func (s *upstreamServer) recordScenario(name string, stream bool) { s.statsMu.Lock() stats := s.scenarioStats[name] stats.Requests++ + now := time.Now().UTC() + if stats.FirstRequestAt.IsZero() { + stats.FirstRequestAt = now + } + stats.LastRequestAt = now if stream { stats.Streams++ s.streams.Add(1) @@ -409,7 +445,7 @@ func (s *upstreamServer) recordError(name string) { } func (s *upstreamServer) selectScenario(model string) (scenario, error) { - name := s.scenarioOverride + name := s.forcedScenario() if name == "" && strings.HasPrefix(model, "mock/") { name = model } @@ -423,6 +459,12 @@ func (s *upstreamServer) selectScenario(model string) (scenario, error) { return selected, nil } +func (s *upstreamServer) forcedScenario() string { + s.scenarioMu.RLock() + defer s.scenarioMu.RUnlock() + return s.scenario +} + func (s *upstreamServer) beginAttempt( requestID string, selected scenario, @@ -836,6 +878,10 @@ type healthResponse struct { ScenarioOverride string `json:"scenario_override,omitempty"` } +type scenarioOverrideRequest struct { + Scenario string `json:"scenario"` +} + type scenarioDescription struct { Name string `json:"name"` Description string `json:"description"` @@ -848,10 +894,12 @@ type scenariosResponse struct { } type scenarioStatsEntry struct { - Name string `json:"name"` - Requests int64 `json:"requests"` - Errors int64 `json:"errors"` - Streams int64 `json:"streams"` + Name string `json:"name"` + Requests int64 `json:"requests"` + Errors int64 `json:"errors"` + Streams int64 `json:"streams"` + FirstRequestAt time.Time `json:"first_request_at"` + LastRequestAt time.Time `json:"last_request_at"` } type upstreamStatsResponse struct { diff --git a/test/fakeupstream/server_test.go b/test/fakeupstream/server_test.go index 4d6db73..b08bb6a 100644 --- a/test/fakeupstream/server_test.go +++ b/test/fakeupstream/server_test.go @@ -208,6 +208,42 @@ func TestUpstreamServerLoadScenariosAndStats(t *testing.T) { } } +func TestUpstreamServerScenarioOverrideCanRecover(t *testing.T) { + upstream, err := newUpstreamServer("recovering-provider", "mock/error-503") + if err != nil { + t.Fatalf("new upstream server: %v", err) + } + testServer := httptest.NewServer(upstream.handler()) + defer testServer.Close() + + failed := postChat(t, testServer.URL, "before-recovery", "mock/instant", false) + _, _ = io.Copy(io.Discard, failed.Body) + _ = failed.Body.Close() + if failed.StatusCode != http.StatusServiceUnavailable { + t.Fatalf("status before recovery = %d, want 503", failed.StatusCode) + } + + response, err := http.Post( + testServer.URL+"/__admin/scenario", + "application/json", + strings.NewReader(`{"scenario":"mock/instant"}`), + ) + if err != nil { + t.Fatalf("set scenario override: %v", err) + } + _, _ = io.Copy(io.Discard, response.Body) + _ = response.Body.Close() + if response.StatusCode != http.StatusNoContent { + t.Fatalf("scenario update status = %d, want 204", response.StatusCode) + } + + recovered := postChat(t, testServer.URL, "after-recovery", "mock/error-503", false) + defer recovered.Body.Close() + if recovered.StatusCode != http.StatusOK { + t.Fatalf("status after recovery = %d, want 200", recovered.StatusCode) + } +} + func requestIDWithFailureOutcome(fail bool) string { for index := range 10_000 { requestID := "failure-outcome-" + strconv.Itoa(index) diff --git a/test/k6/common.js b/test/k6/common.js index bc12cdb..e42506c 100644 --- a/test/k6/common.js +++ b/test/k6/common.js @@ -8,6 +8,7 @@ export const responses200 = new Counter('chat_responses_200'); export const responses429 = new Counter('chat_responses_429'); export const responses5xx = new Counter('chat_responses_5xx'); export const responsesOther = new Counter('chat_responses_other'); +export const gatewayChatRequests = new Counter('gateway_chat_requests'); export const requestTimeout = __ENV.REQUEST_TIMEOUT || '20s'; export const expected200 = http.expectedStatuses(200); @@ -125,6 +126,9 @@ export function sendChat({ stream, }); + if (target === 'gateway') { + gatewayChatRequests.add(1, { model, stream: String(stream) }); + } return http.post(chatURL(targetURL), body, { headers, responseCallback, diff --git a/test/threehost/PERFORMANCE_OPTIMIZATION.md b/test/threehost/PERFORMANCE_OPTIMIZATION.md new file mode 100644 index 0000000..c4dc587 --- /dev/null +++ b/test/threehost/PERFORMANCE_OPTIMIZATION.md @@ -0,0 +1,839 @@ +# Model-Velo 三机性能优化与测试手册 + +这份文件是接下来性能工作的唯一运行清单。测试拓扑固定为: + +```text +k6 客户端机 -> Model-Velo 网关机 -> 假 LLM 上游机 +``` + +不在同一台网关机上轮流运行其他网关,也不把不同机器、不同提交、不同运行参数产生的 +数字放在一起比较。每一轮都保存 k6、三台机器资源、Prometheus、假上游调用数、 +PostgreSQL/Redis/Usage 证据和最终 HTML。 + +## 1. 现在处于什么位置 + +第一轮诊断结果: + +| 负载 | 实际 RPS | 成功率 | dropped | P95 | P99 | +| --- | ---: | ---: | ---: | ---: | ---: | +| 500 RPS | 500.0 | 99.987% | 0 | 83.36 ms | 271.29 ms | +| 750 RPS | 749.5 | 99.959% | 0 | 230.12 ms | 353.18 ms | +| 1000 RPS | 995.0 | 97.044% | 485 | 726.23 ms | 1247.04 ms | +| 1500 RPS | 1460.3 | 44.536% | 3127 | 2934.41 ms | 4110.65 ms | + +500 RPS、10 分钟耐久结果为 99.992% 成功率、P99 221.51 ms。客户端 CPU 最高 +29.61%,因此第一轮的拐点不是客户端 CPU 已经跑满。 + +第一轮性能数字不能直接作为最终成绩,因为 Usage 链路存在两个正确性问题: + +- Cache 状态使用大写 `HIT/MISS/BYPASS`,Usage Event 只接受小写,导致完成事件失败; +- Outbox Relay 重复发布 `published` 记录,Redis Stream 增长到约 607 万条, + Outbox 留下约 64.8 万条记录,Worker 重复消费却没有成功落库。 + +当前修复把 Cache 状态统一为小写,并把链路调整成: + +```text +请求开始 -> PostgreSQL Outbox pending +请求结束 -> Outbox ready +Worker 批量发布 -> Redis consumer group +Worker 批量落 usage_events + 删除 Outbox -> XACK + XDEL +``` + +下一步不是先调 Queue,而是直接用同一组 19 分钟负载验证这次修复。只有 Usage +对账正确、Stream 不再膨胀,后面的 RPS 和延迟数据才可信。 + +## 2. 最快的优化节奏 + +不要为每个猜测新建一个分支,也不要把十个参数一起修改。使用当前性能分支,每个可解释的 +生产代码变化单独提交,每轮测试记录准确 commit: + +1. `D1`:当前 Usage 修复,运行一次 19 分钟诊断; +2. 根据 `D1` 的阶段耗时和连接池指标,只选最大的一个瓶颈; +3. 修改这个瓶颈对应的代码或一个参数,运行 `D2` 19 分钟诊断; +4. 如果 `D2` 明显改善且没有转移成新瓶颈,停止继续试参数; +5. 用最终候选配置再跑一次 19 分钟确认; +6. 最后运行 1.5–2 小时完整套件和单独的 Redis 限流测试。 + +参数实验是条件测试,不是必跑清单。比如 Redis 没有等待和 timeout,就不试 +`100 -> 200`;Provider Queue 没达到 256,就不试 `256 -> 512`。这样通常两到四轮 +19 分钟测试就能得到有证据的优化结果,而不是做一下午无效排列组合。 + +每次代码修改后,先对修改过的 Go 文件执行 `gofmt -w`,再在开发机执行: + +```bash +go test ./... +go vet ./... +``` + +最终版本额外执行: + +```bash +go test -race ./... +``` + +## 3. 每轮测试不变的规则 + +### 3.1 三台服务器必须一致 + +- 三台服务器签出同一个 commit SHA; +- 使用同一个 `RUN_ID`,不要在三台机器分别运行 `date`; +- 网关参数变化后必须重建或重建容器,不能只修改 `.env`; +- 三台服务器时间必须同步; +- 同一轮不更换实例规格、地域、Docker 版本或网络拓扑; +- 正式运行时 Git worktree 应为 clean; +- `.env`、API Key 和指标 Token 不进入结果包。 + +三台机器分别检查: + +```bash +git status --short +git rev-parse HEAD +timedatectl show -p NTPSynchronized --value +``` + +三个 `git rev-parse HEAD` 必须相同,NTP 应显示 `yes`。 + +### 3.2 RUN_ID 命名 + +建议格式: + +```text +diag-usagefix-r1-20260727T230000Z +diag-authcache-r2-20260728T010000Z +diag-final-r1-20260728T030000Z +complete-final-r1-20260728T050000Z +ratelimit-final-r1-20260728T080000Z +``` + +`RUN_ID` 表示一次不可覆盖的实验。失败后重新运行也要换一个 ID,不能覆盖旧目录。 + +### 3.3 什么叫可比较 + +比较两轮时,至少以下项目相同: + +- 服务器规格和地域; +- 网关进程数; +- 假上游镜像和模型场景; +- k6 脚本参数; +- Queue、Breaker、Retry、限流和连接池中没有被实验的参数; +- 测试持续时间; +- 采集器正常覆盖整个 case; +- 客户端 `dropped_iterations` 没有先被客户端资源限制。 + +## 4. 下一轮:Usage 修复后的 19 分钟诊断 + +这一轮编号为 `D1`。它必须使用新 commit、新 `RUN_ID` 和干净 PostgreSQL/Redis。 +旧的 607 万条 Stream 数据不能和修复后的结果共用。 + +### 4.1 先发布准确 commit + +修复代码提交并推送后,在开发机记录 SHA: + +```bash +git status --short +git rev-parse HEAD +``` + +下面所有服务器都把 `REPLACE_WITH_COMMIT_SHA` 替换成这个 SHA。不要把 +`` 原样复制到 Bash,因为尖括号会被 Shell 当成重定向。 + +### 4.2 假上游机 + +```bash +cd ~/model-velo +git fetch origin +COMMIT=REPLACE_WITH_COMMIT_SHA +git checkout "$COMMIT" + +docker compose -f test/threehost/upstream.compose.yaml build main +docker compose -f test/threehost/upstream.compose.yaml up -d --no-build + +curl http://127.0.0.1:9000/healthz +curl http://127.0.0.1:9001/healthz +curl http://127.0.0.1:9002/healthz +``` + +三个端口复用同一个 `model-velo-fake-upstream:local` 镜像,只是启动参数不同。 + +### 4.3 网关机:保留旧数据,创建全新测试卷 + +先停止旧 Compose 项目,但不要使用 `-v`: + +```bash +cd ~/model-velo +git fetch origin +COMMIT=REPLACE_WITH_COMMIT_SHA +git checkout "$COMMIT" + +unset COMPOSE_PROJECT_NAME +docker compose down +``` + +旧卷仍然保留。使用新的 Compose 项目名启动时,会自动创建新的 PostgreSQL/Redis 卷: + +```bash +export COMPOSE_PROJECT_NAME=mv-perf-d1 + +docker compose up -d --build gateway usage-worker +docker compose ps +curl http://127.0.0.1:8080/readyz +curl http://127.0.0.1:9091/readyz +``` + +在新数据库创建测试租户和 API Key: + +```bash +docker compose --profile tools run --rm admin bootstrap-tenant \ + --slug threehost \ + --name "Three Host Benchmark" \ + --label "k6" \ + --models "mock/instant,mock/typical,mock/slow,mock/jitter,mock/spike-5,mock/error-rate-10,mock/payload-10k,mock/payload-50k,mock/retry-2,mock/error-400,mock/error-429,mock/error-503,mock/sse-error,mock/sse-drop,mock/fallback" +``` + +保存只显示一次的 `api_key`,填入客户端 `test/threehost/benchmark.env`。不要重新运行 +`prepare-gateway-env.sh`,否则会更换 Pepper 和其他密钥。 + +当前诊断配置保持: + +```text +MODEL_VELO_POSTGRES_MAX_OPEN_CONNS=50 +MODEL_VELO_POSTGRES_MAX_IDLE_CONNS=10 +MODEL_VELO_REDIS_POOL_SIZE=100 +MODEL_VELO_REDIS_MIN_IDLE_CONNS=10 +MODEL_VELO_RATE_LIMIT_REQUESTS=1000000 +MODEL_VELO_BREAKER_FAILURE_THRESHOLD=5 +MODEL_VELO_QUEUE_MAX_IN_FLIGHT=256 +MODEL_VELO_QUEUE_MAX_WAITING=2048 +MODEL_VELO_QUEUE_WAIT_TIMEOUT=2s +MODEL_VELO_USAGE_BATCH_SIZE=50 +MODEL_VELO_LOG_LEVEL=info +MODEL_VELO_OTEL_SAMPLE_RATIO=0 +``` + +如果 `.env` 没有显式写 `MODEL_VELO_USAGE_BATCH_SIZE`,默认就是 50,不必为了这轮补写。 + +### 4.4 先启动三个监控 + +选择一个固定 ID,并把完全相同的字符串粘贴到三台机器。下面用: + +```text +diag-usagefix-r1-20260727T230000Z +``` + +网关机终端一: + +```bash +cd ~/model-velo +export COMPOSE_PROJECT_NAME=mv-perf-d1 +RUN_ID=diag-usagefix-r1-20260727T230000Z + +DURATION_SECONDS=1800 \ +PROGRESS_SECONDS=10 \ +OUTPUT_FILE="test-results/threehost/$RUN_ID/gateway-stats.jsonl" \ +bash test/threehost/collect-compose-stats.sh +``` + +网关机终端二: + +```bash +cd ~/model-velo +export COMPOSE_PROJECT_NAME=mv-perf-d1 +RUN_ID=diag-usagefix-r1-20260727T230000Z + +DURATION_SECONDS=1800 \ +PROGRESS_SECONDS=10 \ +OUTPUT_DIR="test-results/threehost/$RUN_ID/prometheus" \ +bash test/threehost/collect-prometheus.sh +``` + +假上游机: + +```bash +cd ~/model-velo +RUN_ID=diag-usagefix-r1-20260727T230000Z + +DURATION_SECONDS=1800 \ +PROGRESS_SECONDS=10 \ +COMPOSE_FILE=test/threehost/upstream.compose.yaml \ +SERVICES=main,fail,fallback \ +OUTPUT_FILE="test-results/threehost/$RUN_ID/upstream-stats.jsonl" \ +bash test/threehost/collect-compose-stats.sh +``` + +看到三个终端持续显示 `state=collecting` 后再启动客户端。监控脚本不需要在构建镜像时 +传入 `RUN_ID`;它只用来命名和对齐本轮结果。 + +### 4.5 客户端运行 19 分钟诊断 + +```bash +cd ~/model-velo +git fetch origin +COMMIT=REPLACE_WITH_COMMIT_SHA +git checkout "$COMMIT" + +cp -n test/threehost/benchmark.env.example test/threehost/benchmark.env +editor test/threehost/benchmark.env +``` + +至少填写: + +```text +GATEWAY_URL=http://网关机私网IP:8080 +UPSTREAM_URL=http://假上游机私网IP:9000 +MODEL_VELO_API_KEY=刚刚生成的APIKey +RUN_ID=diag-usagefix-r1-20260727T230000Z +``` + +运行: + +```bash +bash test/threehost/run-diagnostic-client.sh +``` + +该脚本固定运行: + +1. smoke; +2. 100 RPS direct/gateway 预热; +3. 500、750、1000、1500 RPS,各 2 分钟; +4. 500 RPS、10 分钟耐久。 + +诊断脚本会把 k6 最大 VU 提高到 4096,并关闭 Payload、Cache、Fault、Queue overload +等非热路径 case,因此约 19 分钟。 + +### 4.6 客户端结束后收集最终证据 + +网关机第三个终端: + +```bash +cd ~/model-velo +export COMPOSE_PROJECT_NAME=mv-perf-d1 +RUN_ID=diag-usagefix-r1-20260727T230000Z + +bash test/threehost/collect-run-evidence.sh "$RUN_ID" +``` + +这一步不能省略。它按本轮 `bench-$RUN_ID` 等待 Outbox 排空,并保存最终数据库、 +Redis、日志、运行参数和 Worker 指标。 + +先检查: + +```bash +cat "test-results/threehost/$RUN_ID/gateway-evidence/usage-drain.txt" +cat "test-results/threehost/$RUN_ID/gateway-evidence/usage-overview.csv" +cat "test-results/threehost/$RUN_ID/gateway-evidence/usage-outbox.csv" +cat "test-results/threehost/$RUN_ID/gateway-evidence/redis-usage.txt" +``` + +客户端完成且 evidence 已保存后,让监控自然结束;需要提前结束时使用 `Ctrl+C`,不要 +关闭 SSH 窗口强杀进程。正式最终轮最好让状态成为 `completed`。 + +### 4.7 把三台结果合并到客户端 + +在客户端机执行,把主机别名改成自己的 SSH 地址: + +```bash +RUN_ID=diag-usagefix-r1-20260727T230000Z +RESULT_DIR="test-results/threehost/$RUN_ID" + +scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-stats.jsonl" \ + "$RESULT_DIR/" +scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-stats-metadata.txt" \ + "$RESULT_DIR/" +scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-stats-status.json" \ + "$RESULT_DIR/" +scp -r gateway-host:"~/model-velo/$RESULT_DIR/prometheus" \ + "$RESULT_DIR/" +scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-evidence" \ + "$RESULT_DIR/" +scp -r upstream-host:"~/model-velo/$RESULT_DIR/upstream-stats.jsonl" \ + "$RESULT_DIR/" +scp -r upstream-host:"~/model-velo/$RESULT_DIR/upstream-stats-metadata.txt" \ + "$RESULT_DIR/" +scp -r upstream-host:"~/model-velo/$RESULT_DIR/upstream-stats-status.json" \ + "$RESULT_DIR/" + +python3 test/threehost/summarize.py "$RESULT_DIR" +tar -czf "$RUN_ID.tar.gz" -C test-results/threehost "$RUN_ID" +``` + +浏览器打开: + +```text +test-results/threehost//summary.html +``` + +交付分析时只需要 `.tar.gz`,不需要复制整个仓库。 + +## 5. D1 必须通过的门槛 + +### 5.1 正确性门槛 + +任何一项失败都先修正确性,不继续调 RPS: + +| 检查项 | D1 期望 | +| --- | --- | +| `usage-drain.txt` | `state=complete`、`remaining_outbox=0` | +| Usage 对账 | 正常网关 case 的请求数与事件数接近 100% | +| `events` 与 `request_ids` | 相等,不应一条请求多条最终事件 | +| Worker duplicate | 健康新环境中应为 0 | +| Worker failed/dead-letter | 0 | +| Redis pending | 最终为 0 | +| Redis Stream | 测试后能被消费删除,不再持续增长到几十万、几百万 | +| 网关日志 | 没有 `usage cache status invalid` | +| 503 错误码 | 不再由 `usage_accounting_unavailable` 主导 | + +对账不要求把 smoke、健康检查、被认证/限流提前拒绝的请求强行算成 Chat Usage Event。 +以 `summary.html` 的 Usage reconciliation 口径为准。 + +### 5.2 第一阶段性能目标 + +这是当前硬件上的工程目标,不是对所有云服务器都成立的行业标准: + +| 场景 | 合格目标 | +| --- | --- | +| 500 RPS、10 分钟 | 成功率不低于 99.9%,dropped 为 0,Usage 可排空 | +| 750 RPS | 成功率不低于 99.9%,dropped 为 0,P99 小于 500 ms | +| 1000 RPS | 观察拐点;争取成功率不低于 99%,P99 小于 1 s | +| 1500 RPS | 用于观察过载行为,不要求 100% 成功 | + +修好持久化后实际做了更多正确工作,RPS 不一定立刻上升。只要 Usage 正确且阶段指标能明确 +说明成本,就比“忽略落库后得到更高 QPS”更可信。 + +## 6. 看完 HTML 后如何决定改什么 + +先按下面顺序排除外部瓶颈,再看网关: + +| 证据 | 判断 | 下一步 | +| --- | --- | --- | +| direct 也变慢或失败 | 假上游、网络或客户端问题 | 不改网关,先修测试环境 | +| 客户端 CPU 接近满载或 dropped 随客户端资源上涨 | k6 发不出目标流量 | 提升客户端规格或 VU,不能算网关失败 | +| 假上游 CPU 高、`max_active` 异常、direct 恶化 | 假上游成为瓶颈 | 提升上游机,保持场景不变重测 | +| `usage_begin/usage_finalize` P95/P99 高,PostgreSQL waits 增长 | Usage/数据库热路径 | 优先查事务、SQL 次数和连接池 | +| `authentication/authorization` 高,PostgreSQL waits 增长 | 每请求认证和授权查询成为瓶颈 | 做有界认证/授权缓存并明确撤销延迟 | +| `rate_limit` 高,Redis wait/timeout 增长 | Redis 热路径或池耗尽 | 先看 Redis CPU,再决定是否调池或代码 | +| `provider_queue` waiting 高且 active 达到 256 | Provider 并发准入限制 | 判断上游是否有余量,再试 Queue | +| `provider_call` 上升,direct 同时上升 | 上游服务时间增加 | 不归因于网关 | +| 网关 CPU 很高,但连接池和各等待阶段都低 | CPU、分配或锁竞争 | 跑本地 benchmark/pprof 后改代码 | +| Worker pending/lag 增长,网关请求仍正常 | Usage Worker 吞吐不足 | 看 batch、数据库写入和 Redis,而不是 Queue | +| 1500 RPS 延迟不断堆积但 CPU/上游已满 | 过载策略太宽松 | 缩短等待或减少 waiting,快速拒绝 | + +关键指标包括: + +```text +model_velo_request_stage_duration_seconds +model_velo_postgres_connections +model_velo_postgres_waits_total +model_velo_postgres_wait_duration_seconds_total +model_velo_redis_pool_connections +model_velo_redis_pool_events_total +model_velo_redis_pool_wait_duration_seconds_total +model_velo_provider_queue_active +model_velo_provider_queue_waiting +model_velo_usage_worker_events_total +model_velo_usage_worker_pending +process_cpu_seconds_total +process_resident_memory_bytes +go_memstats_heap_inuse_bytes +go_goroutines +``` + +不要只看某一张 CPU 图就下结论。至少要同时有客户端、网关、上游和请求阶段四类证据。 + +## 7. 条件式优化实验 + +每次只选择下面一个实验。修改 `.env` 后保留同一个 Compose 项目和 API Key,但换新的 +`RUN_ID`。前一轮必须已经满足 `remaining_outbox=0`、Redis pending 为 0,才能复用环境。 + +重建网关和 Worker: + +```bash +cd ~/model-velo +export COMPOSE_PROJECT_NAME=mv-perf-d1 +editor .env +docker compose up -d --force-recreate gateway usage-worker +docker compose ps +curl http://127.0.0.1:8080/readyz +curl http://127.0.0.1:9091/readyz +``` + +然后重复“启动三个监控 → 更新客户端 `RUN_ID` → `run-diagnostic-client.sh` → +`collect-run-evidence.sh` → 合并结果”的流程。 + +### 7.1 PostgreSQL 连接池实验 + +只有在下面条件同时出现时才做: + +- `model_velo_postgres_connections{state="in_use"}` 经常接近 50; +- `model_velo_postgres_waits_total` 和 wait duration 在高 RPS case 明显增长; +- PostgreSQL 容器 CPU、内存和连接数仍有余量。 + +候选值: + +```text +MODEL_VELO_POSTGRES_MAX_OPEN_CONNS=100 +MODEL_VELO_POSTGRES_MAX_IDLE_CONNS=25 +``` + +验收不是“连接数更多”,而是: + +- authentication、authorization、usage begin/finalize 延迟下降; +- 1000 RPS 成功率或 P99 改善; +- PostgreSQL CPU 没有被推到满载; +- Usage 仍然完整对账。 + +如果 wait 没下降或 PostgreSQL CPU/延迟更差,恢复 50/10。 + +### 7.2 Redis 连接池实验 + +只有 Redis 出现 `wait`、`timeout`、pending request 或 `rate_limit` 阶段明显升高时才做。 + +候选值: + +```text +MODEL_VELO_REDIS_POOL_SIZE=200 +MODEL_VELO_REDIS_MIN_IDLE_CONNS=20 +``` + +如果 Redis 本身已经 CPU 满载,增加客户端连接只会让它更拥堵,此时不做这个实验。 + +### 7.3 Usage Worker 批量实验 + +只有网关请求正常,但 Worker pending/lag 增长、Outbox 排空慢,并且 PostgreSQL/Redis +还有余量时才做。 + +候选值: + +```text +MODEL_VELO_USAGE_BATCH_SIZE=200 +``` + +验收: + +- Worker `stored` 能跟上本轮网关完成请求数; +- pending 和本轮 Outbox 在 240 秒内清零; +- duplicate、failed、dead-letter 仍为 0; +- PostgreSQL CPU 和事务延迟没有明显恶化。 + +不要一开始直接设到上限 1000。批量越大,单次事务更重,失败重试的工作量也更大。 + +### 7.4 Provider Queue 实验 + +`MODEL_VELO_QUEUE_MAX_IN_FLIGHT=256` 是每个 Provider、每个网关进程的上游并发数, +不是整个网关的总并发数。 + +只有 `provider_queue active` 达到 256、waiting 上升,而假上游仍有 CPU 和并发余量时, +才试: + +```text +MODEL_VELO_QUEUE_MAX_IN_FLIGHT=512 +``` + +`MAX_WAITING=2048` 不会提高吞吐,只决定过载时最多积压多少请求。若 1500 RPS 时吞吐已经 +不再增加、P99 却涨到数秒,优先测试更快失败: + +```text +MODEL_VELO_QUEUE_MAX_WAITING=512 +``` + +如果仍然形成长尾,再单独测试: + +```text +MODEL_VELO_QUEUE_WAIT_TIMEOUT=500ms +``` + +不要把 `MAX_IN_FLIGHT`、`MAX_WAITING` 和 timeout 同时改掉,否则无法判断是哪一个改变了 +吞吐或尾延迟。 + +### 7.5 Breaker 和 Retry + +Breaker 失败阈值 5 不影响健康 `mock/instant` 的吞吐,不作为 RPS 调优项。Retry 也不能 +用来提高健康请求 RPS,它只会在失败场景增加 Attempt。 + +保持: + +```text +MODEL_VELO_BREAKER_FAILURE_THRESHOLD=5 +MODEL_VELO_RETRY_MAX_ATTEMPTS=3 +``` + +它们只在完整套件的 10% 503、Retry、Fallback 和 SSE 故障测试里验收。 + +### 7.6 认证和授权代码优化 + +如果 D1 显示 authentication/authorization 是主要数据库成本,优先考虑带上限、带 TTL +的本地正向缓存,而不是先继续加数据库连接: + +- Cache Key 不能保存明文 API Key; +- 缓存 Identity 和模型授权结果; +- 必须写清楚禁用/撤销 Key 的最大生效延迟; +- 负向结果使用更短 TTL 或不缓存; +- 容量有上限,避免租户/Key 数量导致内存无限增长; +- 认证失败、过期、撤销、租户禁用仍要有测试。 + +实现后先跑 `go test ./...` 和 `go vet ./...`,再运行新的 19 分钟 `D2`。面试时重点讲 +“用阶段指标确认每请求数据库查询是瓶颈,以及如何处理缓存与撤销一致性的取舍”,不要只说 +“加缓存提升性能”。 + +### 7.7 CPU 和内存代码优化 + +只有数据库、Redis、Queue 和上游都没有等待,而网关 CPU/GC 已经很高时才进入这一项。 + +先跑进程内回归 benchmark: + +```bash +go test ./internal/httpapi \ + -run '^$' \ + -bench '^BenchmarkChatCompletions$' \ + -benchmem \ + -benchtime=10s \ + -count=5 +``` + +记录中位 `ns/op`、`B/op`、`allocs/op`,再针对火焰图或 allocation profile 中最大的 +位置修改。常见方向可能是请求体复制、JSON 编解码、日志字段和临时对象,但没有 profile +证据前不预设答案。 + +## 8. 每轮结果记录表 + +每拿到一轮 HTML,就补一行。不要只保存“最大 QPS”。 + +| Run ID | commit | 唯一变化 | 500 成功/P99 | 750 成功/P99 | 1000 成功/P99 | 500×10m | 网关 CPU max | PG wait | Redis wait | Queue wait | Usage 对账 | Outbox drain | +| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | ---: | ---: | ---: | --- | +| 第一轮 | 旧提交 | 修复前基线 | 99.987% / 271 ms | 99.959% / 353 ms | 97.044% / 1247 ms | 99.992% / 222 ms | 待补 | 待补 | 待补 | 待补 | 0% | 失败 | +| D1 | | Usage 链路修复 | | | | | | | | | | | +| D2 | | | | | | | | | | | | | +| Final | | | | | | | | | | | | | + +优化成立至少需要: + +- 正确性门槛不退化; +- 相同负载下成功率、P95/P99、资源或可持续时间中至少一项明确改善; +- 没有把瓶颈从网关悄悄转移到客户端或假上游; +- 新结果至少能由第二轮或最终完整套件复现; +- 运行配置和 commit 都能从结果包中查到。 + +## 9. 最终 19 分钟确认 + +选定代码和参数后,使用新 `RUN_ID`: + +```text +diag-final-r1-20260728T030000Z +``` + +完整重复第 4 节的 19 分钟流程。最终确认轮应使用: + +- clean worktree; +- 最终 commit; +- 最终 `.env`; +- 已排空的 Usage; +- 三个状态正常的监控; +- 新的结果目录。 + +最终确认至少达到第 5 节正确性门槛,并且关键性能改善与上一轮方向一致。若某个改动只在 +一次运行中好看,第二次消失,就不能作为面试中的优化结果。 + +## 10. 最终完整性能与故障套件 + +19 分钟诊断稳定后,才运行完整套件。它大约 1.5–2 小时,默认包含: + +1. smoke 和 direct/gateway 预热; +2. 1–256 VU 闭合模型容量阶梯; +3. 100–2000 RPS 开放模型 SLO 阶梯; +4. 非流式与逐 Chunk SSE; +5. 200 B、10 KiB、50 KiB Payload; +6. Cache bypass、miss/write、warm/hit; +7. 爬坡和突发; +8. 10% 503、抖动、5% 长尾; +9. Queue 过载; +10. 30 分钟耐久; +11. Retry、Fallback、错误映射和 SSE 提交边界。 + +客户端重新从模板准备正式配置,恢复默认三次 repetitions: + +```bash +cd ~/model-velo +cp test/threehost/benchmark.env.example test/threehost/benchmark.env +editor test/threehost/benchmark.env +bash test/threehost/run-complete-client.sh +``` + +至少填写: + +```text +GATEWAY_URL=http://网关机私网IP:8080 +UPSTREAM_URL=http://假上游机私网IP:9000 +MODEL_VELO_API_KEY=测试APIKey +RUN_ID=complete-final-r1-20260728T050000Z +``` + +网关机和上游机采集命令与第 4 节相同,只把: + +```text +DURATION_SECONDS=1800 +``` + +改成: + +```text +DURATION_SECONDS=9000 +``` + +客户端完成后仍必须执行: + +```bash +bash test/threehost/collect-run-evidence.sh complete-final-r1-20260728T050000Z +``` + +再按第 4.7 节合并文件并重新生成 `summary.html`。 + +完整套件的最终验收: + +- smoke 全部通过; +- direct 路径稳定; +- 容量与 RPS 曲线存在清楚拐点,而不是随机抖动; +- Cache hit 确实减少上游调用; +- Retry/Fallback 的上游 Attempt 数与预期一致; +- SSE 首 Content、Chunk 间隔和断流行为正确; +- Queue 过载时错误有界,不出现无限排队; +- 30 分钟耐久无吞吐持续下降、内存持续增长或 Usage 积压; +- Usage 对账完整; +- 三台资源证据覆盖所有 case; +- HTML 不再报告关键 evidence 缺失。 + +## 11. Redis 限流独立测试 + +主容量测试把限流设为每分钟 1,000,000,避免 429 污染容量曲线。最终再单独测试限流, +并使用独立 `RUN_ID`。 + +网关 `.env` 临时设置: + +```text +MODEL_VELO_RATE_LIMIT_REQUESTS=6000 +MODEL_VELO_RATE_LIMIT_WINDOW=1m +``` + +重建网关: + +```bash +export COMPOSE_PROJECT_NAME=mv-perf-d1 +docker compose up -d --force-recreate gateway +curl http://127.0.0.1:8080/readyz +``` + +使用与第 4 节相同的三个监控命令,新 `RUN_ID` 统一设置为 +`ratelimit-final-r1-20260728T080000Z`,`DURATION_SECONDS=600`。 + +客户端: + +```bash +set -a +. test/threehost/benchmark.env +set +a + +export RUN_ID=ratelimit-final-r1-20260728T080000Z +export RUN_CAPACITY=false +export RUN_WARMUP=false +export RUN_RATE_SWEEP=false +export RUN_STREAM_DETAIL=false +export RUN_PAYLOAD=false +export RUN_CACHE=false +export RUN_RAMP=false +export RUN_BURST=false +export RUN_FAULT=false +export RUN_QUEUE_OVERLOAD=false +export RUN_ENDURANCE=false +export RUN_RELIABILITY=false +export RUN_RATE_LIMIT=true +export RATE_LIMIT_TEST_RATE=200 +export RATE_LIMIT_TEST_DURATION=2m +export PRE_ALLOCATED_VUS=256 +export MAX_VUS=1024 + +bash test/threehost/run-complete-client.sh - +``` + +注意:如果 `benchmark.env` 里已经写了旧 `RUN_ID`,上面是在 source 之后重新 export, +所以会使用新的限流 Run ID。 + +验收: + +- 同一 tenant + model 出现明确的 200/429; +- `model_velo_rate_limit_decisions_total` 与 HTTP 结果一致; +- 429 不调用假上游; +- 429 不触发 Retry; +- Redis 无 pool timeout; +- 测试完成后 Usage/Outbox 能排空。 + +完成后把限流恢复: + +```text +MODEL_VELO_RATE_LIMIT_REQUESTS=1000000 +``` + +并重新创建 gateway,防止后续测试意外继续使用低阈值。 + +## 12. 面试时怎么讲 + +不要背“用了 k6、Redis、Prometheus,所以高性能”。按真实调查过程讲: + +### 12.1 一分钟版本 + +> 我把压测拆成三台机器:一台 k6、一台网关、一台可控制延迟和错误的假 LLM 上游。 +> 先用 direct 请求测客户端到上游的基线,再用相同请求经过网关,这样能排除压测机和 +> 假上游先跑满。第一轮在 500 到 750 RPS 比较稳定,1000 RPS 开始出现拐点,但我没有 +> 直接把 Queue 调大,因为 Usage 证据显示事件根本没有落库,Redis Stream 反而增长到 +> 约 607 万条。最后定位到 Cache 状态大小写不一致,以及 Outbox Relay 重复发布 +> published 记录。修复后我用同一组 19 分钟负载复测,对比成功率、P99、阶段耗时、 +> PostgreSQL/Redis 等待和 Usage 对账,再根据最大的阶段瓶颈做下一步优化。 + +### 12.2 为什么不是直接追最大 RPS + +可以说: + +> LLM 网关不能只看一个 QPS。除了吞吐,我同时看固定到达率下的成功率、P95/P99、 +> dropped iterations、过载时是否快速失败、SSE 首 Token、Retry 是否放大上游调用, +> 以及 Usage 是否最终完整落库。否则关闭计费或让请求在后台堆积,也能做出一个很高但 +> 没意义的数字。 + +### 12.3 为什么 Queue 不是越大越好 + +可以说: + +> Queue 的 in-flight 是每个 Provider 的并发准入,不是网关总并发。waiting 只允许更多 +> 请求排队,不会增加上游吞吐。如果上游或数据库已经满了,继续增大 waiting 只会把 P99 +> 拉长。我只在 active 达到上限、上游还有余量时增加 in-flight;如果吞吐不再增长,就 +> 缩短等待,让过载请求尽快失败。 + +### 12.4 为什么保留 Usage 的 fail-closed + +可以说: + +> 第一轮最容易得到高数字的办法是 Usage 写失败时继续放请求,但这会造成计费缺失。 +> 我保留了请求开始时的持久化边界,优化的是多余事务、重复发布和 Worker 批量落库。 +> 性能测试同时做请求数与 Usage Event 对账,确保吞吐提升不是靠丢数据换来的。 + +### 12.5 最终数据怎么填 + +最终只说结果中已经验证的数字: + +```text +修复前最高稳定点: +修复后最高稳定点: +500 RPS / 10 分钟成功率: +750 RPS P99: +1000 RPS 成功率和 P99: +网关 CPU 峰值: +PostgreSQL wait 变化: +Redis wait 变化: +Usage 对账比例: +Outbox 排空时间: +最终优化的 commit: +``` + +最后说明测试边界: + +> 这些结果来自固定规格的三台云服务器和确定性假上游,用于比较 Model-Velo 自己的版本 +> 变化。我没有在同一硬件上运行其他网关,所以不会声称绝对快于某个项目;我能证明的是 +> 测试口径可复现、瓶颈有证据、优化前后能对账。 + +这比引用其他项目 README 里的 QPS 更可信,也更容易在追问时解释清楚。 diff --git a/test/threehost/README.md b/test/threehost/README.md index 17c5e49..49f7cb1 100644 --- a/test/threehost/README.md +++ b/test/threehost/README.md @@ -154,6 +154,26 @@ bash test/threehost/run-client.sh 结果写入 `test-results/threehost//`。每个 case 都有 k6 summary JSON 和文本日志; `SAVE_RAW_METRICS=true` 时还会保存逐指标 JSON,文件会明显变大。 +### 第一轮热路径诊断 + +先不要运行完整矩阵。准备好 `benchmark.env` 后执行约 19 分钟的固定诊断轮: + +```bash +cp test/threehost/benchmark.env.example test/threehost/benchmark.env +editor test/threehost/benchmark.env +bash test/threehost/run-diagnostic-client.sh +``` + +该入口复用完整 runner,只运行 smoke、100 RPS 预热、`500/750/1000/1500 RPS × 2m` +和 `500 RPS × 10m` 耐久。它不会修改网关 Queue、Redis、PostgreSQL 或限流参数。客户端 +最大 VU 提高到 4096,避免把 1500 RPS 过载点过早归因于默认 2048 VU 上限。 + +诊断轮必须与网关机的 Prometheus 和容器采集同时运行;两台服务器的 +`DURATION_SECONDS=1800` 即可覆盖测试和余量。最终 HTML 的“热路径诊断”可以逐 case +切换,显示认证、Usage Begin、授权、Redis 限流、Quota、Provider Queue/调用、Usage +Finalize 的 P50/P95/P99,以及 PostgreSQL/Redis pool wait、Go CPU/RSS/heap/goroutine/GC +和稳定内部错误码。缺少重叠监控时这些字段不会被当成 0,而会给出证据警告。 + ### 完整性能与故障套件 `run-client.sh` 适合先验证三机连通性。正式测试改用完整配置: @@ -215,11 +235,44 @@ PRE_ALLOCATED_VUS >= RATE × 直连 p95 秒数 × 1.2 在客户端开始前,在另外两台服务器的终端中运行采集器。`DURATION_SECONDS` 应覆盖预热、 所有 repetitions、冷却间隔和可靠性测试。 +采集器现在每 10 秒输出一次进度,例如: + +```text +[2026-07-26T15:20:00Z] docker-stats state=collecting elapsed=00:10:00 remaining=02:20:00 snapshots=297 records=1188 errors=0 last=2026-07-26T15:19:59.842Z +``` + +`state=collecting` 表示正在采样,`remaining` 是离计划结束的剩余时间,`last` 是最后一个成功 +样本。对应的 `*-status.json` 会在每轮采样后原子更新;正常结束记为 `completed`,收到 +Ctrl+C、TERM 或 HUP 记为 `interrupted`,其他异常记为 `failed` 并保留原因。用 +`PROGRESS_SECONDS` 可以调整终端刷新间隔。 + +如果状态仍是 `collecting`,但 `updated_at` 已超过两个实际采样周期没有变化,应视为 +采集器已经失联或卡住,不要继续启动正式压测。可以在另一个终端持续查看状态: + +```bash +watch -n 2 'python3 -m json.tool test-results/threehost//gateway-stats-status.json' +``` + +不要让采集器依附于可能关闭的普通 SSH 会话。推荐先进入一个持久的 `tmux` 会话,再执行 +下面的前台命令;这样既能直接看到进度,又能在 SSH 断开后继续采集: + +```bash +tmux new -s model-velo-monitor +# 在 tmux 中执行本机对应的采集命令。 +# 按 Ctrl+B,再按 D,退出但不停止采集。 +tmux attach -t model-velo-monitor +``` + +如果一次客户端运行 smoke 失败或被中断,不要停止采集器后沿用同一批文件假装正式运行。 +修好问题后应使用新的 `RUN_ID`(或新的 attempt 目录),确认三台监控仍显示 +`state=collecting`,再启动正式客户端。 + 网关机: ```bash RUN_ID=20260725T120000Z DURATION_SECONDS=9000 \ +PROGRESS_SECONDS=10 \ OUTPUT_FILE="test-results/threehost/$RUN_ID/gateway-stats.jsonl" \ bash test/threehost/collect-compose-stats.sh ``` @@ -229,6 +282,7 @@ bash test/threehost/collect-compose-stats.sh ```bash RUN_ID=20260725T120000Z DURATION_SECONDS=9000 \ +PROGRESS_SECONDS=10 \ OUTPUT_DIR="test-results/threehost/$RUN_ID/prometheus" \ bash test/threehost/collect-prometheus.sh ``` @@ -238,6 +292,7 @@ bash test/threehost/collect-prometheus.sh ```bash RUN_ID=20260725T120000Z DURATION_SECONDS=9000 \ +PROGRESS_SECONDS=10 \ COMPOSE_FILE=test/threehost/upstream.compose.yaml \ SERVICES=main,fail,fallback \ OUTPUT_FILE="test-results/threehost/$RUN_ID/upstream-stats.jsonl" \ @@ -247,6 +302,10 @@ bash test/threehost/collect-compose-stats.sh 采集器按 `INTERVAL_SECONDS` 使用一次 `docker stats --no-stream` 批量记录所有目标容器的 CPU、内存、网络和块 I/O,同时单独记录服务到容器 ID 的映射、镜像 ID、Docker 版本和 宿主机信息;它不会执行 `docker compose config`,避免把 `.env` 密钥写入结果。 +`INTERVAL_SECONDS` 是最小采样周期:如果一次 `docker stats` 本身超过该时间,下一轮会 +立即开始,不会再额外 sleep;因此 1 秒配置在当前 Docker 环境中通常得到约 2 秒而不是 +原先的约 3 秒实际间隔。Docker 短暂失败时会刷新容器 ID 并继续;默认连续失败 30 次才 +将监控标记为 `failed`。 客户端套件结束后,在网关机保存一次可直接诊断的最终证据: @@ -256,10 +315,11 @@ bash test/threehost/collect-run-evidence.sh 20260725T120000Z 它保存四个服务的末尾日志、容器/镜像、最终指标、Redis Stream/PENDING/Dead Letter, 并按 `bench-` 查询 Usage 的事件数、状态、模型、缓存、Attempts、Retries、 -Fallbacks、平均/P95 延迟和 TTFT。查询前最多等待 240 秒,让本次 Outbox 和 Redis -Stream 排空,并把是否超时写入 `usage-drain.txt`。脚本不会输出 `.env` 或数据库/Redis -密码;`runtime-settings.txt` 只记录连接池、限流、Cache、Breaker、Queue、Retry、 -Timeout 和 Worker 等安全标量,便于将结果对应回实际参数。 +Fallbacks、平均/P95 延迟和 TTFT。查询前最多等待 240 秒,让本次 RUN_ID 对应的 +Outbox 排空;全局 Redis Stream 长度只作为状态记录,不参与本次运行是否排空的判断。 +结果写入 `usage-drain.txt`。脚本不会输出 `.env` 或数据库/Redis 密码; +`runtime-settings.txt` 只记录连接池、限流、Cache、Breaker、Queue、Retry、Timeout +和 Worker 等安全标量,便于将结果对应回实际参数。 三个采集进程不必精确和客户端同时结束;`9000` 秒覆盖默认完整套件,多余的空闲采样不影响 按 case 时间戳分析。如果自定义套件超过 2.5 小时,相应增大 `DURATION_SECONDS`。 @@ -277,6 +337,8 @@ scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-stats.jsonl" \ "$RESULT_DIR/" scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-stats-metadata.txt" \ "$RESULT_DIR/" +scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-stats-status.json" \ + "$RESULT_DIR/" scp -r gateway-host:"~/model-velo/$RESULT_DIR/prometheus" \ "$RESULT_DIR/" scp -r gateway-host:"~/model-velo/$RESULT_DIR/gateway-evidence" \ @@ -285,16 +347,19 @@ scp -r upstream-host:"~/model-velo/$RESULT_DIR/upstream-stats.jsonl" \ "$RESULT_DIR/" scp -r upstream-host:"~/model-velo/$RESULT_DIR/upstream-stats-metadata.txt" \ "$RESULT_DIR/" +scp -r upstream-host:"~/model-velo/$RESULT_DIR/upstream-stats-status.json" \ + "$RESULT_DIR/" python3 test/threehost/summarize.py "$RESULT_DIR" tar -czf "$RUN_ID.tar.gz" -C test-results/threehost "$RUN_ID" ``` -把最后的 `.tar.gz` 给分析者即可。`summary.md` 是人读摘要,`summary.json` 保留 -全部机器可读 case、分位数、状态计数、直连差值、逐 Chunk SSE、资源、Prometheus 和 -Usage 证据。原始 `*.log`、`*-summary.json`、`*-stream.json`、`*-upstream.json` 和 -带时间戳采样也都保留;下一轮可以据此定位是客户端 VU、网关 CPU/Queue/Redis、Usage -Worker、上游放大还是长尾问题,再改对应代码。 +把最后的 `.tar.gz` 给分析者即可。`summary.html` 是不依赖 CDN、可直接在浏览器 +打开的交互报告,`summary.md` 是终端友好的摘要,`summary.json` 保留全部机器可读 case、 +分位数、状态计数、直连差值、逐 Chunk SSE、资源、Prometheus 和 Usage 证据。原始 +`*.log`、`*-summary.json`、`*-stream.json`、`*-upstream.json` 和带时间戳采样也都保留; +下一轮可以据此定位是客户端 VU、网关 CPU/Queue/Redis、Usage Worker、上游放大还是长尾 +问题,再改对应代码。 汇总还会把正常网关 case 的实际 HTTP 请求数和落库 Usage Event 数对账;smoke、 reliability 和独立限流用例因包含非 Chat 请求或前置拒绝,不纳入该等式。 @@ -350,6 +415,7 @@ reliability 和独立限流用例因包含非 Chat 请求或前置拒绝,不 | `prepare-gateway-env.sh` | 生成不入库的三机网关 `.env` | | `run-client.sh` | 固定顺序运行并保存 k6 结果 | | `run-complete-client.sh` | 执行完整矩阵、保存 case 时间与上游计数 | +| `run-diagnostic-client.sh` | 执行约 19 分钟的第一轮热路径诊断 | | `collect-host-stats.sh` | 自动采集客户端 CPU、内存、Load 和负载进程 RSS | | `collect-compose-stats.sh` | 在被测主机本地采集容器资源 | | `collect-prometheus.sh` | 连续采集网关和 Worker 的低基数指标 | diff --git a/test/threehost/benchmark.env.example b/test/threehost/benchmark.env.example index aec3696..cba2516 100644 --- a/test/threehost/benchmark.env.example +++ b/test/threehost/benchmark.env.example @@ -1,6 +1,8 @@ # Private addresses reachable from the client host. GATEWAY_URL=http://10.0.0.20:8080 UPSTREAM_URL=http://10.0.0.30:9000 +UPSTREAM_FAIL_URL=http://10.0.0.30:9001 +UPSTREAM_FALLBACK_URL=http://10.0.0.30:9002 MODEL_VELO_API_KEY=replace-with-model-velo-api-key # Use the same run ID on the client, gateway, and upstream hosts. @@ -45,6 +47,9 @@ PROFILE_MAX_VUS=2048 # Fault and queue-overload cases. FAULT_RATE=100 FAULT_DURATION=2m +FAULT_RECOVERY_RATE=100 +FAULT_RECOVERY_FAILURE_DURATION=45s +FAULT_RECOVERY_HEALTHY_DURATION=45s QUEUE_RATE=1000 QUEUE_DURATION=2m QUEUE_PRE_ALLOCATED_VUS=2048 @@ -71,6 +76,7 @@ RUN_CACHE=true RUN_RAMP=true RUN_BURST=true RUN_FAULT=true +RUN_FAULT_RECOVERY=true RUN_QUEUE_OVERLOAD=true RUN_ENDURANCE=true RUN_RELIABILITY=true diff --git a/test/threehost/collect-compose-stats.sh b/test/threehost/collect-compose-stats.sh index 0508220..51077a6 100644 --- a/test/threehost/collect-compose-stats.sh +++ b/test/threehost/collect-compose-stats.sh @@ -9,6 +9,9 @@ services_csv=${SERVICES:-gateway,usage-worker,postgres,redis} duration_seconds=${DURATION_SECONDS:-600} interval_seconds=${INTERVAL_SECONDS:-1} output_file=${OUTPUT_FILE:-"$repo_root/test-results/threehost/compose-stats.jsonl"} +progress_seconds=${PROGRESS_SECONDS:-10} +max_consecutive_errors=${MAX_CONSECUTIVE_ERRORS:-30} +status_file=${STATUS_FILE:-"${output_file%.jsonl}-status.json"} for command_name in docker date git; do if ! command -v "$command_name" >/dev/null 2>&1; then @@ -28,42 +31,175 @@ if [[ ! "$interval_seconds" =~ ^[1-9][0-9]*$ ]]; then printf 'INTERVAL_SECONDS must be a positive integer\n' >&2 exit 1 fi +if [[ ! "$progress_seconds" =~ ^[1-9][0-9]*$ ]]; then + printf 'PROGRESS_SECONDS must be a positive integer\n' >&2 + exit 1 +fi +if [[ ! "$max_consecutive_errors" =~ ^[1-9][0-9]*$ ]]; then + printf 'MAX_CONSECUTIVE_ERRORS must be a positive integer\n' >&2 + exit 1 +fi IFS=',' read -r -a services <<<"$services_csv" -declare -a container_ids=() declare -a service_names=() for raw_service in "${services[@]}"; do service=$(printf '%s' "$raw_service" | tr -d '[:space:]') if [[ -z "$service" ]]; then continue fi - container_id=$(docker compose -f "$compose_file" ps -q "$service") - if [[ -z "$container_id" ]]; then - printf 'service is not running: %s\n' "$service" >&2 - exit 1 - fi service_names+=("$service") - container_ids+=("$container_id") done -if ((${#container_ids[@]} == 0)); then +if ((${#service_names[@]} == 0)); then printf 'SERVICES did not contain any service names\n' >&2 exit 1 fi -mkdir -p "$(dirname -- "$output_file")" +declare -a container_ids=() +refresh_container_ids() { + local service + local container_id + local -a refreshed_ids=() + for service in "${service_names[@]}"; do + container_id=$(docker compose -f "$compose_file" ps -q "$service") || return 1 + if [[ -z "$container_id" ]]; then + return 1 + fi + refreshed_ids+=("$container_id") + done + container_ids=("${refreshed_ids[@]}") +} +if ! refresh_container_ids; then + printf 'one or more services are not running: %s\n' "$services_csv" >&2 + exit 1 +fi + +mkdir -p "$(dirname -- "$output_file")" "$(dirname -- "$status_file")" metadata_file="${output_file%.jsonl}-metadata.txt" -if [[ -e "$output_file" || -e "$metadata_file" ]]; then - printf 'refusing to overwrite existing stats output: %s or %s\n' \ +if [[ -e "$output_file" || -e "$metadata_file" || -e "$status_file" ]]; then + printf 'refusing to overwrite existing stats output: %s, %s, or %s\n' \ "$output_file" \ - "$metadata_file" >&2 + "$metadata_file" \ + "$status_file" >&2 exit 1 fi + +format_duration() { + local total_seconds=$1 + printf '%02d:%02d:%02d' \ + "$((total_seconds / 3600))" \ + "$(((total_seconds % 3600) / 60))" \ + "$((total_seconds % 60))" +} + +json_escape() { + local value=$1 + value=${value//\\/\\\\} + value=${value//\"/\\\"} + value=${value//$'\n'/\\n} + value=${value//$'\r'/\\r} + value=${value//$'\t'/\\t} + printf '%s' "$value" +} + +monitor_started=$SECONDS +started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ) +deadline=$((monitor_started + duration_seconds)) +next_progress=$monitor_started +snapshots=0 +records=0 +errors=0 +consecutive_errors=0 +last_sample_at= +last_error= +stop_requested=false +termination_signal= +termination_exit_code=1 +completed=false + +write_status() { + local state=$1 + local reason=${2:-} + local elapsed=$((SECONDS - monitor_started)) + local remaining=$((deadline - SECONDS)) + local temporary_status="${status_file}.tmp.$$" + if ((remaining < 0)); then + remaining=0 + fi + printf '{"state":"%s","pid":%d,"started_at":"%s","updated_at":"%s",' \ + "$state" \ + "$$" \ + "$started_at" \ + "$(date -u +%Y-%m-%dT%H:%M:%SZ)" >"$temporary_status" + printf '"duration_seconds":%d,"elapsed_seconds":%d,"remaining_seconds":%d,' \ + "$duration_seconds" \ + "$elapsed" \ + "$remaining" >>"$temporary_status" + printf '"snapshots":%d,"records":%d,"errors":%d,"consecutive_errors":%d,' \ + "$snapshots" \ + "$records" \ + "$errors" \ + "$consecutive_errors" >>"$temporary_status" + printf '"last_sample_at":"%s","reason":"%s","last_error":"%s"}\n' \ + "$last_sample_at" \ + "$(json_escape "$reason")" \ + "$(json_escape "$last_error")" >>"$temporary_status" + mv -- "$temporary_status" "$status_file" +} + +print_progress() { + local state=$1 + local elapsed=$((SECONDS - monitor_started)) + local remaining=$((deadline - SECONDS)) + if ((remaining < 0)); then + remaining=0 + fi + printf '[%s] docker-stats state=%s elapsed=%s remaining=%s snapshots=%d records=%d errors=%d last=%s\n' \ + "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + "$state" \ + "$(format_duration "$elapsed")" \ + "$(format_duration "$remaining")" \ + "$snapshots" \ + "$records" \ + "$errors" \ + "${last_sample_at:-none}" +} + +handle_signal() { + termination_signal=$1 + termination_exit_code=$2 + stop_requested=true +} + +finish_monitor() { + local exit_code=$1 + local state=failed + local reason="exit:$exit_code" + set +e + if [[ "$completed" == "true" ]]; then + state=completed + reason=duration_reached + elif [[ -n "$termination_signal" ]]; then + state=interrupted + reason="signal:$termination_signal" + fi + write_status "$state" "$reason" + print_progress "$state" +} + +trap 'handle_signal INT 130' INT +trap 'handle_signal TERM 143' TERM +trap 'handle_signal HUP 129' HUP +trap 'finish_monitor $?' EXIT + { - printf 'captured_at=%s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" + printf 'captured_at=%s\n' "$started_at" printf 'compose_file=%s\n' "$compose_file" printf 'services=%s\n' "$services_csv" printf 'duration_seconds=%s\n' "$duration_seconds" printf 'interval_seconds=%s\n' "$interval_seconds" + printf 'progress_seconds=%s\n' "$progress_seconds" + printf 'max_consecutive_errors=%s\n' "$max_consecutive_errors" + printf 'status_file=%s\n' "$status_file" printf 'host=%s\n' "$(uname -a)" printf 'commit=%s\n' "$(git -C "$repo_root" rev-parse HEAD)" if [[ -n "$(git -C "$repo_root" status --porcelain)" ]]; then @@ -95,23 +231,66 @@ fi fi } >"$metadata_file" -deadline=$((SECONDS + duration_seconds)) : >"$output_file" -while ((SECONDS < deadline)); do +write_status starting +print_progress starting +while ((SECONDS < deadline)) && [[ "$stop_requested" == "false" ]]; do + iteration_started=$SECONDS timestamp=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ) - while IFS= read -r stats; do - if [[ -n "$stats" ]]; then - printf '{"timestamp":"%s","stats":%s}\n' \ - "$timestamp" \ - "$stats" >>"$output_file" - fi - done < <( - docker stats \ + stats_output= + if stats_output=$(docker stats \ --no-stream \ --format '{{json .}}' \ - "${container_ids[@]}" - ) - sleep "$interval_seconds" + "${container_ids[@]}" 2>&1); then + snapshot_records=0 + while IFS= read -r stats; do + if [[ -n "$stats" ]]; then + printf '{"timestamp":"%s","stats":%s}\n' \ + "$timestamp" \ + "$stats" >>"$output_file" + snapshot_records=$((snapshot_records + 1)) + records=$((records + 1)) + fi + done <<<"$stats_output" + if ((snapshot_records > 0)); then + snapshots=$((snapshots + 1)) + consecutive_errors=0 + last_error= + last_sample_at=$timestamp + else + errors=$((errors + 1)) + consecutive_errors=$((consecutive_errors + 1)) + last_error="docker stats returned no records" + fi + else + errors=$((errors + 1)) + consecutive_errors=$((consecutive_errors + 1)) + last_error=$stats_output + refresh_container_ids || true + fi + + write_status collecting + if ((SECONDS >= next_progress)); then + print_progress collecting + next_progress=$((SECONDS + progress_seconds)) + fi + if ((consecutive_errors >= max_consecutive_errors)); then + printf 'docker stats failed %d consecutive times: %s\n' \ + "$consecutive_errors" \ + "$last_error" >&2 + exit 1 + fi + + iteration_elapsed=$((SECONDS - iteration_started)) + sleep_seconds=$((interval_seconds - iteration_elapsed)) + if ((sleep_seconds > 0)) && ((SECONDS < deadline)); then + sleep "$sleep_seconds" & + wait $! || true + fi done +if [[ "$stop_requested" == "true" ]]; then + exit "$termination_exit_code" +fi +completed=true printf 'wrote %s and %s\n' "$output_file" "$metadata_file" diff --git a/test/threehost/collect-prometheus.sh b/test/threehost/collect-prometheus.sh index 2f7a3e2..f35ac5f 100644 --- a/test/threehost/collect-prometheus.sh +++ b/test/threehost/collect-prometheus.sh @@ -24,6 +24,9 @@ set +a duration_seconds=${DURATION_SECONDS:-7200} interval_seconds=${INTERVAL_SECONDS:-1} output_dir=${OUTPUT_DIR:-"$repo_root/test-results/threehost/metrics"} +progress_seconds=${PROGRESS_SECONDS:-10} +max_consecutive_errors=${MAX_CONSECUTIVE_ERRORS:-30} +status_file=${STATUS_FILE:-"$output_dir/monitor-status.json"} bind_address=${MODEL_VELO_HTTP_BIND:-127.0.0.1} if [[ "$bind_address" == "0.0.0.0" || "$bind_address" == "[::]" ]]; then bind_address=127.0.0.1 @@ -40,6 +43,14 @@ if [[ ! "$interval_seconds" =~ ^[1-9][0-9]*$ ]]; then printf 'INTERVAL_SECONDS must be a positive integer\n' >&2 exit 1 fi +if [[ ! "$progress_seconds" =~ ^[1-9][0-9]*$ ]]; then + printf 'PROGRESS_SECONDS must be a positive integer\n' >&2 + exit 1 +fi +if [[ ! "$max_consecutive_errors" =~ ^[1-9][0-9]*$ ]]; then + printf 'MAX_CONSECUTIVE_ERRORS must be a positive integer\n' >&2 + exit 1 +fi if [[ -e "$output_dir" ]]; then printf 'refusing to overwrite metrics output directory: %s\n' "$output_dir" >&2 exit 1 @@ -52,10 +63,121 @@ metadata_output="$output_dir/prometheus-metadata.txt" : >"$gateway_output" : >"$worker_output" +format_duration() { + local total_seconds=$1 + printf '%02d:%02d:%02d' \ + "$((total_seconds / 3600))" \ + "$(((total_seconds % 3600) / 60))" \ + "$((total_seconds % 60))" +} + +json_escape() { + local value=$1 + value=${value//\\/\\\\} + value=${value//\"/\\\"} + value=${value//$'\n'/\\n} + value=${value//$'\r'/\\r} + value=${value//$'\t'/\\t} + printf '%s' "$value" +} + +monitor_started=$SECONDS +started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ) +deadline=$((monitor_started + duration_seconds)) +next_progress=$monitor_started +snapshots=0 +scrapes=0 +errors=0 +consecutive_errors=0 +last_sample_at= +last_error= +stop_requested=false +termination_signal= +termination_exit_code=1 +completed=false + +write_status() { + local state=$1 + local reason=${2:-} + local elapsed=$((SECONDS - monitor_started)) + local remaining=$((deadline - SECONDS)) + local temporary_status="${status_file}.tmp.$$" + if ((remaining < 0)); then + remaining=0 + fi + printf '{"state":"%s","pid":%d,"started_at":"%s","updated_at":"%s",' \ + "$state" \ + "$$" \ + "$started_at" \ + "$(date -u +%Y-%m-%dT%H:%M:%SZ)" >"$temporary_status" + printf '"duration_seconds":%d,"elapsed_seconds":%d,"remaining_seconds":%d,' \ + "$duration_seconds" \ + "$elapsed" \ + "$remaining" >>"$temporary_status" + printf '"snapshots":%d,"scrapes":%d,"errors":%d,"consecutive_errors":%d,' \ + "$snapshots" \ + "$scrapes" \ + "$errors" \ + "$consecutive_errors" >>"$temporary_status" + printf '"last_sample_at":"%s","reason":"%s","last_error":"%s"}\n' \ + "$last_sample_at" \ + "$(json_escape "$reason")" \ + "$(json_escape "$last_error")" >>"$temporary_status" + mv -- "$temporary_status" "$status_file" +} + +print_progress() { + local state=$1 + local elapsed=$((SECONDS - monitor_started)) + local remaining=$((deadline - SECONDS)) + if ((remaining < 0)); then + remaining=0 + fi + printf '[%s] prometheus state=%s elapsed=%s remaining=%s snapshots=%d scrapes=%d errors=%d last=%s\n' \ + "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + "$state" \ + "$(format_duration "$elapsed")" \ + "$(format_duration "$remaining")" \ + "$snapshots" \ + "$scrapes" \ + "$errors" \ + "${last_sample_at:-none}" +} + +handle_signal() { + termination_signal=$1 + termination_exit_code=$2 + stop_requested=true +} + +finish_monitor() { + local exit_code=$1 + local state=failed + local reason="exit:$exit_code" + set +e + if [[ "$completed" == "true" ]]; then + state=completed + reason=duration_reached + elif [[ -n "$termination_signal" ]]; then + state=interrupted + reason="signal:$termination_signal" + fi + write_status "$state" "$reason" + print_progress "$state" +} + +trap 'handle_signal INT 130' INT +trap 'handle_signal TERM 143' TERM +trap 'handle_signal HUP 129' HUP +trap 'finish_monitor $?' EXIT + { - printf 'captured_at=%s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" + printf 'captured_at=%s\n' "$started_at" printf 'duration_seconds=%s\n' "$duration_seconds" printf 'interval_seconds=%s\n' "$interval_seconds" + printf 'progress_seconds=%s\n' "$progress_seconds" + printf 'max_consecutive_errors=%s\n' "$max_consecutive_errors" + printf 'status_file=%s\n' "$status_file" printf 'gateway_url=%s\n' "$gateway_url" printf 'worker_url=%s\n' "$worker_url" printf 'commit=%s\n' "$(git -C "$repo_root" rev-parse HEAD)" @@ -75,18 +197,60 @@ capture() { printf '# snapshot %s\n' "$timestamp" >>"$output" if ! payload=$(curl "${curl_args[@]}" "$url"); then printf '# scrape_error\n' >>"$output" - return + last_error="scrape failed: $url" + return 1 fi printf '%s\n' "$payload" | - awk '/^model_velo_[A-Za-z0-9_:]+({[^}]*})? [^ ]+/' >>"$output" + awk '/^(model_velo_|go_|process_)[A-Za-z0-9_:]+({[^}]*})? [^ ]+/' >>"$output" + return 0 } -deadline=$((SECONDS + duration_seconds)) -while ((SECONDS < deadline)); do +write_status starting +print_progress starting +while ((SECONDS < deadline)) && [[ "$stop_requested" == "false" ]]; do + iteration_started=$SECONDS timestamp=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ) - capture "$gateway_url" "$gateway_output" "$timestamp" - capture "$worker_url" "$worker_output" "$timestamp" - sleep "$interval_seconds" + snapshot_errors=0 + if ! capture "$gateway_url" "$gateway_output" "$timestamp"; then + snapshot_errors=$((snapshot_errors + 1)) + fi + if ! capture "$worker_url" "$worker_output" "$timestamp"; then + snapshot_errors=$((snapshot_errors + 1)) + fi + snapshots=$((snapshots + 1)) + scrapes=$((scrapes + 2)) + errors=$((errors + snapshot_errors)) + if ((snapshot_errors == 2)); then + consecutive_errors=$((consecutive_errors + 1)) + else + last_sample_at=$timestamp + consecutive_errors=0 + if ((snapshot_errors == 0)); then + last_error= + fi + fi + + write_status collecting + if ((SECONDS >= next_progress)); then + print_progress collecting + next_progress=$((SECONDS + progress_seconds)) + fi + if ((consecutive_errors >= max_consecutive_errors)); then + printf 'both Prometheus scrapes failed %d consecutive times\n' \ + "$consecutive_errors" >&2 + exit 1 + fi + + iteration_elapsed=$((SECONDS - iteration_started)) + sleep_seconds=$((interval_seconds - iteration_elapsed)) + if ((sleep_seconds > 0)) && ((SECONDS < deadline)); then + sleep "$sleep_seconds" & + wait $! || true + fi done +if [[ "$stop_requested" == "true" ]]; then + exit "$termination_exit_code" +fi +completed=true printf 'wrote metrics under %s\n' "$output_dir" diff --git a/test/threehost/collect-run-evidence.sh b/test/threehost/collect-run-evidence.sh index 86d2dd0..bec197f 100644 --- a/test/threehost/collect-run-evidence.sh +++ b/test/threehost/collect-run-evidence.sh @@ -69,6 +69,12 @@ safe_setting_names=( MODEL_VELO_POSTGRES_MAX_IDLE_CONNS MODEL_VELO_POSTGRES_MAX_CONN_IDLE_TIME MODEL_VELO_POSTGRES_MAX_CONN_LIFETIME + MODEL_VELO_AUTH_CACHE_ENABLED + MODEL_VELO_AUTH_CACHE_L1_MAX_ENTRIES + MODEL_VELO_AUTH_CACHE_L1_TTL + MODEL_VELO_AUTH_CACHE_L2_TTL + MODEL_VELO_AUTH_CACHE_KEY_PREFIX + MODEL_VELO_AUTH_CACHE_INVALIDATION_CHANNEL MODEL_VELO_REDIS_DB MODEL_VELO_REDIS_POOL_SIZE MODEL_VELO_REDIS_MIN_IDLE_CONNS @@ -140,17 +146,17 @@ redis_command() { drain_deadline=$((SECONDS + drain_timeout_seconds)) drain_state=timeout -unfinished_outbox=-1 +remaining_outbox=-1 stream_length=-1 while ((SECONDS < drain_deadline)); do - unfinished_outbox=$(docker compose -f "$compose_file" exec -T postgres \ + remaining_outbox=$(docker compose -f "$compose_file" exec -T postgres \ psql \ -U "$postgres_user" \ -d "$postgres_database" \ -At \ - -c "SELECT count(*) FROM usage_outbox WHERE left(request_id, $prefix_length) = '$request_prefix' AND state <> 'published';") + -c "SELECT count(*) FROM usage_outbox WHERE left(request_id, $prefix_length) = '$request_prefix';") stream_length=$(redis_command --raw XLEN "$usage_stream") - if [[ "$unfinished_outbox" == "0" && "$stream_length" == "0" ]]; then + if [[ "$remaining_outbox" == "0" ]]; then drain_state=complete break fi @@ -159,7 +165,7 @@ done { printf 'state=%s\n' "$drain_state" printf 'timeout_seconds=%s\n' "$drain_timeout_seconds" - printf 'unfinished_outbox=%s\n' "$unfinished_outbox" + printf 'remaining_outbox=%s\n' "$remaining_outbox" printf 'stream_length=%s\n' "$stream_length" } >"$output_dir/usage-drain.txt" @@ -189,7 +195,7 @@ docker compose -f "$compose_file" exec -T postgres \ -U "$postgres_user" \ -d "$postgres_database" \ --csv \ - -c "SELECT requested_model, status, stream, cache_status, error_category, error_code, count(*) AS events, sum(attempts) AS attempts, sum(retries) AS retries, sum(fallbacks) AS fallbacks, round(avg(latency_ms)::numeric, 2) AS latency_avg_ms, percentile_cont(0.95) WITHIN GROUP (ORDER BY latency_ms) AS latency_p95_ms, round(avg(first_token_ms)::numeric, 2) AS first_token_avg_ms FROM usage_events WHERE left(request_id, $prefix_length) = '$request_prefix' GROUP BY requested_model, status, stream, cache_status, error_category, error_code ORDER BY requested_model, status, stream, cache_status, error_category, error_code;" \ + -c "SELECT requested_model, status, stream, cache_status, error_category, error_code, count(*) AS events, sum(attempts) AS attempts, sum(retries) AS retries, sum(fallbacks) AS fallbacks, round(avg(latency_ms)::numeric, 2) AS latency_avg_ms, percentile_cont(0.95) WITHIN GROUP (ORDER BY latency_ms) AS latency_p95_ms, percentile_cont(0.99) WITHIN GROUP (ORDER BY latency_ms) AS latency_p99_ms, round(avg(first_token_ms)::numeric, 2) AS first_token_avg_ms, percentile_cont(0.95) WITHIN GROUP (ORDER BY first_token_ms) AS first_token_p95_ms, percentile_cont(0.99) WITHIN GROUP (ORDER BY first_token_ms) AS first_token_p99_ms FROM usage_events WHERE left(request_id, $prefix_length) = '$request_prefix' GROUP BY requested_model, status, stream, cache_status, error_category, error_code ORDER BY requested_model, status, stream, cache_status, error_category, error_code;" \ >"$output_dir/usage-diagnostics.csv" { diff --git a/test/threehost/prepare-gateway-env.sh b/test/threehost/prepare-gateway-env.sh index a9a88a8..e12b621 100644 --- a/test/threehost/prepare-gateway-env.sh +++ b/test/threehost/prepare-gateway-env.sh @@ -76,6 +76,12 @@ umask 077 printf 'MODEL_VELO_POSTGRES_USER=model_velo\n' printf 'MODEL_VELO_POSTGRES_PASSWORD=%s\n' "$postgres_password" printf 'MODEL_VELO_POSTGRES_MAX_OPEN_CONNS=50\n' + printf 'MODEL_VELO_AUTH_CACHE_ENABLED=true\n' + printf 'MODEL_VELO_AUTH_CACHE_L1_MAX_ENTRIES=10000\n' + printf 'MODEL_VELO_AUTH_CACHE_L1_TTL=15s\n' + printf 'MODEL_VELO_AUTH_CACHE_L2_TTL=30s\n' + printf 'MODEL_VELO_AUTH_CACHE_KEY_PREFIX=model-velo:threehost:auth:v1\n' + printf 'MODEL_VELO_AUTH_CACHE_INVALIDATION_CHANNEL=model-velo:threehost:auth:v1:invalidate\n' printf 'MODEL_VELO_POSTGRES_MAX_IDLE_CONNS=10\n' printf 'MODEL_VELO_API_KEY_PEPPER=%s\n' "$api_key_pepper" printf 'MODEL_VELO_ADMIN_KEY_PEPPER=%s\n' "$admin_key_pepper" diff --git a/test/threehost/render_html.py b/test/threehost/render_html.py new file mode 100644 index 0000000..2b31ec7 --- /dev/null +++ b/test/threehost/render_html.py @@ -0,0 +1,256 @@ +#!/usr/bin/env python3 + +import json +import re +import sys +from datetime import datetime, timezone +from pathlib import Path + + +PLACEHOLDER = "__MODEL_VELO_REPORT_DATA__" + + +def read_text(path): + try: + return path.read_text(encoding="utf-8", errors="replace") + except OSError: + return "" + + +def read_json(path): + try: + return json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + + +def parse_metadata(text): + values = {} + for line in text.splitlines(): + if "=" in line: + key, value = line.split("=", 1) + values[key.strip()] = value.strip() + return values + + +def read_json_lines(path): + rows = [] + for line in read_text(path).splitlines(): + try: + rows.append(json.loads(line)) + except json.JSONDecodeError: + continue + return rows + + +def attempt_status(path, current): + if path == current: + return "complete" + name = path.name.lower() + if "smoke-failed" in name: + return "smoke failed" + if "interrupted" in name: + return "interrupted" + if "failed" in name: + return "earlier failed run" + return "other" + + +def collect_attempts(result_dir): + attempts = [] + for path in sorted(item for item in result_dir.parent.iterdir() if item.is_dir()): + files = [item for item in path.rglob("*") if item.is_file()] + metadata_text = read_text(path / "client-metadata.txt") + metadata = parse_metadata(metadata_text) + cases_path = path / "cases.tsv" + case_count = 0 + if cases_path.exists(): + case_count = max(0, len(read_text(cases_path).splitlines()) - 1) + attempts.append( + { + "name": path.name, + "status": attempt_status(path, result_dir), + "case_count": case_count, + "file_count": len(files), + "bytes": sum(item.stat().st_size for item in files), + "commit": metadata.get("commit", ""), + "started_at": metadata.get("started_at", ""), + "ended_at": metadata.get("ended_at", ""), + "has_summary": (path / "summary.json").exists(), + } + ) + return attempts + + +def collect_artifacts(result_dir): + artifacts = [] + for path in sorted( + item + for item in result_dir.rglob("*") + if item.is_file() and item.name != "summary.html" + ): + relative = path.relative_to(result_dir).as_posix() + artifacts.append( + { + "path": relative, + "bytes": path.stat().st_size, + "modified_at": datetime.fromtimestamp( + path.stat().st_mtime, timezone.utc + ).isoformat(), + } + ) + return artifacts + + +def reliability_checks(result_dir): + checks = [] + for line in read_text(result_dir / "reliability.log").splitlines(): + match = re.match(r"^\s*✓\s+(.+?)\s*$", line) + if match and "rate==" not in match.group(1): + checks.append(match.group(1)) + return checks + + +def parse_iso(value): + try: + return datetime.fromisoformat(str(value).replace("Z", "+00:00")) + except (TypeError, ValueError): + return None + + +def capture_relation(first, last, run_start, run_end): + if not all((first, last, run_start, run_end)): + return "unknown" + if last < run_start: + return "before" + if first > run_end: + return "after" + return "overlap" + + +def parse_size(value): + match = re.match(r"^\s*([0-9.]+)\s*([KMGT]?i?B)\s*$", str(value)) + if not match: + return 0.0 + amount = float(match.group(1)) + unit = match.group(2) + powers = {"B": 0, "KB": 1, "KiB": 1, "MB": 2, "MiB": 2, + "GB": 3, "GiB": 3, "TB": 4, "TiB": 4} + base = 1024 if "iB" in unit else 1000 + return amount * base ** powers[unit] + + +def collect_server_captures(result_dir, metadata): + run_start = parse_iso(metadata.get("started_at")) + run_end = parse_iso(metadata.get("ended_at")) + captures = [] + for path in sorted(result_dir.rglob("*stats*.jsonl")): + if path.name == "client-stats.jsonl": + continue + timestamps = [] + containers = {} + for payload in read_json_lines(path): + timestamp = parse_iso(payload.get("timestamp")) + if timestamp: + timestamps.append(timestamp) + stats = payload.get("stats", {}) + name = stats.get("Name") or stats.get("Container") or "unknown" + aggregate = containers.setdefault( + name, + {"samples": 0, "cpu_sum_pct": 0.0, "cpu_max_pct": 0.0, + "memory_max_mb": 0.0}, + ) + cpu = float(str(stats.get("CPUPerc", "0")).rstrip("%") or 0) + memory = str(stats.get("MemUsage", "")).split("/", 1)[0] + aggregate["samples"] += 1 + aggregate["cpu_sum_pct"] += cpu + aggregate["cpu_max_pct"] = max(aggregate["cpu_max_pct"], cpu) + aggregate["memory_max_mb"] = max( + aggregate["memory_max_mb"], parse_size(memory) / (1024 * 1024) + ) + first = min(timestamps) if timestamps else None + last = max(timestamps) if timestamps else None + for aggregate in containers.values(): + aggregate["cpu_avg_pct"] = ( + aggregate.pop("cpu_sum_pct") / max(1, aggregate["samples"]) + ) + captures.append( + { + "path": path.relative_to(result_dir).as_posix(), + "kind": "docker-stats", + "first": first.isoformat() if first else "", + "last": last.isoformat() if last else "", + "relation": capture_relation(first, last, run_start, run_end), + "samples": len(timestamps), + "containers": containers, + } + ) + + for path in sorted(result_dir.rglob("*.promlog")): + timestamps = [ + parse_iso(match.group(1)) + for match in re.finditer( + r"^# snapshot\s+(\S+)\s*$", + read_text(path), + flags=re.MULTILINE, + ) + ] + timestamps = [timestamp for timestamp in timestamps if timestamp] + first = min(timestamps) if timestamps else None + last = max(timestamps) if timestamps else None + captures.append( + { + "path": path.relative_to(result_dir).as_posix(), + "kind": "prometheus", + "first": first.isoformat() if first else "", + "last": last.isoformat() if last else "", + "relation": capture_relation(first, last, run_start, run_end), + "samples": len(timestamps), + "containers": {}, + } + ) + return captures + + +def render(result_dir): + summary_path = result_dir / "summary.json" + if not summary_path.exists(): + raise SystemExit(f"summary not found: {summary_path}") + + template_path = Path(__file__).with_name("report-template.html") + template = read_text(template_path) + if PLACEHOLDER not in template: + raise SystemExit(f"report placeholder not found in {template_path}") + + metadata_text = read_text(result_dir / "client-metadata.txt") + metadata = parse_metadata(metadata_text) + report_data = { + "summary": read_json(summary_path), + "metadata": metadata, + "metadata_text": metadata_text, + "client_stats": read_json_lines(result_dir / "client-stats.jsonl"), + "attempts": collect_attempts(result_dir), + "artifacts": collect_artifacts(result_dir), + "reliability_checks": reliability_checks(result_dir), + "server_captures": collect_server_captures(result_dir, metadata), + "generated_at": datetime.now(timezone.utc).isoformat(), + } + encoded = json.dumps( + report_data, ensure_ascii=False, separators=(",", ":") + ).replace("<", "\\u003c") + output = result_dir / "summary.html" + output.write_text(template.replace(PLACEHOLDER, encoded), encoding="utf-8") + print(f"wrote {output}") + + +def main(): + if len(sys.argv) != 2: + raise SystemExit(f"usage: {Path(sys.argv[0]).name} ") + result_dir = Path(sys.argv[1]).resolve() + if not result_dir.is_dir(): + raise SystemExit(f"result directory not found: {result_dir}") + render(result_dir) + + +if __name__ == "__main__": + main() diff --git a/test/threehost/report-template.html b/test/threehost/report-template.html new file mode 100644 index 0000000..69c74df --- /dev/null +++ b/test/threehost/report-template.html @@ -0,0 +1,1185 @@ + + + + + + + Model‑Velo 三机性能实验报告 + + + +
+ + +
+
+
+
Three‑host benchmark dossier · 2026‑07
+

三机性能实验与热路径证据。

+

客户端 → Model‑Velo → Go 假上游;容量、延迟、依赖池和错误码使用同一 UTC 时间窗。

+
+
+ +
+ 01 / decision line +

当前最可辩护的结论

+
+
+ 0 RPS 严格运行线。 +

严格口径为成功率 ≥ 99.9%、k6 丢弃迭代为 0,并同时观察 P99。

+
+
+
357.56
+
ms · strict point P99
+

低负载 P50 增量 未运行;流式首内容 P50 增量 未运行。

+
+
+
+
0 RPS5007501,0001,5002,000
+
+ + + +
+
+
0–500稳定区
+
500–750严格 SLO 区
+
750–1,000边缘区
+
1,500+过载塌陷区
+
+
+
+
+ +
+
+
02 / load envelope

容量、吞吐与尾延迟

+

闭环并发回答“固定并发下能跑多快”,开环到达率回答“给定业务流量能不能准时接住”。生产容量线应以后者为主,前者用于观察调度和平台形状。

+
+
+
+

闭环网关吞吐

+

每档并发按重复运行取中位数;闭环峰值用于定位吞吐拐点,不等同于生产 SLO。

+
+
+
+

闭环 P99

+

同时展示 P50 与 P99,便于识别并发上升后是否出现排队和尾延迟放大。

+
+
+
+
+
+

开环:目标、实际与丢弃

+
+
+
+

开环:成功率与 P99

+
+
+
+
+ 1,500 RPS 的问题发生在上游之前,但本轮证据还不足以定位代码。 + 一个代表性 case 中,42,938 个 HTTP 请求只有 24,462 个到达假上游并成功,另有 18,476 个 5xx,P99 约 3.0 秒。假上游最大 active 只有个位数;这排除了“假 LLM 被打满”,却不能在缺少与主运行重叠的网关日志、容器资源和 Prometheus 时序时继续下结论。 +
+
+

直连基线 vs 网关增量

+

直连假上游在高并发时把客户端 CPU 打到 99% 以上;网关 case 的客户端 CPU 峰值仅约 10–12%。因此,摘要里的“客户端可能限制容量”只适用于直连高吞吐基线,不足以解释网关约 1.25k RPS 的平台。

+
并发直连 RPS网关 RPS网关/直连P50 增量P99 增量
+
+
+ +
+
+
03 / server-sent events

SSE 回放与逐 Chunk 观测

+

独立 stream loader 记录 headers、首事件、首内容、总耗时和 chunk 间隔;请求数、并发和重复次数均取自本次运行元数据。

+
+
+
+
+

首内容与总完成时间

+
+
+
+

怎么解读

+
+
+
+
+ +
+
+
04 / failure semantics

缓存、故障恢复、队列与耐久

+

这一组不是单纯追求成功率:error10 和 queue overload 会有意制造可接受错误。报告将“测试断言成功”和“业务 HTTP 2xx”分开。

+
+
+
+
+

17 项可靠性断言

+
    +
    +
    +

    本次可靠性诊断

    +
    +
    +
    +
    + +
    +
    +
    05 / hot path

    请求阶段、依赖池与错误码

    +

    每个 case 使用开始前和结束后的 Prometheus 累计值做差。阶段分位数来自 Histogram;reliability 包含 Queue、Provider、Retry 和 Fallback,因此不能与其子阶段相加。

    +
    +
    + +
    +
    +
    +
    +

    阶段 P99

    +
    +
    +
    +

    阶段分位数

    +
    +
    阶段样本avg msP50 msP95 msP99 ms
    +
    +
    +
    +
    +

    该 case 的网关错误码

    +
    +
    HTTP内部错误码次数
    +
    +
    +
    +
    + +
    +
    +
    06 / runtime envelope

    客户端、网关容器与镜像

    +

    资源数据只使用与 case 时间窗重叠的采样;同时展示客户端负载生成器、网关容器内存和最终镜像体积。

    +
    +
    +
    +
    +

    按阶段 CPU 峰值

    +
    +
    +
    +

    按阶段资源明细

    +
    阶段样本CPU avgCPU max内存 maxk6 RSS max
    +
    +
    +
    +

    服务器采集窗口对齐

    +
    文件类型与主运行关系采样开始 UTC结束 UTC资源峰值
    +
    +
    服务器侧文件存在,但没有覆盖主运行。
    +
    + +
    +
    +
    07 / evidence audit

    哪些结论能说,哪些还不能

    +

    “文件不存在”不是“指标为零”。尤其是 Usage:自动摘要把缺失证据显示成 observed=0,但这不能解释为 243 万事件全部丢失。

    +
    +
    +
    +
    +

    四次运行痕迹

    +
    +
    +
    +

    证据审计结论

    +

    本轮是一次完整、可用于建立容量包络的客户端侧实验,但还不是一次可直接驱动代码改动的服务器侧性能剖析。

    +

    下一轮必须让网关/上游容器 stats、Prometheus、gateway evidence、错误响应分类和网关日志真正覆盖客户端主运行窗口。做到后,1,500 RPS 的 5xx 才能直接映射到 Queue wait timeout、Breaker、Redis、DB pool 或 Usage 链路。

    +

    主运行使用提交 ,且客户端 worktree=dirty。当前工作区之后的参数改动不属于这次实测。

    +
    +
    +
    + + + +
    +
    +
    09 / reproducibility

    环境、配置与测量口径

    +

    本报告对应测试提交,不对应当前未提交工作区。所有时间为 UTC;私网地址仅用于复盘三机拓扑。

    +
    +
    +
    +

    测试矩阵

    +
    +
    +
    +

    客户端环境

    +
    +
    +
    +
    查看完整 client-metadata.txt
    +
    + +
    +
    +
    10 / complete case ledger

    全部测试 Case

    +

    这里列出本次实际生成的全部 case;数值保留到足以回查同目录原始 JSON 和日志。

    +
    +
    + + + +
    +
    + + + +
    Case阶段目标负载重复请求RPS成功P50P95P99Dropped上游 calls原始文件
    +
    +

    +
    + +
    +
    +
    11 / raw evidence

    原始产物索引

    +

    主运行目录内的每个文件都列在这里。日志和 JSON 链接使用相对路径,保持本 HTML 与结果目录一起移动即可打开。

    +
    +
    +
    +
    路径大小UTC 修改时间
    +
    +

    +
    查看本报告内嵌数据说明
    summary.html 内嵌 summary.json、client-metadata.txt、client-stats.jsonl 的解析结果、运行尝试目录摘要、可靠性断言名称和产物索引。原始 k6 日志与每个 case 的 JSON 不重复嵌入,使用同目录相对链接访问。
    +
    + +
    +

    Generated · Model‑Velo Performance Lab · 本报告不构成生产 SLA。

    +
    +
    +
    +
    + + + + + diff --git a/test/threehost/run-complete-client.sh b/test/threehost/run-complete-client.sh index 10dacb9..0cc4218 100644 --- a/test/threehost/run-complete-client.sh +++ b/test/threehost/run-complete-client.sh @@ -5,7 +5,7 @@ script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) repo_root=$(CDPATH= cd -- "$script_dir/../.." && pwd) env_file=${1:-"$script_dir/benchmark.env"} -if [[ ! -f "$env_file" ]]; then +if [[ "$env_file" != "-" && ! -f "$env_file" ]]; then printf 'missing benchmark environment file: %s\n' "$env_file" >&2 exit 1 fi @@ -17,13 +17,22 @@ for command_name in curl git k6 python3; do done set -a -# shellcheck disable=SC1090 -. "$env_file" +if [[ "$env_file" != "-" ]]; then + # shellcheck disable=SC1090 + . "$env_file" +fi set +a : "${GATEWAY_URL:?GATEWAY_URL is required}" : "${UPSTREAM_URL:?UPSTREAM_URL is required}" : "${MODEL_VELO_API_KEY:?MODEL_VELO_API_KEY is required}" +UPSTREAM_FAIL_URL=${UPSTREAM_FAIL_URL:-} +UPSTREAM_FALLBACK_URL=${UPSTREAM_FALLBACK_URL:-} +upstream_origin=${UPSTREAM_URL%/} +if [[ "$upstream_origin" =~ ^(https?://.+):9000$ ]]; then + UPSTREAM_FAIL_URL=${UPSTREAM_FAIL_URL:-"${BASH_REMATCH[1]}:9001"} + UPSTREAM_FALLBACK_URL=${UPSTREAM_FALLBACK_URL:-"${BASH_REMATCH[1]}:9002"} +fi REQUEST_TIMEOUT=${REQUEST_TIMEOUT:-30s} COOLDOWN_SECONDS=${COOLDOWN_SECONDS:-3} @@ -50,6 +59,9 @@ PROFILE_PRE_ALLOCATED_VUS=${PROFILE_PRE_ALLOCATED_VUS:-512} PROFILE_MAX_VUS=${PROFILE_MAX_VUS:-2048} FAULT_RATE=${FAULT_RATE:-100} FAULT_DURATION=${FAULT_DURATION:-2m} +FAULT_RECOVERY_RATE=${FAULT_RECOVERY_RATE:-100} +FAULT_RECOVERY_FAILURE_DURATION=${FAULT_RECOVERY_FAILURE_DURATION:-45s} +FAULT_RECOVERY_HEALTHY_DURATION=${FAULT_RECOVERY_HEALTHY_DURATION:-45s} QUEUE_RATE=${QUEUE_RATE:-1000} QUEUE_DURATION=${QUEUE_DURATION:-2m} QUEUE_PRE_ALLOCATED_VUS=${QUEUE_PRE_ALLOCATED_VUS:-2048} @@ -62,6 +74,7 @@ RATE_LIMIT_TEST_RATE=${RATE_LIMIT_TEST_RATE:-100} RATE_LIMIT_TEST_DURATION=${RATE_LIMIT_TEST_DURATION:-2m} RESULTS_ROOT=${RESULTS_ROOT:-"$repo_root/test-results/threehost"} SAVE_RAW_METRICS=${SAVE_RAW_METRICS:-false} +TEST_PROFILE=${TEST_PROFILE:-complete} RUN_CAPACITY=${RUN_CAPACITY:-true} RUN_WARMUP=${RUN_WARMUP:-true} @@ -72,6 +85,7 @@ RUN_CACHE=${RUN_CACHE:-true} RUN_RAMP=${RUN_RAMP:-true} RUN_BURST=${RUN_BURST:-true} RUN_FAULT=${RUN_FAULT:-true} +RUN_FAULT_RECOVERY=${RUN_FAULT_RECOVERY:-true} RUN_QUEUE_OVERLOAD=${RUN_QUEUE_OVERLOAD:-true} RUN_ENDURANCE=${RUN_ENDURANCE:-true} RUN_RELIABILITY=${RUN_RELIABILITY:-true} @@ -87,6 +101,7 @@ for boolean_name in \ RUN_RAMP \ RUN_BURST \ RUN_FAULT \ + RUN_FAULT_RECOVERY \ RUN_QUEUE_OVERLOAD \ RUN_ENDURANCE \ RUN_RELIABILITY \ @@ -117,6 +132,7 @@ for integer_name in \ PROFILE_PRE_ALLOCATED_VUS \ PROFILE_MAX_VUS \ FAULT_RATE \ + FAULT_RECOVERY_RATE \ QUEUE_RATE \ QUEUE_PRE_ALLOCATED_VUS \ QUEUE_MAX_VUS \ @@ -147,6 +163,11 @@ if [[ "$QUEUE_MAX_VUS" -lt "$QUEUE_PRE_ALLOCATED_VUS" ]]; then printf 'QUEUE_MAX_VUS must be greater than or equal to QUEUE_PRE_ALLOCATED_VUS\n' >&2 exit 1 fi +if [[ "$RUN_FAULT_RECOVERY" == "true" ]] && + { [[ -z "$UPSTREAM_FAIL_URL" ]] || [[ -z "$UPSTREAM_FALLBACK_URL" ]]; }; then + printf 'UPSTREAM_FAIL_URL and UPSTREAM_FALLBACK_URL are required when RUN_FAULT_RECOVERY=true\n' >&2 + exit 1 +fi for value in $CAPACITY_VUS $RATE_SWEEP; do if [[ ! "$value" =~ ^[1-9][0-9]*$ ]]; then printf 'capacity and rate sweep lists must contain positive integers\n' >&2 @@ -192,11 +213,14 @@ trap 'cleanup; summarize' EXIT { printf 'run_id=%s\n' "$run_id" + printf 'test_profile=%s\n' "$TEST_PROFILE" printf 'request_prefix=%s\n' "$request_prefix" printf 'commit=%s\n' "$(git -C "$repo_root" rev-parse HEAD)" printf 'started_at=%s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" printf 'gateway_url=%s\n' "$GATEWAY_URL" printf 'upstream_url=%s\n' "$UPSTREAM_URL" + printf 'upstream_fail_url=%s\n' "$UPSTREAM_FAIL_URL" + printf 'upstream_fallback_url=%s\n' "$UPSTREAM_FALLBACK_URL" printf 'repetitions=%s\n' "$REPETITIONS" printf 'capacity_vus=%s\n' "$CAPACITY_VUS" printf 'rate_sweep=%s\n' "$RATE_SWEEP" @@ -205,6 +229,9 @@ trap 'cleanup; summarize' EXIT printf 'ramp_stages=%s\n' "$RAMP_STAGES" printf 'burst_stages=%s\n' "$BURST_STAGES" printf 'queue_rate=%s\n' "$QUEUE_RATE" + printf 'fault_recovery_rate=%s\n' "$FAULT_RECOVERY_RATE" + printf 'fault_recovery_failure_duration=%s\n' "$FAULT_RECOVERY_FAILURE_DURATION" + printf 'fault_recovery_healthy_duration=%s\n' "$FAULT_RECOVERY_HEALTHY_DURATION" printf 'queue_pre_allocated_vus=%s\n' "$QUEUE_PRE_ALLOCATED_VUS" printf 'queue_max_vus=%s\n' "$QUEUE_MAX_VUS" printf 'endurance_rate=%s\n' "$ENDURANCE_RATE" @@ -245,26 +272,60 @@ case_prefix() { } reset_upstream() { + local upstream_url + for upstream_url in "$UPSTREAM_URL" "$UPSTREAM_FAIL_URL" "$UPSTREAM_FALLBACK_URL"; do + if [[ -z "$upstream_url" ]]; then + continue + fi + curl \ + --fail \ + --silent \ + --show-error \ + --max-time 10 \ + -X POST \ + "$(base_url "$upstream_url")/__admin/reset" \ + >/dev/null + done +} + +capture_upstream_url() { + local case_name=$1 + local label=$2 + local upstream_url=$3 + if [[ -z "$upstream_url" ]]; then + return + fi curl \ --fail \ --silent \ --show-error \ --max-time 10 \ - -X POST \ - "$(base_url "$UPSTREAM_URL")/__admin/reset" \ - >/dev/null + "$(base_url "$upstream_url")/__admin/stats" \ + >"$results_dir/$case_name-$label-upstream.json" || + printf '{"capture_error":true}\n' >"$results_dir/$case_name-$label-upstream.json" } capture_upstream() { local case_name=$1 + capture_upstream_url "$case_name" main "$UPSTREAM_URL" + cp "$results_dir/$case_name-main-upstream.json" \ + "$results_dir/$case_name-upstream.json" + capture_upstream_url "$case_name" fail "$UPSTREAM_FAIL_URL" + capture_upstream_url "$case_name" fallback "$UPSTREAM_FALLBACK_URL" +} + +set_upstream_scenario() { + local upstream_url=$1 + local scenario=$2 curl \ --fail \ --silent \ --show-error \ --max-time 10 \ - "$(base_url "$UPSTREAM_URL")/__admin/stats" \ - >"$results_dir/$case_name-upstream.json" || - printf '{"capture_error":true}\n' >"$results_dir/$case_name-upstream.json" + -H 'Content-Type: application/json' \ + -d "{\"scenario\":\"$scenario\"}" \ + "$(base_url "$upstream_url")/__admin/scenario" \ + >/dev/null } record_case() { @@ -673,6 +734,21 @@ if [[ "$RUN_FAULT" == "true" ]]; then run_fault_case fault-spike5-gateway mock/spike-5 "200" fi +if [[ "$RUN_FAULT_RECOVERY" == "true" ]]; then + original_fault_rate=$FAULT_RATE + original_fault_duration=$FAULT_DURATION + FAULT_RATE=$FAULT_RECOVERY_RATE + set_upstream_scenario "$UPSTREAM_FAIL_URL" mock/error-503 + FAULT_DURATION=$FAULT_RECOVERY_FAILURE_DURATION + run_fault_case fault-fallback-5xx-gateway mock/fallback "200" + set_upstream_scenario "$UPSTREAM_FAIL_URL" mock/instant + FAULT_DURATION=$FAULT_RECOVERY_HEALTHY_DURATION + run_fault_case fault-provider-recovery-gateway mock/fallback "200" + set_upstream_scenario "$UPSTREAM_FAIL_URL" mock/error-503 + FAULT_RATE=$original_fault_rate + FAULT_DURATION=$original_fault_duration +fi + if [[ "$RUN_QUEUE_OVERLOAD" == "true" ]]; then original_rate_duration=$RATE_DURATION original_pre_allocated_vus=$PRE_ALLOCATED_VUS diff --git a/test/threehost/run-diagnostic-client.sh b/test/threehost/run-diagnostic-client.sh new file mode 100644 index 0000000..07bd578 --- /dev/null +++ b/test/threehost/run-diagnostic-client.sh @@ -0,0 +1,45 @@ +#!/usr/bin/env bash +set -euo pipefail + +script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) +env_file=${1:-"$script_dir/benchmark.env"} + +if [[ ! -f "$env_file" ]]; then + printf 'missing benchmark environment file: %s\n' "$env_file" >&2 + exit 1 +fi + +set -a +# shellcheck disable=SC1090 +. "$env_file" +set +a + +export REPETITIONS=1 +export TEST_PROFILE=diagnostic +export WARMUP_RATE=100 +export WARMUP_DURATION=15s +export RATE_SWEEP="500 750 1000 1500" +export RATE_DURATION=2m +export PRE_ALLOCATED_VUS=512 +export MAX_VUS=4096 +export ENDURANCE_RATE=500 +export ENDURANCE_DURATION=10m +export ENDURANCE_PRE_ALLOCATED_VUS=512 +export ENDURANCE_MAX_VUS=2048 + +export RUN_CAPACITY=false +export RUN_WARMUP=true +export RUN_RATE_SWEEP=true +export RUN_STREAM_DETAIL=false +export RUN_PAYLOAD=false +export RUN_CACHE=false +export RUN_RAMP=false +export RUN_BURST=false +export RUN_FAULT=false +export RUN_QUEUE_OVERLOAD=false +export RUN_ENDURANCE=true +export RUN_RELIABILITY=false +export RUN_RATE_LIMIT=false +export SAVE_RAW_METRICS=false + +exec bash "$script_dir/run-complete-client.sh" - diff --git a/test/threehost/run-usage-chaos.sh b/test/threehost/run-usage-chaos.sh new file mode 100644 index 0000000..571772a --- /dev/null +++ b/test/threehost/run-usage-chaos.sh @@ -0,0 +1,287 @@ +#!/usr/bin/env bash +set -euo pipefail + +script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) +repo_root=$(CDPATH= cd -- "$script_dir/../.." && pwd) +run_id=${1:-} +env_file=${ENV_FILE:-"$repo_root/.env"} +compose_file=${COMPOSE_FILE:-"$repo_root/compose.yaml"} +output_dir=${OUTPUT_DIR:-"$repo_root/test-results/threehost/$run_id/usage-chaos"} + +if [[ ! "$run_id" =~ ^[A-Za-z0-9._-]+$ || ${#run_id} -gt 48 ]]; then + printf 'usage: MODEL_VELO_BENCH_API_KEY=... %s \n' "$0" >&2 + exit 1 +fi +MODEL_VELO_BENCH_API_KEY=${MODEL_VELO_BENCH_API_KEY:-} +if [[ -z "$MODEL_VELO_BENCH_API_KEY" && -t 0 ]]; then + read -r -s -p 'Benchmark API key: ' MODEL_VELO_BENCH_API_KEY + printf '\n' >&2 +fi +: "${MODEL_VELO_BENCH_API_KEY:?MODEL_VELO_BENCH_API_KEY is required}" +if [[ ! -f "$env_file" ]]; then + printf 'environment file not found: %s\n' "$env_file" >&2 + exit 1 +fi +for command_name in curl docker git; do + if ! command -v "$command_name" >/dev/null 2>&1; then + printf '%s is required\n' "$command_name" >&2 + exit 1 + fi +done +if [[ -e "$output_dir" ]]; then + printf 'refusing to overwrite usage chaos output: %s\n' "$output_dir" >&2 + exit 1 +fi + +set -a +# shellcheck disable=SC1090 +. "$env_file" +set +a + +worker_outage_requests=${WORKER_OUTAGE_REQUESTS:-500} +redis_outage_requests=${REDIS_OUTAGE_REQUESTS:-20} +request_concurrency=${REQUEST_CONCURRENCY:-20} +recovery_timeout_seconds=${RECOVERY_TIMEOUT_SECONDS:-180} +gateway_url=${GATEWAY_LOCAL_URL:-http://127.0.0.1:${MODEL_VELO_HTTP_PORT:-8080}} +postgres_user=${MODEL_VELO_POSTGRES_USER:-model_velo} +postgres_database=${MODEL_VELO_POSTGRES_DB:-model_velo} +environment=${MODEL_VELO_ENVIRONMENT:-development} +usage_stream="model-velo:usage:v1:$environment" +usage_group=${MODEL_VELO_USAGE_GROUP:-model-velo-usage-workers} +prefix="usage-chaos-$run_id" +prefix=${prefix:0:64} + +for integer_name in \ + WORKER_OUTAGE_REQUESTS \ + REDIS_OUTAGE_REQUESTS \ + REQUEST_CONCURRENCY \ + RECOVERY_TIMEOUT_SECONDS; do + value=${!integer_name} + if [[ ! "$value" =~ ^[1-9][0-9]*$ ]]; then + printf '%s must be a positive integer\n' "$integer_name" >&2 + exit 1 + fi +done + +mkdir -p "$output_dir" +services_restored=false +restore_services() { + if [[ "$services_restored" == "true" ]]; then + return + fi + docker compose -f "$compose_file" start redis gateway usage-worker \ + >"$output_dir/restore.log" 2>&1 || true + services_restored=true +} +trap restore_services EXIT + +sql_scalar() { + local query=$1 + docker compose -f "$compose_file" exec -T postgres \ + psql -U "$postgres_user" -d "$postgres_database" -At -c "$query" +} + +redis_command() { + docker compose -f "$compose_file" exec -T redis \ + sh -c \ + 'REDISCLI_AUTH="$MODEL_VELO_REDIS_PASSWORD" redis-cli --no-auth-warning "$@"' \ + redis-cli "$@" +} + +event_count() { + local request_prefix=$1 + local prefix_length=${#request_prefix} + sql_scalar \ + "SELECT count(*) FROM usage_events WHERE left(request_id, $prefix_length) = '$request_prefix';" +} + +outbox_count() { + local request_prefix=$1 + local prefix_length=${#request_prefix} + sql_scalar \ + "SELECT count(*) FROM usage_outbox WHERE left(request_id, $prefix_length) = '$request_prefix';" +} + +wait_for_value() { + local description=$1 + local expected=$2 + shift 2 + local deadline=$((SECONDS + recovery_timeout_seconds)) + local actual= + while ((SECONDS < deadline)); do + actual=$("$@" 2>/dev/null || true) + if [[ "$actual" == "$expected" ]]; then + printf '%s\n' "$actual" + return + fi + sleep 1 + done + printf 'timed out waiting for %s: got %s, want %s\n' \ + "$description" "${actual:-unavailable}" "$expected" >&2 + return 1 +} + +wait_for_redis() { + local deadline=$((SECONDS + recovery_timeout_seconds)) + while ((SECONDS < deadline)); do + if [[ "$(redis_command --raw PING 2>/dev/null || true)" == "PONG" ]]; then + return + fi + sleep 1 + done + printf 'timed out waiting for Redis recovery\n' >&2 + return 1 +} + +send_requests() { + local phase=$1 + local count=$2 + local status_file="$output_dir/$phase-statuses.txt" + : >"$status_file" + local active=0 + local index + for ((index = 1; index <= count; index++)); do + ( + curl \ + --silent \ + --show-error \ + --max-time 20 \ + -o /dev/null \ + -w '%{http_code}\n' \ + -H "Authorization: Bearer $MODEL_VELO_BENCH_API_KEY" \ + -H 'Content-Type: application/json' \ + -H 'Cache-Control: no-store' \ + -H "X-Request-ID: $prefix-$phase-$index" \ + -d '{"model":"mock/instant","messages":[{"role":"user","content":"usage chaos"}]}' \ + "${gateway_url%/}/v1/chat/completions" \ + >>"$status_file" 2>>"$output_dir/$phase-curl-errors.log" || printf '000\n' >>"$status_file" + ) & + active=$((active + 1)) + if ((active >= request_concurrency)); then + wait + active=0 + fi + done + wait + sort "$status_file" | uniq -c >"$output_dir/$phase-status-counts.txt" +} + +{ + printf 'run_id=%s\n' "$run_id" + printf 'request_prefix=%s\n' "$prefix" + printf 'commit=%s\n' "$(git -C "$repo_root" rev-parse HEAD)" + printf 'started_at=%s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" + printf 'worker_outage_requests=%s\n' "$worker_outage_requests" + printf 'redis_outage_requests=%s\n' "$redis_outage_requests" + printf 'request_concurrency=%s\n' "$request_concurrency" +} >"$output_dir/metadata.txt" + +worker_prefix="$prefix-worker" +docker compose -f "$compose_file" stop usage-worker \ + >"$output_dir/worker-stop.log" 2>&1 +send_requests worker "$worker_outage_requests" +worker_backlog=$(outbox_count "$worker_prefix") +docker compose -f "$compose_file" start usage-worker \ + >"$output_dir/worker-start.log" 2>&1 +wait_for_value \ + "worker-outage events" \ + "$worker_outage_requests" \ + event_count "$worker_prefix" \ + >"$output_dir/worker-recovered-events.txt" +wait_for_value "worker-outage outbox drain" 0 outbox_count "$worker_prefix" \ + >"$output_dir/worker-final-outbox.txt" + +redis_prefix="$prefix-redis" +docker compose -f "$compose_file" stop redis \ + >"$output_dir/redis-stop.log" 2>&1 +send_requests redis "$redis_outage_requests" +redis_backlog=$(outbox_count "$redis_prefix") +docker compose -f "$compose_file" start redis \ + >"$output_dir/redis-start.log" 2>&1 +wait_for_redis +wait_for_value \ + "Redis-outage events" \ + "$redis_outage_requests" \ + event_count "$redis_prefix" \ + >"$output_dir/redis-recovered-events.txt" +wait_for_value "Redis-outage outbox drain" 0 outbox_count "$redis_prefix" \ + >"$output_dir/redis-final-outbox.txt" + +duplicate_request_id="$prefix-duplicate-1" +docker compose -f "$compose_file" stop usage-worker \ + >"$output_dir/duplicate-worker-stop.log" 2>&1 +send_requests duplicate 1 +wait_for_value "duplicate test ready outbox" 1 outbox_count "$prefix-duplicate" \ + >"$output_dir/duplicate-ready-outbox.txt" +duplicate_event_id=$(sql_scalar \ + "SELECT event_id FROM usage_outbox WHERE request_id = '$duplicate_request_id';") +duplicate_payload=$(sql_scalar \ + "SELECT payload::text FROM usage_outbox WHERE request_id = '$duplicate_request_id';") +for _ in 1 2; do + printf '%s' "$duplicate_payload" | + docker compose -f "$compose_file" exec -T redis \ + sh -c \ + 'REDISCLI_AUTH="$MODEL_VELO_REDIS_PASSWORD" redis-cli --no-auth-warning -x XADD "$1" "*" event_id "$2" schema_version 2 payload' \ + redis-cli "$usage_stream" "$duplicate_event_id" \ + >>"$output_dir/duplicate-xadd-ids.txt" +done +docker compose -f "$compose_file" start usage-worker \ + >"$output_dir/duplicate-worker-start.log" 2>&1 +wait_for_value \ + "idempotent duplicate storage" \ + 1 \ + sql_scalar "SELECT count(*) FROM usage_events WHERE event_id = '$duplicate_event_id';" \ + >"$output_dir/duplicate-stored-events.txt" +wait_for_value "duplicate outbox drain" 0 outbox_count "$prefix-duplicate" \ + >"$output_dir/duplicate-final-outbox.txt" +wait_for_value "usage stream drain" 0 redis_command --raw XLEN "$usage_stream" \ + >"$output_dir/pre-poison-stream-length.txt" + +dead_letter_before=$(redis_command --raw XLEN "$usage_stream:dead-letter") +docker compose -f "$compose_file" stop usage-worker \ + >"$output_dir/poison-worker-stop.log" 2>&1 +poison_entry_id=$(redis_command --raw XADD "$usage_stream" '*' payload '{') +printf '%s\n' "$poison_entry_id" >"$output_dir/poison-entry-id.txt" +redis_command XREADGROUP \ + GROUP "$usage_group" usage-chaos \ + COUNT 1 STREAMS "$usage_stream" '>' \ + >"$output_dir/poison-initial-delivery.txt" +for _ in 1 2 3 4 5; do + redis_command XCLAIM \ + "$usage_stream" "$usage_group" usage-chaos 0 "$poison_entry_id" JUSTID \ + >>"$output_dir/poison-claims.txt" +done +redis_command XCLAIM \ + "$usage_stream" "$usage_group" usage-chaos 0 "$poison_entry_id" \ + IDLE 31000 JUSTID \ + >>"$output_dir/poison-claims.txt" +docker compose -f "$compose_file" start usage-worker \ + >"$output_dir/poison-worker-start.log" 2>&1 +dead_letter_after=$((dead_letter_before + 1)) +wait_for_value \ + "dead-letter growth" \ + "$dead_letter_after" \ + redis_command --raw XLEN "$usage_stream:dead-letter" \ + >"$output_dir/dead-letter-final-length.txt" + +{ + printf 'phase,requests,outbox_while_dependency_down,stored_after_recovery\n' + printf 'worker_outage,%s,%s,%s\n' \ + "$worker_outage_requests" \ + "$worker_backlog" \ + "$(event_count "$worker_prefix")" + printf 'redis_outage,%s,%s,%s\n' \ + "$redis_outage_requests" \ + "$redis_backlog" \ + "$(event_count "$redis_prefix")" + printf 'duplicate_delivery,3,0,%s\n' \ + "$(sql_scalar "SELECT count(*) FROM usage_events WHERE event_id = '$duplicate_event_id';")" + printf 'poison_dead_letter,1,0,%s\n' "$((dead_letter_after - dead_letter_before))" +} >"$output_dir/summary.csv" + +printf 'ended_at=%s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + >>"$output_dir/metadata.txt" +docker compose -f "$compose_file" ps --format json >"$output_dir/compose-ps-final.json" +services_restored=true +trap - EXIT +printf 'usage chaos evidence: %s\n' "$output_dir" diff --git a/test/threehost/summarize.py b/test/threehost/summarize.py index 2aaf143..c7bef3d 100644 --- a/test/threehost/summarize.py +++ b/test/threehost/summarize.py @@ -7,6 +7,7 @@ import statistics import sys from collections import defaultdict +from datetime import datetime from pathlib import Path @@ -24,6 +25,35 @@ def number(value, default=0.0): return default +def parse_timestamp(value): + if not value: + return None + try: + return datetime.fromisoformat(str(value).replace("Z", "+00:00")) + except ValueError: + return None + + +def case_window(cases): + starts = [ + timestamp + for timestamp in (parse_timestamp(case.get("started_at")) for case in cases) + if timestamp is not None + ] + ends = [ + timestamp + for timestamp in (parse_timestamp(case.get("ended_at")) for case in cases) + if timestamp is not None + ] + return (min(starts) if starts else None, max(ends) if ends else None) + + +def in_window(timestamp, started_at, ended_at): + if timestamp is None or started_at is None or ended_at is None: + return True + return started_at <= timestamp <= ended_at + + def metric_value(metrics, name, field="value"): return number(metrics.get(name, {}).get(field)) @@ -81,10 +111,17 @@ def read_k6_case(result_dir, case): success = metric_value(metrics, name) break upstream = read_json(result_dir / f"{case['case']}-upstream.json") + upstreams = { + label: read_json(result_dir / f"{case['case']}-{label}-upstream.json") + for label in ("main", "fail", "fallback") + } return { **case, "exit_code": int(number(case.get("exit_code"))), "requests": int(metric_value(metrics, "http_reqs", "count")), + "gateway_requests": int( + metric_value(metrics, "gateway_chat_requests", "count") + ), "iterations": int(metric_value(metrics, "iterations", "count")), "throughput_rps": metric_value(metrics, "iterations", "rate"), "success_rate": success, @@ -99,6 +136,7 @@ def read_k6_case(result_dir, case): "other": int(metric_value(metrics, "chat_responses_other", "count")), }, "upstream": upstream, + "upstreams": upstreams, } @@ -272,6 +310,11 @@ def stream_comparisons(rows): - number(direct["total"].get("p99_ms")), "gap_p99_delta_ms": number(gateway["inter_chunk"].get("p99_ms")) - number(direct["inter_chunk"].get("p99_ms")), + "gateway_first_content_p99_ms": number( + gateway["first_content"].get("p99_ms") + ), + "gateway_success_rate": number(gateway.get("success_rate")), + "gateway_requests": int(number(gateway.get("requests"))), } ) comparisons = [] @@ -310,18 +353,25 @@ def parse_bytes(value): return amount * base ** powers.get(unit, 0) -def read_resources(result_dir): +def read_resources(result_dir, started_at=None, ended_at=None): samples = defaultdict(list) - for path in result_dir.rglob("*-stats.jsonl"): + for path in result_dir.rglob("*stats*.jsonl"): + if path.name == "client-stats.jsonl": + continue try: lines = path.read_text(encoding="utf-8").splitlines() except OSError: continue for line in lines: try: - stats = json.loads(line).get("stats", {}) + payload = json.loads(line) except json.JSONDecodeError: continue + if not in_window( + parse_timestamp(payload.get("timestamp")), started_at, ended_at + ): + continue + stats = payload.get("stats", {}) name = stats.get("Name") or stats.get("Container") or stats.get("ID") if not name: continue @@ -375,6 +425,7 @@ def read_client_resource(result_dir): return {} cpu = [number(sample.get("cpu_pct")) for sample in samples] + cpu_peak_sample = max(samples, key=lambda sample: number(sample.get("cpu_pct"))) used_memory = [ max( 0, @@ -387,6 +438,7 @@ def read_client_resource(result_dir): "samples": len(samples), "cpu_avg_pct": statistics.fmean(cpu), "cpu_max_pct": max(cpu), + "cpu_peak_at": cpu_peak_sample.get("timestamp", ""), "memory_used_avg_mb": statistics.fmean(used_memory) / (1024 * 1024), "memory_used_max_mb": max(used_memory) / (1024 * 1024), "load_1m_max": max(number(sample.get("load_1m")) for sample in samples), @@ -400,16 +452,71 @@ def read_client_resource(result_dir): } +def read_images(result_dir): + paths = list(result_dir.rglob("compose-images.json")) + if not paths: + return [] + payload = read_json(paths[0]) + if not isinstance(payload, list): + return [] + images = {} + for row in payload: + if not isinstance(row, dict): + continue + key = ( + str(row.get("Repository", "")), + str(row.get("Tag", "")), + str(row.get("ID", "")), + ) + current = images.setdefault( + key, + { + "repository": key[0], + "tag": key[1], + "id": key[2], + "platform": row.get("Platform", ""), + "size_bytes": int(number(row.get("Size"))), + "containers": [], + }, + ) + container = str(row.get("ContainerName", "")) + if container and container not in current["containers"]: + current["containers"].append(container) + return sorted(images.values(), key=lambda row: (row["repository"], row["tag"])) + + +def read_usage_chaos(result_dir): + paths = list(result_dir.rglob("usage-chaos/summary.csv")) + if not paths: + return [] + try: + with paths[0].open(encoding="utf-8", newline="") as handle: + return list(csv.DictReader(handle)) + except OSError: + return [] + + PROMETHEUS_LINE = re.compile( - r"^(model_velo_[A-Za-z0-9_:]+(?:\{[^}]*\})?)\s+" + r"^((?:model_velo_|go_|process_)[A-Za-z0-9_:]+(?:\{[^}]*\})?)\s+" r"(-?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?)$" ) +PROMETHEUS_SERIES = re.compile( + r"^((?:model_velo_|go_|process_)[A-Za-z0-9_:]+)(?:\{(.*)\})?$" +) +PROMETHEUS_LABEL = re.compile(r'([A-Za-z_][A-Za-z0-9_]*)="((?:\\.|[^"])*)"') -def read_prometheus(result_dir): +def read_prometheus(result_dir, started_at=None, ended_at=None): series = {} + points = defaultdict(list) for path in result_dir.rglob("*.promlog"): + snapshot_at = None for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if line.startswith("# snapshot "): + snapshot_at = parse_timestamp(line.removeprefix("# snapshot ").strip()) + continue + if not in_window(snapshot_at, started_at, ended_at): + continue match = PROMETHEUS_LINE.match(line.strip()) if not match: continue @@ -417,12 +524,369 @@ def read_prometheus(result_dir): value = number(match.group(2)) current = series.setdefault( key, - {"source": path.name, "series": match.group(1), "samples": 0, "max": value}, + { + "source": path.name, + "series": match.group(1), + "samples": 0, + "first": value, + "min": value, + "max": value, + }, ) current["samples"] += 1 current["last"] = value + current["min"] = min(current["min"], value) current["max"] = max(current["max"], value) - return sorted(series.values(), key=lambda item: (item["source"], item["series"])) + if snapshot_at is not None: + points[key].append((snapshot_at, value)) + output = sorted(series.values(), key=lambda item: (item["source"], item["series"])) + for item in output: + item["delta"] = max(0.0, item.get("last", 0.0) - item["first"]) + return output, points + + +def prometheus_identity(key): + _, separator, series = key.partition(":") + if not separator: + return "", {} + match = PROMETHEUS_SERIES.match(series) + if not match: + return "", {} + labels = {} + for label in PROMETHEUS_LABEL.finditer(match.group(2) or ""): + try: + labels[label.group(1)] = json.loads(f'"{label.group(2)}"') + except json.JSONDecodeError: + labels[label.group(1)] = label.group(2) + return match.group(1), labels + + +def window_delta(samples, started_at, ended_at): + if not samples or started_at is None or ended_at is None: + return 0.0 + before = None + after = None + first_inside = None + last_inside = None + for timestamp, value in samples: + if timestamp <= started_at: + before = value + if started_at <= timestamp <= ended_at: + if first_inside is None: + first_inside = value + last_inside = value + if timestamp >= ended_at: + after = value + break + first = before if before is not None else first_inside + last = after if after is not None else last_inside + if first is None or last is None: + return 0.0 + return max(0.0, last - first) + + +def window_max(samples, started_at, ended_at): + values = [ + value + for timestamp, value in samples + if started_at is not None + and ended_at is not None + and started_at <= timestamp <= ended_at + ] + return max(values) if values else 0.0 + + +def prometheus_counter( + points, metric, started_at, ended_at, labels=None, source=None +): + total = 0.0 + labels = labels or {} + for key, samples in points.items(): + if source and not key.startswith(f"{source}:"): + continue + name, series_labels = prometheus_identity(key) + if name != metric: + continue + if any(series_labels.get(name) != value for name, value in labels.items()): + continue + total += window_delta(samples, started_at, ended_at) + return total + + +def prometheus_gauge_max( + points, metric, started_at, ended_at, labels=None, source=None +): + maximum = 0.0 + labels = labels or {} + for key, samples in points.items(): + if source and not key.startswith(f"{source}:"): + continue + name, series_labels = prometheus_identity(key) + if name != metric: + continue + if any(series_labels.get(name) != value for name, value in labels.items()): + continue + maximum = max(maximum, window_max(samples, started_at, ended_at)) + return maximum + + +def histogram_quantile(buckets, quantile): + if not buckets: + return 0.0 + ordered = sorted(buckets.items(), key=lambda item: item[0]) + count = ordered[-1][1] + if count <= 0: + return 0.0 + target = count * quantile + previous_bound = 0.0 + previous_count = 0.0 + for bound, cumulative in ordered: + if cumulative < target: + if math.isfinite(bound): + previous_bound = bound + previous_count = cumulative + continue + if not math.isfinite(bound): + return previous_bound + bucket_count = cumulative - previous_count + if bucket_count <= 0: + return bound + fraction = (target - previous_count) / bucket_count + return previous_bound + (bound - previous_bound) * fraction + return ordered[-1][0] + + +STAGE_ORDER = { + stage: index + for index, stage in enumerate( + ( + "authentication", + "usage_begin", + "authorization", + "rate_limit", + "route_plan", + "quota_reserve", + "cache_lookup", + "provider_queue", + "provider_call", + "reliability", + "cache_store", + "quota_settle", + "usage_finalize", + ) + ) +} + + +def case_stage_metrics(points, started_at, ended_at): + stages = defaultdict( + lambda: {"buckets": defaultdict(float), "sum_seconds": 0.0, "count": 0.0} + ) + for key, samples in points.items(): + metric, labels = prometheus_identity(key) + stage = labels.get("stage", "") + if not stage: + continue + delta = window_delta(samples, started_at, ended_at) + if metric == "model_velo_request_stage_duration_seconds_bucket": + raw_bound = labels.get("le", "") + bound = math.inf if raw_bound == "+Inf" else number(raw_bound, math.nan) + if not math.isnan(bound): + stages[stage]["buckets"][bound] += delta + elif metric == "model_velo_request_stage_duration_seconds_sum": + stages[stage]["sum_seconds"] += delta + elif metric == "model_velo_request_stage_duration_seconds_count": + stages[stage]["count"] += delta + + output = [] + for stage, values in stages.items(): + count = values["count"] + if count <= 0: + continue + output.append( + { + "stage": stage, + "count": int(count), + "avg_ms": values["sum_seconds"] / count * 1000, + "p50_ms": histogram_quantile(values["buckets"], 0.50) * 1000, + "p95_ms": histogram_quantile(values["buckets"], 0.95) * 1000, + "p99_ms": histogram_quantile(values["buckets"], 0.99) * 1000, + } + ) + return sorted( + output, + key=lambda item: (STAGE_ORDER.get(item["stage"], len(STAGE_ORDER)), item["stage"]), + ) + + +def case_error_counts(points, started_at, ended_at): + counts = defaultdict(float) + for key, samples in points.items(): + metric, labels = prometheus_identity(key) + if metric != "model_velo_http_errors_total": + continue + count = window_delta(samples, started_at, ended_at) + if count > 0: + counts[(labels.get("status", ""), labels.get("code", ""))] += count + return [ + {"status": status, "code": code, "count": int(count)} + for (status, code), count in sorted( + counts.items(), key=lambda item: (-item[1], item[0]) + ) + ] + + +def build_performance_diagnostics(points, rows): + diagnostics = [] + for row in rows: + if row.get("target") != "gateway": + continue + started_at = parse_timestamp(row.get("started_at")) + ended_at = parse_timestamp(row.get("ended_at")) + if started_at is None or ended_at is None: + continue + duration_seconds = max(0.0, (ended_at - started_at).total_seconds()) + stages = case_stage_metrics(points, started_at, ended_at) + diagnostics.append( + { + "case": row.get("case", ""), + "phase": row.get("phase", ""), + "load": row.get("load", ""), + "duration_seconds": duration_seconds, + "stages": stages, + "errors": case_error_counts(points, started_at, ended_at), + "postgres": { + "waits": int( + prometheus_counter( + points, + "model_velo_postgres_waits_total", + started_at, + ended_at, + ) + ), + "wait_ms": prometheus_counter( + points, + "model_velo_postgres_wait_duration_seconds_total", + started_at, + ended_at, + ) + * 1000, + "in_use_max": int( + prometheus_gauge_max( + points, + "model_velo_postgres_connections", + started_at, + ended_at, + {"state": "in_use"}, + ) + ), + "open_max": int( + prometheus_gauge_max( + points, + "model_velo_postgres_connections", + started_at, + ended_at, + {"state": "open"}, + ) + ), + }, + "redis": { + "waits": int( + prometheus_counter( + points, + "model_velo_redis_pool_events_total", + started_at, + ended_at, + {"event": "wait"}, + ) + ), + "timeouts": int( + prometheus_counter( + points, + "model_velo_redis_pool_events_total", + started_at, + ended_at, + {"event": "timeout"}, + ) + ), + "wait_ms": prometheus_counter( + points, + "model_velo_redis_pool_wait_duration_seconds_total", + started_at, + ended_at, + ) + * 1000, + "pending_max": int( + prometheus_gauge_max( + points, + "model_velo_redis_pool_connections", + started_at, + ended_at, + {"state": "pending"}, + ) + ), + "total_max": int( + prometheus_gauge_max( + points, + "model_velo_redis_pool_connections", + started_at, + ended_at, + {"state": "total"}, + ) + ), + }, + "runtime": { + "process_cpu_avg_pct": ( + prometheus_counter( + points, + "process_cpu_seconds_total", + started_at, + ended_at, + source="gateway-metrics.promlog", + ) + / duration_seconds + * 100 + if duration_seconds > 0 + else 0.0 + ), + "rss_max_mb": prometheus_gauge_max( + points, + "process_resident_memory_bytes", + started_at, + ended_at, + source="gateway-metrics.promlog", + ) + / (1024 * 1024), + "heap_max_mb": prometheus_gauge_max( + points, + "go_memstats_heap_alloc_bytes", + started_at, + ended_at, + source="gateway-metrics.promlog", + ) + / (1024 * 1024), + "goroutines_max": int( + prometheus_gauge_max( + points, + "go_goroutines", + started_at, + ended_at, + source="gateway-metrics.promlog", + ) + ), + "gc_cycles": int( + prometheus_counter( + points, + "go_gc_duration_seconds_count", + started_at, + ended_at, + source="gateway-metrics.promlog", + ) + ), + }, + } + ) + return diagnostics def read_usage_evidence(result_dir): @@ -463,12 +927,21 @@ def find_series(prometheus, prefix): def reconcile_usage(k6_rows, stream_rows, usage): - excluded_phases = {"smoke", "reliability", "rate-limit"} - expected = sum( + measured_gateway_requests = sum( + int(number(row.get("gateway_requests"))) for row in k6_rows + ) + if measured_gateway_requests == 0: + excluded_phases = {"smoke", "reliability", "rate-limit"} + measured_gateway_requests = sum( + int(number(row.get("requests"))) + for row in k6_rows + if row.get("target") == "gateway" + and row.get("phase") not in excluded_phases + ) + expected = measured_gateway_requests + sum( int(number(row.get("requests"))) - for row in k6_rows + stream_rows + for row in stream_rows if row.get("target") == "gateway" - and row.get("phase") not in excluded_phases ) observed = int(number(usage.get("overview", {}).get("events"))) ratio = float(observed) / expected if expected > 0 else 0.0 @@ -521,20 +994,99 @@ def capacity_findings(capacity, rate, comparisons, stream_pairs): ) if stream_pairs: findings.append( - f"SSE first-content P50 overhead: " + f"SSE gateway TTFT P99: " + f"{stream_pairs[0]['gateway_first_content_p99_ms']:.2f} ms across " + f"{stream_pairs[0]['gateway_requests']} requests with " + f"{stream_pairs[0]['gateway_success_rate'] * 100:.3f}% success; " + f"first-content P50 overhead: " f"{stream_pairs[0]['first_content_p50_delta_ms']:.2f} ms; " f"inter-chunk P99 overhead: {stream_pairs[0]['gap_p99_delta_ms']:.2f} ms." ) return findings +def fault_recovery_findings(rows): + findings = [] + by_case = {row.get("case"): row for row in rows} + failure = by_case.get("fault-fallback-5xx-gateway") + if failure: + fail_calls = int( + number(failure.get("upstreams", {}).get("fail", {}).get("requests")) + ) + fallback_calls = int( + number(failure.get("upstreams", {}).get("fallback", {}).get("requests")) + ) + requests = max(1, int(number(failure.get("requests")))) + amplification = (fail_calls + fallback_calls) / requests + findings.append( + f"Full-503 fallback success: {failure['success_rate'] * 100:.3f}% " + f"across {requests} requests; failed provider calls {fail_calls}, " + f"fallback calls {fallback_calls}, upstream amplification " + f"{amplification:.3f}x." + ) + + recovery = by_case.get("fault-provider-recovery-gateway") + if recovery: + fail_stats = recovery.get("upstreams", {}).get("fail", {}) + first_request = None + scenarios = fail_stats.get("scenarios", []) + if scenarios: + first_request = parse_timestamp(scenarios[0].get("first_request_at")) + case_start = parse_timestamp(recovery.get("started_at")) + recovery_seconds = ( + max(0.0, (first_request - case_start).total_seconds()) + if first_request is not None and case_start is not None + else None + ) + fail_calls = int(number(fail_stats.get("requests"))) + fallback_calls = int( + number(recovery.get("upstreams", {}).get("fallback", {}).get("requests")) + ) + timing = ( + f"{recovery_seconds:.2f} s after recovery injection" + if recovery_seconds is not None + else "timing unavailable" + ) + findings.append( + f"Provider recovery: first half-open traffic reached the restored " + f"primary {timing}; primary handled {fail_calls} calls and fallback " + f"handled {fallback_calls}, with " + f"{recovery['success_rate'] * 100:.3f}% client success." + ) + return findings + + def actionable_findings(resources, client_resource, prometheus, rows): findings = [] if client_resource.get("cpu_max_pct", 0) >= 90: - findings.append( - f"Client host CPU reached {client_resource['cpu_max_pct']:.1f}%; " - "capacity results may be load-generator limited." + peak_at = parse_timestamp(client_resource.get("cpu_peak_at")) + peak_case = next( + ( + row + for row in rows + if peak_at is not None + and ( + parse_timestamp(row.get("started_at")) or peak_at + ) <= peak_at + <= (parse_timestamp(row.get("ended_at")) or peak_at) + ), + {}, ) + if peak_case.get("target") == "direct": + findings.append( + f"Client host CPU reached {client_resource['cpu_max_pct']:.1f}% " + f"during direct baseline {peak_case.get('case')}; " + "that direct-throughput point may be load-generator limited." + ) + else: + findings.append( + f"Client host CPU reached {client_resource['cpu_max_pct']:.1f}%" + + ( + f" during {peak_case.get('case')}." + if peak_case.get("case") + else "." + ) + ) for resource in resources: name = resource["container"].lower() if "gateway" in name and resource["cpu_max_pct"] >= 90: @@ -586,6 +1138,61 @@ def actionable_findings(resources, client_resource, prometheus, rows): return findings +def performance_findings(diagnostics): + findings = [] + diagnosed = [item for item in diagnostics if item["stages"]] + if not diagnosed: + return findings + + pre_provider_names = { + "authentication", + "usage_begin", + "authorization", + "rate_limit", + "route_plan", + "quota_reserve", + } + representative = max( + diagnosed, + key=lambda item: ( + number(item.get("load")), + item.get("duration_seconds", 0), + ), + ) + pre_provider = [ + stage + for stage in representative["stages"] + if stage["stage"] in pre_provider_names + ] + if pre_provider: + slowest = max(pre_provider, key=lambda stage: stage["p99_ms"]) + findings.append( + f"{representative['case']} slowest pre-provider stage was " + f"{slowest['stage']} at P99 {slowest['p99_ms']:.2f} ms." + ) + + postgres_waits = sum(item["postgres"]["waits"] for item in diagnosed) + postgres_wait_ms = sum(item["postgres"]["wait_ms"] for item in diagnosed) + redis_waits = sum(item["redis"]["waits"] for item in diagnosed) + redis_timeouts = sum(item["redis"]["timeouts"] for item in diagnosed) + findings.append( + f"Diagnosed cases recorded {postgres_waits} PostgreSQL pool waits " + f"({postgres_wait_ms:.1f} ms cumulative) and {redis_waits} Redis pool waits " + f"with {redis_timeouts} timeouts." + ) + + errors = defaultdict(int) + for item in diagnosed: + for error in item["errors"]: + errors[error["code"]] += error["count"] + if errors: + code, count = max(errors.items(), key=lambda item: item[1]) + findings.append( + f"Most frequent classified gateway error was {code}: {count} responses." + ) + return findings + + def fmt(value, digits=2): if value is None: return "-" @@ -726,7 +1333,9 @@ def write_markdown(path, summary): "200", "429", "5xx", - "upstream calls", + "main calls", + "failed-provider calls", + "fallback calls", ], [ [ @@ -739,6 +1348,20 @@ def write_markdown(path, summary): row["status_counts"]["429"], row["status_counts"]["5xx"], int(number(row.get("upstream", {}).get("requests"))), + int( + number( + row.get("upstreams", {}) + .get("fail", {}) + .get("requests") + ) + ), + int( + number( + row.get("upstreams", {}) + .get("fallback", {}) + .get("requests") + ) + ), ] for row in diagnostics ], @@ -747,6 +1370,85 @@ def write_markdown(path, summary): ] ) + performance = summary.get("performance_diagnostics", []) + stage_rows = [ + [ + item["case"], + stage["stage"], + stage["count"], + fmt(stage["avg_ms"], 3), + fmt(stage["p50_ms"], 3), + fmt(stage["p95_ms"], 3), + fmt(stage["p99_ms"], 3), + ] + for item in performance + for stage in item["stages"] + ] + if stage_rows: + lines.extend( + [ + "## Hot-path stage timing", + "", + "The `reliability` row contains queue, provider calls, retries, and " + "fallbacks; it overlaps the `provider_queue` and `provider_call` rows.", + "", + markdown_table( + ["case", "stage", "count", "avg ms", "P50 ms", "P95 ms", "P99 ms"], + stage_rows, + ), + "", + "### Dependency pools and Go runtime", + "", + markdown_table( + [ + "case", + "PG waits", + "PG wait ms", + "PG in-use max", + "Redis waits", + "Redis wait ms", + "Redis pending max", + "process CPU avg", + "RSS max MB", + "goroutines max", + ], + [ + [ + item["case"], + item["postgres"]["waits"], + fmt(item["postgres"]["wait_ms"], 2), + item["postgres"]["in_use_max"], + item["redis"]["waits"], + fmt(item["redis"]["wait_ms"], 2), + item["redis"]["pending_max"], + fmt(item["runtime"]["process_cpu_avg_pct"], 1) + "%", + fmt(item["runtime"]["rss_max_mb"], 1), + item["runtime"]["goroutines_max"], + ] + for item in performance + ], + ), + "", + ] + ) + errors = [ + [item["case"], error["status"], error["code"], error["count"]] + for item in performance + for error in item["errors"] + ] + if errors: + lines.extend( + [ + "### Gateway error codes", + "", + markdown_table( + ["case", "HTTP status", "error code", "count"], + errors, + ), + "", + ] + ) + resources = summary["resources"] if resources: lines.extend( @@ -771,6 +1473,28 @@ def write_markdown(path, summary): ] ) + images = summary.get("images", []) + if images: + lines.extend( + [ + "## Container images", + "", + markdown_table( + ["image", "platform", "size MB", "containers"], + [ + [ + f"{row['repository']}:{row['tag']}", + row["platform"], + fmt(row["size_bytes"] / 1_000_000, 2), + ", ".join(row["containers"]), + ] + for row in images + ], + ), + "", + ] + ) + client = summary["client_resource"] if client: lines.extend( @@ -826,6 +1550,33 @@ def write_markdown(path, summary): ] ) + usage_chaos = summary.get("usage_chaos", []) + if usage_chaos: + lines.extend( + [ + "## Usage failure recovery", + "", + markdown_table( + [ + "phase", + "requests/deliveries", + "outbox while down", + "stored/recovered", + ], + [ + [ + row.get("phase", ""), + row.get("requests", ""), + row.get("outbox_while_dependency_down", ""), + row.get("stored_after_recovery", ""), + ] + for row in usage_chaos + ], + ), + "", + ] + ) + lines.extend(["## Evidence", ""]) for warning in summary["warnings"]: lines.append(f"- {warning}") @@ -871,19 +1622,50 @@ def main(): "reliability", } ] - resources = read_resources(result_dir) + started_at, ended_at = case_window(cases) + resources = read_resources(result_dir, started_at, ended_at) client_resource = read_client_resource(result_dir) - prometheus = read_prometheus(result_dir) + images = read_images(result_dir) + prometheus, prometheus_points = read_prometheus( + result_dir, started_at, ended_at + ) + performance_diagnostics = build_performance_diagnostics( + prometheus_points, k6_rows + ) usage = read_usage_evidence(result_dir) usage_reconciliation = reconcile_usage(k6_rows, stream_rows, usage) + usage_chaos = read_usage_chaos(result_dir) warnings = [] - if not resources: + stats_paths = [ + path + for path in result_dir.rglob("*stats*.jsonl") + if path.name != "client-stats.jsonl" + ] + prometheus_paths = list(result_dir.rglob("*.promlog")) + if stats_paths and not resources: + warnings.append( + "Docker stats files exist, but their timestamps do not overlap " + "the benchmark case window." + ) + elif not resources: warnings.append("Missing gateway/upstream Docker stats JSONL files.") + if not images: + warnings.append("Missing Docker image-size evidence.") if not client_resource: warnings.append("Missing client host resource samples.") - if not prometheus: + if prometheus_paths and not prometheus: + warnings.append( + "Prometheus capture exists, but its timestamps do not overlap " + "the benchmark case window." + ) + elif not prometheus: warnings.append("Missing Prometheus time-series capture.") + elif not any(item["stages"] for item in performance_diagnostics): + warnings.append( + "Prometheus capture does not contain per-stage request histograms; " + "confirm the gateway image includes diagnostic metrics." + ) if not usage: warnings.append("Missing post-run Usage/PostgreSQL/Redis evidence.") elif usage.get("drain", {}).get("state") not in {"", "complete", None}: @@ -894,7 +1676,11 @@ def main(): ): warnings.append( "Stored Usage event count does not match measured gateway requests " - "after excluding smoke, reliability, and rate-limit cases." + "after the Usage stream and outbox drained." + ) + if not usage_chaos: + warnings.append( + "Missing Usage worker/Redis outage, duplicate, and dead-letter evidence." ) metadata = (result_dir / "client-metadata.txt").read_text( encoding="utf-8", errors="replace" @@ -910,6 +1696,7 @@ def main(): warnings.append("Fewer than three trials were recorded.") findings = capacity_findings(capacity, rate_sweep, comparisons, stream_pairs) + findings.extend(fault_recovery_findings(k6_rows)) findings.extend( actionable_findings( resources, @@ -918,6 +1705,7 @@ def main(): k6_rows + stream_rows, ) ) + findings.extend(performance_findings(performance_diagnostics)) if not findings: if k6_rows or stream_rows: findings.append( @@ -940,9 +1728,12 @@ def main(): "rate_sweep": rate_sweep, "resources": resources, "client_resource": client_resource, + "images": images, "prometheus": prometheus, + "performance_diagnostics": performance_diagnostics, "usage_evidence": usage, "usage_reconciliation": usage_reconciliation, + "usage_chaos": usage_chaos, "findings": findings, "warnings": warnings, } @@ -951,7 +1742,13 @@ def main(): encoding="utf-8", ) write_markdown(result_dir / "summary.md", summary) - print(f"wrote {result_dir / 'summary.json'} and {result_dir / 'summary.md'}") + from render_html import render + + render(result_dir) + print( + f"wrote {result_dir / 'summary.json'}, " + f"{result_dir / 'summary.md'}, and {result_dir / 'summary.html'}" + ) if __name__ == "__main__": diff --git a/threehost.tar.gz b/threehost.tar.gz deleted file mode 100644 index 39ac73e..0000000 Binary files a/threehost.tar.gz and /dev/null differ diff --git a/threehost/threehost/20260725T120000Z/client-metadata.txt b/threehost/threehost/20260725T120000Z/client-metadata.txt deleted file mode 100644 index d90ac81..0000000 --- a/threehost/threehost/20260725T120000Z/client-metadata.txt +++ /dev/null @@ -1,80 +0,0 @@ -run_id=20260725T120000Z -commit=510e4d408291e5ddd410a6e44c94dcb6e7909036 -started_at=2026-07-26T08:00:03Z -gateway_url=http://10.206.0.9:8080 -upstream_url=http://10.206.0.10:9000 -duration=60s -rate=50 -vus=20 -pre_allocated_vus=50 -max_vus=50 -non_stream_model=mock/instant -stream_model=mock/typical -repetitions=1 -warmup_duration=10s -warmup_rate=5 -k6_version=k6 v2.0.0 (commit/8c3be52cc1, go1.26.3, linux/amd64) -host=Linux VM-0-5-ubuntu 5.15.0-181-generic #191-Ubuntu SMP Fri May 22 19:09:02 UTC 2026 x86_64 x86_64 x86_64 GNU/Linux -worktree=dirty - -[os-release] -PRETTY_NAME="Ubuntu 22.04.5 LTS" -NAME="Ubuntu" -VERSION_ID="22.04" -VERSION="22.04.5 LTS (Jammy Jellyfish)" -VERSION_CODENAME=jammy -ID=ubuntu -ID_LIKE=debian -HOME_URL="https://www.ubuntu.com/" -SUPPORT_URL="https://help.ubuntu.com/" -BUG_REPORT_URL="https://bugs.launchpad.net/ubuntu/" -PRIVACY_POLICY_URL="https://www.ubuntu.com/legal/terms-and-policies/privacy-policy" -UBUNTU_CODENAME=jammy - -[cpu] -Architecture: x86_64 -CPU op-mode(s): 32-bit, 64-bit -Address sizes: 52 bits physical, 48 bits virtual -Byte Order: Little Endian -CPU(s): 4 -On-line CPU(s) list: 0-3 -Vendor ID: AuthenticAMD -Model name: AMD EPYC 9K65 192-Core Processor -CPU family: 26 -Model: 0 -Thread(s) per core: 2 -Core(s) per socket: 2 -Socket(s): 1 -Stepping: 0 -BogoMIPS: 4500.31 -Flags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm rep_good nopl cpuid extd_apicid tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt aes xsave avx f16c rdrand hypervisor lahf_lm cmp_legacy cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw topoext perfctr_core invpcid_single ssbd ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 erms invpcid avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx512_bf16 clzero xsaveerptr wbnoinvd arat avx512vbmi umip avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq rdpid movdiri movdir64b fsrm -Hypervisor vendor: KVM -Virtualization type: full -L1d cache: 96 KiB (2 instances) -L1i cache: 64 KiB (2 instances) -L2 cache: 2 MiB (2 instances) -L3 cache: 32 MiB (1 instance) -NUMA node(s): 1 -NUMA node0 CPU(s): 0-3 -Vulnerability Gather data sampling: Not affected -Vulnerability Indirect target selection: Not affected -Vulnerability Itlb multihit: Not affected -Vulnerability L1tf: Not affected -Vulnerability Mds: Not affected -Vulnerability Meltdown: Not affected -Vulnerability Mmio stale data: Not affected -Vulnerability Reg file data sampling: Not affected -Vulnerability Retbleed: Not affected -Vulnerability Spec rstack overflow: Not affected -Vulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp -Vulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization -Vulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; IBRS_FW; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected -Vulnerability Srbds: Not affected -Vulnerability Tsa: Not affected -Vulnerability Tsx async abort: Not affected -Vulnerability Vmscape: Not affected - -[memory-bytes] - total used free shared buff/cache available -Mem: 7989432320 370229248 5943021568 2695168 1676181504 7337566208 -Swap: 0 0 0 diff --git a/threehost/threehost/20260725T120000Z/rate-nonstream-direct-r1-summary.json b/threehost/threehost/20260725T120000Z/rate-nonstream-direct-r1-summary.json deleted file mode 100644 index f30c00c..0000000 --- a/threehost/threehost/20260725T120000Z/rate-nonstream-direct-r1-summary.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "root_group": { - "groups": {}, - "checks": { - "chat status is 200": { - "id": "8b366dfc28447643d7a100f81d49e9cb", - "passes": 3001, - "fails": 0, - "name": "chat status is 200", - "path": "::chat status is 200" - }, - "chat body is a completion": { - "name": "chat body is a completion", - "path": "::chat body is a completion", - "id": "dd8a724b149dbb7fedc9ddf36bff0341", - "passes": 3001, - "fails": 0 - } - }, - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e" - }, - "metrics": { - "dropped_iterations": { - "count": 0, - "rate": 0, - "thresholds": { - "count==0": false - } - }, - "http_req_duration{expected_response:true}": { - "p(95)": 0.406356, - "p(99)": 0.451744, - "max": 2.379825, - "avg": 0.3526029416861055, - "med": 0.342081, - "p(90)": 0.389398 - }, - "http_req_receiving": { - "avg": 0.018326097634121973, - "med": 0.016958, - "p(90)": 0.025299, - "p(95)": 0.029637, - "p(99)": 0.038787, - "max": 0.163426 - }, - "http_req_sending": { - "p(99)": 0.029887, - "max": 0.124381, - "avg": 0.011586590803065648, - "med": 0.010639, - "p(90)": 0.014868, - "p(95)": 0.018019 - }, - "http_reqs": { - "rate": 50.015880622088396, - "count": 3001 - }, - "checks": { - "passes": 6002, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_failed": { - "passes": 0, - "fails": 3001, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "vus_max": { - "value": 50, - "min": 50, - "max": 50 - }, - "http_req_waiting": { - "avg": 0.322690253248917, - "med": 0.312725, - "p(90)": 0.356081, - "p(95)": 0.371829, - "p(99)": 0.416807, - "max": 2.219467 - }, - "data_sent": { - "count": 1086899, - "rate": 18114.698644540906 - }, - "iteration_duration": { - "avg": 0.4785522865711419, - "med": 0.459112, - "p(90)": 0.529166, - "p(95)": 0.559463, - "p(99)": 0.935094, - "max": 2.687921 - }, - "http_req_duration": { - "p(90)": 0.389398, - "p(95)": 0.406356, - "p(99)": 0.451744, - "max": 2.379825, - "avg": 0.3526029416861055, - "med": 0.342081 - }, - "http_req_blocked": { - "avg": 0.008254388537154286, - "med": 0.002449, - "p(90)": 0.00346, - "p(95)": 0.00403, - "p(99)": 0.334693, - "max": 0.419546 - }, - "http_req_connecting": { - "max": 0.360901, - "avg": 0.005106820393202265, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0.301436 - }, - "http_req_tls_handshaking": { - "p(90)": 0, - "p(95)": 0, - "p(99)": 0, - "max": 0, - "avg": 0, - "med": 0 - }, - "iterations": { - "count": 3001, - "rate": 50.015880622088396 - }, - "vus": { - "value": 0, - "min": 0, - "max": 0 - }, - "data_received": { - "count": 1577294, - "rate": 26287.820196579905 - }, - "chat_success": { - "passes": 3001, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/rate-nonstream-gateway-r1-summary.json b/threehost/threehost/20260725T120000Z/rate-nonstream-gateway-r1-summary.json deleted file mode 100644 index 5c54e95..0000000 --- a/threehost/threehost/20260725T120000Z/rate-nonstream-gateway-r1-summary.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "root_group": { - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "chat status is 200": { - "name": "chat status is 200", - "path": "::chat status is 200", - "id": "8b366dfc28447643d7a100f81d49e9cb", - "passes": 3001, - "fails": 0 - }, - "chat body is a completion": { - "path": "::chat body is a completion", - "id": "dd8a724b149dbb7fedc9ddf36bff0341", - "passes": 3001, - "fails": 0, - "name": "chat body is a completion" - } - }, - "name": "" - }, - "metrics": { - "http_req_sending": { - "avg": 0.012152875374875027, - "med": 0.011029, - "p(90)": 0.016399, - "p(95)": 0.019089, - "p(99)": 0.029298, - "max": 0.109401 - }, - "data_received": { - "rate": 31717.191981283613, - "count": 1903175 - }, - "iterations": { - "count": 3001, - "rate": 50.01289589020038 - }, - "checks": { - "passes": 6002, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_waiting": { - "max": 11.455487, - "avg": 3.908035195601455, - "med": 3.809531, - "p(90)": 4.348087, - "p(95)": 4.573169, - "p(99)": 5.593113 - }, - "http_req_failed": { - "passes": 0, - "fails": 3001, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "http_reqs": { - "count": 3001, - "rate": 50.01289589020038 - }, - "http_req_connecting": { - "avg": 0.0056433572142619145, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0.325693, - "max": 0.463163 - }, - "vus": { - "max": 0, - "value": 0, - "min": 0 - }, - "http_req_blocked": { - "p(99)": 0.359101, - "max": 0.524217, - "avg": 0.009080566811062972, - "med": 0.00259, - "p(90)": 0.003769, - "p(95)": 0.00487 - }, - "data_sent": { - "count": 1341988, - "rate": 22364.78044981614 - }, - "iteration_duration": { - "avg": 4.0835600373208925, - "med": 3.985937, - "p(90)": 4.525721, - "p(95)": 4.79503, - "p(99)": 5.737023, - "max": 11.639402 - }, - "http_req_tls_handshaking": { - "avg": 0, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0, - "max": 0 - }, - "http_req_receiving": { - "p(95)": 0.032827, - "p(99)": 0.042466, - "max": 0.126809, - "avg": 0.020622816394535146, - "med": 0.019268, - "p(90)": 0.029077 - }, - "chat_success": { - "fails": 0, - "passes": 3001, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "vus_max": { - "max": 50, - "value": 50, - "min": 50 - }, - "http_req_duration": { - "avg": 3.9408108873708696, - "med": 3.842798, - "p(90)": 4.377105, - "p(95)": 4.605346, - "p(99)": 5.621792, - "max": 11.494974 - }, - "http_req_duration{expected_response:true}": { - "p(99)": 5.621792, - "max": 11.494974, - "avg": 3.9408108873708696, - "med": 3.842798, - "p(90)": 4.377105, - "p(95)": 4.605346 - }, - "dropped_iterations": { - "count": 0, - "rate": 0, - "thresholds": { - "count==0": false - } - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/rate-stream-direct-r1-summary.json b/threehost/threehost/20260725T120000Z/rate-stream-direct-r1-summary.json deleted file mode 100644 index 6115d46..0000000 --- a/threehost/threehost/20260725T120000Z/rate-stream-direct-r1-summary.json +++ /dev/null @@ -1,170 +0,0 @@ -{ - "root_group": { - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "stream status is 200": { - "fails": 0, - "name": "stream status is 200", - "path": "::stream status is 200", - "id": "d15eb5c4982b20ba39bac56e96c9eff4", - "passes": 3001 - }, - "stream content type is SSE": { - "fails": 0, - "name": "stream content type is SSE", - "path": "::stream content type is SSE", - "id": "50a79a216c54aaeb47d07c1cbfcc3332", - "passes": 3001 - }, - "stream has terminal marker": { - "fails": 0, - "name": "stream has terminal marker", - "path": "::stream has terminal marker", - "id": "787cf8391739503c3d3fdb967bb841df", - "passes": 3001 - } - } - }, - "metrics": { - "iterations": { - "count": 3001, - "rate": 49.57791117757754 - }, - "vus_max": { - "value": 50, - "min": 50, - "max": 50 - }, - "http_reqs": { - "count": 3001, - "rate": 49.57791117757754 - }, - "iteration_duration": { - "avg": 527.6777057697436, - "med": 527.308188, - "p(90)": 530.622684, - "p(95)": 531.838161, - "p(99)": 533.679385, - "max": 536.362367 - }, - "http_req_sending": { - "p(95)": 0.020578, - "p(99)": 0.031627, - "max": 0.090443, - "avg": 0.013504630456514486, - "med": 0.012559, - "p(90)": 0.017157 - }, - "http_req_receiving": { - "p(99)": 332.389053, - "max": 335.519585, - "avg": 326.7528786914361, - "med": 326.434028, - "p(90)": 329.636601, - "p(95)": 330.853928 - }, - "chat_success": { - "passes": 3001, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_blocked": { - "avg": 0.0087074285238254, - "med": 0.00289, - "p(90)": 0.004039, - "p(95)": 0.00502, - "p(99)": 0.322554, - "max": 0.511538 - }, - "http_req_connecting": { - "p(99)": 0.287817, - "max": 0.440195, - "avg": 0.004978521826057981, - "med": 0, - "p(90)": 0, - "p(95)": 0 - }, - "data_sent": { - "count": 1083896, - "rate": 17906.464383116156 - }, - "http_req_tls_handshaking": { - "p(95)": 0, - "p(99)": 0, - "max": 0, - "avg": 0, - "med": 0, - "p(90)": 0 - }, - "http_req_failed": { - "passes": 0, - "fails": 3001, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "http_req_duration": { - "p(99)": 533.307341, - "max": 536.201601, - "avg": 527.5194397757409, - "med": 527.15393, - "p(90)": 530.439469, - "p(95)": 531.678116 - }, - "http_req_duration{expected_response:true}": { - "avg": 527.5194397757409, - "med": 527.15393, - "p(90)": 530.439469, - "p(95)": 531.678116, - "p(99)": 533.307341, - "max": 536.201601 - }, - "checks": { - "passes": 9003, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_waiting": { - "avg": 200.75305645384827, - "med": 200.686027, - "p(90)": 201.260264, - "p(95)": 201.314153, - "p(99)": 201.389237, - "max": 202.691555 - }, - "data_received": { - "count": 11836032, - "rate": 195536.73548516008 - }, - "stream_first_byte_ms": { - "max": 202.691555, - "avg": 200.75305645384827, - "med": 200.686027, - "p(90)": 201.260264, - "p(95)": 201.314153, - "p(99)": 201.389237 - }, - "dropped_iterations": { - "rate": 0, - "count": 0, - "thresholds": { - "count==0": false - } - }, - "vus": { - "value": 26, - "min": 26, - "max": 27 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/rate-stream-gateway-r1-summary.json b/threehost/threehost/20260725T120000Z/rate-stream-gateway-r1-summary.json deleted file mode 100644 index 5c9f1b5..0000000 --- a/threehost/threehost/20260725T120000Z/rate-stream-gateway-r1-summary.json +++ /dev/null @@ -1,170 +0,0 @@ -{ - "root_group": { - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "stream status is 200": { - "path": "::stream status is 200", - "id": "d15eb5c4982b20ba39bac56e96c9eff4", - "passes": 3001, - "fails": 0, - "name": "stream status is 200" - }, - "stream content type is SSE": { - "path": "::stream content type is SSE", - "id": "50a79a216c54aaeb47d07c1cbfcc3332", - "passes": 3001, - "fails": 0, - "name": "stream content type is SSE" - }, - "stream has terminal marker": { - "name": "stream has terminal marker", - "path": "::stream has terminal marker", - "id": "787cf8391739503c3d3fdb967bb841df", - "passes": 3001, - "fails": 0 - } - } - }, - "metrics": { - "http_req_connecting": { - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0.294876, - "max": 0.470652, - "avg": 0.005100930023325559 - }, - "http_req_tls_handshaking": { - "avg": 0, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0, - "max": 0 - }, - "stream_first_byte_ms": { - "med": 203.865335, - "p(90)": 204.554208, - "p(95)": 204.807119, - "p(99)": 206.031977, - "max": 375.70377, - "avg": 204.45041005564804 - }, - "http_reqs": { - "count": 3001, - "rate": 49.574898057465354 - }, - "dropped_iterations": { - "count": 0, - "rate": 0, - "thresholds": { - "count==0": false - } - }, - "http_req_blocked": { - "avg": 0.008374614461846066, - "med": 0.00256, - "p(90)": 0.00356, - "p(95)": 0.00408, - "p(99)": 0.324495, - "max": 0.536447 - }, - "data_received": { - "count": 12296837, - "rate": 203137.10120102236 - }, - "checks": { - "passes": 9003, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_duration": { - "avg": 535.3030622512492, - "med": 534.813828, - "p(90)": 537.097121, - "p(95)": 537.726448, - "p(99)": 539.760855, - "max": 705.607716 - }, - "http_req_sending": { - "med": 0.011689, - "p(90)": 0.015738, - "p(95)": 0.019289, - "p(99)": 0.028627, - "max": 0.111071, - "avg": 0.012622578473842034 - }, - "vus": { - "value": 27, - "min": 27, - "max": 27 - }, - "iterations": { - "count": 3001, - "rate": 49.574898057465354 - }, - "http_req_failed": { - "passes": 0, - "fails": 3001, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "data_sent": { - "count": 1338925, - "rate": 22118.317354745686 - }, - "http_req_receiving": { - "p(90)": 332.970178, - "p(95)": 333.544879, - "p(99)": 334.798451, - "max": 346.08687, - "avg": 330.8400296171276, - "med": 330.920133 - }, - "http_req_waiting": { - "p(99)": 206.031977, - "max": 375.70377, - "avg": 204.45041005564804, - "med": 203.865335, - "p(90)": 204.554208, - "p(95)": 204.807119 - }, - "iteration_duration": { - "p(99)": 539.92304, - "max": 705.718386, - "avg": 535.45108287604, - "med": 534.959801, - "p(90)": 537.241732, - "p(95)": 537.863787 - }, - "chat_success": { - "passes": 3001, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "vus_max": { - "value": 50, - "min": 50, - "max": 50 - }, - "http_req_duration{expected_response:true}": { - "avg": 535.3030622512492, - "med": 534.813828, - "p(90)": 537.097121, - "p(95)": 537.726448, - "p(99)": 539.760855, - "max": 705.607716 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/reliability-summary.json b/threehost/threehost/20260725T120000Z/reliability-summary.json deleted file mode 100644 index cbd4633..0000000 --- a/threehost/threehost/20260725T120000Z/reliability-summary.json +++ /dev/null @@ -1,314 +0,0 @@ -{ - "root_group": { - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": { - "reset deterministic fake state": { - "name": "reset deterministic fake state", - "path": "::reset deterministic fake state", - "id": "029c2d06817864c089890b95425158a2", - "groups": {}, - "checks": { - "fake upstream state reset": { - "passes": 1, - "fails": 0, - "name": "fake upstream state reset", - "path": "::reset deterministic fake state::fake upstream state reset", - "id": "eafa33b107980d184697057ac3daea13" - } - } - }, - "retry succeeds on third attempt": { - "id": "570e46e690a29d15bbe53a18450a93d2", - "groups": {}, - "checks": { - "retry sequence returns 200": { - "name": "retry sequence returns 200", - "path": "::retry succeeds on third attempt::retry sequence returns 200", - "id": "5368b3f11eaa9289a0caaf4f2262ade5", - "passes": 1, - "fails": 0 - }, - "retry sequence returns a completion": { - "name": "retry sequence returns a completion", - "path": "::retry succeeds on third attempt::retry sequence returns a completion", - "id": "53c684a71e57f18b36f63720cfe8c4dd", - "passes": 1, - "fails": 0 - } - }, - "name": "retry succeeds on third attempt", - "path": "::retry succeeds on third attempt" - }, - "fallback uses the healthy second provider": { - "name": "fallback uses the healthy second provider", - "path": "::fallback uses the healthy second provider", - "id": "c06bef701ef9cfdcb6e16e879f2f6ac5", - "groups": {}, - "checks": { - "fallback sequence returns 200": { - "name": "fallback sequence returns 200", - "path": "::fallback uses the healthy second provider::fallback sequence returns 200", - "id": "869b779e72cf80304d4121f543839cc0", - "passes": 1, - "fails": 0 - }, - "fallback response came from fallback provider": { - "name": "fallback response came from fallback provider", - "path": "::fallback uses the healthy second provider::fallback response came from fallback provider", - "id": "302945ac24be60f7d90bcef3326951ce", - "passes": 1, - "fails": 0 - } - } - }, - "upstream HTTP errors are normalized": { - "name": "upstream HTTP errors are normalized", - "path": "::upstream HTTP errors are normalized", - "id": "32a2aad37a49c753878cc677cacc74f2", - "groups": {}, - "checks": { - "mock/error-400 maps to HTTP 400": { - "name": "mock/error-400 maps to HTTP 400", - "path": "::upstream HTTP errors are normalized::mock/error-400 maps to HTTP 400", - "id": "c7879436a115417be678ddc6599ac85b", - "passes": 1, - "fails": 0 - }, - "mock/error-400 maps to upstream_rejected_request": { - "path": "::upstream HTTP errors are normalized::mock/error-400 maps to upstream_rejected_request", - "id": "0ff165804aee476e99c0f183a8e47194", - "passes": 1, - "fails": 0, - "name": "mock/error-400 maps to upstream_rejected_request" - }, - "mock/error-503 maps to HTTP 502": { - "fails": 0, - "name": "mock/error-503 maps to HTTP 502", - "path": "::upstream HTTP errors are normalized::mock/error-503 maps to HTTP 502", - "id": "c8cc09e5630d8ea345216978a16a5758", - "passes": 1 - }, - "mock/error-503 maps to upstream_http_error": { - "name": "mock/error-503 maps to upstream_http_error", - "path": "::upstream HTTP errors are normalized::mock/error-503 maps to upstream_http_error", - "id": "7e9c8d43397b04850b58146835104dbb", - "passes": 1, - "fails": 0 - } - } - }, - "invalid first SSE event remains an HTTP error": { - "path": "::invalid first SSE event remains an HTTP error", - "id": "f363df470e062154860663761186013b", - "groups": {}, - "checks": { - "invalid first event returns 502": { - "path": "::invalid first SSE event remains an HTTP error::invalid first event returns 502", - "id": "cc3f5f8af185e392c88c313ef7235c0a", - "passes": 1, - "fails": 0, - "name": "invalid first event returns 502" - }, - "invalid first event is a protocol error": { - "name": "invalid first event is a protocol error", - "path": "::invalid first SSE event remains an HTTP error::invalid first event is a protocol error", - "id": "0ec7fd8bfe0ce80934f8323c329a9ff5", - "passes": 1, - "fails": 0 - } - }, - "name": "invalid first SSE event remains an HTTP error" - }, - "committed SSE stream never switches provider": { - "id": "ab0b6df520ceb7cc018fa70249f08c75", - "groups": {}, - "checks": { - "dropped stream was already committed": { - "name": "dropped stream was already committed", - "path": "::committed SSE stream never switches provider::dropped stream was already committed", - "id": "b97aa7caa5265f5bd57d1daefe9d294f", - "passes": 1, - "fails": 0 - }, - "dropped stream contains the first chunk": { - "fails": 0, - "name": "dropped stream contains the first chunk", - "path": "::committed SSE stream never switches provider::dropped stream contains the first chunk", - "id": "d3be32634465149ce992660ccd48bed9", - "passes": 1 - }, - "dropped stream has no terminal marker": { - "name": "dropped stream has no terminal marker", - "path": "::committed SSE stream never switches provider::dropped stream has no terminal marker", - "id": "c56b585a5317a2d6de7337067b595d14", - "passes": 1, - "fails": 0 - } - }, - "name": "committed SSE stream never switches provider", - "path": "::committed SSE stream never switches provider" - }, - "rate limit preserves Retry-After": { - "groups": {}, - "checks": { - "upstream rate limit returns 429": { - "passes": 1, - "fails": 0, - "name": "upstream rate limit returns 429", - "path": "::rate limit preserves Retry-After::upstream rate limit returns 429", - "id": "2f6579d8d6b38804d7433310c64ab817" - }, - "upstream rate limit has the gateway error code": { - "fails": 0, - "name": "upstream rate limit has the gateway error code", - "path": "::rate limit preserves Retry-After::upstream rate limit has the gateway error code", - "id": "3d51dbe72ebaffa776184897c0a934a9", - "passes": 1 - }, - "upstream rate limit preserves Retry-After": { - "id": "0c5e5336bd996b00ee343bd6f07cb7b9", - "passes": 1, - "fails": 0, - "name": "upstream rate limit preserves Retry-After", - "path": "::rate limit preserves Retry-After::upstream rate limit preserves Retry-After" - } - }, - "name": "rate limit preserves Retry-After", - "path": "::rate limit preserves Retry-After", - "id": "0a41eb17ed50fb9bf10e3fc8f139997e" - } - }, - "checks": {}, - "name": "", - "path": "" - }, - "metrics": { - "data_received": { - "count": 3724, - "rate": 1916.8837141982758 - }, - "http_req_sending": { - "p(90)": 0.05284209999999999, - "p(95)": 0.06750254999999998, - "p(99)": 0.07923090999999999, - "max": 0.082163, - "avg": 0.029845125, - "med": 0.023618 - }, - "http_req_blocked": { - "med": 0.0043095, - "p(90)": 0.463826, - "p(95)": 0.46475350000000004, - "p(99)": 0.4654955, - "max": 0.465681, - "avg": 0.11934374999999998 - }, - "http_req_receiving": { - "med": 0.078164, - "p(90)": 0.2532670999999999, - "p(95)": 0.43640354999999975, - "p(99)": 0.5829127099999999, - "max": 0.61954, - "avg": 0.1289085 - }, - "vus_max": { - "value": 1, - "min": 1, - "max": 1 - }, - "http_req_duration{expected_response:true}": { - "p(90)": 517.5200582999998, - "p(95)": 761.3863771499996, - "p(99)": 956.4794322299999, - "max": 1005.252696, - "avg": 242.44273262499996, - "med": 154.739508 - }, - "http_req_tls_handshaking": { - "p(95)": 0, - "p(99)": 0, - "max": 0, - "avg": 0, - "med": 0, - "p(90)": 0 - }, - "checks": { - "passes": 17, - "fails": 0, - "thresholds": { - "rate==1": false - }, - "value": 1 - }, - "http_req_duration": { - "avg": 242.44273262499996, - "med": 154.739508, - "p(90)": 517.5200582999998, - "p(95)": 761.3863771499996, - "p(99)": 956.4794322299999, - "max": 1005.252696 - }, - "group_duration": { - "avg": 277.50142285714287, - "med": 308.737737, - "p(90)": 588.1673946000003, - "p(95)": 796.8726422999996, - "p(99)": 963.8368404599996, - "max": 1005.57789 - }, - "data_sent": { - "count": 3242, - "rate": 1668.780075572183 - }, - "reliability_success": { - "passes": 8, - "fails": 0, - "thresholds": { - "rate==1": false - }, - "value": 1 - }, - "iterations": { - "count": 1, - "rate": 0.5147378394732212 - }, - "http_req_waiting": { - "avg": 242.283979, - "med": 154.67119350000002, - "p(90)": 517.4175926999999, - "p(95)": 761.2851078499996, - "p(99)": 956.3791199699998, - "max": 1005.152623 - }, - "http_req_connecting": { - "max": 0.420416, - "avg": 0.10306425, - "med": 0, - "p(90)": 0.4089934, - "p(95)": 0.4147047, - "p(99)": 0.41927374 - }, - "vus": { - "min": 1, - "max": 1, - "value": 1 - }, - "http_reqs": { - "count": 8, - "rate": 4.117902715785769 - }, - "http_req_failed": { - "passes": 0, - "fails": 8, - "value": 0 - }, - "iteration_duration": { - "avg": 1942.62506, - "med": 1942.62506, - "p(90)": 1942.62506, - "p(95)": 1942.62506, - "p(99)": 1942.62506, - "max": 1942.62506 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/smoke-summary.json b/threehost/threehost/20260725T120000Z/smoke-summary.json deleted file mode 100644 index 3796a62..0000000 --- a/threehost/threehost/20260725T120000Z/smoke-summary.json +++ /dev/null @@ -1,178 +0,0 @@ -{ - "root_group": { - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "fake upstream is healthy": { - "fails": 0, - "name": "fake upstream is healthy", - "path": "::fake upstream is healthy", - "id": "7930b329c03b9ff24cacf29e7222e1a3", - "passes": 1 - }, - "gateway is ready": { - "id": "f94d33b229bbbdcbeef6a450cb5fc984", - "passes": 1, - "fails": 0, - "name": "gateway is ready", - "path": "::gateway is ready" - }, - "chat status is 200": { - "name": "chat status is 200", - "path": "::chat status is 200", - "id": "8b366dfc28447643d7a100f81d49e9cb", - "passes": 2, - "fails": 0 - }, - "chat body is a completion": { - "path": "::chat body is a completion", - "id": "dd8a724b149dbb7fedc9ddf36bff0341", - "passes": 2, - "fails": 0, - "name": "chat body is a completion" - }, - "stream status is 200": { - "fails": 0, - "name": "stream status is 200", - "path": "::stream status is 200", - "id": "d15eb5c4982b20ba39bac56e96c9eff4", - "passes": 1 - }, - "stream content type is SSE": { - "name": "stream content type is SSE", - "path": "::stream content type is SSE", - "id": "50a79a216c54aaeb47d07c1cbfcc3332", - "passes": 1, - "fails": 0 - }, - "stream has terminal marker": { - "fails": 0, - "name": "stream has terminal marker", - "path": "::stream has terminal marker", - "id": "787cf8391739503c3d3fdb967bb841df", - "passes": 1 - } - } - }, - "metrics": { - "http_reqs": { - "count": 5, - "rate": 9.16406775996011 - }, - "http_req_blocked": { - "avg": 0.4294272, - "med": 0.00422, - "p(90)": 1.1923028000000002, - "p(95)": 1.4397524, - "p(99)": 1.63771208, - "max": 1.687202 - }, - "chat_success": { - "passes": 3, - "fails": 0, - "thresholds": { - "rate==1": false - }, - "value": 1 - }, - "http_req_sending": { - "p(90)": 0.053858600000000006, - "p(95)": 0.06094179999999999, - "p(99)": 0.06660836, - "max": 0.068025, - "avg": 0.030132200000000005, - "med": 0.019629 - }, - "iteration_duration": { - "p(95)": 545.455489, - "p(99)": 545.455489, - "max": 545.455489, - "avg": 545.455489, - "med": 545.455489, - "p(90)": 545.455489 - }, - "checks": { - "passes": 9, - "fails": 0, - "thresholds": { - "rate==1": false - }, - "value": 1 - }, - "http_req_receiving": { - "avg": 64.9019556, - "med": 0.041297, - "p(90)": 194.62608740000005, - "p(95)": 259.47496519999993, - "p(99)": 311.35406744, - "max": 324.323843 - }, - "http_req_duration{expected_response:true}": { - "p(90)": 322.0048118, - "p(95)": 425.4668333999999, - "p(99)": 508.23645067999996, - "max": 528.928855, - "avg": 108.4359378, - "med": 0.875398 - }, - "http_req_tls_handshaking": { - "p(95)": 0, - "p(99)": 0, - "max": 0, - "avg": 0, - "med": 0, - "p(90)": 0 - }, - "stream_first_byte_ms": { - "avg": 204.585383, - "med": 204.585383, - "p(90)": 204.585383, - "p(95)": 204.585383, - "p(99)": 204.585383, - "max": 204.585383 - }, - "http_req_failed": { - "passes": 0, - "fails": 5, - "value": 0 - }, - "data_received": { - "count": 5588, - "rate": 10241.76212853142 - }, - "http_req_duration": { - "avg": 108.4359378, - "med": 0.875398, - "p(90)": 322.0048118, - "p(95)": 425.4668333999999, - "p(99)": 508.23645067999996, - "max": 528.928855 - }, - "http_req_connecting": { - "avg": 0.1532252, - "med": 0, - "p(90)": 0.3838178, - "p(95)": 0.3853274, - "p(99)": 0.38653508, - "max": 0.386837 - }, - "iterations": { - "count": 1, - "rate": 1.8328135519920221 - }, - "http_req_waiting": { - "avg": 43.50385, - "med": 0.807143, - "p(90)": 127.3825498, - "p(95)": 165.98396639999999, - "p(99)": 196.86509968000001, - "max": 204.585383 - }, - "data_sent": { - "count": 1409, - "rate": 2582.4342947567593 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/vus-nonstream-direct-r1-summary.json b/threehost/threehost/20260725T120000Z/vus-nonstream-direct-r1-summary.json deleted file mode 100644 index 25dda97..0000000 --- a/threehost/threehost/20260725T120000Z/vus-nonstream-direct-r1-summary.json +++ /dev/null @@ -1,148 +0,0 @@ -{ - "metrics": { - "http_req_tls_handshaking": { - "p(90)": 0, - "p(95)": 0, - "p(99)": 0, - "max": 0, - "avg": 0, - "med": 0 - }, - "http_req_waiting": { - "avg": 0.5028096898741096, - "med": 0.408877, - "p(90)": 0.8539768000000008, - "p(95)": 1.1027205999999996, - "p(99)": 1.9330259599999986, - "max": 16.299486 - }, - "vus": { - "value": 20, - "min": 20, - "max": 20 - }, - "http_req_failed": { - "passes": 0, - "fails": 1960665, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "http_req_blocked": { - "med": 0.00101, - "p(90)": 0.00147, - "p(95)": 0.00168, - "p(99)": 0.00573, - "max": 9.438044, - "avg": 0.0012812534772655298 - }, - "data_sent": { - "rate": 12016879.70553216, - "count": 721019369 - }, - "data_received": { - "count": 1035958777, - "rate": 17265821.88820397 - }, - "vus_max": { - "max": 20, - "value": 20, - "min": 20 - }, - "http_req_receiving": { - "avg": 0.013690428945283037, - "med": 0.010199, - "p(90)": 0.015279, - "p(95)": 0.017168799999999814, - "p(99)": 0.03783735999999987, - "max": 15.950354 - }, - "http_req_connecting": { - "max": 1.24023, - "avg": 0.000005123439241277832, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0 - }, - "http_req_sending": { - "max": 19.423403, - "avg": 0.005744150261258389, - "med": 0.00323, - "p(90)": 0.004469, - "p(95)": 0.006949, - "p(99)": 0.013189 - }, - "iterations": { - "count": 1960665, - "rate": 32677.45148167748 - }, - "checks": { - "passes": 3921330, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "chat_success": { - "passes": 1960665, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_duration": { - "p(99)": 2.0318672399999924, - "max": 20.581399, - "avg": 0.5222442690806934, - "med": 0.423675, - "p(90)": 0.875549, - "p(95)": 1.1343778 - }, - "http_req_duration{expected_response:true}": { - "p(90)": 0.875549, - "p(95)": 1.1343778, - "p(99)": 2.0318672399999924, - "max": 20.581399, - "avg": 0.5222442690806934, - "med": 0.423675 - }, - "iteration_duration": { - "avg": 0.6081731183358622, - "med": 0.499959, - "p(90)": 0.964022, - "p(95)": 1.2441645999999968, - "p(99)": 2.27606543999999, - "max": 20.675921 - }, - "http_reqs": { - "rate": 32677.45148167748, - "count": 1960665 - } - }, - "root_group": { - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "chat status is 200": { - "name": "chat status is 200", - "path": "::chat status is 200", - "id": "8b366dfc28447643d7a100f81d49e9cb", - "passes": 1960665, - "fails": 0 - }, - "chat body is a completion": { - "name": "chat body is a completion", - "path": "::chat body is a completion", - "id": "dd8a724b149dbb7fedc9ddf36bff0341", - "passes": 1960665, - "fails": 0 - } - }, - "name": "" - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/vus-nonstream-gateway-r1-summary.json b/threehost/threehost/20260725T120000Z/vus-nonstream-gateway-r1-summary.json deleted file mode 100644 index 315228d..0000000 --- a/threehost/threehost/20260725T120000Z/vus-nonstream-gateway-r1-summary.json +++ /dev/null @@ -1,148 +0,0 @@ -{ - "root_group": { - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "chat status is 200": { - "id": "8b366dfc28447643d7a100f81d49e9cb", - "passes": 55606, - "fails": 0, - "name": "chat status is 200", - "path": "::chat status is 200" - }, - "chat body is a completion": { - "path": "::chat body is a completion", - "id": "dd8a724b149dbb7fedc9ddf36bff0341", - "passes": 55606, - "fails": 0, - "name": "chat body is a completion" - } - } - }, - "metrics": { - "http_req_receiving": { - "avg": 0.013720952109484458, - "med": 0.012519, - "p(90)": 0.017609, - "p(95)": 0.021175, - "p(99)": 0.03204795, - "max": 1.067383 - }, - "http_req_failed": { - "passes": 0, - "fails": 55606, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "chat_success": { - "passes": 55606, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_reqs": { - "count": 55606, - "rate": 926.199508382272 - }, - "checks": { - "passes": 111212, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "data_sent": { - "count": 25032424, - "rate": 416951.7462578964 - }, - "http_req_blocked": { - "avg": 0.0013424362299032442, - "med": 0.00098, - "p(90)": 0.001371, - "p(95)": 0.001639, - "p(99)": 0.0028599499999999974, - "max": 1.041957 - }, - "http_req_duration{expected_response:true}": { - "avg": 21.508987718447628, - "med": 16.8623485, - "p(90)": 33.8611355, - "p(95)": 52.73719425, - "p(99)": 105.64271049999984, - "max": 237.637214 - }, - "http_req_tls_handshaking": { - "p(90)": 0, - "p(95)": 0, - "p(99)": 0, - "max": 0, - "avg": 0, - "med": 0 - }, - "vus_max": { - "max": 20, - "value": 20, - "min": 20 - }, - "data_received": { - "count": 35430746, - "rate": 590151.0543253813 - }, - "http_req_waiting": { - "p(90)": 33.8405485, - "p(95)": 52.719040250000006, - "p(99)": 105.62022294999983, - "max": 237.610416, - "avg": 21.490067717638286, - "med": 16.843105 - }, - "http_req_duration": { - "p(90)": 33.8611355, - "p(95)": 52.73719425, - "p(99)": 105.64271049999984, - "max": 237.637214, - "avg": 21.508987718447628, - "med": 16.8623485 - }, - "http_req_connecting": { - "max": 0.865549, - "avg": 0.0001866500917167212, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0 - }, - "iteration_duration": { - "med": 16.9368625, - "p(90)": 33.933643000000004, - "p(95)": 52.8381055, - "p(99)": 105.72411394999986, - "max": 237.710367, - "avg": 21.583570985649022 - }, - "iterations": { - "count": 55606, - "rate": 926.199508382272 - }, - "http_req_sending": { - "p(90)": 0.007309, - "p(95)": 0.00852, - "p(99)": 0.017139949999999998, - "max": 1.092461, - "avg": 0.00519904869978063, - "med": 0.00446 - }, - "vus": { - "value": 20, - "min": 20, - "max": 20 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/vus-stream-direct-r1-summary.json b/threehost/threehost/20260725T120000Z/vus-stream-direct-r1-summary.json deleted file mode 100644 index eb9756c..0000000 --- a/threehost/threehost/20260725T120000Z/vus-stream-direct-r1-summary.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "root_group": { - "groups": {}, - "checks": { - "stream status is 200": { - "fails": 0, - "name": "stream status is 200", - "path": "::stream status is 200", - "id": "d15eb5c4982b20ba39bac56e96c9eff4", - "passes": 2280 - }, - "stream content type is SSE": { - "path": "::stream content type is SSE", - "id": "50a79a216c54aaeb47d07c1cbfcc3332", - "passes": 2280, - "fails": 0, - "name": "stream content type is SSE" - }, - "stream has terminal marker": { - "fails": 0, - "name": "stream has terminal marker", - "path": "::stream has terminal marker", - "id": "787cf8391739503c3d3fdb967bb841df", - "passes": 2280 - } - }, - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e" - }, - "metrics": { - "checks": { - "passes": 6840, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_reqs": { - "count": 2280, - "rate": 37.858220930436026 - }, - "http_req_connecting": { - "p(90)": 0, - "p(95)": 0, - "p(99)": 0, - "max": 2.992108, - "avg": 0.02203977807017544, - "med": 0 - }, - "http_req_duration": { - "max": 531.805825, - "avg": 528.0914095609654, - "med": 528.0730925, - "p(90)": 528.9627703, - "p(95)": 529.28909835, - "p(99)": 530.16431677 - }, - "http_req_tls_handshaking": { - "max": 0, - "avg": 0, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0 - }, - "data_received": { - "count": 9008280, - "rate": 149577.83089615277 - }, - "http_req_waiting": { - "p(95)": 201.23695795, - "p(99)": 201.5198839, - "max": 202.581315, - "avg": 200.72924511622784, - "med": 200.73288150000002, - "p(90)": 201.0841552 - }, - "vus_max": { - "value": 20, - "min": 20, - "max": 20 - }, - "data_sent": { - "count": 823134, - "rate": 13667.714397962074 - }, - "http_req_failed": { - "passes": 0, - "fails": 2280, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "http_req_blocked": { - "med": 0.00131, - "p(90)": 0.0040991000000000005, - "p(95)": 0.004710049999999994, - "p(99)": 0.02366491000000001, - "max": 3.036584, - "avg": 0.02417528377192985 - }, - "iterations": { - "count": 2280, - "rate": 37.858220930436026 - }, - "http_req_duration{expected_response:true}": { - "max": 531.805825, - "avg": 528.0914095609654, - "med": 528.0730925, - "p(90)": 528.9627703, - "p(95)": 529.28909835, - "p(99)": 530.16431677 - }, - "iteration_duration": { - "max": 531.993939, - "avg": 528.2720779100878, - "med": 528.219664, - "p(90)": 529.1674115999999, - "p(95)": 529.5399081500001, - "p(99)": 530.42293998 - }, - "vus": { - "value": 20, - "min": 20, - "max": 20 - }, - "http_req_receiving": { - "avg": 327.35131136096476, - "med": 327.397225, - "p(90)": 328.1362136, - "p(95)": 328.5709741, - "p(99)": 329.23601348, - "max": 331.075162 - }, - "stream_first_byte_ms": { - "max": 202.581315, - "avg": 200.72924511622787, - "med": 200.73288150000002, - "p(90)": 201.0841552, - "p(95)": 201.23695795, - "p(99)": 201.5198839 - }, - "chat_success": { - "passes": 2280, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_sending": { - "med": 0.006269, - "p(90)": 0.021588, - "p(95)": 0.025271449999999987, - "p(99)": 0.06644169000000011, - "max": 0.328003, - "avg": 0.010853083771929825 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/vus-stream-gateway-r1-summary.json b/threehost/threehost/20260725T120000Z/vus-stream-gateway-r1-summary.json deleted file mode 100644 index ce684e3..0000000 --- a/threehost/threehost/20260725T120000Z/vus-stream-gateway-r1-summary.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "metrics": { - "http_req_duration{expected_response:true}": { - "max": 577.868892, - "avg": 537.705397901785, - "med": 539.4157605, - "p(90)": 543.6027708, - "p(95)": 545.2504474, - "p(99)": 559.26164579 - }, - "data_received": { - "count": 9188393, - "rate": 152472.57702156113 - }, - "http_req_blocked": { - "avg": 0.005836951339285714, - "med": 0.00116, - "p(90)": 0.0032410000000000013, - "p(95)": 0.003950449999999998, - "p(99)": 0.013665710000000764, - "max": 0.891858 - }, - "iteration_duration": { - "p(95)": 545.3530045, - "p(99)": 559.35368109, - "max": 578.3869, - "avg": 537.8332772727675, - "med": 539.5350169999999, - "p(90)": 543.7136159 - }, - "chat_success": { - "passes": 2240, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_sending": { - "avg": 0.009649153125, - "med": 0.007609, - "p(90)": 0.018202000000000003, - "p(95)": 0.02170039999999999, - "p(99)": 0.034496800000000015, - "max": 0.145877 - }, - "stream_first_byte_ms": { - "avg": 205.58899070133907, - "med": 204.75843700000001, - "p(90)": 207.1935972, - "p(95)": 209.3904336, - "p(99)": 222.17192246000002, - "max": 249.742346 - }, - "http_req_connecting": { - "max": 0.840453, - "avg": 0.004052936607142857, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0 - }, - "vus": { - "value": 20, - "min": 20, - "max": 20 - }, - "checks": { - "passes": 6720, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_duration": { - "avg": 537.705397901785, - "med": 539.4157605, - "p(90)": 543.6027708, - "p(95)": 545.2504474, - "p(99)": 559.26164579, - "max": 577.868892 - }, - "http_req_receiving": { - "p(90)": 337.1059902, - "p(95)": 337.69769410000004, - "p(99)": 340.94168205, - "max": 363.686844, - "avg": 332.1067580473211, - "med": 334.771739 - }, - "http_req_waiting": { - "med": 204.75843700000001, - "p(90)": 207.1935972, - "p(95)": 209.3904336, - "p(99)": 222.17192246000002, - "max": 249.742346, - "avg": 205.58899070133907 - }, - "data_sent": { - "count": 999076, - "rate": 16578.708851525313 - }, - "http_req_failed": { - "passes": 0, - "fails": 2240, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "http_reqs": { - "count": 2240, - "rate": 37.1706535112611 - }, - "vus_max": { - "value": 20, - "min": 20, - "max": 20 - }, - "http_req_tls_handshaking": { - "max": 0, - "avg": 0, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0 - }, - "iterations": { - "count": 2240, - "rate": 37.1706535112611 - } - }, - "root_group": { - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "stream status is 200": { - "passes": 2240, - "fails": 0, - "name": "stream status is 200", - "path": "::stream status is 200", - "id": "d15eb5c4982b20ba39bac56e96c9eff4" - }, - "stream content type is SSE": { - "name": "stream content type is SSE", - "path": "::stream content type is SSE", - "id": "50a79a216c54aaeb47d07c1cbfcc3332", - "passes": 2240, - "fails": 0 - }, - "stream has terminal marker": { - "name": "stream has terminal marker", - "path": "::stream has terminal marker", - "id": "787cf8391739503c3d3fdb967bb841df", - "passes": 2240, - "fails": 0 - } - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/warmup-direct-summary.json b/threehost/threehost/20260725T120000Z/warmup-direct-summary.json deleted file mode 100644 index 7c36014..0000000 --- a/threehost/threehost/20260725T120000Z/warmup-direct-summary.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "root_group": { - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {}, - "checks": { - "chat status is 200": { - "id": "8b366dfc28447643d7a100f81d49e9cb", - "passes": 51, - "fails": 0, - "name": "chat status is 200", - "path": "::chat status is 200" - }, - "chat body is a completion": { - "path": "::chat body is a completion", - "id": "dd8a724b149dbb7fedc9ddf36bff0341", - "passes": 51, - "fails": 0, - "name": "chat body is a completion" - } - }, - "name": "" - }, - "metrics": { - "chat_success": { - "passes": 51, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_sending": { - "avg": 0.0490731568627451, - "med": 0.047286, - "p(90)": 0.057525, - "p(95)": 0.070269, - "p(99)": 0.0798885, - "max": 0.085873 - }, - "http_req_failed": { - "passes": 0, - "fails": 51, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "http_req_tls_handshaking": { - "p(95)": 0, - "p(99)": 0, - "max": 0, - "avg": 0, - "med": 0, - "p(90)": 0 - }, - "data_received": { - "count": 26761, - "rate": 2675.773303599116 - }, - "iterations": { - "count": 51, - "rate": 5.099377395596386 - }, - "http_req_duration{expected_response:true}": { - "p(95)": 0.505534, - "p(99)": 0.5225420000000001, - "max": 0.537176, - "avg": 0.47504952941176465, - "med": 0.476071, - "p(90)": 0.499369 - }, - "data_sent": { - "count": 18383, - "rate": 1838.075581632321 - }, - "http_reqs": { - "count": 51, - "rate": 5.099377395596386 - }, - "http_req_blocked": { - "p(99)": 0.5202315, - "max": 0.520576, - "avg": 0.4571129803921567, - "med": 0.464823, - "p(90)": 0.493419, - "p(95)": 0.5149079999999999 - }, - "http_req_receiving": { - "max": 0.064355, - "avg": 0.04207807843137255, - "med": 0.041256, - "p(90)": 0.045836, - "p(95)": 0.050225, - "p(99)": 0.058470499999999995 - }, - "http_req_connecting": { - "p(95)": 0.45422850000000004, - "p(99)": 0.464271, - "max": 0.469201, - "avg": 0.40334200000000014, - "med": 0.410246, - "p(90)": 0.437663 - }, - "iteration_duration": { - "p(95)": 1.3608985, - "p(99)": 1.40417, - "max": 1.407584, - "avg": 1.2936127254901961, - "med": 1.296853, - "p(90)": 1.35466 - }, - "vus": { - "max": 0, - "value": 0, - "min": 0 - }, - "http_req_duration": { - "p(90)": 0.499369, - "p(95)": 0.505534, - "p(99)": 0.5225420000000001, - "max": 0.537176, - "avg": 0.47504952941176465, - "med": 0.476071 - }, - "http_req_waiting": { - "med": 0.387538, - "p(90)": 0.411696, - "p(95)": 0.41874500000000003, - "p(99)": 0.432915, - "max": 0.444185, - "avg": 0.38389829411764714 - }, - "checks": { - "passes": 102, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "dropped_iterations": { - "count": 0, - "rate": 0, - "thresholds": { - "count==0": false - } - }, - "vus_max": { - "min": 50, - "max": 50, - "value": 50 - } - } -} \ No newline at end of file diff --git a/threehost/threehost/20260725T120000Z/warmup-gateway-summary.json b/threehost/threehost/20260725T120000Z/warmup-gateway-summary.json deleted file mode 100644 index cbec808..0000000 --- a/threehost/threehost/20260725T120000Z/warmup-gateway-summary.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "root_group": { - "checks": { - "chat status is 200": { - "passes": 51, - "fails": 0, - "name": "chat status is 200", - "path": "::chat status is 200", - "id": "8b366dfc28447643d7a100f81d49e9cb" - }, - "chat body is a completion": { - "name": "chat body is a completion", - "path": "::chat body is a completion", - "id": "dd8a724b149dbb7fedc9ddf36bff0341", - "passes": 51, - "fails": 0 - } - }, - "name": "", - "path": "", - "id": "d41d8cd98f00b204e9800998ecf8427e", - "groups": {} - }, - "metrics": { - "vus_max": { - "value": 50, - "min": 50, - "max": 50 - }, - "checks": { - "passes": 102, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_tls_handshaking": { - "avg": 0, - "med": 0, - "p(90)": 0, - "p(95)": 0, - "p(99)": 0, - "max": 0 - }, - "http_req_sending": { - "avg": 0.05004149019607842, - "med": 0.048206, - "p(90)": 0.062345, - "p(95)": 0.06609999999999999, - "p(99)": 0.071644, - "max": 0.073454 - }, - "http_req_waiting": { - "max": 6.055635, - "avg": 4.7223325686274515, - "med": 4.65387, - "p(90)": 5.084563, - "p(95)": 5.37992, - "p(99)": 5.937994 - }, - "data_sent": { - "count": 22722, - "rate": 2270.945810450304 - }, - "dropped_iterations": { - "count": 0, - "rate": 0, - "thresholds": { - "count==0": false - } - }, - "iterations": { - "count": 51, - "rate": 5.097184945557852 - }, - "http_req_duration": { - "p(95)": 5.4831805, - "p(99)": 6.040051, - "max": 6.147937, - "avg": 4.817414215686273, - "med": 4.749031, - "p(90)": 5.182436 - }, - "iteration_duration": { - "avg": 5.679276372549019, - "med": 5.660856, - "p(90)": 6.073183, - "p(95)": 6.3768435, - "p(99)": 6.8888549999999995, - "max": 6.948491 - }, - "http_req_failed": { - "passes": 0, - "fails": 51, - "thresholds": { - "rate<=0.01": false - }, - "value": 0 - }, - "http_reqs": { - "count": 51, - "rate": 5.097184945557852 - }, - "http_req_connecting": { - "p(99)": 0.5063735, - "max": 0.514878, - "avg": 0.44594156862745093, - "med": 0.455163, - "p(90)": 0.4865, - "p(95)": 0.4933745 - }, - "vus": { - "value": 0, - "min": 0, - "max": 0 - }, - "http_req_duration{expected_response:true}": { - "max": 6.147937, - "avg": 4.817414215686273, - "med": 4.749031, - "p(90)": 5.182436, - "p(95)": 5.4831805, - "p(99)": 6.040051 - }, - "http_req_receiving": { - "avg": 0.0450401568627451, - "med": 0.044315, - "p(90)": 0.053115, - "p(95)": 0.053985500000000006, - "p(99)": 0.060354500000000005, - "max": 0.065734 - }, - "data_received": { - "count": 32259, - "rate": 3224.119395269622 - }, - "chat_success": { - "passes": 51, - "fails": 0, - "thresholds": { - "rate>=0.99": false - }, - "value": 1 - }, - "http_req_blocked": { - "p(99)": 0.5723739999999999, - "max": 0.584303, - "avg": 0.4988157058823529, - "med": 0.509478, - "p(90)": 0.539877, - "p(95)": 0.5491355 - } - } -} \ No newline at end of file