diff --git a/.env.example b/.env.example index 0972348..7e21eea 100644 --- a/.env.example +++ b/.env.example @@ -26,8 +26,10 @@ LLMBENCHLAB_WORKER_MAX_ATTEMPTS=3 LLMBENCHLAB_WORKER_SHUTDOWN_GRACE_SECONDS=30 LLMBENCHLAB_WORKER_PROGRESS_FLUSH_SECONDS=5 LLMBENCHLAB_WORKER_PROGRESS_STALE_SECONDS=60 +LLMBENCHLAB_DEV_WORKER_PROCESSES=1 LLMBENCHLAB_WORKER_EXPECTED_PROCESSES=1 LLMBENCHLAB_WORKER_RECOVERY_ALERT_SECONDS=60 +LLMBENCHLAB_COMPOSE_WORKER_PROCESSES=2 LLMBENCHLAB_REDIS_PUBLISH_TIMEOUT_SECONDS=1 LLMBENCHLAB_REDIS_OPERATION_TIMEOUT_SECONDS=2 diff --git a/CHANGELOG.md b/CHANGELOG.md index 177c14b..d7eb40c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,9 @@ Phase 0 and the Phase 1 MVP vertical slice are complete. The Phase 2 reliable-ex ### Added +- **Completed:** added explicit remote API protocol adapters under ADR-0019. Existing `openai_compatible` remains the backward-compatible Chat Completions value; new `openai_responses` and `anthropic_messages` types provide protocol-specific endpoints, request fields, authentication headers, ordinary JSON and typed SSE parsing, usage normalization, terminal-event checks and pre-send parameter rejection. Responses/Messages omit `temperature`, `top_p`, and `seed` when neither the request nor Model defaults supply them; Messages also enforces `temperature<=1` and finite `max_tokens`. Model/API/Run snapshots, trusted-local CLI preflight, protocol-authenticated `/models` discovery with bounded Messages pagination, Alembic `20260830_0008`, and the Web model/run forms now carry the explicit adapter type. Responses rate-limit/server typed errors and Messages `rate_limit_error`/`api_error`/`overloaded_error`/`timeout_error` plus HTTP `529` use bounded retry with an independently settled attempt ledger row; unknown typed errors fail closed. The implementation keeps the existing HTTPS/loopback, redirect, identity-only, bounded-body and current-Key redaction boundaries; automated verification uses MockTransport only and makes no real Provider call. Implementation SHA [`6943aa29a154c82bdfbe5efb2578c916c3cbf632`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/6943aa29a154c82bdfbe5efb2578c916c3cbf632) was pushed to `codex/complete-evaluation-workflow`, and its exact-SHA GitHub Actions [run `33304667092`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33304667092) passed all four required jobs. +- **Completed:** added daily multi-Worker execution entrypoints. `make dev DEV_WORKERS=N` manages independent local Worker processes only when the effective database is PostgreSQL, while `make dev-multi` / `make docker-up WORKERS=N` defaults to two Compose Workers, derives scale and API expected from one bounded input, counts exited replicas when choosing the safe direction, proves fresh/new generation scans with a database-time watermark, and requires exact `expected/registered/live/stalled/shortfall=N/N/N/0/0`. Launcher tests pass `42/42`; migrated PostgreSQL cross-Benchmark lease tests pass `2/2`; isolated real Compose `2→1→2` converges to `2/2/2/0/0 → 1/1/1/0/0 → 2/2/2/0/0` with exact cleanup; final local lint/test (`1003 passed, 35 skipped` backend and `64 passed` frontend), offline Mock smoke, build and config gates are green. Implementation SHA [`b06594c2df67d6e2a8b117651b193cd0fa409bf5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/b06594c2df67d6e2a8b117651b193cd0fa409bf5) was pushed, and its exact-SHA GitHub Actions [run `33299883513`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33299883513) passed all four required jobs. SQLite remains single-Worker, existing SQLite data is not auto-migrated, and 3+ Workers and real Providers are not qualified or exercised. +- **Completed:** added the P3-06 Run Detail heatmap/live-metrics slice. `GET /api/v1/runs/{run_id}/progress` uses a fixed 512-question absolute-position block index with evidence-derived metrics and counts from one database read snapshot; `/progress/blocks/{block_index}` returns the requested block's cells from a strict no-body/no-Provider-metadata allowlist, and the UI refetches only blocks whose counts changed. The UI uses a virtualized accessible four-state grid and keeps known Token/cost subtotals distinct from exact nullable Run totals. Local target/full gates and the target Run browser check pass; implementation SHA [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) was pushed to `codex/complete-evaluation-workflow`, is tracked by [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5), and its exact-SHA GitHub Actions [run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) passed all four required jobs. Phase 3 as a whole remains `in_progress`. - Added the P2-07 recovery-operations kickoff package: ADR-0016, an independent execution plan, and a work log now freeze a deliberately narrow PostgreSQL 16 dump/restore verification, separate keyring pairing, Redis-as-notification rebuild, alert-drill, destructive-operation, and evidence boundary. This commit contains documentation only; P2-07 implementation and qualification have not started, Phase 2 remains `in_progress`, and no production DR/PITR/RPO/RTO/HA claim is made. - Added the completed P2-06 implementation under ADR-0015: `GET /api/v1/metrics/prometheus` renders a fixed Prometheus text `0.0.4` gauge set from one DB-time snapshot, with enum-only labels, bounded 15-minute audit and one-hour latency windows, fail-closed audit validation, and one in-flight collection per API process. Repository-owned configuration supplies exactly eight alert rules and fixed Operations runbook links; Prometheus, Alertmanager, and notification delivery are not deployed by this repository. - Added Alembic revision `20260828_0005`, the `worker_processes` generation table, bounded `(expires_at,id)` / `(occurred_at,id)` audit scan indexes, and DB-time coalesced Worker scan/claim/lease-heartbeat/progress recording. `/tasks/metrics` and the exporter expose only low-cardinality aggregates; the dependency probe remains capability-only. The stopped SQLite-to-PostgreSQL importer now copies and reconciles all 13 application tables, rejects a still-live Worker generation, and preserves stopped/stale process facts. @@ -32,7 +35,7 @@ Phase 0 and the Phase 1 MVP vertical slice are complete. The Phase 2 reliable-ex - An explicit stopped-SQLite to offline-empty-PostgreSQL importer with read-only source validation, transactional locking/copy, content-free reconciliation digests, and distinct rollback/commit-uncertainty/post-commit-verification outcomes. - Public organization repository at `CWNU-Open-Source-Community/LLMBenchLab`, a visible CI badge, and a repository-wide stage gate requiring each commit to be pushed and its exact GitHub Actions SHA to pass all required jobs. - Pinned-source MMLU-Pro and GPQA-Diamond converters with source/archive SHA-256 verification, validated caches, deterministic filtering/shuffling, reproducible dataset-v1 ZIPs, and source/license/profile evidence without committing third-party questions. -- A trusted-local `llmbenchlab-evaluate` CLI with `prepare`, `run`, `resume`, and `report`; OpenAI-compatible model discovery and canary preflight; hidden/environment-only API keys; explicit request-bound confirmation; direct database execution; and missing-question recovery. Remote Provider endpoints require HTTPS, while plain HTTP is accepted only for loopback hosts; discovery rejects a model ID that reflects the current Key, and canary rejects a returned model that differs from the requested target. +- A trusted-local `llmbenchlab-evaluate` CLI with `prepare`, `run`, `resume`, and `report`; explicit Chat Completions, OpenAI Responses, or Anthropic Messages model discovery and canary preflight; hidden/environment-only API keys; explicit request-bound confirmation; direct database execution; and missing-question recovery. Remote Provider endpoints require HTTPS, while plain HTTP is accepted only for loopback hosts; discovery rejects a model ID that reflects the current Key, and canary rejects a returned model that differs from the requested target. - Atomic, non-overwriting terminal Run reports containing a protocol/source/model/execution summary, optional metadata groups, and every persisted per-question Response in paginated JSONL. Report metrics are derived from planned questions plus persisted Responses, and `metrics_provenance` identifies drift from persisted Run aggregate fields. - Web/API write-only Provider credentials: the Models password field accepts an 8–8192-byte visible-ASCII `api_key` directly, never reads it back, and distinguishes `stored`, legacy `environment`, and `none` sources without displaying the legacy environment-variable name. A one-row-per-model `model_credentials` table stores only AES-256-GCM ciphertext, nonce, algorithm and key ID; API and Worker share a deployment keyring while the existing environment-variable and trusted-local CLI paths remain compatible. - A first-class Evaluation Runs page in the main navigation with all-status history, status filtering, 20-item pagination, manual refresh, two-second polling only while the current page contains pending/running work, and stable links back to Run evidence. @@ -40,6 +43,8 @@ Phase 0 and the Phase 1 MVP vertical slice are complete. The Phase 2 reliable-ex ### Changed +- Prepared and loaded six source-pinned, Git-ignored 100-question small Benchmarks into the default personal SQLite through the existing validated import API: five deterministic mini subsets (GSM8K, Chinese MGSM, HellaSwag, WinoGrande, and TruthfulQA Binary) plus the complete 100-question Chinese XCOPA validation split. The 600 questions cover English and Chinese numeric/multiple-choice evaluation; deterministic seed-42 selection, source/archive/Dataset hashes, licenses, and balanced option distributions are recorded locally. This is personal data maintenance, not redistribution, a product/protocol change, or an official full-benchmark score; the pre-import online backup remains local. The import task did not create, cancel, or reset a Run; concurrent Run changes after the import are recorded separately in its work log. Documentation commit [`8faa2093b2c3308994d50e42a31063cdbf5264a6`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/8faa2093b2c3308994d50e42a31063cdbf5264a6) was pushed, and its exact-SHA GitHub Actions [run `33296049611`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33296049611) passed all four required jobs. +- Replaced the P3-06 draft `(created_at,id)` progress cursor contract before production implementation. Its four failure-first tests exposed the missing API, then design review established that application timestamps and UUIDs are not a monotonic database commit sequence and cannot prove no-loss concurrent pagination. The accepted design uses immutable-response counts in fixed 512-question absolute-position blocks and backend-derived same-snapshot live metrics; it needs no migration, ADR, API-version or protocol-version bump. - Loaded the three already prepared, Git-ignored standard dataset ZIPs into the default personal SQLite through the existing validated import API: GPQA-Diamond (198 questions), MMLU-Pro Direct (12,032), and MMLU-Pro Official-CoT (12,032). This is a local data-maintenance result, not redistributed third-party data or a product/protocol change; the pre-import database backup remains local and no Provider was called. Documentation commit `0163b67c00eb59ae59db5f3adb679ad85c799142` was pushed, and its exact-SHA GitHub Actions run `33266167547` passed all four required jobs. - Changed the combined `make dev` launcher to keep the console quiet after a concise address/log summary. API, Worker, and Vite output is appended to separate Git-ignored `artifacts/dev-logs` files with UTC session markers and private local permissions; individual service Make targets remain foreground diagnostics, and child failures still propagate their status. - Governed production logging at the source boundary: application calls use literal argument-free messages, structured values remain allowlisted and finite, third-party logger text/identity is replaced by fixed classification, and the raw Uvicorn access handler is disabled rather than bypassing the JSON redaction contract. @@ -66,6 +71,7 @@ Phase 0 and the Phase 1 MVP vertical slice are complete. The Phase 2 reliable-ex ### Fixed +- Fixed Run Detail treating `error_questions` as the total number of wrong answers. It now reports unscored, ordinary-wrong, and execution-error counts separately. The paginated Responses API also returns Run-wide known input/output Token subtotals and independent coverage counts, allowing the UI to show partial usage as an explicitly incomplete subtotal while preserving protocol-v1 exact Run Token `null` semantics. - Fixed managed Runs without an explicit `input_token_reservation` being permanently exhausted when Provider actual input exceeded a UTF-8 observational estimate. New attempts now leave input reservation and reserved cost unset in that case while preserving Provider actual usage; explicit input/output reservations and reserved cost derived from complete bounds plus frozen prices still enforce overdraw. Added data-only Alembic head `20260830_0007` to recompute only `governance_scopes.overdrawn`, preserve historical ledger/actual/Response/Run facts, reject upgrade or downgrade while any reservation is active, and restore the old derived predicate on downgrade. Run Detail now uses neutral historical wording for overdraw instead of attributing every case to conservative settlement. The real-Compose acceptance seam now verifies nullable input/cost bounds without coercing a valid `null` settlement through `float()`. - Added ADR-0017 and the forward-only, schema-equivalent Alembic repair revision `20260829_0006` for databases that executed an early `0004` variant before three governance indexes were present. Migration preflight accepts canonical schemas or only a missing subset of those three indexes, so an interrupted SQLite repair is resumable; it backs up SQLite, validates newly historical PostgreSQL `0005` metadata, rejects multiple active policies, and continues to reject every other schema drift. - Prevented external `LogRecord` extras from injecting allowlisted request/run fields, rejected FIFO/non-regular archive inputs without blocking by opening them nonblocking before `fstat`, and made oversized audit archives fail on their global line cap before decoding attacker-controlled lines. @@ -112,6 +118,7 @@ Phase 0 and the Phase 1 MVP vertical slice are complete. The Phase 2 reliable-ex ### Verification +- P3-06 verification and repository gate are complete. The abandoned cursor-first backend suite produced the expected `4 failed` before implementation; the replacement fixed-block backend/frontend targets passed `37/32` (`20` Run Detail + `12` heatmap). `make test` passed backend `964 passed, 33 skipped` and frontend `64 passed`; `make lint`, offline Mock smoke (`1 passed, 7 deselected`), frontend production build and `docker compose config --quiet` passed. The final frontend regressions ensure that terminal + reconciled progress triggers exactly one final Run/current-evidence refresh, and that a same-route `runId` change resets evidence offset to zero. Trusted-loopback browser verification of Run `a3de7e4d-40b2-4d8c-994b-c713047393ae` showed 179 passed / 17 wrong / 2 error, known input/output Token `45,509 / 4,561,625` with `196/198` coverage, no horizontal overflow at desktop/768/375, no console warning/error, and working keyboard/tooltip behavior. The 12,032/20,000 virtualization boundary is automated coverage, not a large-Run manual DevTools performance claim. Commit `99791964621165c9cc7ec36b4b2d27fe04e6acd5` was normally pushed to `codex/complete-evaluation-workflow`; PR #5 exact-SHA Actions run `33289522923` passed backend, backend-integration, full-stack-reliability, and frontend, completing this P3-06 slice without completing Phase 3. - P2-06 implementation gates are complete. Commit [`9a20676dcf545040782f04c166205d0043345753`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/9a20676dcf545040782f04c166205d0043345753) was pushed to `codex/complete-evaluation-workflow` and is tracked by [PR #3](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/3); exact-SHA GitHub Actions [run `33164609388`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33164609388) passed all four required jobs. Its local gates include green combined targeted suites, `make lint` across 152 Python files plus ESLint/TypeScript, backend `916 passed, 33 skipped`, frontend `38 passed`, offline Mock smoke `1 passed, 7 deselected`, PostgreSQL 16/Redis 7 migration/check and `33 passed, 0 skipped` integration, isolated SQLite migration/check, frontend build, Compose config, eight-rule Prometheus `v3.5.0` validation, and a 76-file technical/security review with 0 Blocker/High/Medium. Clean-SHA Compose acceptance passed 9/9 with `.pytest_cache/artifacts/phase2-acceptance/llmbenchlab-p2-92e173eeee28/evidence.json` (SHA-256 `e4ffb8668fd3fa62d59b5d83f5c29eede35b327d88e6099345acd5950670fc47`), Worker gauges `2/2/2/0/0`, and empty container/volume/network cleanup. Clean capacity passed with `.pytest_cache/artifacts/phase2-capacity/llmbenchlab-p2-ca5673061b0f/evidence.json` (SHA-256 `2382f9138f09028f269d76c341b236dd4089d678c8a2323582045fac2b4f5039`): 1W/2W/burst QPS `7.267474/12.962228/9.333604`, wall `8.255963/4.628834/6.428385s`, 18 Runs/270 Responses/270 question executions/271 reservations/1230 audit events, zero question error/drift/duplicate/PEL/lag, Worker expected 2 with shortfall 0, and empty container/volume/network/image cleanup with image counters `1/1/0/0`. The earlier dirty acceptance `.pytest_cache/artifacts/phase2-acceptance/llmbenchlab-p2-11554c25ec2d/evidence.json` (SHA-256 `d5f058457dbc29875cbac4bc38345b810b5ed556ea538862d309116ceb629fde`) and dirty capacity `.pytest_cache/artifacts/phase2-capacity/llmbenchlab-p2-c6de062ab77e/evidence.json` (SHA-256 `4aeb8271dd81e8671fc287942839f8d06862140ea9a6bf1d7ee5660265aa8453`, 1229 audit events) remain historical evidence rather than being overwritten. These are offline Mock observations, not production or real-Provider SLOs. Evidence-documentation commit [`ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6) was pushed, and its exact-SHA GitHub Actions [run `33165775037`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33165775037) passed all four required jobs, completing P2-06. Phase 2 stays `in_progress`; P2-07 now has a `planned` work package but remains unimplemented. The default user SQLite was below head, so its direct `alembic check` failed and it was deliberately not migrated. - Recorded two corrected local gate invocations without hiding their first outcomes: the initial integration cleanup command was rejected by the local safety policy before any container started, then the explicit cleanup/run passed; an over-broad Ruff invocation over `scripts/` surfaced 93 existing modernization warnings, while the intended `--select E,F,I` correctness/import gate passed. - On the governance/audit candidate at `665244e095905083b606b8e98e946ed1a02dc0fc`, `make test` passed with `604 passed, 29 skipped` on the backend and `38 passed` on the frontend; a separate real PostgreSQL/Redis integration run passed `29/29`. That frozen implementation also passed `make lint`, offline `make smoke` (`1 passed, 7 deselected`), the frontend production build, isolated SQLite and PostgreSQL migration/check gates, `docker compose config --quiet`, targeted governance/API/Worker suites, enhanced capacity, and 9/9 Compose acceptance. diff --git a/Makefile b/Makefile index 08dacb5..713fcef 100644 --- a/Makefile +++ b/Makefile @@ -1,12 +1,13 @@ SHELL := /bin/bash .DEFAULT_GOAL := help -.PHONY: help setup dev backend worker frontend test lint format smoke migrate phase2-acceptance phase2-capacity phase2-slo docker-up docker-down +.PHONY: help setup dev dev-multi backend worker frontend test lint format smoke migrate phase2-acceptance phase2-capacity phase2-slo docker-up docker-down help: @echo "LLMBenchLab developer commands:" @echo " make setup Install dependencies and initialize the local database" - @echo " make dev Start API, independent Worker, and frontend together" + @echo " make dev Start local API, Worker(s), and frontend (DEV_WORKERS=N needs PostgreSQL)" + @echo " make dev-multi Start PostgreSQL/Redis/API/frontend with two Workers (WORKERS=N)" @echo " make backend Start only the FastAPI development server" @echo " make worker Start only the independent task Worker" @echo " make frontend Start only the Vite development server" @@ -25,7 +26,13 @@ setup: @./scripts/setup.sh dev: - @./scripts/dev.sh + @if [[ "$(origin DEV_WORKERS)" != "undefined" ]]; then \ + LLMBENCHLAB_DEV_WORKER_PROCESSES="$(DEV_WORKERS)" ./scripts/dev.sh; \ + else \ + ./scripts/dev.sh; \ + fi + +dev-multi: docker-up backend: @./scripts/bootstrap_credential_keyring.sh @@ -82,7 +89,11 @@ phase2-slo: docker-up: @./scripts/bootstrap_credential_keyring.sh - @docker compose up --build --wait --wait-timeout 180 --remove-orphans + @if [[ "$(origin WORKERS)" != "undefined" ]]; then \ + LLMBENCHLAB_COMPOSE_WORKER_PROCESSES="$(WORKERS)" ./scripts/compose_up.sh; \ + else \ + ./scripts/compose_up.sh; \ + fi docker-down: @docker compose down --remove-orphans diff --git a/README.md b/README.md index 6457c2a..d9aab9c 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ GitHub:[`CWNU-Open-Source-Community/LLMBenchLab`](https://github.com/CWNU-Open LLMBenchLab 是一个面向个人开发者与研究人员的轻量级 LLM 评测工作台。它把模型注册、版本化 Benchmark、后台评测、逐题证据、汇总指标和排行榜放进一条可审计的本地流程,并以“默认离线、严格评分、结果可复现”为首要约束。 -当前版本在保留 SQLite 单机兼容路径的同时,已经交付 Phase 2 的可靠执行与治理候选;P2-06 可观测性/审计保留切片已完成实现、clean evidence 与 evidence-doc 精确 SHA CI,状态为 `completed`。PostgreSQL 是 Compose/部署目标和任务、四层治理、逐 HTTP attempt ledger、typed audit 与 Worker progress 的事实来源,Redis Streams 只提供可重复、可丢失的低延迟通知,独立 Worker 通过数据库租约和公平 question quantum 执行任务。完全不需要 API Key 的 Mock Demo 仍是默认验收路径;OpenAI-compatible Chat Completions 适配器及真实 Provider 调用始终是用户主动启用的可选能力。 +当前版本在保留 SQLite 单机兼容路径的同时,已经交付 Phase 2 的可靠执行与治理候选;P2-06 可观测性/审计保留切片已完成实现、clean evidence 与 evidence-doc 精确 SHA CI,状态为 `completed`。PostgreSQL 是 Compose/部署目标和任务、四层治理、逐 HTTP attempt ledger、typed audit 与 Worker progress 的事实来源,Redis Streams 只提供可重复、可丢失的低延迟通知,独立 Worker 通过数据库租约和公平 question quantum 执行任务。完全不需要 API Key 的 Mock Demo 仍是默认验收路径;Chat Completions、OpenAI Responses、Anthropic Messages 三类远程适配器及真实 Provider 调用始终是用户主动启用的可选能力。 ## 当前状态 @@ -15,31 +15,32 @@ LLMBenchLab 是一个面向个人开发者与研究人员的轻量级 LLM 评测 - Phase 0(治理、需求、架构和协议)已完成。 - Phase 1 MVP 已具备完整垂直链路:注册模型、载入/导入 Benchmark、创建 Run、逐题持久化、结果聚合和前端展示。 - Phase 2 候选已通过真实 PostgreSQL/Redis 和进程故障验证:除租约、心跳、fencing、幂等 Response 与数据库恢复外,Web/API managed Run 还具有 global/provider/model/run 四层数据库 admission、fixed-minute RPM/TPM、lifetime request/Token/cost budget、有限 backlog、公平 slice、逐 attempt reservation/settlement、typed audit 和历史延迟。 -- 当前 Alembic head 为 data-only `20260830_0007`:没有显式 `input_token_reservation` 时,输入 Token 估算只用于观测,不再写成 hard reservation 或参与 cost/overdraw 裁决;Provider actual usage 仍完整保存。显式 input/output 预留及由完整上界和价格计算的 reserved cost 超额仍按原规则 fail closed。 -- 可信本地 CLI 已提供 MMLU-Pro 与 GPQA-Diamond 的固定来源转换、真实 OpenAI-compatible 预检、可恢复执行和完整报告导出;这是 Phase 3 的客观题垂直切片,不代表 Phase 2 或 Phase 3 已完成。 +- 当前 Alembic head 为 `20260830_0008`:它将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,同时替换 Provider 类型 check 与远程配置 check;既有 `mock`/`openai_compatible` 行不改写,存在新协议 Model 时 downgrade 会先拒绝。其上游 data-only `0007` 继续保证没有显式 `input_token_reservation` 时,输入 Token 估算只用于观测,不写成 hard reservation 或参与 cost/overdraw 裁决;Provider actual usage 仍完整保存。 +- 可信本地 CLI 已提供 MMLU-Pro 与 GPQA-Diamond 的固定来源转换、三类显式远程协议预检、可恢复执行和完整报告导出;这是 Phase 3 的客观题垂直切片,不代表 Phase 2 或 Phase 3 已完成。 - 自动化、CI、Compose 故障验收和容量演练的模型执行都只使用 Mock;根据层级使用临时 SQLite 或隔离 PostgreSQL 16/Redis 7,不访问真实模型服务,也不产生模型费用。 - Phase 2 仍为 `in_progress`。P2-01 已完整交付:`P2-local-control-plane-v2` 在 clean commit `b6a35fef1dd069ebb54b69955058915c722aa34d` 从零完成 1 次 warm-up + 5 次 measured trial,23/23 SLO 与逐轮硬门禁全部通过,容量模型为 `qualified`;该实现的 [GitHub Actions run 33146681285](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33146681285) 4/4 成功,证据文档收尾 commit `875f13a253c40b7573d45c6287385e60f2bb8f04` 的 [run 33150080341](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33150080341) 也已 4/4 成功。 - P2-06 状态为 `completed`。实现 SHA [`9a20676dcf545040782f04c166205d0043345753`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/9a20676dcf545040782f04c166205d0043345753) 已 push 到当前分支并进入 [PR #3](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/3),其精确 SHA 的 [GitHub Actions run 33164609388](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33164609388) 4/4 成功。绑定同一 clean SHA 的 Compose acceptance 9/9 通过,evidence 为 `.pytest_cache/artifacts/phase2-acceptance/llmbenchlab-p2-92e173eeee28/evidence.json`(SHA-256 `e4ffb8668fd3fa62d59b5d83f5c29eede35b327d88e6099345acd5950670fc47`),Worker expected/registered/live/stalled/shortfall=`2/2/2/0/0` 且 cleanup C/V/N 全空;clean capacity evidence 为 `.pytest_cache/artifacts/phase2-capacity/llmbenchlab-p2-ca5673061b0f/evidence.json`(SHA-256 `2382f9138f09028f269d76c341b236dd4089d678c8a2323582045fac2b4f5039`),1W/2W/burst QPS=`7.267474/12.962228/9.333604`、wall=`8.255963/4.628834/6.428385s`,最终 18 Runs/270 Responses/270 QuestionExecutions/271 reservations/1230 audit,0 question error/drift/duplicate/PEL/lag,Worker expected=2、shortfall=0,cleanup C/V/N/image 全零且 image counters=`1/1/0/0`。这是 Mock-only 单机观测,不是生产或真实 Provider SLO;此前 dirty evidence 继续保留为历史,不替代该 clean-SHA 结果。Evidence-doc commit [`ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6) 已 push,其精确 SHA 的 [run 33165775037](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33165775037) 4/4 成功,完成 P2-06 仓库级收尾。 -- P2-07 状态为 `planned`,ADR-0016、独立计划和工作日志已建立,但功能尚未实现;后续才开始数据库/keyring 配对 backup/restore、Redis 重建、Worker 扩缩/告警处置和剩余故障矩阵,因此 Phase 2 仍为 `in_progress`。Phase 3 也只交付客观数据垂直切片,其余 Phase 3–6 能力仍未完成。 +- P3-06 Run Detail 热力图/live metrics 状态为 `completed`:backend/frontend target `37/32 passed`,完整 backend `964 passed, 33 skipped`、frontend `64 passed`,lint、Mock smoke、build、Compose config 和目标 198 题 Run 的 desktop/768/375、键盘/Tooltip/console 实页验收通过;12,032/20,000 题边界为自动化虚拟化测试。实现 SHA [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) 已普通 push 到 `codex/complete-evaluation-workflow` 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5);精确 SHA 的 [GitHub Actions run 33289522923](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) 四个必需 job 全部成功。 +- P2-07 状态为 `planned`,ADR-0016、独立计划和工作日志已建立,但功能尚未实现;后续才开始数据库/keyring 配对 backup/restore、Redis 重建、Worker 扩缩/告警处置和剩余故障矩阵,因此 Phase 2 仍为 `in_progress`。Phase 3 已有客观数据与已完成的 P3-06 UI 切片,但其余 Phase 3–6 能力仍未完成。 最新、可复核的完成状态与测试证据以 [`docs/PROJECT_STATUS.md`](docs/PROJECT_STATUS.md) 和 [`docs/worklogs/`](docs/worklogs/) 为准;Roadmap 中的计划能力不等于已交付能力。 ## 核心特性 - **完全离线的 Mock Demo**:15 道原创双语演示题,覆盖 exact match、multiple choice 和 numeric;结果必须明确标记为 Demo,不能当作正式模型能力结论。 -- **模型注册表**:支持 `mock` 与 `openai_compatible`,记录远端模型名、默认生成参数和可选价格信息。 +- **模型注册表**:支持 `mock`、Chat Completions、OpenAI Responses 与 Anthropic Messages 四种显式 Adapter 类型,记录远端模型名、默认生成参数和可选价格信息。 - **版本化 Benchmark**:严格校验 `manifest.json` 与 `questions.jsonl`,支持受限 ZIP 导入、稳定 SHA-256 和导入冲突检测。 - **固定标准数据集供应链**:MMLU-Pro test/validation 与 GPQA-Diamond 使用固定 revision、源文件 SHA-256、转换器版本和确定性 profile/选项重排;第三方题目只落在 Git 忽略的本地 `artifacts/`。 - **真实模型本地入口**:`llmbenchlab-evaluate prepare/run/resume/report` 完成下载、模型发现、付费 canary、显式确认、有界执行、缺失题恢复和全量证据导出;Key 只来自环境变量或隐藏输入,远端 Provider 必须使用 HTTPS,明文 HTTP 仅允许 loopback。 - **确定性评分**:内置三类 Evaluator;解析失败和单题调用失败严格计 0 分,并保留错误证据。 -- **可解释指标**:严格总分 `score`、完成率 `completion_rate` 和已回答准确率 `answered_accuracy` 分开呈现,避免把缺失回答隐藏在成功样本中。 -- **可靠任务执行基础**:API 先提交 Run,再 best-effort 发送 Redis Streams 通知;独立 Worker 以数据库时间、租约和 fencing token 领取任务,并通过数据库扫描从通知丢失或进程故障中恢复;大快照加载移出事件循环,已领取 Run 在物化题目时仍可续租。 +- **可解释指标与动态进度**:严格总分 `score`、完成率 `completion_rate` 和已回答准确率 `answered_accuracy` 分开呈现,避免把缺失回答隐藏在成功样本中;Run Detail 以固定 512 题轻量 block 呈现通过、普通答错、执行异常、未执行四态热力图,并从后端同快照证据实时刷新主指标,避免把 `error_questions` 误读为全部错题。 +- **可靠任务执行基础**:API 先提交 Run,再 best-effort 发送 Redis Streams 通知;独立 Worker 以数据库时间、租约和 fencing token 领取任务,并通过数据库扫描从通知丢失或进程故障中恢复;标准 Compose PostgreSQL 入口默认提供两个 Worker,使不同 Run 可占用不同执行槽;大快照加载移出事件循环,已领取 Run 在物化题目时仍可续租。 - **幂等与恢复**:同一 Run/Question 只有一条计分证据;租约心跳、有限 attempt、退避、取消、过期接管和 dead-letter 都由数据库裁决,Redis 不是状态数据库。 - **数据库权威治理**:Web/API admission 把版本化完整 policy 冻结进 Run;global/provider/model/run 四层并发、RPM/TPM 和累计预算在固定锁序中共同裁决,backlog 满时在提交前稳定拒绝,Token/cost hard limit 缺少显式上界或价格时 fail closed。非显式输入估算不会冒充 hard reservation;actual usage 仍保留,只有实际用量超过显式预留才触发对应 overdraw。 - **逐 Provider attempt 账本与公平调度**:每次 HTTP attempt 先 reserve、再持久化 `send_started`、最后 actual/conservative settlement;可证明未发送的 release 保留终态 ledger,另起 generation 并重试当前未发送 ordinal,不重置之前已发送的 HTTP retry。Worker 每个 lease 只新增有界 question quantum,按最久未获服务顺序 cooperative yield,不把让出误计为失败。 - **可审计观测与受控保留**:typed、应用 append-only audit 以稳定 event key 去重;`/tasks/history` 在同一读取快照中校验 retained audit 后给出 counters 与 Run latency,`/metrics/prometheus` 用固定 gauge/enum label、硬样本上限和进程内 single-flight 暴露同源快照。Worker generation 只在真实 scan/claim/lease-heartbeat/progress 后按数据库 UTC 合并刷新,dependency probe 仍明确不检查主循环。八条 Prometheus 规则附固定 Runbook;`llmbenchlab-audit-retention` 提供 canonical JSONL archive、离线 verify、reconcile、精确 restore/delete,默认不删除且不把普通 hash 冒充 WORM。Run created/finished、credential audit 和逐题 Provider 元数据继续遵守非秘密边界。 - **可复现记录**:持久化模型参数、Prompt、Benchmark Hash、协议版本、代码 commit(可用时)、raw response、parsed answer、参考答案快照和逐题评分。 -- **七个前端页面**:Dashboard、Models、Benchmarks、Evaluation Runs、New Run、Run Detail 和 Leaderboard;评测记录页可找回全部状态的 Run,详情证据按 100 条分页。Run Detail 会明确区分 `managed`、`delayed`、`exhausted` 和 `legacy_unmanaged`,对可公开的稳定 reason 给出中文说明并以 UTC 显示最早重调度时间;未知 reason 不原样反射。 +- **七个前端页面**:Dashboard、Models、Benchmarks、Evaluation Runs、New Run、Run Detail 和 Leaderboard;评测记录页可找回全部状态的 Run,详情证据按 100 条分页。Run Detail 会明确区分 `managed`、`delayed`、`exhausted` 和 `legacy_unmanaged`,对可公开的稳定 reason 给出中文说明并以 UTC 显示最早重调度时间;未知 reason 不原样反射。热力图每秒只读取小型 block index 和变化 block,不下载全量题目/回答正文;精确 Run Token 因部分 usage 缺失而未知时,页面仍显示全量 Response 的已知小计、输入/输出覆盖率与“完整总量未知”,不会把部分证据冒充账单真值。 - **Web 只写凭据**:用户可在 Models 表单直接粘贴 API Key;API 不把凭据流中的原值复制到公开 Model/Run-model 字段,数据库只保存由独立 keyring 加密的 AES-GCM 密文。旧 `api_key_env` 模型仍兼容;Provider 返回证据会递归检查对象键/JSON 标量,当前 Key 的精确回显会在进入 Runner/持久化前替换为 `[REDACTED]`。这不是对无关 Benchmark/Question 内容的全局字面扫描。 - **开发交付完整**:Alembic、Ruff、pytest、ESLint、TypeScript、Vitest、Vite production build、GitHub Actions、Makefile,以及 PostgreSQL、Redis、API、Worker、frontend 和一次性 migrate 组成的 Docker Compose。 @@ -74,11 +75,11 @@ flowchart LR Runner --> Adapters[Adapter Registry] Runner --> Evaluators[Evaluator Registry] Adapters --> Mock[Mock / 无网络] - Adapters -->|仅用户主动配置| Provider[OpenAI-compatible API] + Adapters -->|仅用户主动配置| Provider[Chat / Responses / Messages API] WorkerEnv[Worker 环境变量 / 旧配置] -.->|兼容读取| Adapters ``` -API 创建 managed Run 时先在数据库锁内检查 backlog、冻结 active policy 与 Run override,再提交数据库、best-effort 发送 Redis 通知并返回 `202`;通知失败不回滚数据库事实。Worker 优先从数据库对账,并可消费重复 Redis 消息;每次写入都校验当前租约 owner/token,每个 Provider HTTP attempt 由数据库 ledger 单独 admission/结算,恢复时跳过已有 Response。数据库因此是唯一事实来源,Redis、日志、指标和 Worker 内存都不能覆盖 Run 状态。前端轮询 Run,进入 `completed`、`failed` 或 `cancelled` 终态后停止。详细设计见 [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md),治理语义见 [`ADR-0009`](docs/decisions/ADR-0009-database-governance-audit-fair-scheduling.md)、交付边界修正 [`ADR-0010`](docs/decisions/ADR-0010-phase-2-governance-delivery-boundaries.md)、pre-send retry generation 修正 [`ADR-0011`](docs/decisions/ADR-0011-confirmed-pre-send-release-retry-generation.md) 和 observational reservation 修正 [`ADR-0018`](docs/decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md),评分语义见 [`docs/BENCHMARK_PROTOCOL.md`](docs/BENCHMARK_PROTOCOL.md)。 +API 创建 managed Run 时先在数据库锁内检查 backlog、冻结 active policy 与 Run override,再提交数据库、best-effort 发送 Redis 通知并返回 `202`;通知失败不回滚数据库事实。Worker 优先从数据库对账,并可消费重复 Redis 消息;每次写入都校验当前租约 owner/token,每个 Provider HTTP attempt 由数据库 ledger 单独 admission/结算,恢复时跳过已有 Response。数据库因此是唯一事实来源,Redis、日志、指标和 Worker 内存都不能覆盖 Run 状态。前端轮询 Run,进入 `completed`、`failed` 或 `cancelled` 终态后停止。详细设计见 [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md),治理语义见 [`ADR-0009`](docs/decisions/ADR-0009-database-governance-audit-fair-scheduling.md)、交付边界修正 [`ADR-0010`](docs/decisions/ADR-0010-phase-2-governance-delivery-boundaries.md)、pre-send retry generation 修正 [`ADR-0011`](docs/decisions/ADR-0011-confirmed-pre-send-release-retry-generation.md)、observational reservation 修正 [`ADR-0018`](docs/decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 和显式 Provider 协议选择 [`ADR-0019`](docs/decisions/ADR-0019-explicit-provider-api-protocol-adapters.md),评分语义见 [`docs/BENCHMARK_PROTOCOL.md`](docs/BENCHMARK_PROTOCOL.md)。 任务投递是 at-least-once,本地 Response、ledger 状态转换和聚合是幂等的;这不等于 Provider exactly-once。若 Worker 在 `send_started` 后崩溃,本地会保守结算并最终释放 admission permit,但远端幽灵请求可能仍在运行;若 Provider 已响应而本地 Response 尚未提交,接管 Worker 还可能再次调用并产生额外费用。本地 consumed 数是保守预算证据,不是 Provider 账单真值。 @@ -89,7 +90,7 @@ LLMBenchLab/ ├── backend/ │ ├── alembic/ # 数据库迁移 │ ├── app/ -│ │ ├── adapters/ # Mock / OpenAI-compatible +│ │ ├── adapters/ # Mock / Chat / Responses / Messages │ │ ├── api/v1/ # REST 路由 │ │ ├── cli/ # 可信本地正式评测入口 │ │ ├── core/ # 配置、日志、常量、时间 @@ -97,7 +98,7 @@ LLMBenchLab/ │ │ ├── evaluators/ # 三类确定性评分器 │ │ ├── governance/ # Policy、四层 admission、attempt ledger 与审计 │ │ ├── models/ # SQLAlchemy 实体 -│ │ ├── providers/ # 模型发现与最小 Chat canary +│ │ ├── providers/ # 模型发现与显式协议 canary │ │ ├── reports/ # 完整 Run 报告导出 │ │ ├── runners/ # 租约仓储与评测 Runner │ │ ├── schemas/ # Pydantic API Schema @@ -136,9 +137,18 @@ make setup make dev ``` -`make setup` 会让 `uv` 显式选择 CPython,按锁文件安装前后端依赖、仅在 `.env` 不存在时从 `.env.example` 创建它,并执行 Alembic migration;已有 `.env` 不会被覆盖。该命令可重复执行。若检测到由早期开发版自动建表留下的未版本化 SQLite,只有在结构与完整性严格匹配已知版本时才会先创建同目录 `.bak` 一致性备份并无损收养;未知或部分结构会在写入版本标记前停止。普通 API/Worker 启动不会隐式建表,未迁移时会提示先运行 `make setup` 或 `make migrate`。`make dev` 在一个终端启动 API、独立 Worker 和 frontend,控制台只显示地址与日志位置;三个服务的详细输出分别追加到 Git 忽略的 `artifacts/dev-logs/api.log`、`worker.log` 和 `frontend.log`,`Ctrl-C` 会一起停止。 +`make setup` 会让 `uv` 显式选择 CPython,按锁文件安装前后端依赖、仅在 `.env` 不存在时从 `.env.example` 创建它,并执行 Alembic migration;已有 `.env` 不会被覆盖。该命令可重复执行。若检测到由早期开发版自动建表留下的未版本化 SQLite,只有在结构与完整性严格匹配已知版本时才会先创建同目录 `.bak` 一致性备份并无损收养;未知或部分结构会在写入版本标记前停止。普通 API/Worker 启动不会隐式建表,未迁移时会提示先运行 `make setup` 或 `make migrate`。`make dev` 在一个终端启动 API、独立 Worker 和 frontend,控制台只显示地址与日志位置;单 Worker 日志保持为 `worker.log`,PostgreSQL 下使用 `make dev DEV_WORKERS=2` 时改为私有的 `worker-1.log`、`worker-2.log`。`Ctrl-C` 或任一子进程退出会停止同一开发会话中的全部服务。 -默认地址: +需要并行执行多个 Benchmark Run 时,使用已支持多 Worker 租约的 PostgreSQL 模式。最简单的入口默认启动两个 Worker,并在返回前校验 `expected/registered/live/stalled/shortfall=2/2/2/0/0`: + +```bash +make dev-multi +# 或显式设置:make dev-multi WORKERS=2 +``` + +该模式的 Web 地址为 `http://127.0.0.1:8080`。本地 `make dev` 仍默认使用 SQLite 单 Worker;若它连接的是 PostgreSQL,也可用 `DEV_WORKERS=2` 启动两个本地 Worker 进程。请求多个 Worker而数据库不是 PostgreSQL 时,启动器会在创建日志或启动服务前拒绝。现有 SQLite 数据不会自动复制到 Compose PostgreSQL;需要保留它时必须在活动 Run、reservation 和 Worker 全部收敛并停写后,使用文档化的 SQLite→空 PostgreSQL importer。 + +普通 `make dev` 的默认地址: - Web:`http://127.0.0.1:5173` - API:`http://127.0.0.1:8000` @@ -159,7 +169,7 @@ curl -sS http://127.0.0.1:8000/api/v1/metrics/prometheus API 为每个请求自行生成 `X-Request-ID` 并在响应中返回,不信任或回显客户端提供的同名 header。LLMBenchLab 生产日志调用只允许无格式参数的字面量消息;结构化 extra 除字段白名单外还逐字段执行固定枚举、UUID/Redis stream ID 与有限数值规范化,非法 ID 被省略,未知 method/code 只输出固定 `unsupported`。Redis Run 通知本身也只接受 canonical UUID。外部 logger 的动态消息不进入 JSON,原始 Uvicorn access handler 关闭。凭据和敏感内容仍绝不得放在 URL、header、请求路径或日志字段中。 -需要跟踪组合启动日志时可运行 `tail -f artifacts/dev-logs/api.log artifacts/dev-logs/worker.log artifacts/dev-logs/frontend.log`;需要在前台分别观察时,可在三个终端运行 `make backend`、`make worker` 和 `make frontend`。只启动 API 时,新 Run 会持久化为 `pending`,但不会在 API 进程内执行。本地 SQLite 只支持一个 Worker;Redis URL 可留空,Worker 将使用数据库对账。所有命令可通过 `make help` 查看;更完整的环境变量、迁移和排障说明见 [`docs/DEPLOYMENT.md`](docs/DEPLOYMENT.md)。 +需要跟踪组合启动日志时可运行 `tail -f artifacts/dev-logs/api.log artifacts/dev-logs/worker*.log artifacts/dev-logs/frontend.log`;需要在前台分别观察时,可在三个终端运行 `make backend`、`make worker` 和 `make frontend`。只启动 API 时,新 Run 会持久化为 `pending`,但不会在 API 进程内执行。本地 SQLite 只支持一个 Worker;Redis URL 可留空,Worker 将使用数据库对账。所有命令可通过 `make help` 查看;更完整的环境变量、迁移和排障说明见 [`docs/DEPLOYMENT.md`](docs/DEPLOYMENT.md)。 ## Mock Demo:完整离线流程 @@ -169,8 +179,8 @@ API 为每个请求自行生成 `X-Request-ID` 并在响应中返回,不信任 2. 进入 **模型** 页面,新建模型;名称可填 `Offline Mock`,Provider 选择 `mock`,保持启用。Mock 不需要 Base URL、远端模型名或 API Key。 3. 进入 **评测集** 页面,点击重载/载入内置 Demo。确认它显示 `demo-general`、版本 `1.0.0`、15 道题,以及“Demo 数据,不代表正式模型能力”的提示。 4. 点击 **新建评测**,选择刚注册的 Mock 和 Demo Benchmark。默认参数可直接使用;推荐可复现基线为 `temperature=0`、`top_p=1`、`max_tokens=256`、`seed=42`、`concurrency=1`。 -5. 提交后进入 Run Detail。页面会轮询 `pending/running` 状态,展示进度、配置快照和逐题结果,并在终态停止轮询;超过 100 条证据时使用页尾按钮翻页。 -6. 确定性 Mock Demo 正常应完成 15/15,严格总分、完成率和已回答准确率均为 100;逐题区域会分别显示 raw response、parsed answer、reference、score 和 error。 +5. 提交后进入 Run Detail。页面会轮询 `pending/running` 状态,以绿/红/黑/白热力格显示通过、普通答错、执行异常和未执行,并动态刷新严格总分、完成率、准确率、延迟及 usage/cost 已知覆盖;初始 block 尚未追齐时会明确显示“同步中”。 +6. 确定性 Mock Demo 正常应完成 15/15,严格总分、完成率和已回答准确率均为 100;悬停、键盘聚焦或移动端点按热力格可查看该题 score、Token、延迟、成本/错误类型。逐题证据区继续分别显示 raw response、parsed answer、reference、score 和 error,超过 100 条时使用页尾按钮翻页。 7. 离开详情后可从主导航 **评测记录** 找回等待中、运行中、已完成、失败或已取消的 Run;列表支持状态筛选、20 条分页、手动刷新,并在当前页存在活动 Run 时自动更新。 8. 进入 **排行榜**,按模型或 Benchmark 筛选,核对协议版本、数据集 Hash、完成率和醒目的 Demo 标识。该成绩只证明本地垂直链路可工作。 @@ -231,9 +241,9 @@ uv run llmbenchlab-evaluate run \ --concurrency 1 ``` -`base_url` 可以是兼容根地址(如 `https://host/v1`,实际 POST 到 `/v1/chat/completions`),也可以直接是以 `/chat/completions` 结尾的完整端点。远端主机只接受 HTTPS;`http://localhost`、`http://127.0.0.1` 或 `http://[::1]` 仅用于本机推理服务。CLI 默认先请求同一根路径的 `GET /models`:只发现一个模型时可省略 `--model`;多个模型时必须显式选择。若 Provider 不实现 `/models`,只有已经给出 `--model` 时才会继续;也可显式使用 `--no-model-discovery --model ...`。发现结果中任何模型 ID 若反射当前 Key,预检会直接失败且不会把该值写入诊断信息。 +`base_url` 可以是协议根地址,也可以直接是匹配的完整 endpoint。CLI 的 `--provider-type` 可选 `openai_compatible`(默认 Chat Completions)、`openai_responses` 或 `anthropic_messages`;它分别请求 `/chat/completions`、`/responses`、`/messages`,不会根据模型名猜测。远端主机只接受 HTTPS;`http://localhost`、`http://127.0.0.1` 或 `http://[::1]` 仅用于本机推理服务。CLI 默认先请求同一根路径的 `GET /models`:只发现一个模型时可省略 `--model`;多个模型时必须显式选择。若 Provider 不实现 `/models`,只有已经给出 `--model` 时才会继续;也可显式使用 `--no-model-discovery --model ...`。发现结果中任何模型 ID 若反射当前 Key,预检会直接失败且不会把该值写入诊断信息。 -在创建 Run 前,CLI 会打印目标 host、模型、题数、剩余 Run attempts 和 Chat Completion HTTP 尝试次数上界,等待输入 `RUN`,然后发送一个可能计费的最小 canary。canary 必须可解析为预期答案;若成功体明确返回的模型名不同于请求目标,也会失败。当前上界按 `(计分题数 × 剩余 Run attempts + 1 个 canary) × 3 次 HTTP attempts` 保守计算;自动化脚本只有显式传入 `--yes` 才能越过交互确认。该直连 CLI 当前创建 `legacy_unmanaged` Run,不经过 Web/API governance admission;这不是 Token、RPM/TPM 或金额预算上限。建议确认少量题结果后再运行全量: +在创建 Run 前,CLI 会打印目标 host、显式协议、模型、题数、剩余 failed-attempt 预算和 Provider HTTP 尝试次数上界,等待输入 `RUN`,然后发送一个可能计费的最小 canary。剩余预算严格为 `max_attempts - failed_attempt_count`,不把 cooperative yield 算作失败;canary 必须可解析为预期答案,且成功体明确返回的模型名不同于请求目标时失败。当前上界按 `(计分题数 × 剩余 failed-attempt 预算 + 1 个 canary) × 3 次 HTTP attempts` 保守计算;自动化脚本只有显式传入 `--yes` 才能越过交互确认。该直连 CLI 当前创建 `legacy_unmanaged` Run,不经过 Web/API governance admission;这不是 Token、RPM/TPM 或金额预算上限。建议确认少量题结果后再运行全量: ```bash cd backend @@ -266,17 +276,25 @@ uv run llmbenchlab-evaluate report \ 输出目录必须尚不存在。每份报告包含 `summary.json`、`groups.csv` 和覆盖全部已持久化 Response 的 `responses.jsonl`;全局与分组指标统一从计划题和这些 Response 证据派生,`metrics_provenance` 会标出数据库 Run 汇总字段是否发生漂移。默认输出、下载缓存和转换 ZIP 都在 Git 忽略的 `artifacts/`。完整来源、profile、Hash 与比较规则见 [`docs/DATASET_FORMAT.md`](docs/DATASET_FORMAT.md) 和 [`docs/BENCHMARK_PROTOCOL.md`](docs/BENCHMARK_PROTOCOL.md),操作与安全边界见 [`docs/DEPLOYMENT.md`](docs/DEPLOYMENT.md) 和 [`docs/SECURITY.md`](docs/SECURITY.md)。 -## 接入 OpenAI-compatible Provider +## 接入远程 Provider + +本节描述通过 Web/API 加常驻 Worker 的服务路径;一次性正式评测优先使用上一节的可信本地 CLI。远程协议必须显式选择,不按模型名或失败结果自动猜测,也不会在一次失败后切换 endpoint: + +| Provider 类型 | 请求 endpoint | 主要参数边界 | +| --- | --- | --- | +| `openai_compatible` | `/chat/completions` | 默认 `temperature=0/top_p=1/seed=42`;`max_tokens` 可由 Provider 决定 | +| `openai_responses` | `/responses` | 未显式配置时省略 `temperature/top_p/seed`;输出预算映射为 `max_output_tokens` | +| `anthropic_messages` | `/messages` | 未显式配置时省略 `temperature/top_p/seed`;`temperature<=1`,且必须提供有限 `max_tokens` | -本节描述通过 Web/API 加常驻 Worker 的服务路径;一次性正式评测优先使用上一节的可信本地 CLI。LLMBenchLab 使用 Chat Completions 风格接口。若 `base_url` 为 `https://provider.example/v1`,Adapter 会请求 `https://provider.example/v1/chat/completions`;如果填写的 URL 已以 `/chat/completions` 结尾,则不会重复追加。 +`base_url` 可以填写协议根地址(例如 `https://provider.example/v1`),也可以填写与所选类型一致的完整 endpoint。填写其他已知协议的完整后缀会在发起网络请求前被拒绝,避免构造 `/responses/chat/completions` 一类错误地址。OpenCode Go 当前文档中的模型也必须按其所列 endpoint 选择对应类型。模型发现同样按该显式协议鉴权:Chat/Responses 使用 `Authorization: Bearer`,Messages 使用 `x-api-key` 与 `anthropic-version`;Messages 的 `has_more/last_id` 通过 `after_id` 分页并受累计 100 页、60 秒 wall-clock、2 MiB、10,000 项与重复 cursor 门禁保护。 1. 运行 `make setup && make dev`,打开 `http://127.0.0.1:5173`,进入 **模型**,点击新建模型。 -2. Provider 选择 `openai_compatible`,填写 API Base URL、远端模型名,并把真实 Key 直接粘贴到 **API Key** 密码框;这里不再填写环境变量名称。 +2. Provider 选择 **Chat Completions**、**OpenAI Responses** 或 **Anthropic Messages**,填写 API Base URL、远端模型名,并把真实 Key 直接粘贴到 **API Key** 密码框;这里不再填写环境变量名称。 3. 保存后密码框立即清空,卡片只显示“已安全保存”,GET/list/编辑表单都不会回填原 Key。进入 **新建评测** 选择模型和 Benchmark 后,独立 Worker 才解密并调用 Provider。 4. 在 **新建评测** 检查页面给出的输出预算和单次读取超时建议。Demo 默认建议 `256 / 60s`,MMLU-Pro Direct 为 `1024 / 180s`,MMLU-Pro official CoT 为 `4000 / 300s`,GPQA-Diamond 为 `8192 / 600s`;未知正式集使用保守起点 `4096 / 300s`。建议值可调整,也不代表 Provider 一定支持相同上限。 -5. 创建后可离开详情页;主导航 **评测记录** 会列出全部状态并重新进入详情。逐题证据每页 100 条,上一页/下一页不会丢失全量计数。 +5. 创建后可离开详情页;主导航 **评测记录** 会列出全部状态并重新进入详情。Run Detail 的全题热力图使用 absolute-position block 独立同步,逐题正文证据仍每页 100 条;上一页/下一页不会改变全 Run 动态指标或热力图计数。 -Web 的数字 `max_tokens` 允许 `1..131072`;选择“由 Provider 决定”会保存 `null` 并在 Chat Completions 请求中省略 `max_tokens`,含义是采用 Provider 自身默认值,**不是无限输出**。未显式提供该字段的通用 API 和 `llmbenchlab-protocol-v1` 兼容路径仍默认 `256`,Benchmark 建议只影响 Web 表单起点。OpenAI-compatible Chat 请求会发送 `stream:true` 与 `stream_options.include_usage:true`,持续消费 SSE token/心跳;看到 finish 后不会提前结束,若 Provider 发送 usage-only 尾块也会继续读取,直到 `[DONE]` 才完成本题。usage 缺失时 Token 统计保持未知;忽略流式参数而返回普通 JSON 的 Provider 仍兼容。 +Web 的数字 `max_tokens` 允许 `1..131072`;Chat Completions 与 Responses 可选择“由 Provider 决定”,保存的 `null` 只表示省略相应输出预算字段,**不是无限输出**。Anthropic Messages 要求有限 `max_tokens`,页面会禁用该选项。未显式提供该字段的通用 API 和 `llmbenchlab-protocol-v1` 兼容路径仍默认 `256`,Benchmark 建议只影响 Web 表单起点。Responses/Messages 若请求和 Model 默认都没有显式采样值,会把 `temperature/top_p/seed` 冻结为 `null` 并从 Provider payload 省略;非空 seed 会在外发前拒绝。Chat 请求以 `[DONE]`、Responses 以 `response.completed`、Messages 以 `message_stop` 作为流式成功终止证据;缺少相应终止事件不会保存部分答案。三类 Adapter 都兼容各自协议的普通 JSON 成功响应,usage 缺失时 Token 统计保持未知。Responses 的 rate-limit/server typed error,以及 Messages 的 `rate_limit_error`、`api_error`、`overloaded_error`、`timeout_error` 和 HTTP `529` 才进入新增协议的有限 retry 白名单;未知流内错误 fail closed,每次重试独立结算 attempt ledger。 `read_timeout_seconds` 允许 `1..1800` 秒并随 Run 的 `execution.timeouts_seconds.read` 快照保存。它是等待下一批响应字节的**空闲读取上限**,不是整个生成的总墙钟上限;只要 token 或 SSE comment 持续到达,总生成时间可以超过这个数值。Provider 以 `finish_reason="length"` 截断空内容或在截断后无法解析最终答案时,逐题证据会标记 `output_truncated`,提示提高输出预算或改由 Provider 决定。 @@ -286,8 +304,8 @@ Web 的数字 `max_tokens` 允许 `1..131072`;选择“由 Provider 决定” ```json { - "name": "My Compatible Model", - "provider_type": "openai_compatible", + "name": "My Responses Model", + "provider_type": "openai_responses", "base_url": "https://provider.example.invalid/v1", "remote_model_name": "replace-with-provider-model-name", "api_key": "", @@ -296,7 +314,7 @@ Web 的数字 `max_tokens` 允许 `1..131072`;选择“由 Provider 决定” } ``` -真实 Key 会且只会在创建/替换模型时进入本机 API 请求体,随后以 AES-256-GCM 密文落库;它不应进入 Git、Issue、日志、截图、URL、命令行或 `VITE_*` 变量。浏览器不会直接调用 Provider。Model Schema 会拒绝 `base_url` query,拒绝远端明文 HTTP(仅 loopback 可用 HTTP),并将 `default_parameters` 限定为 `temperature`、`top_p`、`max_tokens`、`seed` 四个严格校验的生成字段;其中 `max_tokens=null` 同样只表示请求时省略该字段。当前 MVP 尚无 SSRF allowlist;managed Run 虽支持数据库权威 RPM/TPM 与累计费用 hard limit,但默认 policy 可关闭限制,且 hard Token/cost 要求显式 input reservation、有限 `max_tokens` 和价格,否则在外发前 fail closed。操作者仍须审查 Provider 地址、题目外发许可、数据政策并独立核对真实账单。 +真实 Key 会且只会在创建/替换模型时进入本机 API 请求体,随后以 AES-256-GCM 密文落库;它不应进入 Git、Issue、日志、截图、URL、命令行或 `VITE_*` 变量。浏览器不会直接调用 Provider。Model Schema 会拒绝 `base_url` query,拒绝远端明文 HTTP(仅 loopback 可用 HTTP),并将 `default_parameters` 限定为 `temperature`、`top_p`、`max_tokens`、`seed` 四个严格校验的生成字段;Responses/Messages 的非空 `seed`、Messages 的 `temperature>1` 和 `max_tokens=null` 会在外发前稳定拒绝。当前 MVP 尚无 SSRF allowlist;managed Run 虽支持数据库权威 RPM/TPM 与累计费用 hard limit,但默认 policy 可关闭限制,且 hard Token/cost 要求显式 input reservation、有限 `max_tokens` 和价格,否则在外发前 fail closed。操作者仍须审查 Provider 地址、题目外发许可、数据政策并独立核对真实账单。 治理 API 只面向可信 loopback:首次 Run/policy apply 前,`GET /api/v1/governance/policy` 不产生隐式写入并返回 `404 governance_policy_not_initialized`;`PUT` 必须提交全部 policy 字段,原子激活一个不可变、内容寻址的版本,不能当作局部 PATCH。Run 创建会在没有 policy 时引导默认版本并冻结其 ID/hash;之后修改 policy 不会追溯改变已提交 Run。完整字段和错误码见 [`docs/API.md`](docs/API.md),限流、预算、backlog、settlement 与恢复操作见 [`docs/OPERATIONS.md`](docs/OPERATIONS.md);当前 Mock 容量基线见 [`docs/PERFORMANCE.md`](docs/PERFORMANCE.md)。 @@ -324,10 +342,11 @@ npm run build ## Docker Compose -Docker 模式包含六个 service:长运行的 `postgres`、`redis`、`api`、`worker`、`frontend`,以及一次性 `migrate`。`migrate` 是 Compose 中唯一执行 Alembic 升级的服务;API/Worker 只在启动时检查 schema 已在 head。PostgreSQL 和开启 AOF 的 Redis 分别使用 named volume: +Docker 模式包含六类 service:长运行的 `postgres`、`redis`、`api`、可横向复制的 `worker`、`frontend`,以及一次性 `migrate`。`migrate` 是 Compose 中唯一执行 Alembic 升级的服务;API/Worker 只在启动时检查 schema 已在 head。PostgreSQL 和开启 AOF 的 Redis 分别使用 named volume。标准入口默认启动两个 Worker;显式设置只能改变部署规模,不会自动提高 governance limit: ```bash make docker-up +make docker-up WORKERS=2 ``` 默认地址: @@ -374,7 +393,7 @@ uv run python -m app.db.import_sqlite \ | `3` | `committed_but_verification_failed` | 目标已提交完整 precommit 快照,但 postcommit 验证或报告失败;停止服务并独立对账 | | `4` | `commit_outcome_unknown` | PostgreSQL 未确认 COMMIT;原子事务意味着目标可能为空,也可能已完整提交 | -退出码 `3` 或 `4` 后**禁止盲目重试**。应先隔离目标,检查 Alembic head、13 表行数/主键集/canonical hash 和工具已输出的对账证据;非空目标会拒绝再次导入。工具不提供 PostgreSQL → SQLite 反向同步,回滚依赖保留的 SQLite 源/备份或单独验证的导出流程。当前 head `20260830_0007` 降到 `0006` 会按旧谓词重算 `governance_scopes.overdrawn`,不删除 ledger 或 actual usage;`0006 → 0005` 不删除索引对象。继续跨过 `20260828_0005` 时,只要 `worker_processes` 有事实就会拒绝。先停止 Worker、保存必要事实并显式清空该表后,才能进入 `0005 → 0004`,而 `0004` 原有 ledger/audit downgrade guard 仍继续生效。只有隔离空库用于完整降级/升级往返,处理见 [`docs/OPERATIONS.md`](docs/OPERATIONS.md)。 +退出码 `3` 或 `4` 后**禁止盲目重试**。应先隔离目标,检查 Alembic head、13 表行数/主键集/canonical hash 和工具已输出的对账证据;非空目标会拒绝再次导入。工具不提供 PostgreSQL → SQLite 反向同步,回滚依赖保留的 SQLite 源/备份或单独验证的导出流程。当前 head `20260830_0008` 降到 `0007` 会把 `models.provider_type` 从 `VARCHAR(18)` 收回 `VARCHAR(17)`,并恢复旧 Provider 类型 check 与远程配置 check;存在 `openai_responses` 或 `anthropic_messages` Model 时会在 DDL 前拒绝。`0007 → 0006` 按旧谓词重算 `governance_scopes.overdrawn`,不删除 ledger 或 actual usage;`0006 → 0005` 不删除索引对象。继续跨过 `20260828_0005` 时,只要 `worker_processes` 有事实就会拒绝。先停止 Worker、保存必要事实并显式清空该表后,才能进入 `0005 → 0004`,而 `0004` 原有 ledger/audit downgrade guard 仍继续生效。只有隔离空库用于完整降级/升级往返,处理见 [`docs/OPERATIONS.md`](docs/OPERATIONS.md)。 ## Audit retention 维护 @@ -398,7 +417,7 @@ uv run llmbenchlab-audit-retention delete \ | Phase 0 | 项目治理、需求、架构、协议 | 已完成 | | Phase 1 | FastAPI + React + SQLite 的 MVP 垂直链路 | 已完成 | | Phase 2 | PostgreSQL、Redis、独立 Worker、治理、恢复与可观测性 | `in_progress`:P2-01 已闭环;P2-06 实现与 evidence-doc 精确 SHA CI 均全绿,状态为 `completed`;P2-07 工作包已建立,状态为 `planned`,功能尚未实现 | -| Phase 3 | 合规标准 Benchmark 与隔离代码评测 | MMLU-Pro/GPQA 客观数据切片已交付;IFEval、沙箱及其余验收未完成 | +| Phase 3 | 合规标准 Benchmark 与隔离代码评测 | `in_progress`:MMLU-Pro/GPQA 客观数据与 P3-06 热力图/live metrics 切片已交付;IFEval、沙箱及其余验收未完成 | | Phase 4 | LLM Judge、人工校准与 Arena | 计划中 | | Phase 5 | Agent、工具调用与 Live Benchmark | 计划中 | | Phase 6 | 公共发布、多用户、安全与运营加固 | 计划中 | @@ -413,19 +432,19 @@ uv run llmbenchlab-audit-retention delete \ - 服务仅限可信 loopback 本机;Host allowlist 与 Nginx 请求流式转发不能替代公网认证、授权或 KMS。 - `VITE_*` 会进入浏览器构建产物,永远不能用于存放秘密。 - Benchmark ZIP 会做路径、文件类型、大小、压缩比、Schema 和题数校验,但导入者仍需审查来源、许可证、敏感数据和提示注入风险。 -- 任意 OpenAI-compatible `base_url` 存在 SSRF 与数据外发风险;公开部署前必须加入地址策略、出站隔离、鉴权和费用控制。 +- 任意远程 Provider `base_url` 都存在 SSRF 与数据外发风险;公开部署前必须加入地址策略、出站隔离、鉴权和费用控制。 - SQLite/PostgreSQL 会保存题目、参考答案、原始回答和错误证据;数据库、volume、导入源与备份需使用最小权限和加密存储保护。 威胁模型、秘密轮换和公开部署前门槛见 [`docs/SECURITY.md`](docs/SECURITY.md)。 ## 当前限制 -- SQLite 兼容路径只支持一个 Worker 和低并发;多 Worker 的真实租约协调路径以 PostgreSQL 为目标。 +- SQLite 兼容路径只支持一个 Worker 和低并发;多 Worker 的真实租约协调路径以 PostgreSQL 为目标,普通 Compose 入口默认 2 个 Worker。当前容量资格只覆盖 1–2 个 Worker,3 个以上必须重新测量。 - API 重启不拥有或改写 Run;Worker 异常退出后由数据库租约过期和对账恢复。这是受限的可靠基础,不是 HA/SLA 保证。 - 取消是协作式的;已经发出的上游请求可能要等到返回或超时。 - at-least-once 恢复不保证 Provider 调用或计费 exactly-once;数据库只保留一份幂等的 Response/费用证据。 - managed Web/API Run 已有可配置的本地 admission/预算/背压,但默认限制可关闭;它不覆盖 `legacy_unmanaged` 直连 CLI,也不提供端到端账单保证、Provider 幽灵请求终止或远端 exactly-once。 -- OpenAI-compatible 只实现 Chat Completions 共同子集,不保证覆盖各供应商私有参数和响应扩展。 +- 远程 Adapter 只实现 Chat Completions、OpenAI Responses 与 Anthropic Messages 的文本生成共同子集,不支持 tools、多模态,也不保证覆盖供应商私有参数和响应扩展。 - 真 SSE 可避免慢生成在完成前长时间没有响应字节,但 Worker 到 Provider 之间的 Cloudflare/Caddy/其他 Gateway 仍有独立的缓冲、空闲或绝对总时长配置;Run 的 `read_timeout_seconds` 不会改写这些代理限制。 - 当前标准数据垂直切片只含 MMLU-Pro 和 GPQA-Diamond 客观选择题转换器;IFEval 专用规则 Evaluator、代码沙箱和完整标准 Benchmark 插件体系尚未交付,也不执行任何不可信代码。 - 全量真实 CLI 评测可能产生大量 Token、时间和费用;它当前只有预检、显式确认、限题与 1–4 有界并发,不继承 Web/API managed Run 的 RPM/TPM 或全局费用 hard limit。 diff --git a/backend/alembic/versions/20260830_0008_provider_api_protocols.py b/backend/alembic/versions/20260830_0008_provider_api_protocols.py new file mode 100644 index 0000000..0d49b20 --- /dev/null +++ b/backend/alembic/versions/20260830_0008_provider_api_protocols.py @@ -0,0 +1,98 @@ +"""Expand registered Model provider types to explicit API protocols. + +Revision ID: 20260830_0008 +Revises: 20260830_0007 +Create Date: 2026-08-30 12:00:00 UTC +""" + +from collections.abc import Sequence + +import sqlalchemy as sa + +from alembic import op + +revision: str = "20260830_0008" +down_revision: str | None = "20260830_0007" +branch_labels: str | Sequence[str] | None = None +depends_on: str | Sequence[str] | None = None + +_OLD_PROVIDER_VALUES = "provider_type IN ('mock', 'openai_compatible')" +_NEW_PROVIDER_VALUES = ( + "provider_type IN ('mock', 'openai_compatible', 'openai_responses', 'anthropic_messages')" +) +_OLD_REMOTE_CONFIGURATION = ( + "provider_type != 'openai_compatible' OR " + "(base_url IS NOT NULL AND remote_model_name IS NOT NULL AND " + "((credential_source = 'environment' AND api_key_env IS NOT NULL) OR " + "(credential_source = 'stored' AND api_key_env IS NULL)))" +) +_NEW_REMOTE_CONFIGURATION = ( + "provider_type NOT IN " + "('openai_compatible', 'openai_responses', 'anthropic_messages') OR " + "(base_url IS NOT NULL AND remote_model_name IS NOT NULL AND " + "((credential_source = 'environment' AND api_key_env IS NOT NULL) OR " + "(credential_source = 'stored' AND api_key_env IS NULL)))" +) + + +def _replace_constraints( + *, + provider_values: str, + remote_configuration: str, + existing_length: int, + target_length: int, +) -> None: + with op.batch_alter_table("models") as batch_op: + batch_op.drop_constraint(op.f("ck_models_provider_type_values"), type_="check") + batch_op.drop_constraint( + op.f("ck_models_openai_configuration_required"), + type_="check", + ) + batch_op.alter_column( + "provider_type", + existing_type=sa.String(length=existing_length), + type_=sa.String(length=target_length), + existing_nullable=False, + ) + batch_op.create_check_constraint( + op.f("ck_models_provider_type_values"), + provider_values, + ) + batch_op.create_check_constraint( + op.f("ck_models_openai_configuration_required"), + remote_configuration, + ) + + +def upgrade() -> None: + _replace_constraints( + provider_values=_NEW_PROVIDER_VALUES, + remote_configuration=_NEW_REMOTE_CONFIGURATION, + existing_length=17, + target_length=18, + ) + + +def downgrade() -> None: + connection = op.get_bind() + if connection.dialect.name == "sqlite": + connection.exec_driver_sql("BEGIN IMMEDIATE") + elif connection.dialect.name == "postgresql": + connection.exec_driver_sql("LOCK TABLE models IN ACCESS EXCLUSIVE MODE") + new_type_count = connection.scalar( + sa.text( + "SELECT COUNT(*) FROM models " + "WHERE provider_type IN ('openai_responses', 'anthropic_messages')" + ) + ) + if int(new_type_count or 0) != 0: + raise RuntimeError( + "Cannot downgrade while openai_responses or anthropic_messages Models exist; " + "delete or convert those Model configurations first" + ) + _replace_constraints( + provider_values=_OLD_PROVIDER_VALUES, + remote_configuration=_OLD_REMOTE_CONFIGURATION, + existing_length=18, + target_length=17, + ) diff --git a/backend/app/adapters/__init__.py b/backend/app/adapters/__init__.py index 2b8fdf9..6ca55c8 100644 --- a/backend/app/adapters/__init__.py +++ b/backend/app/adapters/__init__.py @@ -20,6 +20,49 @@ ) from .mock import MockModelAdapter from .openai_compatible import OpenAICompatibleAdapter, sanitize_error_message +from .provider_protocols import AnthropicMessagesAdapter, OpenAIResponsesAdapter + + +def _remote_adapter_options(kwargs: dict[str, Any]) -> dict[str, Any]: + """Normalize the shared constructor surface for every remote adapter.""" + + options = dict(kwargs) + if "remote_model_name" not in options and "model_name" in options: + options["remote_model_name"] = options.pop("model_name") + aliases = { + "connect_timeout": "connect_timeout_seconds", + "read_timeout": "read_timeout_seconds", + "write_timeout": "write_timeout_seconds", + "pool_timeout": "pool_timeout_seconds", + "retry_count": "max_retries", + "backoff_base_seconds": "retry_backoff_base_seconds", + "backoff_cap_seconds": "retry_backoff_cap_seconds", + } + for old_name, new_name in aliases.items(): + if old_name in options and new_name not in options: + options[new_name] = options.pop(old_name) + allowed = { + "base_url", + "remote_model_name", + "api_key_env", + "api_key", + "connect_timeout_seconds", + "read_timeout_seconds", + "write_timeout_seconds", + "pool_timeout_seconds", + "max_retries", + "retry_backoff_base_seconds", + "retry_backoff_cap_seconds", + "client", + "sleep", + "attempt_controller", + } + required = {"base_url", "remote_model_name"} + return { + key: value + for key, value in options.items() + if key in allowed and (value is not None or key in required) + } def build_adapter(provider_type: str, **kwargs: Any) -> ModelAdapter: @@ -34,44 +77,11 @@ def build_adapter(provider_type: str, **kwargs: Any) -> ModelAdapter: } return MockModelAdapter(**mock_options) if normalized in {"openai", "openai_compatible"}: - options = dict(kwargs) - if "remote_model_name" not in options and "model_name" in options: - options["remote_model_name"] = options.pop("model_name") - aliases = { - "connect_timeout": "connect_timeout_seconds", - "read_timeout": "read_timeout_seconds", - "write_timeout": "write_timeout_seconds", - "pool_timeout": "pool_timeout_seconds", - "retry_count": "max_retries", - "backoff_base_seconds": "retry_backoff_base_seconds", - "backoff_cap_seconds": "retry_backoff_cap_seconds", - } - for old_name, new_name in aliases.items(): - if old_name in options and new_name not in options: - options[new_name] = options.pop(old_name) - allowed = { - "base_url", - "remote_model_name", - "api_key_env", - "api_key", - "connect_timeout_seconds", - "read_timeout_seconds", - "write_timeout_seconds", - "pool_timeout_seconds", - "max_retries", - "retry_backoff_base_seconds", - "retry_backoff_cap_seconds", - "client", - "sleep", - "attempt_controller", - } - required = {"base_url", "remote_model_name"} - adapter_options = { - key: value - for key, value in options.items() - if key in allowed and (value is not None or key in required) - } - return OpenAICompatibleAdapter(**adapter_options) + return OpenAICompatibleAdapter(**_remote_adapter_options(kwargs)) + if normalized == "openai_responses": + return OpenAIResponsesAdapter(**_remote_adapter_options(kwargs)) + if normalized == "anthropic_messages": + return AnthropicMessagesAdapter(**_remote_adapter_options(kwargs)) raise ValueError(f"Unsupported provider_type: {provider_type!r}") @@ -80,6 +90,7 @@ def build_adapter(provider_type: str, **kwargs: Any) -> ModelAdapter: __all__ = [ "AdapterError", + "AnthropicMessagesAdapter", "GenerationConfig", "Message", "MockModelAdapter", @@ -87,6 +98,7 @@ def build_adapter(provider_type: str, **kwargs: Any) -> ModelAdapter: "ModelGenerationResult", "OpenAICompatibleAdapter", "OpenAICompatibleModelAdapter", + "OpenAIResponsesAdapter", "ProviderAttemptContext", "ProviderAttemptController", "ProviderAttemptDisposition", diff --git a/backend/app/adapters/openai_compatible.py b/backend/app/adapters/openai_compatible.py index 33c130d..32f239b 100644 --- a/backend/app/adapters/openai_compatible.py +++ b/backend/app/adapters/openai_compatible.py @@ -68,6 +68,7 @@ "provider_http_error", } ) +_KNOWN_PROVIDER_ENDPOINT_SUFFIXES = frozenset({"/chat/completions", "/responses", "/messages"}) class _ResponseBodyTooLarge(RuntimeError): @@ -102,17 +103,25 @@ def content(self) -> str: def consume_event(self, raw_event: bytes) -> None: """Consume one decoded SSE ``data`` event without retaining raw bytes.""" + invalid_utf8 = False try: event_text = raw_event.decode("utf-8") - except UnicodeError as exc: - raise self._invalid("Upstream SSE data was not valid UTF-8.") from exc + except UnicodeError: + invalid_utf8 = True + event_text = "" + if invalid_utf8: + raise self._invalid("Upstream SSE data was not valid UTF-8.") if event_text.strip() == "[DONE]": self.done = True return + invalid_json = False try: body = json.loads(event_text) - except ValueError as exc: - raise self._invalid("Upstream SSE data was not valid JSON.") from exc + except ValueError: + invalid_json = True + body = None + if invalid_json: + raise self._invalid("Upstream SSE data was not valid JSON.") if not isinstance(body, Mapping): raise self._invalid("Upstream SSE data was not a JSON object.") @@ -169,10 +178,14 @@ def consume_event(self, raw_event: bytes) -> None: return if not isinstance(content, str): raise self._invalid("Upstream SSE content delta was not text.") + invalid_content = False try: encoded_size = len(content.encode("utf-8")) - except UnicodeError as exc: - raise self._invalid("Upstream SSE content was not valid Unicode text.") from exc + except UnicodeError: + invalid_content = True + encoded_size = 0 + if invalid_content: + raise self._invalid("Upstream SSE content was not valid Unicode text.") if self.content_bytes + encoded_size > MAX_CHAT_SUCCESS_RESPONSE_BYTES: raise _ResponseBodyTooLarge( limit=MAX_CHAT_SUCCESS_RESPONSE_BYTES, @@ -227,11 +240,17 @@ def _is_loopback_host(hostname: str) -> bool: def _validated_base_url(base_url: str) -> str: normalized = base_url.strip().rstrip("/") + invalid_url = False try: parsed = urlsplit(normalized) hostname = parsed.hostname - except ValueError as exc: - raise ValueError("base_url must be an absolute HTTP(S) URL") from exc + except ValueError: + invalid_url = True + parsed = None + hostname = None + if invalid_url: + raise ValueError("base_url must be an absolute HTTP(S) URL") + assert parsed is not None if parsed.scheme not in {"http", "https"} or not hostname: raise ValueError("base_url must be an absolute HTTP(S) URL") if parsed.username or parsed.password: @@ -245,6 +264,18 @@ def _validated_base_url(base_url: str) -> str: return normalized +def _validate_protocol_endpoint(base_url: str, expected_suffix: str) -> None: + """Reject a full endpoint that belongs to a different known protocol.""" + + path = urlsplit(base_url).path.rstrip("/") + for suffix in _KNOWN_PROVIDER_ENDPOINT_SUFFIXES: + if path.endswith(suffix) and suffix != expected_suffix: + raise ValueError( + f"base_url endpoint {suffix!r} does not match the selected " + f"{expected_suffix!r} protocol" + ) + + def sanitize_error_message(message: object, *secrets: str) -> str: """Return a bounded diagnostic string with common credentials removed.""" @@ -294,6 +325,9 @@ class OpenAICompatibleAdapter(ModelAdapter): closed by this adapter. """ + _endpoint_suffix = "/chat/completions" + _transport_error_label = "OpenAI-compatible" + def __init__( self, base_url: str, @@ -319,6 +353,7 @@ def __init__( if api_key is not None and not isinstance(api_key, (SecretStr, str)): raise ValueError("api_key must be a secret string or null") self.base_url = _validated_base_url(base_url) + _validate_protocol_endpoint(self.base_url, self._endpoint_suffix) self.remote_model_name = remote_model_name self.api_key_env = api_key_env self._api_key = SecretStr(api_key) if isinstance(api_key, str) else api_key @@ -372,9 +407,21 @@ def __init__( @property def chat_completions_url(self) -> str: - if self.base_url.endswith("/chat/completions"): + return self.endpoint_url + + @property + def endpoint_url(self) -> str: + if urlsplit(self.base_url).path.rstrip("/").endswith(self._endpoint_suffix): return self.base_url - return f"{self.base_url}/chat/completions" + return f"{self.base_url}{self._endpoint_suffix}" + + def _build_headers(self, api_key: str) -> dict[str, str]: + return { + "Authorization": f"Bearer {api_key}", + "Accept": "text/event-stream", + "Accept-Encoding": "identity", + "Content-Type": "application/json", + } async def generate( self, @@ -400,12 +447,7 @@ async def generate( ) payload = self._build_payload(messages, generation_config) - headers = { - "Authorization": f"Bearer {api_key}", - "Accept": "text/event-stream", - "Accept-Encoding": "identity", - "Content-Type": "application/json", - } + headers = self._build_headers(api_key) first_attempt = self._first_provider_attempt(attempt_context) started = time.perf_counter() client = self._client @@ -440,11 +482,12 @@ async def generate( stream_result: _SSEAccumulator | None = None response_body: bytes | None = None + terminal_response_error: AdapterError | None = None terminal_transport_error: AdapterError | None = None try: async with client.stream( "POST", - self.chat_completions_url, + self.endpoint_url, json=payload, headers=headers, timeout=self._timeout, @@ -535,7 +578,7 @@ async def generate( disposition=ProviderAttemptDisposition.SETTLED_CONSERVATIVE, outcome=ProviderAttemptOutcome.PROVIDER_RESPONSE_ERROR, ) - raise AdapterError( + terminal_response_error = AdapterError( "provider_response_too_large", sanitize_error_message( f"Upstream response exceeded the {exc.limit}-byte safety limit.", @@ -543,7 +586,7 @@ async def generate( ), status_code=_safe_status_code(exc.status_code, api_key), attempts=attempt, - ) from exc + ) except httpx.TransportError as exc: await _finish_provider_attempt( self._attempt_controller, @@ -558,7 +601,7 @@ async def generate( continue terminal_transport_error = AdapterError( error_type, - f"OpenAI-compatible request failed: {safe_message}", + f"{self._transport_error_label} request failed: {safe_message}", retryable=True, attempts=attempt, ) @@ -587,9 +630,11 @@ async def generate( ) raise - # Raise only after leaving the TransportError handler. httpx transport - # exceptions retain their request (including Authorization), and Python - # otherwise attaches that exception as both cause/context to AdapterError. + # Raise only after leaving handlers whose exceptions retain the request + # (including credentials) or raw response chunks. Otherwise Python would + # attach those exceptions and their traceback frames to AdapterError. + if terminal_response_error is not None: + raise terminal_response_error if terminal_transport_error is not None: raise terminal_transport_error @@ -916,28 +961,40 @@ def _parse_success_response( attempts: int, api_key: str, ) -> ModelGenerationResult: + invalid_json = False try: body = json.loads(response_body) - except (ValueError, UnicodeError) as exc: + except (ValueError, UnicodeError): + invalid_json = True + body = None + if invalid_json: raise AdapterError( "invalid_provider_response", "Upstream returned a non-JSON success response.", status_code=_safe_status_code(response.status_code, api_key), attempts=attempts, - ) from exc + ) + missing_choice = False try: choice = body["choices"][0] - except (KeyError, IndexError, TypeError) as exc: + except (KeyError, IndexError, TypeError): + missing_choice = True + choice = None + if missing_choice: raise AdapterError( "invalid_provider_response", "Upstream response did not contain choices[0].", status_code=_safe_status_code(response.status_code, api_key), attempts=attempts, - ) from exc + ) finish_reason = choice.get("finish_reason") if isinstance(choice, Mapping) else None + missing_content = False try: content = choice["message"]["content"] - except (KeyError, TypeError) as exc: + except (KeyError, TypeError): + missing_content = True + content = None + if missing_content: if finish_reason == "length": raise AdapterError( "output_truncated", @@ -945,13 +1002,13 @@ def _parse_success_response( "Increase max_tokens or let the Provider choose its default.", status_code=_safe_status_code(response.status_code, api_key), attempts=attempts, - ) from exc + ) raise AdapterError( "invalid_provider_response", "Upstream response did not contain choices[0].message.content.", status_code=_safe_status_code(response.status_code, api_key), attempts=attempts, - ) from exc + ) usage_obj = body.get("usage") if isinstance(body, Mapping) else None provider_request_id = body.get("id") if isinstance(body, Mapping) else None returned_model = body.get("model") if isinstance(body, Mapping) else None diff --git a/backend/app/adapters/provider_protocols.py b/backend/app/adapters/provider_protocols.py new file mode 100644 index 0000000..8f94a3b --- /dev/null +++ b/backend/app/adapters/provider_protocols.py @@ -0,0 +1,1178 @@ +"""Explicit OpenAI Responses and Anthropic Messages protocol adapters.""" + +from __future__ import annotations + +import io +import json +from collections.abc import Mapping, Sequence +from contextlib import aclosing +from typing import Any, Protocol, TypeVar + +import httpx + +from .base import AdapterError, GenerationConfig, Message, ModelGenerationResult +from .openai_compatible import ( + MAX_CHAT_STREAM_EVENT_BYTES, + MAX_CHAT_STREAM_WIRE_BYTES, + MAX_CHAT_SUCCESS_RESPONSE_BYTES, + OpenAICompatibleAdapter, + _redact_json_secret, + _ResponseBodyTooLarge, + _safe_status_code, + sanitize_error_message, +) + + +class _TypedSSEAccumulator(Protocol): + done: bool + attempts: int + api_key: str + + def consume_event(self, raw_event: bytes, event_name: str | None) -> None: ... + + +_AccumulatorT = TypeVar("_AccumulatorT", bound=_TypedSSEAccumulator) +_RESPONSES_RATE_LIMIT_ERROR_CODES = frozenset( + {"rate_limit", "rate_limit_error", "rate_limit_exceeded"} +) +_RESPONSES_SERVER_ERROR_CODES = frozenset({"server_error"}) +_MESSAGES_RATE_LIMIT_ERROR_CODES = frozenset({"rate_limit_error"}) +_MESSAGES_SERVER_ERROR_CODES = frozenset({"api_error", "overloaded_error", "timeout_error"}) + + +def _invalid_stream(message: str, *, status_code: int, attempts: int, api_key: str) -> AdapterError: + return AdapterError( + "invalid_provider_stream", + message, + status_code=_safe_status_code(status_code, api_key), + attempts=attempts, + ) + + +def _decode_typed_event( + raw_event: bytes, + event_name: str | None, + *, + status_code: int, + attempts: int, + api_key: str, +) -> tuple[str, Mapping[str, Any]]: + invalid_utf8 = False + try: + event_text = raw_event.decode("utf-8") + except UnicodeError: + invalid_utf8 = True + event_text = "" + if invalid_utf8: + raise _invalid_stream( + "Upstream SSE data was not valid UTF-8.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + invalid_json = False + try: + body = json.loads(event_text) + except ValueError: + invalid_json = True + body = None + if invalid_json: + raise _invalid_stream( + "Upstream SSE data was not valid JSON.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + if not isinstance(body, Mapping): + raise _invalid_stream( + "Upstream SSE data was not a JSON object.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + + body_type = body.get("type") + if body_type is not None and not isinstance(body_type, str): + raise _invalid_stream( + "Upstream SSE event type was not text.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + if event_name and body_type and event_name != body_type: + raise _invalid_stream( + "Upstream SSE event name conflicted with its JSON type.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + event_type = body_type or event_name + if not event_type: + raise _invalid_stream( + "Upstream SSE event did not declare a type.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + return event_type, body + + +async def _consume_typed_sse( + response: httpx.Response, + accumulator: _AccumulatorT, + *, + terminal_event: str, +) -> _AccumulatorT: + """Consume bounded SSE framing while delegating typed event semantics.""" + + content_length = response.headers.get("content-length") + if content_length is not None: + try: + declared_length = int(content_length) + except ValueError: + declared_length = None + if declared_length is not None and declared_length > MAX_CHAT_STREAM_WIRE_BYTES: + raise _ResponseBodyTooLarge( + limit=MAX_CHAT_STREAM_WIRE_BYTES, + status_code=response.status_code, + ) + + line_buffer = bytearray() + data_lines: list[bytes] = [] + event_name: str | None = None + event_bytes = 0 + wire_bytes = 0 + first_line = True + + def dispatch_event() -> None: + nonlocal event_name + if data_lines: + accumulator.consume_event(b"\n".join(data_lines), event_name) + data_lines.clear() + event_name = None + + def process_line(line: bytes) -> None: + nonlocal event_bytes, event_name, first_line + raw_line_size = len(line) + if first_line: + first_line = False + if line.startswith(b"\xef\xbb\xbf"): + line = line[3:] + if not line: + dispatch_event() + event_bytes = 0 + return + + event_bytes += raw_line_size + 1 + if event_bytes > MAX_CHAT_STREAM_EVENT_BYTES: + raise _ResponseBodyTooLarge( + limit=MAX_CHAT_STREAM_EVENT_BYTES, + status_code=response.status_code, + ) + if line.startswith(b":"): + return + field, separator, value = line.partition(b":") + if separator and value.startswith(b" "): + value = value[1:] + if field == b"data": + data_lines.append(value if separator else b"") + return + if field != b"event": + return + invalid_event_name = False + try: + decoded_name = (value if separator else b"").decode("utf-8") + except UnicodeError: + invalid_event_name = True + decoded_name = "" + if invalid_event_name: + raise _invalid_stream( + "Upstream SSE event name was not valid UTF-8.", + status_code=response.status_code, + attempts=accumulator.attempts, + api_key=accumulator.api_key, + ) + event_name = decoded_name or None + + def process_chunk(chunk: bytes) -> bool: + nonlocal wire_bytes + wire_bytes += len(chunk) + if wire_bytes > MAX_CHAT_STREAM_WIRE_BYTES: + raise _ResponseBodyTooLarge( + limit=MAX_CHAT_STREAM_WIRE_BYTES, + status_code=response.status_code, + ) + line_buffer.extend(chunk) + while True: + line = OpenAICompatibleAdapter._pop_sse_line(line_buffer) + if line is None: + break + process_line(line) + if accumulator.done: + return True + if event_bytes + len(line_buffer) > MAX_CHAT_STREAM_EVENT_BYTES: + raise _ResponseBodyTooLarge( + limit=MAX_CHAT_STREAM_EVENT_BYTES, + status_code=response.status_code, + ) + return False + + if response.is_stream_consumed: + process_chunk(response.content) + else: + async with aclosing(response.aiter_raw()) as chunks: + async for chunk in chunks: + if process_chunk(chunk): + break + if accumulator.done: + return accumulator + + while True: + line = OpenAICompatibleAdapter._pop_sse_line(line_buffer, at_eof=True) + if line is None: + break + process_line(line) + if accumulator.done: + return accumulator + dispatch_event() + if not accumulator.done: + raise AdapterError( + "incomplete_provider_stream", + f"Upstream SSE response ended before the {terminal_event} event.", + status_code=_safe_status_code( + response.status_code, + accumulator.api_key, + ), + attempts=accumulator.attempts, + ) + return accumulator + + +def _parse_json_object( + response: httpx.Response, + response_body: bytes, + *, + attempts: int, + api_key: str, +) -> Mapping[str, Any]: + invalid_json = False + try: + body = json.loads(response_body) + except (ValueError, UnicodeError): + invalid_json = True + body = None + if invalid_json: + raise AdapterError( + "invalid_provider_response", + "Upstream returned a non-JSON success response.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + if not isinstance(body, Mapping): + raise AdapterError( + "invalid_provider_response", + "Upstream success response was not a JSON object.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + return body + + +def _append_text( + content: io.StringIO, + text: object, + *, + content_bytes: int, + status_code: int, + attempts: int, + api_key: str, + source: str, +) -> int: + if not isinstance(text, str): + raise _invalid_stream( + f"Upstream SSE {source} was not text.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + invalid_text = False + try: + encoded_size = len(text.encode("utf-8")) + except UnicodeError: + invalid_text = True + encoded_size = 0 + if invalid_text: + raise _invalid_stream( + f"Upstream SSE {source} was not valid Unicode text.", + status_code=status_code, + attempts=attempts, + api_key=api_key, + ) + if content_bytes + encoded_size > MAX_CHAT_SUCCESS_RESPONSE_BYTES: + raise _ResponseBodyTooLarge( + limit=MAX_CHAT_SUCCESS_RESPONSE_BYTES, + status_code=status_code, + ) + content.write(text) + return content_bytes + encoded_size + + +def _collect_response_output_text(output: object) -> str: + if not isinstance(output, list): + raise ValueError("output was not an array") + parts: list[str] = [] + for item in output: + if not isinstance(item, Mapping): + raise ValueError("output item was not an object") + item_type = item.get("type") + if item_type == "output_text": + text = item.get("text") + if not isinstance(text, str): + raise ValueError("output_text item did not contain text") + parts.append(text) + continue + if item_type != "message": + continue + blocks = item.get("content") + if not isinstance(blocks, list): + raise ValueError("message content was not an array") + for block in blocks: + if not isinstance(block, Mapping): + raise ValueError("message content item was not an object") + if block.get("type") != "output_text": + continue + text = block.get("text") + if not isinstance(text, str): + raise ValueError("output_text block did not contain text") + parts.append(text) + return "".join(parts) + + +def _provider_failure_detail(body: Mapping[str, Any], fallback: str) -> object: + error = body.get("error") + if isinstance(error, Mapping): + return error.get("message") or error.get("type") or fallback + if error is not None: + return error + details = body.get("incomplete_details") + if isinstance(details, Mapping): + return details.get("reason") or fallback + return fallback + + +def _provider_error_codes(body: Mapping[str, Any]) -> frozenset[str]: + error = body.get("error") + candidates: tuple[object, ...] + if isinstance(error, Mapping): + candidates = (error.get("type"), error.get("code"), body.get("code")) + else: + candidates = (body.get("code"),) + return frozenset(value for value in candidates if isinstance(value, str)) + + +def _responses_stream_error_type(body: Mapping[str, Any]) -> tuple[str, bool]: + codes = _provider_error_codes(body) + if codes & _RESPONSES_RATE_LIMIT_ERROR_CODES: + return "rate_limited", True + if codes & _RESPONSES_SERVER_ERROR_CODES: + return "provider_5xx", True + return "provider_stream_error", False + + +def _messages_stream_error_type(body: Mapping[str, Any]) -> tuple[str, bool]: + codes = _provider_error_codes(body) + if codes & _MESSAGES_RATE_LIMIT_ERROR_CODES: + return "rate_limited", True + if codes & _MESSAGES_SERVER_ERROR_CODES: + return "provider_5xx", True + return "provider_stream_error", False + + +def _build_result( + response: httpx.Response, + *, + content: object, + usage_obj: object, + provider_request_id: object, + returned_model: object, + finish_reason: object, + latency_ms: float, + attempts: int, + api_key: str, + response_mode: str, + adapter_name: str, +) -> ModelGenerationResult: + if not isinstance(content, str): + raise AdapterError( + "invalid_provider_response", + "Upstream response content was not text.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + if not content.strip(): + if finish_reason in {"length", "max_tokens", "max_output_tokens"}: + raise AdapterError( + "output_truncated", + "Upstream exhausted the output token budget before returning text. " + "Increase max_tokens.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + raise AdapterError( + "empty_response", + "Upstream returned an empty model response.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + safe_content = content.replace(api_key, "[REDACTED]") + raw_usage = ( + _redact_json_secret(dict(usage_obj), api_key) if isinstance(usage_obj, Mapping) else None + ) + input_tokens = OpenAICompatibleAdapter._usage_int(raw_usage, "input_tokens") + output_tokens = OpenAICompatibleAdapter._usage_int(raw_usage, "output_tokens") + if provider_request_id is None: + provider_request_id = response.headers.get("x-request-id") or response.headers.get( + "request-id" + ) + return ModelGenerationResult( + text=safe_content, + input_tokens=input_tokens, + output_tokens=output_tokens, + latency_ms=max(0.0, latency_ms), + provider_request_id=( + None + if provider_request_id is None + else sanitize_error_message(provider_request_id, api_key) + ), + raw_usage=raw_usage, + metadata={ + "adapter": adapter_name, + "attempts": attempts, + "response_mode": response_mode, + "finish_reason": ( + sanitize_error_message(finish_reason, api_key) + if isinstance(finish_reason, str) + else None + ), + "returned_model": ( + sanitize_error_message(returned_model, api_key) + if isinstance(returned_model, str) + else None + ), + }, + ) + + +class _ResponsesSSEAccumulator: + def __init__(self, *, status_code: int, attempts: int, api_key: str) -> None: + self.status_code = status_code + self.attempts = attempts + self.api_key = api_key + self._content = io.StringIO() + self.content_bytes = 0 + self.usage: Mapping[str, Any] | None = None + self.provider_request_id: str | None = None + self.returned_model: str | None = None + self.system_fingerprint: str | None = None + self.finish_reason: str | None = None + self.done = False + + @property + def content(self) -> str: + return self._content.getvalue() + + def consume_event(self, raw_event: bytes, event_name: str | None) -> None: + event_type, body = _decode_typed_event( + raw_event, + event_name, + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + if event_type == "error": + detail = _provider_failure_detail(body, "Provider reported a Responses stream error.") + error_type, retryable = _responses_stream_error_type(body) + raise AdapterError( + error_type, + sanitize_error_message(f"Upstream SSE error: {detail}", self.api_key), + retryable=retryable, + status_code=_safe_status_code(self.status_code, self.api_key), + attempts=self.attempts, + ) + if event_type in {"response.failed", "response.incomplete"}: + response_obj = body.get("response") + detail_body = response_obj if isinstance(response_obj, Mapping) else body + detail = _provider_failure_detail(detail_body, event_type) + if event_type == "response.failed": + error_type, retryable = _responses_stream_error_type(detail_body) + elif detail == "max_output_tokens": + error_type = "output_truncated" + retryable = False + else: + error_type = "incomplete_provider_stream" + retryable = False + raise AdapterError( + error_type, + sanitize_error_message( + f"Upstream Responses stream did not complete: {detail}", + self.api_key, + ), + retryable=retryable, + status_code=_safe_status_code(self.status_code, self.api_key), + attempts=self.attempts, + ) + if event_type == "response.output_text.delta": + self.content_bytes = _append_text( + self._content, + body.get("delta"), + content_bytes=self.content_bytes, + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + source="output_text delta", + ) + return + if event_type not in { + "response.created", + "response.in_progress", + "response.completed", + }: + return + response_obj = body.get("response") + if not isinstance(response_obj, Mapping): + raise _invalid_stream( + f"Upstream SSE {event_type} response was not an object.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + self._capture_response(response_obj) + if event_type != "response.completed": + return + status = response_obj.get("status") + if status is not None and status != "completed": + detail = _provider_failure_detail(response_obj, str(status)) + raise AdapterError( + "incomplete_provider_stream", + sanitize_error_message( + f"Upstream Responses stream did not complete: {detail}", self.api_key + ), + status_code=_safe_status_code(self.status_code, self.api_key), + attempts=self.attempts, + ) + if not self.content: + invalid_completed_output = False + try: + completed_text = _collect_response_output_text(response_obj.get("output")) + except ValueError: + invalid_completed_output = True + completed_text = "" + if invalid_completed_output: + raise _invalid_stream( + "Upstream SSE response.completed output was invalid.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + self.content_bytes = _append_text( + self._content, + completed_text, + content_bytes=self.content_bytes, + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + source="completed output", + ) + self.finish_reason = "completed" + self.done = True + + def _capture_response(self, response_obj: Mapping[str, Any]) -> None: + self.provider_request_id = self._stable_string( + response_obj.get("id"), self.provider_request_id, "response id" + ) + self.returned_model = self._stable_string( + response_obj.get("model"), self.returned_model, "response model" + ) + usage = response_obj.get("usage") + if usage is not None: + if not isinstance(usage, Mapping): + raise _invalid_stream( + "Upstream SSE Responses usage was not an object.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + self.usage = dict(usage) + + def _stable_string( + self, + value: object, + existing: str | None, + label: str, + ) -> str | None: + if value is None: + return existing + if not isinstance(value, str): + raise _invalid_stream( + f"Upstream SSE {label} was not text.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + if existing is not None and existing != value: + raise _invalid_stream( + f"Upstream SSE returned conflicting {label} values.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + return value + + +class OpenAIResponsesAdapter(OpenAICompatibleAdapter): + """Call an explicit OpenAI Responses ``/responses`` endpoint.""" + + _endpoint_suffix = "/responses" + _transport_error_label = "OpenAI Responses" + + @property + def responses_url(self) -> str: + return self.endpoint_url + + def _build_payload( + self, + messages: Sequence[Message], + generation_config: GenerationConfig, + ) -> dict[str, Any]: + if generation_config.get("seed") is not None: + raise AdapterError( + "invalid_request", + "OpenAI Responses does not support the configured seed parameter.", + ) + payload = super()._build_payload(messages, generation_config) + payload["input"] = payload.pop("messages") + payload.pop("stream_options", None) + payload.pop("seed", None) + if "max_tokens" in payload: + payload["max_output_tokens"] = payload.pop("max_tokens") + return payload + + @staticmethod + async def _consume_sse_response( + response: httpx.Response, + *, + attempts: int, + api_key: str, + ) -> _ResponsesSSEAccumulator: + accumulator = _ResponsesSSEAccumulator( + status_code=response.status_code, + attempts=attempts, + api_key=api_key, + ) + return await _consume_typed_sse( + response, + accumulator, + terminal_event="response.completed", + ) + + @staticmethod + def _parse_success_response( + response: httpx.Response, + response_body: bytes, + *, + latency_ms: float, + attempts: int, + api_key: str, + ) -> ModelGenerationResult: + body = _parse_json_object( + response, + response_body, + attempts=attempts, + api_key=api_key, + ) + status = body.get("status") + if status is not None and not isinstance(status, str): + raise AdapterError( + "invalid_provider_response", + "Upstream Responses status was not text.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + if status != "completed" and status is not None: + detail = _provider_failure_detail(body, status) + retryable = False + if status == "failed": + error_type, retryable = _responses_stream_error_type(body) + if error_type == "provider_stream_error": + error_type = "invalid_provider_response" + elif detail == "max_output_tokens": + error_type = "output_truncated" + else: + error_type = "invalid_provider_response" + raise AdapterError( + error_type, + sanitize_error_message( + f"Upstream Responses request did not complete: {detail}", api_key + ), + retryable=retryable, + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + invalid_output = False + try: + content = _collect_response_output_text(body.get("output")) + except ValueError: + invalid_output = True + content = "" + if invalid_output: + raise AdapterError( + "invalid_provider_response", + "Upstream Responses output did not contain valid output text items.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + return _build_result( + response, + content=content, + usage_obj=body.get("usage"), + provider_request_id=body.get("id"), + returned_model=body.get("model"), + finish_reason=status, + latency_ms=latency_ms, + attempts=attempts, + api_key=api_key, + response_mode="json", + adapter_name="openai_responses", + ) + + @staticmethod + def _build_generation_result( + response: httpx.Response, + *, + content: object, + finish_reason: object, + usage_obj: object, + provider_request_id: object, + returned_model: object, + system_fingerprint: object, + latency_ms: float, + attempts: int, + api_key: str, + response_mode: str, + ) -> ModelGenerationResult: + del system_fingerprint + return _build_result( + response, + content=content, + usage_obj=usage_obj, + provider_request_id=provider_request_id, + returned_model=returned_model, + finish_reason=finish_reason, + latency_ms=latency_ms, + attempts=attempts, + api_key=api_key, + response_mode=response_mode, + adapter_name="openai_responses", + ) + + +class _MessagesSSEAccumulator: + def __init__(self, *, status_code: int, attempts: int, api_key: str) -> None: + self.status_code = status_code + self.attempts = attempts + self.api_key = api_key + self._content = io.StringIO() + self.content_bytes = 0 + self.usage: dict[str, Any] = {} + self.provider_request_id: str | None = None + self.returned_model: str | None = None + self.system_fingerprint: str | None = None + self.finish_reason: str | None = None + self.saw_message_start = False + self.done = False + + @property + def content(self) -> str: + return self._content.getvalue() + + def consume_event(self, raw_event: bytes, event_name: str | None) -> None: + event_type, body = _decode_typed_event( + raw_event, + event_name, + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + if event_type == "error": + detail = _provider_failure_detail(body, "Provider reported a Messages stream error.") + error_type, retryable = _messages_stream_error_type(body) + raise AdapterError( + error_type, + sanitize_error_message(f"Upstream SSE error: {detail}", self.api_key), + retryable=retryable, + status_code=_safe_status_code(self.status_code, self.api_key), + attempts=self.attempts, + ) + if event_type == "message_start": + if self.saw_message_start: + raise _invalid_stream( + "Upstream SSE returned more than one message_start event.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + message = body.get("message") + if not isinstance(message, Mapping): + raise _invalid_stream( + "Upstream SSE message_start message was not an object.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + self.saw_message_start = True + self.provider_request_id = self._optional_string(message.get("id"), "message id") + self.returned_model = self._optional_string(message.get("model"), "message model") + self.finish_reason = self._optional_string(message.get("stop_reason"), "stop reason") + self._merge_usage(message.get("usage")) + self._consume_initial_content(message.get("content")) + return + if event_type == "content_block_delta": + if not self.saw_message_start: + raise _invalid_stream( + "Upstream SSE content_block_delta preceded message_start.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + delta = body.get("delta") + if not isinstance(delta, Mapping): + raise _invalid_stream( + "Upstream SSE content_block_delta delta was not an object.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + if delta.get("type") == "text_delta": + self.content_bytes = _append_text( + self._content, + delta.get("text"), + content_bytes=self.content_bytes, + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + source="text delta", + ) + return + if event_type == "message_delta": + if not self.saw_message_start: + raise _invalid_stream( + "Upstream SSE message_delta preceded message_start.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + delta = body.get("delta") + if not isinstance(delta, Mapping): + raise _invalid_stream( + "Upstream SSE message_delta delta was not an object.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + stop_reason = self._optional_string(delta.get("stop_reason"), "stop reason") + if ( + stop_reason is not None + and self.finish_reason is not None + and stop_reason != self.finish_reason + ): + raise _invalid_stream( + "Upstream SSE returned conflicting stop reason values.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + if stop_reason is not None: + self.finish_reason = stop_reason + self._merge_usage(body.get("usage")) + return + if event_type == "message_stop": + if not self.saw_message_start: + raise _invalid_stream( + "Upstream SSE message_stop preceded message_start.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + self.done = True + + def _optional_string(self, value: object, label: str) -> str | None: + if value is None: + return None + if not isinstance(value, str): + raise _invalid_stream( + f"Upstream SSE {label} was not text.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + return value + + def _merge_usage(self, usage: object) -> None: + if usage is None: + return + if not isinstance(usage, Mapping): + raise _invalid_stream( + "Upstream SSE Messages usage was not an object.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + # Anthropic stream usage is cumulative/delta-by-field. Overwrite each + # supplied field; summing repeated output_tokens would double count. + self.usage.update(usage) + + def _consume_initial_content(self, content: object) -> None: + if content is None: + return + if not isinstance(content, list): + raise _invalid_stream( + "Upstream SSE initial message content was not an array.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + for block in content: + if not isinstance(block, Mapping): + raise _invalid_stream( + "Upstream SSE initial content item was not an object.", + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + ) + if block.get("type") != "text": + continue + self.content_bytes = _append_text( + self._content, + block.get("text"), + content_bytes=self.content_bytes, + status_code=self.status_code, + attempts=self.attempts, + api_key=self.api_key, + source="initial text", + ) + + +class AnthropicMessagesAdapter(OpenAICompatibleAdapter): + """Call an explicit Anthropic Messages ``/messages`` endpoint.""" + + _endpoint_suffix = "/messages" + _transport_error_label = "Anthropic Messages" + + @property + def messages_url(self) -> str: + return self.endpoint_url + + @staticmethod + def _status_error_type(status_code: int) -> tuple[str, bool]: + if status_code == 529: + return "provider_5xx", True + return OpenAICompatibleAdapter._status_error_type(status_code) + + def _build_headers(self, api_key: str) -> dict[str, str]: + return { + "x-api-key": api_key, + "anthropic-version": "2023-06-01", + "Accept": "text/event-stream", + "Accept-Encoding": "identity", + "Content-Type": "application/json", + } + + def _build_payload( + self, + messages: Sequence[Message], + generation_config: GenerationConfig, + ) -> dict[str, Any]: + if generation_config.get("seed") is not None: + raise AdapterError( + "invalid_request", + "Anthropic Messages does not support the configured seed parameter.", + ) + temperature = generation_config.get("temperature") + if temperature is not None and ( + isinstance(temperature, bool) + or not isinstance(temperature, (int, float)) + or temperature != temperature + or temperature in {float("inf"), float("-inf")} + or not 0 <= temperature <= 1 + ): + raise AdapterError( + "invalid_request", + "Anthropic Messages requires temperature to be between 0 and 1.", + ) + top_p = generation_config.get("top_p") + if top_p is not None and ( + isinstance(top_p, bool) + or not isinstance(top_p, (int, float)) + or top_p != top_p + or top_p in {float("inf"), float("-inf")} + or not 0 < top_p <= 1 + ): + raise AdapterError( + "invalid_request", + "Anthropic Messages requires top_p to be greater than 0 and at most 1.", + ) + max_tokens = generation_config.get("max_tokens") + if isinstance(max_tokens, bool) or not isinstance(max_tokens, int) or max_tokens <= 0: + raise AdapterError( + "invalid_request", + "Anthropic Messages requires max_tokens to be a positive integer.", + ) + + chat_payload = super()._build_payload(messages, generation_config) + prepared_messages = chat_payload["messages"] + system_parts: list[str] = [] + body_messages: list[dict[str, Any]] = [] + for index, message in enumerate(prepared_messages): + role = message.get("role") + if role == "system": + if body_messages: + raise AdapterError( + "invalid_request", + "Anthropic Messages accepts system instructions only before messages.", + ) + content = message.get("content") + if not isinstance(content, str): + raise AdapterError( + "invalid_request", + f"System message at index {index} must contain text.", + ) + if content.strip(): + system_parts.append(content) + continue + if role not in {"user", "assistant"}: + raise AdapterError( + "invalid_request", + f"Anthropic message at index {index} has unsupported role {role!r}.", + ) + body_messages.append({"role": role, "content": message.get("content")}) + if not body_messages: + raise AdapterError("invalid_request", "At least one non-system message is required.") + + payload: dict[str, Any] = { + "model": self.remote_model_name, + "messages": body_messages, + "stream": True, + "max_tokens": max_tokens, + } + if system_parts: + payload["system"] = "\n\n".join(system_parts) + for key in ("temperature", "top_p"): + value = generation_config.get(key) + if value is not None: + payload[key] = value + return payload + + @staticmethod + async def _consume_sse_response( + response: httpx.Response, + *, + attempts: int, + api_key: str, + ) -> _MessagesSSEAccumulator: + accumulator = _MessagesSSEAccumulator( + status_code=response.status_code, + attempts=attempts, + api_key=api_key, + ) + return await _consume_typed_sse( + response, + accumulator, + terminal_event="message_stop", + ) + + @staticmethod + def _parse_success_response( + response: httpx.Response, + response_body: bytes, + *, + latency_ms: float, + attempts: int, + api_key: str, + ) -> ModelGenerationResult: + body = _parse_json_object( + response, + response_body, + attempts=attempts, + api_key=api_key, + ) + content_obj = body.get("content") + if not isinstance(content_obj, list): + raise AdapterError( + "invalid_provider_response", + "Upstream Messages response content was not an array.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + parts: list[str] = [] + for block in content_obj: + if not isinstance(block, Mapping): + raise AdapterError( + "invalid_provider_response", + "Upstream Messages content item was not an object.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + if block.get("type") != "text": + continue + text = block.get("text") + if not isinstance(text, str): + raise AdapterError( + "invalid_provider_response", + "Upstream Messages text block did not contain text.", + status_code=_safe_status_code(response.status_code, api_key), + attempts=attempts, + ) + parts.append(text) + return _build_result( + response, + content="".join(parts), + usage_obj=body.get("usage"), + provider_request_id=body.get("id"), + returned_model=body.get("model"), + finish_reason=body.get("stop_reason"), + latency_ms=latency_ms, + attempts=attempts, + api_key=api_key, + response_mode="json", + adapter_name="anthropic_messages", + ) + + @staticmethod + def _build_generation_result( + response: httpx.Response, + *, + content: object, + finish_reason: object, + usage_obj: object, + provider_request_id: object, + returned_model: object, + system_fingerprint: object, + latency_ms: float, + attempts: int, + api_key: str, + response_mode: str, + ) -> ModelGenerationResult: + del system_fingerprint + return _build_result( + response, + content=content, + usage_obj=usage_obj, + provider_request_id=provider_request_id, + returned_model=returned_model, + finish_reason=finish_reason, + latency_ms=latency_ms, + attempts=attempts, + api_key=api_key, + response_mode=response_mode, + adapter_name="anthropic_messages", + ) diff --git a/backend/app/api/v1/health.py b/backend/app/api/v1/health.py index 0c17c04..b34e96b 100644 --- a/backend/app/api/v1/health.py +++ b/backend/app/api/v1/health.py @@ -308,7 +308,12 @@ def info(settings: SettingsDep) -> InfoResponse: protocol_version=PROTOCOL_VERSION, environment=settings.environment, capabilities={ - "providers": ["mock", "openai_compatible"], + "providers": [ + "mock", + "openai_compatible", + "openai_responses", + "anthropic_messages", + ], "question_types": ["exact_match", "multiple_choice", "numeric"], "runner": "independent_database_lease_worker", }, diff --git a/backend/app/api/v1/models.py b/backend/app/api/v1/models.py index 222f74b..762c437 100644 --- a/backend/app/api/v1/models.py +++ b/backend/app/api/v1/models.py @@ -26,6 +26,7 @@ ) from app.models import Model as RegisteredModel from app.schemas.model import ( + REMOTE_PROVIDER_TYPES, ModelCreate, ModelList, ModelRead, @@ -497,13 +498,12 @@ def update_model( ) old_origin = ( normalize_provider_origin(model.base_url) - if model.provider_type == ProviderType.OPENAI_COMPATIBLE and model.base_url is not None + if model.provider_type in REMOTE_PROVIDER_TYPES and model.base_url is not None else None ) new_origin = ( normalize_provider_origin(validated.base_url) - if validated.provider_type == ProviderType.OPENAI_COMPATIBLE - and validated.base_url is not None + if validated.provider_type in REMOTE_PROVIDER_TYPES and validated.base_url is not None else None ) if ( diff --git a/backend/app/api/v1/runs.py b/backend/app/api/v1/runs.py index 112bc58..be2f72c 100644 --- a/backend/app/api/v1/runs.py +++ b/backend/app/api/v1/runs.py @@ -4,8 +4,9 @@ import logging from datetime import timedelta +from typing import Annotated -from fastapi import APIRouter, HTTPException, Request, status +from fastapi import APIRouter, HTTPException, Path, Request, Response, status from sqlalchemy import func, select from sqlalchemy.orm import Session, sessionmaker @@ -30,8 +31,20 @@ ) from app.runners.run_leases import CancelDisposition, RunLeaseRepository from app.schemas.audit import AuditEventList +from app.schemas.evaluation_progress import ( + PROGRESS_BLOCK_SIZE, + EvaluationProgressBlock, + EvaluationProgressBlockSummary, + EvaluationProgressIndex, +) from app.schemas.evaluation_response import EvaluationResponseList from app.schemas.evaluation_run import EvaluationRunCreate, EvaluationRunList, EvaluationRunRead +from app.services.run_progress import ( + RunProgressIntegrityError, + load_progress_block, + load_progress_index, + progress_block_count, +) from app.services.run_service import build_evaluation_run from app.task_queue import QueueUnavailable @@ -193,7 +206,17 @@ async def create_run( detail={"code": "benchmark_not_found", "message": "Benchmark was not found"}, ) - run = build_evaluation_run(model, benchmark, payload, settings) + try: + run = build_evaluation_run(model, benchmark, payload, settings) + except ValueError as exc: + session.rollback() + raise HTTPException( + status_code=status.HTTP_422_UNPROCESSABLE_CONTENT, + detail={ + "code": "invalid_provider_generation_parameters", + "message": str(exc), + }, + ) from None try: repository.admit_run( session, @@ -368,6 +391,83 @@ def cancel_run(run_id: str, session: SessionDep, settings: SettingsDep) -> Evalu return run +def _raise_progress_integrity_error() -> None: + raise HTTPException( + status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, + detail={ + "code": "run_progress_integrity_error", + "message": "Persisted Run progress does not match its frozen question plan", + }, + ) + + +@router.get( + "/{run_id}/progress", + response_model=EvaluationProgressIndex, + summary="查看 Run 实时进度索引", +) +def get_run_progress( + run_id: str, + response: Response, + session: SessionDep, +) -> EvaluationProgressIndex: + response.headers["Cache-Control"] = "no-store" + run = _get_run_or_404(session, run_id) + try: + projection = load_progress_index(session, run) + except RunProgressIntegrityError: + _raise_progress_integrity_error() + metrics = projection.metrics + return EvaluationProgressIndex( + block_size=PROGRESS_BLOCK_SIZE, + total_questions=run.total_questions, + completed_questions=metrics.completed_questions, + correct_questions=metrics.correct_questions, + error_questions=metrics.error_questions, + score=metrics.score, + completion_rate=metrics.completion_rate, + answered_accuracy=metrics.answered_accuracy, + average_latency_ms=metrics.average_latency_ms, + known_input_tokens=projection.known_input_tokens, + known_output_tokens=projection.known_output_tokens, + input_token_reported_responses=projection.input_token_reported_responses, + output_token_reported_responses=projection.output_token_reported_responses, + known_estimated_cost=float(projection.known_estimated_cost), + estimated_cost_reported_responses=projection.estimated_cost_reported_responses, + blocks=[ + EvaluationProgressBlockSummary(block_index=index, response_count=count) + for index, count in enumerate(projection.block_response_counts) + ], + ) + + +@router.get( + "/{run_id}/progress/blocks/{block_index}", + response_model=EvaluationProgressBlock, + summary="查看 Run 单个进度区块", +) +def get_run_progress_block( + run_id: str, + block_index: Annotated[int, Path(ge=0)], + response: Response, + session: SessionDep, +) -> EvaluationProgressBlock: + response.headers["Cache-Control"] = "no-store" + run = _get_run_or_404(session, run_id) + if block_index >= progress_block_count(run.total_questions): + raise HTTPException( + status_code=status.HTTP_422_UNPROCESSABLE_CONTENT, + detail={ + "code": "progress_block_out_of_range", + "message": "Run progress block is outside the frozen question plan", + }, + ) + return EvaluationProgressBlock( + block_index=block_index, + items=load_progress_block(session, run, block_index), + ) + + @router.get("/{run_id}/responses", response_model=EvaluationResponseList, summary="查看逐题结果") def list_responses( run_id: str, @@ -375,14 +475,20 @@ def list_responses( pagination: PaginationDep, ) -> EvaluationResponseList: _get_run_or_404(session, run_id) - total = ( - session.scalar( - select(func.count()) - .select_from(EvaluationResponse) - .where(EvaluationResponse.run_id == run_id) - ) - or 0 - ) + usage_summary = session.execute( + select( + func.count(EvaluationResponse.id), + func.coalesce(func.sum(EvaluationResponse.input_tokens), 0), + func.count(EvaluationResponse.input_tokens), + func.coalesce(func.sum(EvaluationResponse.output_tokens), 0), + func.count(EvaluationResponse.output_tokens), + ).where(EvaluationResponse.run_id == run_id) + ).one() + total = int(usage_summary[0] or 0) + known_input_tokens = int(usage_summary[1] or 0) + input_token_reported_responses = int(usage_summary[2] or 0) + known_output_tokens = int(usage_summary[3] or 0) + output_token_reported_responses = int(usage_summary[4] or 0) rows = session.execute( select(EvaluationResponse, Question) .join(Question, Question.id == EvaluationResponse.question_id) @@ -423,5 +529,12 @@ def list_responses( for response, question in rows ] return EvaluationResponseList( - items=items, total=total, offset=pagination.offset, limit=pagination.limit + items=items, + total=total, + offset=pagination.offset, + limit=pagination.limit, + known_input_tokens=known_input_tokens, + known_output_tokens=known_output_tokens, + input_token_reported_responses=input_token_reported_responses, + output_token_reported_responses=output_token_reported_responses, ) diff --git a/backend/app/cli/evaluate.py b/backend/app/cli/evaluate.py index 886de50..2b67531 100644 --- a/backend/app/cli/evaluate.py +++ b/backend/app/cli/evaluate.py @@ -41,13 +41,14 @@ ProviderPreflightError, discover_models, run_chat_canary, + run_provider_canary, select_remote_model, ) from app.reports import GROUP_FIELD_WHITELIST, ReportExportError, export_run_report from app.runners.evaluation_runner import EvaluationRunner from app.runners.run_leases import RunLeaseRepository from app.schemas.evaluation_run import EvaluationRunCreate -from app.schemas.model import ENV_VAR_PATTERN, ModelCreate +from app.schemas.model import ENV_VAR_PATTERN, ModelCreate, validate_provider_generation_parameters from app.services.benchmark_service import persist_dataset from app.services.run_service import build_evaluation_run from app.standard_datasets import prepare_gpqa_diamond, prepare_mmlu_pro @@ -126,8 +127,8 @@ def build_parser() -> argparse.ArgumentParser: prog="llmbenchlab-evaluate", allow_abbrev=False, description=( - "Pinned standard-dataset evaluation for a trusted local OpenAI-compatible API. " - "API keys are never accepted as command-line arguments." + "API keys are never accepted as command-line arguments. " + "Run trusted local Provider evaluations." ), ) subparsers = parser.add_subparsers(dest="command", required=True) @@ -146,6 +147,16 @@ def build_parser() -> argparse.ArgumentParser: ) _add_dataset_arguments(run, require_scope=True) run.add_argument("--base-url", required=True) + run.add_argument( + "--provider-type", + choices=( + ProviderType.OPENAI_COMPATIBLE.value, + ProviderType.OPENAI_RESPONSES.value, + ProviderType.ANTHROPIC_MESSAGES.value, + ), + default=ProviderType.OPENAI_COMPATIBLE.value, + help="Explicit Provider API protocol; defaults to OpenAI Chat Completions.", + ) run.add_argument( "--model", help="Remote model ID; auto-selected only when discovery is unique." ) @@ -154,13 +165,14 @@ def build_parser() -> argparse.ArgumentParser: run.add_argument( "--no-model-discovery", action="store_true", - help="Skip GET /models; requires --model. The billed Chat canary still runs.", + help="Skip GET /models; requires --model. The billed protocol canary still runs.", ) run.add_argument("--temperature", type=float) - run.add_argument("--top-p", type=float, default=1.0) + run.add_argument("--top-p", type=float) run.add_argument("--max-tokens", type=_positive_int) - run.add_argument("--no-seed", action="store_true") - run.add_argument("--generation-seed", type=int, default=42) + seed = run.add_mutually_exclusive_group() + seed.add_argument("--no-seed", action="store_true") + seed.add_argument("--generation-seed", type=int) run.add_argument("--concurrency", type=int, choices=(1, 2, 3, 4), default=1) run.add_argument("--input-price-per-million", type=_price) run.add_argument("--output-price-per-million", type=_price) @@ -289,24 +301,43 @@ def _secret_environment( def _provider_defaults(args: argparse.Namespace) -> dict[str, Any]: + provider_type = ProviderType(args.provider_type) official_cot = args.dataset == "mmlu-pro" and args.profile == "official_cot" - temperature = args.temperature if args.temperature is not None else 0 + supports_protocol_sampling_defaults = provider_type == ProviderType.OPENAI_COMPATIBLE + temperature = ( + args.temperature + if args.temperature is not None or not supports_protocol_sampling_defaults + else 0 + ) + top_p = args.top_p if args.top_p is not None or not supports_protocol_sampling_defaults else 1 max_tokens = ( args.max_tokens if args.max_tokens is not None else (4000 if official_cot else 1024) ) + if provider_type != ProviderType.OPENAI_COMPATIBLE and args.generation_seed is not None: + raise EvaluationCLIError(f"{provider_type.value} does not support --generation-seed.") + seed = ( + None + if args.no_seed or provider_type != ProviderType.OPENAI_COMPATIBLE + else (42 if args.generation_seed is None else args.generation_seed) + ) candidate = EvaluationRunCreate( model_id="preflight-model", benchmark_id="preflight-benchmark", temperature=temperature, - top_p=args.top_p, + top_p=top_p, max_tokens=max_tokens, - seed=None if args.no_seed else args.generation_seed, + seed=seed, concurrency=args.concurrency, ) - return { + defaults = { field: getattr(candidate, field) for field in ("temperature", "top_p", "max_tokens", "seed", "concurrency") } + try: + validate_provider_generation_parameters(provider_type, defaults) + except ValueError as exc: + raise EvaluationCLIError(str(exc)) from None + return defaults async def _resolve_and_canary( @@ -321,6 +352,7 @@ async def _resolve_and_canary( run_attempts: int, yes: bool, before_canary: Callable[[str], None] | None = None, + provider_type: ProviderType = ProviderType.OPENAI_COMPATIBLE, ) -> tuple[str, ModelDiscoveryResult | None, CanaryResult]: if skip_discovery and not requested_model: raise EvaluationCLIError("--no-model-discovery requires --model.") @@ -332,7 +364,11 @@ async def _resolve_and_canary( discovery: ModelDiscoveryResult | None = None if not skip_discovery: try: - discovery = await discover_models(base_url, api_key) + discovery = await discover_models( + base_url, + api_key, + provider_type=provider_type, + ) except ProviderPreflightError as exc: if exc.code == "model_discovery_unsupported" and requested_model: print( @@ -356,13 +392,23 @@ async def _resolve_and_canary( question_count, run_attempts=run_attempts, yes=yes, + provider_type=provider_type, ) - canary = await run_chat_canary( - base_url, - remote_model, - api_key_env, - generation, - ) + if provider_type == ProviderType.OPENAI_COMPATIBLE: + canary = await run_chat_canary( + base_url, + remote_model, + api_key_env, + generation, + ) + else: + canary = await run_provider_canary( + provider_type, + base_url, + remote_model, + api_key_env, + generation, + ) return remote_model, discovery, canary @@ -373,18 +419,23 @@ def _confirm_real_calls( *, run_attempts: int, yes: bool, + provider_type: ProviderType = ProviderType.OPENAI_COMPATIBLE, ) -> None: if run_attempts < 1: raise EvaluationCLIError("The Run has no remaining execution attempts.") adapter_attempts = DEFAULT_MAX_RETRIES + 1 maximum_requests = (question_count * run_attempts + 1) * adapter_attempts + request_label = ( + "Chat Completion" if provider_type == ProviderType.OPENAI_COMPATIBLE else "Provider API" + ) host = urlsplit(base_url).hostname or "unknown-host" message = ( f"Provider host: {host}\n" f"Remote model: {remote_model}\n" f"Scored questions: {question_count}\n" f"Remaining Run attempts included in upper bound: {run_attempts}\n" - f"Maximum billed Chat Completion attempts this invocation: {maximum_requests} " + f"Provider protocol: {provider_type.value}\n" + f"Maximum billed {request_label} attempts this invocation: {maximum_requests} " f"(one canary plus up to {adapter_attempts} HTTP attempts per question per Run attempt)\n" "Pricing or usage may be unknown; LLMBenchLab cannot enforce a global Provider " "budget yet.\n" @@ -419,10 +470,11 @@ def _model_payload( display_name: str | None, input_price: Decimal | None, output_price: Decimal | None, + provider_type: ProviderType = ProviderType.OPENAI_COMPATIBLE, ) -> ModelCreate: return ModelCreate( name=_model_name(base_url, remote_model, display_name), - provider_type=ProviderType.OPENAI_COMPATIBLE, + provider_type=provider_type, base_url=base_url, remote_model_name=remote_model, api_key_env=api_key_env, @@ -495,6 +547,7 @@ def _find_or_create_model( display_name: str | None, input_price: Decimal | None, output_price: Decimal | None, + provider_type: ProviderType = ProviderType.OPENAI_COMPATIBLE, ) -> tuple[Model, bool]: payload = _model_payload( base_url=base_url, @@ -503,6 +556,7 @@ def _find_or_create_model( display_name=display_name, input_price=input_price, output_price=output_price, + provider_type=provider_type, ) existing, payload = _resolve_model_registration(session, payload) if existing is not None: @@ -524,6 +578,7 @@ def _validate_model_registration( display_name: str | None, input_price: Decimal | None, output_price: Decimal | None, + provider_type: ProviderType = ProviderType.OPENAI_COMPATIBLE, ) -> None: payload = _model_payload( base_url=base_url, @@ -532,6 +587,7 @@ def _validate_model_registration( display_name=display_name, input_price=input_price, output_price=output_price, + provider_type=provider_type, ) with SessionLocal() as session: _resolve_model_registration(session, payload) @@ -540,7 +596,24 @@ def _validate_model_registration( def _preflight_snapshot( discovery: ModelDiscoveryResult | None, canary: CanaryResult, + provider_type: ProviderType = ProviderType.OPENAI_COMPATIBLE, ) -> dict[str, Any]: + canary_evidence = { + "status": "passed", + "model": canary.model, + "returned_model": canary.returned_model, + "system_fingerprint": canary.system_fingerprint, + "finish_reason": canary.finish_reason, + "provider_request_id": canary.provider_request_id, + "input_tokens": canary.input_tokens, + "output_tokens": canary.output_tokens, + "latency_ms": canary.latency_ms, + "attempts": canary.attempts, + } + canary_key = "chat_canary" + if provider_type != ProviderType.OPENAI_COMPATIBLE: + canary_key = "provider_canary" + canary_evidence["provider_type"] = provider_type.value return { "performed_at": utc_now().isoformat(), "model_discovery": { @@ -548,18 +621,7 @@ def _preflight_snapshot( "model_count": len(discovery.models) if discovery is not None else None, "request_id": discovery.request_id if discovery is not None else None, }, - "chat_canary": { - "status": "passed", - "model": canary.model, - "returned_model": canary.returned_model, - "system_fingerprint": canary.system_fingerprint, - "finish_reason": canary.finish_reason, - "provider_request_id": canary.provider_request_id, - "input_tokens": canary.input_tokens, - "output_tokens": canary.output_tokens, - "latency_ms": canary.latency_ms, - "attempts": canary.attempts, - }, + canary_key: canary_evidence, } @@ -595,6 +657,7 @@ def _create_persisted_run( generation: dict[str, Any], discovery: ModelDiscoveryResult | None, canary: CanaryResult, + provider_type: ProviderType = ProviderType.OPENAI_COMPATIBLE, ) -> EvaluationRun: with SessionLocal() as session: _raise_if_run_is_active(session) @@ -607,6 +670,7 @@ def _create_persisted_run( display_name=display_name, input_price=input_price, output_price=output_price, + provider_type=provider_type, ) run_payload = EvaluationRunCreate( model_id=model.id, @@ -619,7 +683,7 @@ def _create_persisted_run( ) run = build_evaluation_run(model, benchmark, run_payload, settings) snapshot = dict(run.model_parameters_snapshot) - snapshot["preflight"] = _preflight_snapshot(discovery, canary) + snapshot["preflight"] = _preflight_snapshot(discovery, canary, provider_type) preparation = _prepared_summary(prepared) preparation.pop("archive_path", None) snapshot["dataset_preparation"] = preparation @@ -639,16 +703,25 @@ def _load_run(run_id: str) -> EvaluationRun: return run -def _run_provider_configuration(run: EvaluationRun) -> tuple[str, str, str, dict[str, Any]]: +def _run_provider_configuration( + run: EvaluationRun, +) -> tuple[ProviderType, str, str, str, dict[str, Any]]: snapshot = dict(run.model_parameters_snapshot or {}) model = dict(snapshot.get("model", {})) generation = dict(snapshot.get("generation", {})) base_url = model.get("base_url") remote_model = model.get("remote_model_name") api_key_env = model.get("api_key_env") + adapter_type = model.get("adapter_type", ProviderType.OPENAI_COMPATIBLE.value) if not all(isinstance(value, str) and value for value in (base_url, remote_model, api_key_env)): raise EvaluationCLIError("Run does not contain a complete compatible Provider snapshot.") - return str(base_url), str(remote_model), str(api_key_env), generation + try: + provider_type = ProviderType(str(adapter_type)) + except ValueError as exc: + raise EvaluationCLIError("Run contains an unsupported Provider protocol snapshot.") from exc + if provider_type == ProviderType.MOCK: + raise EvaluationCLIError("Run does not contain a remote Provider protocol snapshot.") + return provider_type, str(base_url), str(remote_model), str(api_key_env), generation async def _execute_with_progress( @@ -760,9 +833,10 @@ async def _run_new(args: argparse.Namespace, settings: Settings) -> int: prepared = await asyncio.to_thread(_prepare_dataset, args) prepared_summary = _prepared_summary(prepared) generation = _provider_defaults(args) + provider_type = ProviderType(args.provider_type) validated_model = ModelCreate( name="validation-only", - provider_type=ProviderType.OPENAI_COMPATIBLE, + provider_type=provider_type, base_url=args.base_url, remote_model_name=args.model or "pending-discovery", api_key_env=args.api_key_env, @@ -779,6 +853,7 @@ async def _run_new(args: argparse.Namespace, settings: Settings) -> int: question_count=prepared_summary["question_count"], run_attempts=settings.worker_max_attempts, yes=args.yes, + provider_type=provider_type, before_canary=lambda remote_model: _validate_model_registration( base_url=base_url, remote_model=remote_model, @@ -786,6 +861,7 @@ async def _run_new(args: argparse.Namespace, settings: Settings) -> int: display_name=args.display_name, input_price=args.input_price_per_million, output_price=args.output_price_per_million, + provider_type=provider_type, ), ) run = _create_persisted_run( @@ -800,6 +876,7 @@ async def _run_new(args: argparse.Namespace, settings: Settings) -> int: generation=generation, discovery=discovery, canary=canary, + provider_type=provider_type, ) print(f"Run created: {run.id}", file=sys.stderr) terminal = await _drive_run(run.id, settings) @@ -851,9 +928,9 @@ async def _resume(args: argparse.Namespace, settings: Settings) -> int: ) ) return 0 - base_url, remote_model, target_env, generation = _run_provider_configuration(run) + provider_type, base_url, remote_model, target_env, generation = _run_provider_configuration(run) missing = max(0, run.total_questions - run.completed_questions) - remaining_attempts = run.max_attempts - run.attempt_count + remaining_attempts = run.max_attempts - run.failed_attempt_count if remaining_attempts < 1: raise EvaluationCLIError( f"Run {run.id} has exhausted all {run.max_attempts} execution attempts. " @@ -870,6 +947,7 @@ async def _resume(args: argparse.Namespace, settings: Settings) -> int: question_count=missing, run_attempts=remaining_attempts, yes=args.yes, + provider_type=provider_type, ) terminal = await _drive_run(run.id, settings) if terminal.status in TERMINAL_STATUSES: diff --git a/backend/app/core/logging_contract.py b/backend/app/core/logging_contract.py index d6e2ff0..c853a64 100644 --- a/backend/app/core/logging_contract.py +++ b/backend/app/core/logging_contract.py @@ -113,6 +113,8 @@ "/runs/{run_id}", "/runs/{run_id}/audit", "/runs/{run_id}/cancel", + "/runs/{run_id}/progress", + "/runs/{run_id}/progress/blocks/{block_index}", "/runs/{run_id}/responses", "/tasks/history", "/tasks/metrics", diff --git a/backend/app/db/prepare_migrations.py b/backend/app/db/prepare_migrations.py index 79b14b8..a2cca05 100644 --- a/backend/app/db/prepare_migrations.py +++ b/backend/app/db/prepare_migrations.py @@ -33,6 +33,7 @@ GOVERNANCE_REVISION = "20260827_0004" WORKER_PROGRESS_REVISION = "20260828_0005" INDEX_REPAIR_REVISION = "20260829_0006" +OBSERVATIONAL_OVERDRAW_REVISION = "20260830_0007" _WORKER_PROGRESS_TABLES = {"worker_processes"} _WORKER_PROGRESS_INDEXES = { @@ -63,6 +64,7 @@ _KNOWN_GOVERNANCE_MISSING_INDEX_DIFFERENCES = frozenset( _KNOWN_GOVERNANCE_MISSING_INDEX_DIFFERENCE_BY_NAME.values() ) +_PROVIDER_PROTOCOL_TYPE_DIFFERENCE = "modify_type:models.provider_type:VARCHAR(17)->VARCHAR(18)" _WEB_CREDENTIAL_DIFFERENCES = { "add_table:model_credentials", @@ -145,6 +147,7 @@ "remove_constraint:evaluation_runs.ck_evaluation_runs_attempt_within_limit", } | _WORKER_PROGRESS_DIFFERENCE_SET + | {_PROVIDER_PROTOCOL_TYPE_DIFFERENCE} ) _GOVERNANCE_DIFFERENCES = tuple(sorted(_GOVERNANCE_DIFFERENCE_SET)) _GOVERNANCE_CHECK_NAMES_BY_TABLE = { @@ -310,6 +313,10 @@ def _difference_fingerprint(difference: tuple[Any, ...]) -> str: operation = str(difference[0]) if operation == "modify_nullable": return f"modify_nullable:{difference[2]}.{difference[3]}:{difference[5]}->{difference[6]}" + if operation == "modify_type": + existing_type = "".join(str(difference[5]).upper().split()) + target_type = "".join(str(difference[6]).upper().split()) + return f"modify_type:{difference[2]}.{difference[3]}:{existing_type}->{target_type}" if operation == "add_column": column = difference[3] return f"add_column:{difference[2]}.{column.name}" @@ -664,6 +671,16 @@ def _validate_check_constraints(connection: Connection, *, schema_revision: str) RELIABILITY_REVISION, CREDENTIAL_REVISION, } + pre_provider_protocols = schema_revision in { + LEGACY_REVISION, + PHASE_1_REVISION, + RELIABILITY_REVISION, + CREDENTIAL_REVISION, + GOVERNANCE_REVISION, + WORKER_PROGRESS_REVISION, + INDEX_REPAIR_REVISION, + OBSERVATIONAL_OVERDRAW_REVISION, + } for table_name, table in _schema_tables_for_revision(schema_revision).items(): expected_entries = [ (str(constraint.name), _normalized_check_sql(constraint.sqltext)) @@ -706,6 +723,37 @@ def _validate_check_constraints(connection: Connection, *, schema_revision: str) ), ] ) + if pre_provider_protocols and table_name == "models": + expected_entries = [ + entry + for entry in expected_entries + if entry[0] + not in { + "ck_models_provider_type_values", + "ck_models_openai_configuration_required", + } + ] + expected_entries.append( + ( + "ck_models_provider_type_values", + "provider_type IN ('mock', 'openai_compatible')", + ) + ) + if not legacy: + expected_entries.append( + ( + "ck_models_openai_configuration_required", + ( + "provider_type != 'openai_compatible' OR (base_url IS NOT NULL " + "AND remote_model_name IS NOT NULL AND api_key_env IS NOT NULL)" + if pre_credentials + else "provider_type != 'openai_compatible' OR (base_url IS NOT NULL " + "AND remote_model_name IS NOT NULL AND ((credential_source = " + "'environment' AND api_key_env IS NOT NULL) OR " + "(credential_source = 'stored' AND api_key_env IS NULL)))" + ), + ) + ) if pre_governance and table_name in _GOVERNANCE_CHECK_NAMES_BY_TABLE: expected_entries = [ entry @@ -874,6 +922,7 @@ def prepare_database( GOVERNANCE_REVISION, WORKER_PROGRESS_REVISION, INDEX_REPAIR_REVISION, + OBSERVATIONAL_OVERDRAW_REVISION, ) historical_heads = {(revision,) for revision in historical_revisions} @@ -924,6 +973,7 @@ def prepare_database( if not sqlite_locked and source_revision not in { WORKER_PROGRESS_REVISION, INDEX_REPAIR_REVISION, + OBSERVATIONAL_OVERDRAW_REVISION, }: return PreparationResult(action="versioned") expected_differences = { @@ -931,9 +981,12 @@ def prepare_database( PHASE_1_REVISION: _PHASE_1_DIFFERENCES, RELIABILITY_REVISION: _RELIABILITY_DIFFERENCES, CREDENTIAL_REVISION: _CREDENTIAL_DIFFERENCES, - GOVERNANCE_REVISION: _WORKER_PROGRESS_DIFFERENCES, - WORKER_PROGRESS_REVISION: (), - INDEX_REPAIR_REVISION: (), + GOVERNANCE_REVISION: tuple( + sorted((*_WORKER_PROGRESS_DIFFERENCES, _PROVIDER_PROTOCOL_TYPE_DIFFERENCE)) + ), + WORKER_PROGRESS_REVISION: (_PROVIDER_PROTOCOL_TYPE_DIFFERENCE,), + INDEX_REPAIR_REVISION: (_PROVIDER_PROTOCOL_TYPE_DIFFERENCE,), + OBSERVATIONAL_OVERDRAW_REVISION: (_PROVIDER_PROTOCOL_TYPE_DIFFERENCE,), }[source_revision] differences = _schema_differences(connection) expected_missing_index_names: frozenset[str] = frozenset() @@ -982,6 +1035,7 @@ def prepare_database( GOVERNANCE_REVISION: "versioned_governance", WORKER_PROGRESS_REVISION: "versioned_worker_progress", INDEX_REPAIR_REVISION: "versioned_index_repair", + OBSERVATIONAL_OVERDRAW_REVISION: "versioned_observational_overdraw", }[source_revision] else: if not sqlite_locked: @@ -991,6 +1045,8 @@ def prepare_database( ) differences = _schema_differences(connection) if not differences: + # Even a metadata-current unversioned schema must pass through + # the idempotent 0007 data repair before reaching the head. target_revision = INDEX_REPAIR_REVISION action = "stamped_index_repair" source_revision = INDEX_REPAIR_REVISION @@ -1006,10 +1062,18 @@ def prepare_database( target_revision = CREDENTIAL_REVISION action = "stamped_credentials" source_revision = CREDENTIAL_REVISION - elif differences == _WORKER_PROGRESS_DIFFERENCES: + elif differences == tuple( + sorted((*_WORKER_PROGRESS_DIFFERENCES, _PROVIDER_PROTOCOL_TYPE_DIFFERENCE)) + ): target_revision = GOVERNANCE_REVISION action = "stamped_governance" source_revision = GOVERNANCE_REVISION + elif differences == (_PROVIDER_PROTOCOL_TYPE_DIFFERENCE,): + # Re-run the idempotent 0007 data repair before widening the + # provider type when an unversioned 0005-0007 schema is found. + target_revision = INDEX_REPAIR_REVISION + action = "stamped_index_repair" + source_revision = INDEX_REPAIR_REVISION elif differences == _LEGACY_DIFFERENCES: target_revision = LEGACY_REVISION action = "stamped_legacy" @@ -1021,7 +1085,13 @@ def prepare_database( "schema; no migration marker was written. Differences: " f"{rendered_differences}" ) - _validate_sqlite_database(connection, schema_revision=source_revision) + _validate_sqlite_database( + connection, + # A metadata-current schema is intentionally stamped at 0006 + # only so the idempotent 0007 data repair still executes. + # Validate its actual DDL against the real current head. + schema_revision=head_revision if not differences else source_revision, + ) backup_path = _backup_sqlite_database(database_url) if target_revision is not None: diff --git a/backend/app/governance/audit_archive.py b/backend/app/governance/audit_archive.py index 1543e8e..339e898 100644 --- a/backend/app/governance/audit_archive.py +++ b/backend/app/governance/audit_archive.py @@ -35,6 +35,7 @@ "20260828_0005", "20260829_0006", "20260830_0007", + "20260830_0008", } ) ARCHIVE_V1_RETENTION_VALUES = ("operational", "security") diff --git a/backend/app/models/enums.py b/backend/app/models/enums.py index 1e2a640..5953f0d 100644 --- a/backend/app/models/enums.py +++ b/backend/app/models/enums.py @@ -6,6 +6,8 @@ class ProviderType(StrEnum): MOCK = "mock" OPENAI_COMPATIBLE = "openai_compatible" + OPENAI_RESPONSES = "openai_responses" + ANTHROPIC_MESSAGES = "anthropic_messages" class CredentialSource(StrEnum): diff --git a/backend/app/models/model.py b/backend/app/models/model.py index 109f37e..f819ecc 100644 --- a/backend/app/models/model.py +++ b/backend/app/models/model.py @@ -39,14 +39,17 @@ class Model(TimestampMixin, Base): __table_args__ = ( UniqueConstraint("name", name="uq_models_name"), CheckConstraint( - "provider_type IN ('mock', 'openai_compatible')", name="provider_type_values" + "provider_type IN ('mock', 'openai_compatible', 'openai_responses', " + "'anthropic_messages')", + name="provider_type_values", ), CheckConstraint( "credential_source IN ('none', 'environment', 'stored')", name="credential_source_values", ), CheckConstraint( - "provider_type != 'openai_compatible' OR " + "provider_type NOT IN " + "('openai_compatible', 'openai_responses', 'anthropic_messages') OR " "(base_url IS NOT NULL AND remote_model_name IS NOT NULL AND " "((credential_source = 'environment' AND api_key_env IS NOT NULL) OR " "(credential_source = 'stored' AND api_key_env IS NULL)))", diff --git a/backend/app/providers/__init__.py b/backend/app/providers/__init__.py index 0b8be3c..b7556fc 100644 --- a/backend/app/providers/__init__.py +++ b/backend/app/providers/__init__.py @@ -7,6 +7,7 @@ discover_models, models_url, run_chat_canary, + run_provider_canary, select_remote_model, ) @@ -17,5 +18,6 @@ "discover_models", "models_url", "run_chat_canary", + "run_provider_canary", "select_remote_model", ] diff --git a/backend/app/providers/preflight.py b/backend/app/providers/preflight.py index ad7434c..6569c7d 100644 --- a/backend/app/providers/preflight.py +++ b/backend/app/providers/preflight.py @@ -1,4 +1,4 @@ -"""Secret-safe discovery and a minimal billed Chat Completions canary. +"""Secret-safe discovery and a minimal billed Provider-protocol canary. These helpers are intentionally used only by the trusted-local CLI. The API key is supplied by the caller and is never returned, logged, or persisted. @@ -9,6 +9,7 @@ import ipaddress import json import re +import time from collections.abc import Mapping, Sequence from dataclasses import dataclass from typing import Any @@ -16,13 +17,17 @@ import httpx -from app.adapters import AdapterError, OpenAICompatibleAdapter, sanitize_error_message +from app.adapters import AdapterError, build_adapter, sanitize_error_message from app.evaluators import get_evaluator +from app.models import ProviderType MAX_DISCOVERY_RESPONSE_BYTES = 2 * 1024 * 1024 MAX_DISCOVERED_MODELS = 10_000 +MAX_DISCOVERY_PAGES = 100 +MAX_DISCOVERY_WALL_SECONDS = 60 _DISCOVERY_READ_CHUNK_BYTES = 64 * 1024 _MODEL_ID_RE = re.compile(r"^[^\x00-\x1f\x7f]{1,256}$") +_KNOWN_ENDPOINT_SUFFIXES = ("/chat/completions", "/responses", "/messages") class ProviderPreflightError(RuntimeError): @@ -96,9 +101,10 @@ def models_url(base_url: str) -> str: """Return the sibling ``/models`` URL for a compatible endpoint.""" scheme, netloc, path, query, fragment = _validated_base_url(base_url) - suffix = "/chat/completions" - if path.endswith(suffix): - path = path[: -len(suffix)] + for suffix in _KNOWN_ENDPOINT_SUFFIXES: + if path.endswith(suffix): + path = path[: -len(suffix)] + break return urlunsplit((scheme, netloc, f"{path}/models", query, fragment)) @@ -131,13 +137,46 @@ async def discover_models( base_url: str, api_key: str, *, + provider_type: ProviderType | str = ProviderType.OPENAI_COMPATIBLE, client: httpx.AsyncClient | None = None, ) -> ModelDiscoveryResult: - """Authenticate against ``GET /models`` and return bounded model IDs.""" + """Authenticate against ``GET /models`` and return bounded model IDs. + + OpenAI-style protocols use bearer authentication. Anthropic Messages uses + the same ``x-api-key`` and version headers as its generation endpoint and + follows the native ``after_id`` cursor when the Provider paginates models. + """ if not api_key: raise ProviderPreflightError("missing_api_key", "API key is empty.") + try: + normalized_provider = ( + provider_type + if isinstance(provider_type, ProviderType) + else ProviderType(provider_type) + ) + except ValueError: + raise ProviderPreflightError( + "invalid_provider_type", "Model discovery requires a known remote protocol." + ) from None + if normalized_provider == ProviderType.MOCK: + raise ProviderPreflightError( + "invalid_provider_type", "Model discovery is unavailable for the mock Provider." + ) endpoint = models_url(base_url) + headers = { + "Accept": "application/json", + "Accept-Encoding": "identity", + } + if normalized_provider == ProviderType.ANTHROPIC_MESSAGES: + headers.update( + { + "x-api-key": api_key, + "anthropic-version": "2023-06-01", + } + ) + else: + headers["Authorization"] = f"Bearer {api_key}" owns_client = client is None active_client = client or httpx.AsyncClient( timeout=httpx.Timeout(connect=5, read=30, write=10, pool=5), @@ -145,114 +184,171 @@ async def discover_models( trust_env=False, ) try: - try: - async with active_client.stream( - "GET", - endpoint, - headers={ - "Authorization": f"Bearer {api_key}", - "Accept": "application/json", - "Accept-Encoding": "identity", - }, - ) as response: - content_encoding = response.headers.get("content-encoding", "identity") - if content_encoding.strip().lower() not in {"", "identity"}: - raise ProviderPreflightError( - "unsupported_model_discovery_response_encoding", - "Model discovery returned a compressed response despite the " - "identity-only request.", - status_code=response.status_code, - ) - content_length = response.headers.get("content-length") - if content_length is not None: - try: - declared_length = int(content_length) - except ValueError: - declared_length = None - if ( - declared_length is not None - and declared_length > MAX_DISCOVERY_RESPONSE_BYTES - ): - raise ProviderPreflightError( - "model_discovery_response_too_large", - "Model discovery response exceeds the 2 MiB safety limit.", - ) - content = bytearray() - if response.is_stream_consumed: - buffered = response.content - if len(buffered) > MAX_DISCOVERY_RESPONSE_BYTES: + model_ids: set[str] = set() + request_id: str | None = None + after_id: str | None = None + seen_cursors: set[str] = set() + total_entries = 0 + total_response_bytes = 0 + page_count = 0 + discovery_started = time.monotonic() + while True: + if page_count >= MAX_DISCOVERY_PAGES: + raise ProviderPreflightError( + "model_discovery_too_many_pages", + f"Model discovery exceeded the {MAX_DISCOVERY_PAGES}-page safety limit.", + ) + if time.monotonic() - discovery_started >= MAX_DISCOVERY_WALL_SECONDS: + raise ProviderPreflightError( + "model_discovery_deadline_exceeded", + f"Model discovery exceeded the {MAX_DISCOVERY_WALL_SECONDS}-second " + "wall-clock safety limit.", + ) + page_count += 1 + network_error: str | None = None + try: + async with active_client.stream( + "GET", + endpoint, + headers=headers, + params={"after_id": after_id} if after_id is not None else None, + ) as response: + content_encoding = response.headers.get("content-encoding", "identity") + if content_encoding.strip().lower() not in {"", "identity"}: raise ProviderPreflightError( - "model_discovery_response_too_large", - "Model discovery response exceeds the 2 MiB safety limit.", + "unsupported_model_discovery_response_encoding", + "Model discovery returned a compressed response despite the " + "identity-only request.", + status_code=response.status_code, ) - content.extend(buffered) - else: - async for chunk in response.aiter_raw(chunk_size=_DISCOVERY_READ_CHUNK_BYTES): - if len(content) + len(chunk) > MAX_DISCOVERY_RESPONSE_BYTES: + content_length = response.headers.get("content-length") + if content_length is not None: + try: + declared_length = int(content_length) + except ValueError: + declared_length = None + if ( + declared_length is not None + and total_response_bytes + declared_length + > MAX_DISCOVERY_RESPONSE_BYTES + ): raise ProviderPreflightError( "model_discovery_response_too_large", "Model discovery response exceeds the 2 MiB safety limit.", ) - content.extend(chunk) - response_content = bytes(content) - except ProviderPreflightError: - raise - except httpx.TransportError as exc: - raise ProviderPreflightError( - "model_discovery_network_error", - "Model discovery request failed: " + sanitize_error_message(exc, api_key), - ) from exc - if response.status_code >= 400: - code = ( - "model_discovery_authentication_error" - if response.status_code in {401, 403} - else "model_discovery_unsupported" - if response.status_code in {404, 405} - else "model_discovery_http_error" - ) - raise ProviderPreflightError( - code, - f"Model discovery returned HTTP {response.status_code}: " - f"{_safe_error_detail(response, api_key, response_content)}", - status_code=response.status_code, - ) - try: - body = json.loads(response_content) - except (ValueError, UnicodeError) as exc: - raise ProviderPreflightError( - "invalid_model_discovery_response", - "Model discovery returned a non-JSON response.", - ) from exc - if not isinstance(body, Mapping) or not isinstance(body.get("data"), list): - raise ProviderPreflightError( - "invalid_model_discovery_response", - "Model discovery response must contain a data array.", - ) - raw_models = body["data"] - if len(raw_models) > MAX_DISCOVERED_MODELS: - raise ProviderPreflightError( - "too_many_discovered_models", - f"Model discovery returned more than {MAX_DISCOVERED_MODELS} entries.", - ) - model_ids: set[str] = set() - for item in raw_models: - if not isinstance(item, Mapping): - continue - model_id = item.get("id") - if isinstance(model_id, str) and _MODEL_ID_RE.fullmatch(model_id): + content = bytearray() + if response.is_stream_consumed: + buffered = response.content + if total_response_bytes + len(buffered) > MAX_DISCOVERY_RESPONSE_BYTES: + raise ProviderPreflightError( + "model_discovery_response_too_large", + "Model discovery response exceeds the 2 MiB safety limit.", + ) + content.extend(buffered) + else: + async for chunk in response.aiter_raw( + chunk_size=_DISCOVERY_READ_CHUNK_BYTES + ): + if ( + total_response_bytes + len(content) + len(chunk) + > MAX_DISCOVERY_RESPONSE_BYTES + ): + raise ProviderPreflightError( + "model_discovery_response_too_large", + "Model discovery response exceeds the 2 MiB safety limit.", + ) + content.extend(chunk) + response_content = bytes(content) + except ProviderPreflightError: + raise + except httpx.TransportError as exc: + network_error = sanitize_error_message(exc, api_key) + if network_error is not None: + raise ProviderPreflightError( + "model_discovery_network_error", + "Model discovery request failed: " + network_error, + ) + + total_response_bytes += len(response_content) + if response.status_code >= 400: + code = ( + "model_discovery_authentication_error" + if response.status_code in {401, 403} + else "model_discovery_unsupported" + if response.status_code in {404, 405} + else "model_discovery_http_error" + ) + raise ProviderPreflightError( + code, + f"Model discovery returned HTTP {response.status_code}: " + f"{_safe_error_detail(response, api_key, response_content)}", + status_code=response.status_code, + ) + invalid_json = False + try: + body = json.loads(response_content) + except (ValueError, UnicodeError): + invalid_json = True + body = None + if invalid_json: + raise ProviderPreflightError( + "invalid_model_discovery_response", + "Model discovery returned a non-JSON response.", + ) + if not isinstance(body, Mapping) or not isinstance(body.get("data"), list): + raise ProviderPreflightError( + "invalid_model_discovery_response", + "Model discovery response must contain a data array.", + ) + if request_id is None: + request_id = _safe_request_id(response, api_key) + raw_models = body["data"] + total_entries += len(raw_models) + if total_entries > MAX_DISCOVERED_MODELS: + raise ProviderPreflightError( + "too_many_discovered_models", + f"Model discovery returned more than {MAX_DISCOVERED_MODELS} entries.", + ) + for item in raw_models: + if not isinstance(item, Mapping): + continue + model_id = item.get("id") + if not isinstance(model_id, str) or not _MODEL_ID_RE.fullmatch(model_id): + continue if api_key in model_id: raise ProviderPreflightError( "model_discovery_secret_reflection", "Model discovery returned an unsafe model ID.", ) model_ids.add(model_id) + + has_more = body.get("has_more", False) + if not isinstance(has_more, bool): + raise ProviderPreflightError( + "invalid_model_discovery_response", + "Model discovery has_more must be a boolean.", + ) + if normalized_provider != ProviderType.ANTHROPIC_MESSAGES or not has_more: + break + last_id = body.get("last_id") + if ( + not isinstance(last_id, str) + or not _MODEL_ID_RE.fullmatch(last_id) + or last_id in seen_cursors + or api_key in last_id + ): + raise ProviderPreflightError( + "invalid_model_discovery_response", + "Paginated model discovery returned an unsafe or repeated cursor.", + ) + seen_cursors.add(last_id) + after_id = last_id + if not model_ids: raise ProviderPreflightError( "no_models_discovered", "Model discovery returned no usable model IDs." ) - return ModelDiscoveryResult( - models=tuple(sorted(model_ids)), request_id=_safe_request_id(response, api_key) - ) + return ModelDiscoveryResult(models=tuple(sorted(model_ids)), request_id=request_id) finally: if owns_client: await active_client.aclose() @@ -293,7 +389,8 @@ def select_remote_model( ) -async def run_chat_canary( +async def run_provider_canary( + provider_type: ProviderType | str, base_url: str, remote_model_name: str, api_key_env: str, @@ -301,13 +398,17 @@ async def run_chat_canary( *, client: httpx.AsyncClient | None = None, ) -> CanaryResult: - """Make one minimal billed request using the same request fields as a Run.""" + """Make one minimal billed request with the Run's explicit Adapter type.""" _validated_base_url(base_url) - adapter = OpenAICompatibleAdapter( - base_url, - remote_model_name, - api_key_env, + normalized_provider = ( + provider_type.value if isinstance(provider_type, ProviderType) else str(provider_type) + ) + adapter = build_adapter( + normalized_provider, + base_url=base_url, + remote_model_name=remote_model_name, + api_key_env=api_key_env, client=client, ) config = { @@ -315,20 +416,31 @@ async def run_chat_canary( for key in ("temperature", "top_p", "seed") if generation_config.get(key) is not None } - config["max_tokens"] = min(int(generation_config.get("max_tokens", 16)), 16) + configured_max_tokens = generation_config.get("max_tokens") + config["max_tokens"] = ( + 16 if configured_max_tokens is None else min(int(configured_max_tokens), 16) + ) + canary_error: ProviderPreflightError | None = None try: result = await adapter.generate( [{"role": "user", "content": "Compatibility check: reply with exactly A."}], config, ) except AdapterError as exc: - raise ProviderPreflightError( + canary_label = { + ProviderType.OPENAI_COMPATIBLE.value: "Chat Completions", + ProviderType.OPENAI_RESPONSES.value: "OpenAI Responses", + ProviderType.ANTHROPIC_MESSAGES.value: "Anthropic Messages", + }.get(normalized_provider, "Provider") + canary_error = ProviderPreflightError( f"canary_{exc.error_type}", - f"Chat Completions canary failed: {exc.error_message}", + f"{canary_label} canary failed: {exc.error_message}", status_code=exc.status_code, - ) from exc + ) finally: await adapter.aclose() + if canary_error is not None: + raise canary_error returned_model = ( str(result.metadata["returned_model"]) @@ -371,3 +483,23 @@ async def run_chat_canary( latency_ms=float(result.latency_ms), attempts=int(attempts) if isinstance(attempts, int) else 1, ) + + +async def run_chat_canary( + base_url: str, + remote_model_name: str, + api_key_env: str, + generation_config: Mapping[str, Any], + *, + client: httpx.AsyncClient | None = None, +) -> CanaryResult: + """Backward-compatible Chat Completions canary entry point.""" + + return await run_provider_canary( + ProviderType.OPENAI_COMPATIBLE, + base_url, + remote_model_name, + api_key_env, + generation_config, + client=client, + ) diff --git a/backend/app/runners/evaluation_runner.py b/backend/app/runners/evaluation_runner.py index fd2bdbe..36341cd 100644 --- a/backend/app/runners/evaluation_runner.py +++ b/backend/app/runners/evaluation_runner.py @@ -65,6 +65,7 @@ logger = logging.getLogger(__name__) _OPAQUE_PROVIDER_SCOPE_PATTERN = re.compile(r"[0-9a-f]{64}") +_REMOTE_PROVIDER_TYPES = frozenset({"openai_compatible", "openai_responses", "anthropic_messages"}) @dataclass(frozen=True, slots=True) @@ -776,7 +777,7 @@ def _load_snapshots( or model_snapshot.api_key_env is not None ): raise ValueError("run_credential_snapshot_invalid") - elif model_snapshot.provider_type == "openai_compatible": + elif model_snapshot.provider_type in _REMOTE_PROVIDER_TYPES: if model_snapshot.base_url is None or model_snapshot.remote_model_name is None: raise ValueError("run_model_snapshot_incomplete") if credential_source == "environment": @@ -1142,7 +1143,7 @@ def _parse_error_evidence( if parse_error is None: return None, None - if generation_metadata.get("finish_reason") == "length": + if generation_metadata.get("finish_reason") in {"length", "max_tokens"}: return ( "output_truncated", "Provider stopped at the output token limit before a valid final " diff --git a/backend/app/runners/run_leases.py b/backend/app/runners/run_leases.py index 26181e5..0b22d5f 100644 --- a/backend/app/runners/run_leases.py +++ b/backend/app/runners/run_leases.py @@ -26,6 +26,7 @@ QuestionExecution, RunStatus, ) +from app.services.run_evidence import canonical_run_evidence DatabaseClock = Callable[[Session], datetime] _GOVERNANCE_INTEGRITY_REASON = "governance_integrity_error" @@ -145,31 +146,32 @@ def aggregate_run_evidence(session: Session, run: EvaluationRun) -> int: cost, cost_reports, ) = aggregate - planned = run.total_questions completed_response_count = int(response_count or 0) - correct = round(float(score_sum or 0)) - run.completed_questions = completed_response_count - run.correct_questions = correct - run.error_questions = int(errors or 0) - run.score = (float(score_sum or 0) / planned * 100) if planned else 0.0 - run.completion_rate = (int(completed_outputs or 0) / planned * 100) if planned else 0.0 - run.answered_accuracy = (correct / int(evaluable) * 100) if int(evaluable or 0) else None - run.average_latency_ms = float(avg_latency) if avg_latency is not None else None - run.input_tokens = ( - int(in_tok or 0) - if completed_response_count and int(in_reports or 0) == completed_response_count - else None - ) - run.output_tokens = ( - int(out_tok or 0) - if completed_response_count and int(out_reports or 0) == completed_response_count - else None - ) - run.estimated_cost = ( - Decimal(cost or 0) - if completed_response_count and int(cost_reports or 0) == completed_response_count - else None + metrics = canonical_run_evidence( + planned_questions=run.total_questions, + response_count=completed_response_count, + score_sum=float(score_sum or 0), + completed_outputs=int(completed_outputs or 0), + evaluable_responses=int(evaluable or 0), + error_responses=int(errors or 0), + average_latency_ms=float(avg_latency) if avg_latency is not None else None, + known_input_tokens=int(in_tok or 0), + input_token_reports=int(in_reports or 0), + known_output_tokens=int(out_tok or 0), + output_token_reports=int(out_reports or 0), + known_estimated_cost=Decimal(cost or 0), + estimated_cost_reports=int(cost_reports or 0), ) + run.completed_questions = metrics.completed_questions + run.correct_questions = metrics.correct_questions + run.error_questions = metrics.error_questions + run.score = metrics.score + run.completion_rate = metrics.completion_rate + run.answered_accuracy = metrics.answered_accuracy + run.average_latency_ms = metrics.average_latency_ms + run.input_tokens = metrics.input_tokens + run.output_tokens = metrics.output_tokens + run.estimated_cost = metrics.estimated_cost return completed_response_count diff --git a/backend/app/schemas/__init__.py b/backend/app/schemas/__init__.py index 070af2c..fd7040b 100644 --- a/backend/app/schemas/__init__.py +++ b/backend/app/schemas/__init__.py @@ -3,6 +3,14 @@ from app.schemas.audit import AuditEventList, AuditEventRead from app.schemas.base import Pagination from app.schemas.benchmark import BenchmarkCreate, BenchmarkList, BenchmarkRead +from app.schemas.evaluation_progress import ( + PROGRESS_BLOCK_SIZE, + EvaluationProgressBlock, + EvaluationProgressBlockSummary, + EvaluationProgressCell, + EvaluationProgressIndex, + EvaluationProgressOutcome, +) from app.schemas.evaluation_response import ( EvaluationResponseDetail, EvaluationResponseList, @@ -25,12 +33,18 @@ ) __all__ = [ + "PROGRESS_BLOCK_SIZE", "AuditEventList", "AuditEventRead", "BenchmarkCreate", "BenchmarkList", "BenchmarkRead", "DashboardSummary", + "EvaluationProgressBlock", + "EvaluationProgressBlockSummary", + "EvaluationProgressCell", + "EvaluationProgressIndex", + "EvaluationProgressOutcome", "EvaluationResponseDetail", "EvaluationResponseList", "EvaluationResponseRead", diff --git a/backend/app/schemas/evaluation_progress.py b/backend/app/schemas/evaluation_progress.py new file mode 100644 index 0000000..2988d05 --- /dev/null +++ b/backend/app/schemas/evaluation_progress.py @@ -0,0 +1,66 @@ +"""Compact, body-free progress projections for one Evaluation Run.""" + +from enum import StrEnum +from typing import Literal + +from pydantic import Field + +from app.schemas.base import ORMModel + +PROGRESS_BLOCK_SIZE = 512 + + +class EvaluationProgressOutcome(StrEnum): + """Disjoint terminal outcome for one persisted Response.""" + + PASSED = "passed" + WRONG = "wrong" + ERROR = "error" + + +class EvaluationProgressBlockSummary(ORMModel): + """Number of persisted Responses currently present in one planned block.""" + + block_index: int = Field(ge=0) + response_count: int = Field(ge=0, le=PROGRESS_BLOCK_SIZE) + + +class EvaluationProgressIndex(ORMModel): + """Canonical live metrics plus a compact change index for fixed-size blocks.""" + + block_size: Literal[PROGRESS_BLOCK_SIZE] + total_questions: int = Field(ge=0) + completed_questions: int = Field(ge=0) + correct_questions: int = Field(ge=0) + error_questions: int = Field(ge=0) + score: float = Field(ge=0, le=100) + completion_rate: float = Field(ge=0, le=100) + answered_accuracy: float | None = Field(ge=0, le=100) + average_latency_ms: float | None = Field(ge=0) + known_input_tokens: int = Field(ge=0) + known_output_tokens: int = Field(ge=0) + input_token_reported_responses: int = Field(ge=0) + output_token_reported_responses: int = Field(ge=0) + known_estimated_cost: float = Field(ge=0) + estimated_cost_reported_responses: int = Field(ge=0) + blocks: list[EvaluationProgressBlockSummary] + + +class EvaluationProgressCell(ORMModel): + """Allowlisted tooltip facts for one completed absolute question position.""" + + position: int = Field(ge=0) + outcome: EvaluationProgressOutcome + score: float = Field(ge=0, le=1) + latency_ms: float | None = Field(ge=0) + input_tokens: int | None = Field(ge=0) + output_tokens: int | None = Field(ge=0) + estimated_cost: float | None = Field(ge=0) + error_type: str | None = Field(max_length=128) + + +class EvaluationProgressBlock(ORMModel): + """All persisted compact cells for one fixed absolute-position block.""" + + block_index: int = Field(ge=0) + items: list[EvaluationProgressCell] diff --git a/backend/app/schemas/evaluation_response.py b/backend/app/schemas/evaluation_response.py index f725fa5..717ef10 100644 --- a/backend/app/schemas/evaluation_response.py +++ b/backend/app/schemas/evaluation_response.py @@ -3,7 +3,7 @@ from datetime import datetime from typing import Any -from pydantic import field_validator +from pydantic import Field, field_validator from app.schemas.base import ORMModel from app.security import normalize_http_attempt_count, normalize_provider_metadata @@ -66,3 +66,19 @@ class EvaluationResponseList(ORMModel): total: int offset: int limit: int + known_input_tokens: int = Field( + ge=0, + description="Sum of non-null input tokens across every Response in the Run.", + ) + known_output_tokens: int = Field( + ge=0, + description="Sum of non-null output tokens across every Response in the Run.", + ) + input_token_reported_responses: int = Field( + ge=0, + description="Run-wide Response count with non-null input tokens.", + ) + output_token_reported_responses: int = Field( + ge=0, + description="Run-wide Response count with non-null output tokens.", + ) diff --git a/backend/app/schemas/evaluation_run.py b/backend/app/schemas/evaluation_run.py index 7a6c342..7aa1dfc 100644 --- a/backend/app/schemas/evaluation_run.py +++ b/backend/app/schemas/evaluation_run.py @@ -20,8 +20,8 @@ class EvaluationRunCreate(APIModel): model_id: str = Field(min_length=1, max_length=36) benchmark_id: str = Field(min_length=1, max_length=36) - temperature: float = Field(default=0.0, ge=0, le=2) - top_p: float = Field(default=1.0, gt=0, le=1) + temperature: float | None = Field(default=0.0, ge=0, le=2) + top_p: float | None = Field(default=1.0, gt=0, le=1) max_tokens: int | None = Field(default=256, ge=1, le=MAX_GENERATION_TOKENS) seed: int | None = Field(default=42, ge=-(2**31), le=2**31 - 1) system_prompt: str | None = Field(default=None, max_length=4000) diff --git a/backend/app/schemas/model.py b/backend/app/schemas/model.py index 0cb90ac..5dd77e7 100644 --- a/backend/app/schemas/model.py +++ b/backend/app/schemas/model.py @@ -19,6 +19,18 @@ ALLOWED_DEFAULT_PARAMETERS = frozenset({"temperature", "top_p", "max_tokens", "seed"}) API_KEY_MIN_BYTES = 8 API_KEY_MAX_BYTES = 8192 +REMOTE_PROVIDER_TYPES = frozenset( + { + ProviderType.OPENAI_COMPATIBLE, + ProviderType.OPENAI_RESPONSES, + ProviderType.ANTHROPIC_MESSAGES, + } +) +_PROVIDER_ENDPOINT_SUFFIXES = { + ProviderType.OPENAI_COMPATIBLE: "/chat/completions", + ProviderType.OPENAI_RESPONSES: "/responses", + ProviderType.ANTHROPIC_MESSAGES: "/messages", +} def _is_loopback_host(hostname: str) -> bool: @@ -150,6 +162,67 @@ def _validate_default_parameters(value: dict[str, Any] | None) -> dict[str, Any] return validated +def _validate_provider_endpoint(provider_type: ProviderType, base_url: str | None) -> None: + """Reject a full endpoint that belongs to a different explicit protocol.""" + + if provider_type not in REMOTE_PROVIDER_TYPES or base_url is None: + return + path = urlsplit(base_url).path.rstrip("/") + matching_type = next( + ( + candidate + for candidate, suffix in _PROVIDER_ENDPOINT_SUFFIXES.items() + if path.endswith(suffix) + ), + None, + ) + if matching_type is not None and matching_type != provider_type: + expected = _PROVIDER_ENDPOINT_SUFFIXES[provider_type] + raise ValueError( + f"{provider_type.value} requires a compatible root URL or an endpoint ending " + f"in {expected}" + ) + + +def validate_provider_generation_parameters( + provider_type: ProviderType | str, + parameters: Mapping[str, Any], +) -> None: + """Validate protocol-specific fields without weakening Adapter final checks.""" + + normalized = ( + provider_type if isinstance(provider_type, ProviderType) else ProviderType(provider_type) + ) + if normalized in {ProviderType.MOCK, ProviderType.OPENAI_COMPATIBLE}: + for field in ("temperature", "top_p"): + if field in parameters and parameters[field] is None: + raise ValueError(f"{normalized.value} requires a non-null {field} value") + if ( + normalized in {ProviderType.OPENAI_RESPONSES, ProviderType.ANTHROPIC_MESSAGES} + and parameters.get("seed") is not None + ): + raise ValueError(f"{normalized.value} does not support a non-null seed") + if ( + normalized == ProviderType.ANTHROPIC_MESSAGES + and "max_tokens" in parameters + and parameters.get("max_tokens") is None + ): + raise ValueError("anthropic_messages requires a finite max_tokens value") + temperature = parameters.get("temperature") + if ( + normalized == ProviderType.ANTHROPIC_MESSAGES + and temperature is not None + and ( + isinstance(temperature, bool) + or not isinstance(temperature, (int, float)) + or temperature != temperature + or temperature in {float("inf"), float("-inf")} + or not 0 <= temperature <= 1 + ) + ): + raise ValueError("anthropic_messages temperature must be between 0 and 1") + + class ModelFields(APIModel): """Fields shared by create validation and reconstructed PATCH payloads.""" @@ -225,16 +298,23 @@ def validate_provider_requirements(self) -> "ModelFields": self.input_price_per_million = 0 if self.output_price_per_million is None: self.output_price_per_million = 0 - if self.provider_type == ProviderType.OPENAI_COMPATIBLE: + if self.provider_type in REMOTE_PROVIDER_TYPES: missing = [ field for field in ("base_url", "remote_model_name") if getattr(self, field) is None ] if self.api_key is None and self.api_key_env is None: missing.append("api_key or api_key_env") if missing: - raise ValueError("openai_compatible requires " + ", ".join(missing)) + raise ValueError(f"{self.provider_type.value} requires " + ", ".join(missing)) if self.api_key is not None and self.api_key_env is not None: - raise ValueError("openai_compatible accepts api_key or api_key_env, not both") + raise ValueError( + f"{self.provider_type.value} accepts api_key or api_key_env, not both" + ) + _validate_provider_endpoint(self.provider_type, self.base_url) + validate_provider_generation_parameters( + self.provider_type, + self.default_parameters, + ) return self diff --git a/backend/app/services/run_evidence.py b/backend/app/services/run_evidence.py new file mode 100644 index 0000000..5751df0 --- /dev/null +++ b/backend/app/services/run_evidence.py @@ -0,0 +1,67 @@ +"""Pure protocol-v1 metric derivation shared by writers and read projections.""" + +from __future__ import annotations + +from dataclasses import dataclass +from decimal import Decimal + + +@dataclass(frozen=True, slots=True) +class CanonicalRunEvidence: + """Derived Run metrics, including exact all-or-nothing usage fields.""" + + completed_questions: int + correct_questions: int + error_questions: int + score: float + completion_rate: float + answered_accuracy: float | None + average_latency_ms: float | None + input_tokens: int | None + output_tokens: int | None + estimated_cost: Decimal | None + + +def canonical_run_evidence( + *, + planned_questions: int, + response_count: int, + score_sum: float, + completed_outputs: int, + evaluable_responses: int, + error_responses: int, + average_latency_ms: float | None, + known_input_tokens: int, + input_token_reports: int, + known_output_tokens: int, + output_token_reports: int, + known_estimated_cost: Decimal, + estimated_cost_reports: int, +) -> CanonicalRunEvidence: + """Apply the protocol-v1 denominators and exact usage coverage rule.""" + + correct_questions = round(float(score_sum)) + return CanonicalRunEvidence( + completed_questions=response_count, + correct_questions=correct_questions, + error_questions=error_responses, + score=(float(score_sum) / planned_questions * 100) if planned_questions else 0.0, + completion_rate=(completed_outputs / planned_questions * 100) if planned_questions else 0.0, + answered_accuracy=(correct_questions / evaluable_responses * 100) + if evaluable_responses + else None, + average_latency_ms=average_latency_ms, + input_tokens=( + known_input_tokens if response_count and input_token_reports == response_count else None + ), + output_tokens=( + known_output_tokens + if response_count and output_token_reports == response_count + else None + ), + estimated_cost=( + known_estimated_cost + if response_count and estimated_cost_reports == response_count + else None + ), + ) diff --git a/backend/app/services/run_progress.py b/backend/app/services/run_progress.py new file mode 100644 index 0000000..3caa8b4 --- /dev/null +++ b/backend/app/services/run_progress.py @@ -0,0 +1,220 @@ +"""Read-only fixed-block projections for live Run Detail polling.""" + +from __future__ import annotations + +from dataclasses import dataclass +from decimal import Decimal +from math import ceil + +from sqlalchemy import func, select +from sqlalchemy.orm import Session + +from app.models import EvaluationResponse, EvaluationRun, Question +from app.schemas.evaluation_progress import ( + PROGRESS_BLOCK_SIZE, + EvaluationProgressOutcome, +) +from app.services.run_evidence import CanonicalRunEvidence, canonical_run_evidence + + +class RunProgressIntegrityError(RuntimeError): + """Persisted Response positions do not belong to the frozen Run plan.""" + + +@dataclass(frozen=True, slots=True) +class RunProgressIndexProjection: + metrics: CanonicalRunEvidence + known_input_tokens: int + known_output_tokens: int + input_token_reported_responses: int + output_token_reported_responses: int + known_estimated_cost: Decimal + estimated_cost_reported_responses: int + block_response_counts: tuple[int, ...] + + +def progress_block_count(total_questions: int) -> int: + """Return the number of fixed blocks needed for the frozen plan.""" + + return ceil(total_questions / PROGRESS_BLOCK_SIZE) if total_questions else 0 + + +def load_progress_index(session: Session, run: EvaluationRun) -> RunProgressIndexProjection: + """Scan compact Response facts once and derive global and per-block aggregates.""" + + has_output = EvaluationResponse.raw_response.is_not(None) & ( + EvaluationResponse.raw_response != "" + ) + evaluable = EvaluationResponse.error_type.is_(None) & has_output + block_index = (Question.position // PROGRESS_BLOCK_SIZE).label("block_index") + rows = session.execute( + select( + Question.benchmark_id.label("benchmark_id"), + block_index, + func.min(Question.position).label("minimum_position"), + func.max(Question.position).label("maximum_position"), + func.count(EvaluationResponse.id).label("response_count"), + func.coalesce(func.sum(EvaluationResponse.score), 0.0).label("score_sum"), + func.count(EvaluationResponse.id).filter(has_output).label("completed_outputs"), + func.count(EvaluationResponse.id).filter(evaluable).label("evaluable_responses"), + func.count(EvaluationResponse.id) + .filter(EvaluationResponse.error_type.is_not(None)) + .label("error_responses"), + func.coalesce(func.sum(EvaluationResponse.latency_ms), 0.0).label("latency_sum"), + func.count(EvaluationResponse.latency_ms).label("latency_reports"), + func.coalesce(func.sum(EvaluationResponse.input_tokens), 0).label("known_input_tokens"), + func.count(EvaluationResponse.input_tokens).label("input_token_reports"), + func.coalesce(func.sum(EvaluationResponse.output_tokens), 0).label( + "known_output_tokens" + ), + func.count(EvaluationResponse.output_tokens).label("output_token_reports"), + func.coalesce(func.sum(EvaluationResponse.estimated_cost), 0).label( + "known_estimated_cost" + ), + func.count(EvaluationResponse.estimated_cost).label("estimated_cost_reports"), + ) + .select_from(EvaluationRun) + .outerjoin(EvaluationResponse, EvaluationResponse.run_id == EvaluationRun.id) + .outerjoin(Question, Question.id == EvaluationResponse.question_id) + .where(EvaluationRun.id == run.id) + .group_by(Question.benchmark_id, block_index) + .order_by(Question.benchmark_id, block_index) + ).all() + + block_counts = [0] * progress_block_count(run.total_questions) + response_count = 0 + score_sum = 0.0 + completed_outputs = 0 + evaluable_responses = 0 + error_responses = 0 + latency_sum = 0.0 + latency_reports = 0 + known_input_tokens = 0 + input_token_reports = 0 + known_output_tokens = 0 + output_token_reports = 0 + known_estimated_cost = Decimal(0) + estimated_cost_reports = 0 + + for row in rows: + current_count = int(row.response_count or 0) + # The outer-join sentinel makes an empty Run return one zero-count group. + # A positive count without a joined Question is orphaned evidence and must + # fail closed rather than disappearing from the block index. + if current_count == 0: + continue + if ( + row.benchmark_id is None + or row.block_index is None + or row.minimum_position is None + or row.maximum_position is None + ): + raise RunProgressIntegrityError("run_progress_response_mapping_missing") + current_block = int(row.block_index) + block_start = current_block * PROGRESS_BLOCK_SIZE + block_end = min(block_start + PROGRESS_BLOCK_SIZE, run.total_questions) + if ( + row.benchmark_id != run.benchmark_id + or not 0 <= current_block < len(block_counts) + or int(row.minimum_position) < block_start + or int(row.maximum_position) >= block_end + ): + raise RunProgressIntegrityError("run_progress_response_outside_frozen_plan") + block_counts[current_block] += current_count + if block_counts[current_block] > block_end - block_start: + raise RunProgressIntegrityError("run_progress_block_response_count_invalid") + response_count += current_count + score_sum += float(row.score_sum or 0.0) + completed_outputs += int(row.completed_outputs or 0) + evaluable_responses += int(row.evaluable_responses or 0) + error_responses += int(row.error_responses or 0) + latency_sum += float(row.latency_sum or 0.0) + latency_reports += int(row.latency_reports or 0) + known_input_tokens += int(row.known_input_tokens or 0) + input_token_reports += int(row.input_token_reports or 0) + known_output_tokens += int(row.known_output_tokens or 0) + output_token_reports += int(row.output_token_reports or 0) + known_estimated_cost += Decimal(row.known_estimated_cost or 0) + estimated_cost_reports += int(row.estimated_cost_reports or 0) + + if response_count > run.total_questions: + raise RunProgressIntegrityError("run_progress_response_count_invalid") + average_latency_ms = latency_sum / latency_reports if latency_reports else None + metrics = canonical_run_evidence( + planned_questions=run.total_questions, + response_count=response_count, + score_sum=score_sum, + completed_outputs=completed_outputs, + evaluable_responses=evaluable_responses, + error_responses=error_responses, + average_latency_ms=average_latency_ms, + known_input_tokens=known_input_tokens, + input_token_reports=input_token_reports, + known_output_tokens=known_output_tokens, + output_token_reports=output_token_reports, + known_estimated_cost=known_estimated_cost, + estimated_cost_reports=estimated_cost_reports, + ) + return RunProgressIndexProjection( + metrics=metrics, + known_input_tokens=known_input_tokens, + known_output_tokens=known_output_tokens, + input_token_reported_responses=input_token_reports, + output_token_reported_responses=output_token_reports, + known_estimated_cost=known_estimated_cost, + estimated_cost_reported_responses=estimated_cost_reports, + block_response_counts=tuple(block_counts), + ) + + +def load_progress_block( + session: Session, + run: EvaluationRun, + block_index: int, +) -> list[dict[str, object]]: + """Return only allowlisted compact facts for one absolute-position block.""" + + start_position = block_index * PROGRESS_BLOCK_SIZE + end_position = min(start_position + PROGRESS_BLOCK_SIZE, run.total_questions) + rows = session.execute( + select( + Question.position, + EvaluationResponse.score, + EvaluationResponse.latency_ms, + EvaluationResponse.input_tokens, + EvaluationResponse.output_tokens, + EvaluationResponse.estimated_cost, + EvaluationResponse.error_type, + ) + .join(Question, Question.id == EvaluationResponse.question_id) + .where( + EvaluationResponse.run_id == run.id, + Question.benchmark_id == run.benchmark_id, + Question.position >= start_position, + Question.position < end_position, + ) + .order_by(Question.position) + ).all() + items: list[dict[str, object]] = [] + for row in rows: + if row.error_type is not None: + outcome = EvaluationProgressOutcome.ERROR + elif float(row.score) == 1.0: + outcome = EvaluationProgressOutcome.PASSED + else: + outcome = EvaluationProgressOutcome.WRONG + items.append( + { + "position": int(row.position), + "outcome": outcome, + "score": float(row.score), + "latency_ms": float(row.latency_ms) if row.latency_ms is not None else None, + "input_tokens": row.input_tokens, + "output_tokens": row.output_tokens, + "estimated_cost": ( + float(row.estimated_cost) if row.estimated_cost is not None else None + ), + "error_type": row.error_type, + } + ) + return items diff --git a/backend/app/services/run_service.py b/backend/app/services/run_service.py index 97fd953..b0f8d48 100644 --- a/backend/app/services/run_service.py +++ b/backend/app/services/run_service.py @@ -17,9 +17,12 @@ PROTOCOL_VERSION, RETRYABLE_PROVIDER_STATUS_CODES, ) -from app.models import Benchmark, EvaluationRun, Model, RunStatus +from app.models import Benchmark, EvaluationRun, Model, ProviderType, RunStatus from app.schemas.evaluation_run import EvaluationRunCreate -from app.schemas.model import model_run_snapshot_values +from app.schemas.model import ( + model_run_snapshot_values, + validate_provider_generation_parameters, +) PROJECT_ROOT = Path(__file__).resolve().parents[3] @@ -63,6 +66,17 @@ def build_evaluation_run( ) for field in ("temperature", "top_p", "max_tokens", "seed") } + if model.provider_type in { + ProviderType.OPENAI_RESPONSES, + ProviderType.ANTHROPIC_MESSAGES, + }: + for field in ("temperature", "top_p", "seed"): + if field not in requested_fields and field not in model_defaults: + generation[field] = None + validate_provider_generation_parameters(model.provider_type, generation) + retryable_status_codes = list(RETRYABLE_PROVIDER_STATUS_CODES) + if model.provider_type == ProviderType.ANTHROPIC_MESSAGES: + retryable_status_codes.append(529) snapshot: dict[str, Any] = { "generation": generation, "model": model_run_snapshot_values(model), @@ -95,7 +109,7 @@ def build_evaluation_run( "max_attempts": DEFAULT_MAX_RETRIES + 1, "backoff_base_seconds": DEFAULT_RETRY_BACKOFF_BASE_SECONDS, "backoff_cap_seconds": DEFAULT_RETRY_BACKOFF_CAP_SECONDS, - "retryable_status_codes": list(RETRYABLE_PROVIDER_STATUS_CODES), + "retryable_status_codes": retryable_status_codes, }, "task_delivery": "at_least_once", "task_max_attempts": settings.worker_max_attempts, diff --git a/backend/tests/integration/test_postgres_leases.py b/backend/tests/integration/test_postgres_leases.py index f92db81..7dc99de 100644 --- a/backend/tests/integration/test_postgres_leases.py +++ b/backend/tests/integration/test_postgres_leases.py @@ -189,6 +189,50 @@ def _response() -> EvaluationResponse: ) +def _add_second_due_run(postgres_store) -> None: + with postgres_store() as session, session.begin(): + benchmark = Benchmark( + id="benchmark-pg-b", + slug="postgres-lease-fixture-b", + name="PostgreSQL lease fixture B", + version="1.0.0", + description="fixture", + dimension="general", + language="en", + license="MIT", + source="local", + evaluator_type="exact_match", + evaluator_config={}, + prompt_template={}, + dataset_hash="postgres-lease-hash-b", + question_count=1, + ) + session.add_all( + [ + benchmark, + Question( + id="question-pg-b", + benchmark_id=benchmark.id, + external_id="q1", + position=0, + question_type="exact_match", + prompt="Two?", + reference_answer="two", + ), + EvaluationRun( + id="run-pg-b", + model_id="model-pg", + benchmark_id=benchmark.id, + status=RunStatus.PENDING, + model_parameters_snapshot={}, + benchmark_hash_snapshot=benchmark.dataset_hash, + prompt_template_snapshot={}, + total_questions=1, + ), + ] + ) + + def _full_policy(**overrides: object) -> dict[str, object]: values: dict[str, object] = { "global_concurrency_limit": None, @@ -405,6 +449,54 @@ def compete(owner: str): assert run.completed_questions == response_count == 1 +def test_postgres_two_workers_claim_distinct_due_runs_across_benchmarks( + postgres_store, +) -> None: + _add_second_due_run(postgres_store) + repository = RunLeaseRepository(postgres_store, lease_for=timedelta(seconds=30)) + barrier = Barrier(2) + owners = ("worker-pg-a", "worker-pg-b") + + def claim_next(owner: str): + barrier.wait(timeout=5) + return repository.claim_next(owner=owner) + + with ThreadPoolExecutor(max_workers=2) as executor: + leases = list(executor.map(claim_next, owners)) + + assert all(lease is not None for lease in leases) + claimed = [lease for lease in leases if lease is not None] + assert {lease.owner for lease in claimed} == set(owners) + assert {lease.run_id for lease in claimed} == {"run-pg", "run-pg-b"} + with postgres_store() as session: + benchmark_ids = set( + session.scalars( + select(EvaluationRun.benchmark_id).where( + EvaluationRun.id.in_(tuple(lease.run_id for lease in claimed)) + ) + ) + ) + assert benchmark_ids == {"benchmark-pg", "benchmark-pg-b"} + + +def test_postgres_active_lease_does_not_block_another_run_claim(postgres_store) -> None: + _add_second_due_run(postgres_store) + repository = RunLeaseRepository(postgres_store, lease_for=timedelta(seconds=30)) + first = repository.claim("run-pg", owner="worker-pg-a") + assert first is not None + + with ThreadPoolExecutor(max_workers=1) as executor: + second = executor.submit( + repository.claim_next, + owner="worker-pg-b", + ).result(timeout=10) + + assert second is not None + assert second.run_id == "run-pg-b" + assert second.owner == "worker-pg-b" + assert repository.claim("run-pg", owner="worker-pg-b") is None + + def test_postgres_claim_and_cancel_race_preserves_state_constraints(postgres_store) -> None: repository = RunLeaseRepository(postgres_store, lease_for=timedelta(seconds=30)) barrier = Barrier(2) @@ -1120,3 +1212,51 @@ def patch_provider_origin() -> str: run = session.get(EvaluationRun, run_id) assert model is not None and model.base_url == "https://provider.example/v1" assert run is not None and run.status == RunStatus.PENDING + + +def test_postgres_provider_protocol_constraint_accepts_explicit_remote_adapters( + postgres_store, +) -> None: + with postgres_store() as session, session.begin(): + session.add_all( + [ + Model( + id="model-pg-responses", + name="Postgres Responses", + provider_type=ProviderType.OPENAI_RESPONSES, + base_url="https://provider.example/v1", + remote_model_name="responses-model", + api_key_env="PROVIDER_KEY", + credential_source=CredentialSource.ENVIRONMENT, + ), + Model( + id="model-pg-messages", + name="Postgres Messages", + provider_type=ProviderType.ANTHROPIC_MESSAGES, + base_url="https://provider.example/v1", + remote_model_name="messages-model", + api_key_env="PROVIDER_KEY", + credential_source=CredentialSource.ENVIRONMENT, + ), + ] + ) + + with postgres_store() as session: + assert set( + session.scalars( + select(Model.provider_type).where( + Model.id.in_(("model-pg-responses", "model-pg-messages")) + ) + ) + ) == {ProviderType.OPENAI_RESPONSES, ProviderType.ANTHROPIC_MESSAGES} + + with pytest.raises(IntegrityError), postgres_store() as session, session.begin(): + session.execute( + text( + "INSERT INTO models (" + "id, name, provider_type, credential_source, enabled, " + "default_parameters, created_at, updated_at" + ") VALUES ('model-pg-invalid-protocol', 'Invalid protocol', 'bad', " + "'none', true, '{}', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)" + ) + ) diff --git a/backend/tests/test_adapters.py b/backend/tests/test_adapters.py index 79528c7..7d5eb9e 100644 --- a/backend/tests/test_adapters.py +++ b/backend/tests/test_adapters.py @@ -682,6 +682,44 @@ async def test_openai_compatible_rejects_invalid_or_error_sse_events( assert stream.closed is True +@pytest.mark.parametrize("response_mode", ["json", "sse"]) +@pytest.mark.asyncio +async def test_openai_compatible_parse_errors_do_not_retain_provider_secrets( + response_mode: str, +) -> None: + secret = "malformed-chat-provider-secret" + + def handler(request: httpx.Request) -> httpx.Response: + if response_mode == "sse": + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=f'data: {{"reflected":"{secret}"\n\n'.encode(), + request=request, + ) + return httpx.Response( + 200, + content=f'{{"reflected":"{secret}"'.encode(), + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + with pytest.raises(AdapterError) as caught: + await OpenAICompatibleAdapter( + "https://provider.example/v1", + "model", + api_key=secret, + client=client, + ).generate([{"role": "user", "content": "question"}], {}) + + expected = "invalid_provider_stream" if response_mode == "sse" else "invalid_provider_response" + assert caught.value.error_type == expected + assert caught.value.__cause__ is None + assert caught.value.__context__ is None + formatted = "".join(traceback.format_exception(caught.type, caught.value, caught.tb)) + assert secret not in formatted + + @pytest.mark.asyncio @pytest.mark.parametrize("limit_kind", ["wire", "event", "content"]) async def test_openai_compatible_enforces_sse_size_limits( @@ -741,6 +779,8 @@ async def test_openai_compatible_enforces_sse_size_limits( assert caught.value.error_type == "provider_response_too_large" assert caught.value.attempts == 1 assert f"{expected_limit}-byte safety limit" in caught.value.error_message + assert caught.value.__cause__ is None + assert caught.value.__context__ is None assert stream.closed is True @@ -1061,6 +1101,10 @@ def handler(_request: httpx.Request) -> httpx.Response: assert caught.value.retryable is False assert "32-byte safety limit" in caught.value.error_message assert secret not in str(caught.value) + assert caught.value.__cause__ is None + assert caught.value.__context__ is None + formatted = "".join(traceback.format_exception(caught.type, caught.value, caught.tb)) + assert secret not in formatted assert stream.yielded < len(stream.chunks) assert stream.closed is True diff --git a/backend/tests/test_api.py b/backend/tests/test_api.py index 35f15e9..d6edd58 100644 --- a/backend/tests/test_api.py +++ b/backend/tests/test_api.py @@ -25,6 +25,7 @@ Question, RunStatus, ) +from app.runners.evaluation_runner import EvaluationRunner from app.runners.run_leases import RunLeaseRepository from app.task_queue import QueueUnavailable, RedisRunQueue @@ -47,6 +48,21 @@ def test_health_and_info_do_not_require_provider(client) -> None: info = client.get("/api/v1/info") assert info.status_code == 200 assert info.json()["protocol_version"] == "llmbenchlab-protocol-v1" + assert info.json()["capabilities"]["providers"] == [ + "mock", + "openai_compatible", + "openai_responses", + "anthropic_messages", + ] + + openapi = client.get("/openapi.json") + assert openapi.status_code == 200 + assert set(openapi.json()["components"]["schemas"]["ProviderType"]["enum"]) == { + "mock", + "openai_compatible", + "openai_responses", + "anthropic_messages", + } def test_liveness_readiness_and_request_id_are_componentized(client, monkeypatch) -> None: @@ -376,6 +392,254 @@ def test_openai_model_provider_fields_are_validated(client) -> None: assert "base_url" in response.text +@pytest.mark.parametrize( + ("provider_type", "endpoint"), + [ + ("openai_compatible", "https://provider.example/v1/chat/completions"), + ("openai_responses", "https://provider.example/v1/responses"), + ("anthropic_messages", "https://provider.example/v1/messages"), + ], +) +def test_remote_provider_protocols_persist_and_filter( + client, + provider_type: str, + endpoint: str, +) -> None: + created = client.post( + "/api/v1/models", + json={ + "name": f"Protocol {provider_type}", + "provider_type": provider_type, + "base_url": endpoint, + "remote_model_name": "offline-model", + "api_key_env": "PROTOCOL_PROVIDER_KEY", + }, + ) + + assert created.status_code == 201, created.text + assert created.json()["provider_type"] == provider_type + listed = client.get(f"/api/v1/models?provider_type={provider_type}") + assert listed.status_code == 200 + assert listed.json()["total"] == 1 + assert listed.json()["items"][0]["id"] == created.json()["id"] + + +@pytest.mark.parametrize( + ("provider_type", "wrong_endpoint"), + [ + ("openai_compatible", "https://provider.example/v1/responses"), + ("openai_responses", "https://provider.example/v1/messages"), + ("anthropic_messages", "https://provider.example/v1/chat/completions"), + ], +) +def test_remote_provider_protocol_rejects_a_mismatched_known_endpoint( + client, + provider_type: str, + wrong_endpoint: str, +) -> None: + response = client.post( + "/api/v1/models", + json={ + "name": "Mismatched Protocol", + "provider_type": provider_type, + "base_url": wrong_endpoint, + "remote_model_name": "offline-model", + "api_key_env": "PROTOCOL_PROVIDER_KEY", + }, + ) + + assert response.status_code == 422 + assert "compatible root URL" in response.text + + +@pytest.mark.parametrize( + ("provider_type", "default_parameters", "expected_message"), + [ + ("openai_responses", {"seed": 42}, "does not support a non-null seed"), + ("anthropic_messages", {"seed": 42}, "does not support a non-null seed"), + ("anthropic_messages", {"max_tokens": None}, "requires a finite max_tokens"), + ("anthropic_messages", {"temperature": 1.5}, "temperature must be between 0 and 1"), + ], +) +def test_remote_provider_protocol_rejects_invalid_model_defaults( + client, + provider_type: str, + default_parameters: dict[str, object], + expected_message: str, +) -> None: + response = client.post( + "/api/v1/models", + json={ + "name": f"Invalid defaults {provider_type}", + "provider_type": provider_type, + "base_url": "https://provider.example/v1", + "remote_model_name": "offline-model", + "api_key_env": "PROTOCOL_PROVIDER_KEY", + "default_parameters": default_parameters, + }, + ) + + assert response.status_code == 422 + assert expected_message in response.text + + +def test_run_rejects_invalid_effective_provider_protocol_parameters_before_queueing( + client, +) -> None: + benchmark = client.post("/api/v1/benchmarks/reload-demo").json() + responses_model = client.post( + "/api/v1/models", + json={ + "name": "Responses protocol run", + "provider_type": "openai_responses", + "base_url": "https://provider.example/v1", + "remote_model_name": "responses-model", + "api_key_env": "PROTOCOL_PROVIDER_KEY", + }, + ).json() + messages_model = client.post( + "/api/v1/models", + json={ + "name": "Messages protocol run", + "provider_type": "anthropic_messages", + "base_url": "https://provider.example/v1", + "remote_model_name": "messages-model", + "api_key_env": "PROTOCOL_PROVIDER_KEY", + }, + ).json() + + responses_with_omitted_seed = client.post( + "/api/v1/runs", + json={"model_id": responses_model["id"], "benchmark_id": benchmark["id"]}, + ) + assert responses_with_omitted_seed.status_code == 202, responses_with_omitted_seed.text + assert ( + responses_with_omitted_seed.json()["model_parameters_snapshot"]["generation"]["seed"] + is None + ) + assert responses_with_omitted_seed.json()["model_parameters_snapshot"]["generation"] == { + "temperature": None, + "top_p": None, + "max_tokens": 256, + "seed": None, + } + + responses_with_explicit_seed = client.post( + "/api/v1/runs", + json={ + "model_id": responses_model["id"], + "benchmark_id": benchmark["id"], + "seed": 42, + }, + ) + assert responses_with_explicit_seed.status_code == 422 + assert responses_with_explicit_seed.json()["detail"]["code"] == ( + "invalid_provider_generation_parameters" + ) + + accepted_responses = client.post( + "/api/v1/runs", + json={ + "model_id": responses_model["id"], + "benchmark_id": benchmark["id"], + "seed": None, + }, + ) + assert accepted_responses.status_code == 202, accepted_responses.text + assert ( + accepted_responses.json()["model_parameters_snapshot"]["model"]["adapter_type"] + == "openai_responses" + ) + + messages_without_limit = client.post( + "/api/v1/runs", + json={ + "model_id": messages_model["id"], + "benchmark_id": benchmark["id"], + "seed": None, + "max_tokens": None, + }, + ) + assert messages_without_limit.status_code == 422 + assert messages_without_limit.json()["detail"]["code"] == ( + "invalid_provider_generation_parameters" + ) + + messages_with_invalid_temperature = client.post( + "/api/v1/runs", + json={ + "model_id": messages_model["id"], + "benchmark_id": benchmark["id"], + "seed": None, + "temperature": 1.5, + }, + ) + assert messages_with_invalid_temperature.status_code == 422 + assert messages_with_invalid_temperature.json()["detail"]["code"] == ( + "invalid_provider_generation_parameters" + ) + + accepted_messages = client.post( + "/api/v1/runs", + json={ + "model_id": messages_model["id"], + "benchmark_id": benchmark["id"], + "seed": None, + }, + ) + assert accepted_messages.status_code == 202, accepted_messages.text + assert accepted_messages.json()["model_parameters_snapshot"]["generation"] == { + "temperature": None, + "top_p": None, + "max_tokens": 256, + "seed": None, + } + assert accepted_messages.json()["model_parameters_snapshot"]["execution"]["retry_policy"][ + "retryable_status_codes" + ] == [408, 429, 500, 502, 503, 504, 529] + + runner = EvaluationRunner(SessionLocal) + responses_snapshot = runner._load_snapshots(accepted_responses.json()["id"])[0] + messages_snapshot = runner._load_snapshots(accepted_messages.json()["id"])[0] + assert responses_snapshot.provider_type == "openai_responses" + assert messages_snapshot.provider_type == "anthropic_messages" + + +@pytest.mark.parametrize("provider_type", ["mock", "openai_compatible"]) +@pytest.mark.parametrize("field", ["temperature", "top_p"]) +def test_run_preserves_nonnullable_sampling_contract_for_legacy_protocols( + client, + provider_type: str, + field: str, +) -> None: + benchmark = client.post("/api/v1/benchmarks/reload-demo").json() + model_payload: dict[str, object] = { + "name": f"Legacy sampling {provider_type} {field}", + "provider_type": provider_type, + } + if provider_type == "openai_compatible": + model_payload.update( + { + "base_url": "https://provider.example/v1", + "remote_model_name": "chat-model", + "api_key_env": "PROTOCOL_PROVIDER_KEY", + } + ) + model = client.post("/api/v1/models", json=model_payload).json() + + response = client.post( + "/api/v1/runs", + json={ + "model_id": model["id"], + "benchmark_id": benchmark["id"], + field: None, + }, + ) + + assert response.status_code == 422 + assert response.json()["detail"]["code"] == "invalid_provider_generation_parameters" + + def test_mock_model_rejects_remote_connection_fields(client) -> None: response = client.post( "/api/v1/models", diff --git a/backend/tests/test_audit_archive.py b/backend/tests/test_audit_archive.py index 05fdeb9..42daee6 100644 --- a/backend/tests/test_audit_archive.py +++ b/backend/tests/test_audit_archive.py @@ -282,7 +282,10 @@ def test_archive_v1_rejects_unsupported_source_revision_on_write_and_read(tmp_pa verify_archive(path) -@pytest.mark.parametrize("source_revision", ["20260829_0006", "20260830_0007"]) +@pytest.mark.parametrize( + "source_revision", + ["20260829_0006", "20260830_0007", "20260830_0008"], +) def test_archive_v1_accepts_compatible_repair_revision(source_revision: str) -> None: data, _digest = build_archive_bytes( (), diff --git a/backend/tests/test_compose_up_script.py b/backend/tests/test_compose_up_script.py new file mode 100644 index 0000000..fa39788 --- /dev/null +++ b/backend/tests/test_compose_up_script.py @@ -0,0 +1,577 @@ +"""Offline checks for the bounded multi-Worker Compose launcher.""" + +from __future__ import annotations + +import os +import stat +import subprocess +from pathlib import Path + +import pytest + +REPOSITORY_ROOT = Path(__file__).resolve().parents[2] +COMPOSE_UP_SCRIPT = REPOSITORY_ROOT / "scripts" / "compose_up.sh" + + +def _write_executable(path: Path, contents: str) -> None: + path.write_text(contents, encoding="utf-8") + path.chmod(path.stat().st_mode | stat.S_IXUSR) + + +def _prepare_project(tmp_path: Path, *, dotenv: str | None = None) -> tuple[Path, Path]: + project = tmp_path / "project" + scripts = project / "scripts" + fake_bin = tmp_path / "bin" + scripts.mkdir(parents=True) + fake_bin.mkdir() + (scripts / "compose_up.sh").write_text( + COMPOSE_UP_SCRIPT.read_text(encoding="utf-8"), + encoding="utf-8", + ) + if dotenv is not None: + (project / ".env").write_text(dotenv, encoding="utf-8") + + _write_executable( + fake_bin / "docker", + """#!/usr/bin/env bash +set -euo pipefail +if [[ "$1" != "compose" ]]; then + exit 90 +fi +shift +case "${1:-}" in + ps) + if [[ "${2:-}" == "--all" && "${3:-}" == "-q" && "${4:-}" == "worker" ]]; then + printf 'ps:all\n' >>"$DOCKER_CALLS_FILE" + current_workers="${FAKE_ALL_WORKERS:-0}" + elif [[ "${2:-}" == "--status" && "${3:-}" == "running" \ + && "${4:-}" == "-q" && "${5:-}" == "worker" ]]; then + printf 'ps:running\n' >>"$DOCKER_CALLS_FILE" + current_workers="${FAKE_RUNNING_WORKERS:-0}" + else + exit 95 + fi + for ((worker_index = 1; worker_index <= current_workers; worker_index += 1)); do + printf 'fake-worker-%s\n' "$worker_index" + done + ;; + build) + printf 'build\n' >>"$DOCKER_CALLS_FILE" + ;; + up) + if [[ -n "${FAKE_EXPECTED_DATABASE_URL:-}" \ + && "${LLMBENCHLAB_COMPOSE_DATABASE_URL:-}" != "$FAKE_EXPECTED_DATABASE_URL" ]]; then + exit 94 + fi + printf 'up:%s\n' "$*" >>"$DOCKER_CALLS_FILE" + for argument in "$@"; do + if [[ "$argument" == "api" ]]; then + printf '%s\n' "${LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES-unset}" \ + >"$EXPECTED_PROCESSES_FILE" + fi + done + exit "${FAKE_UP_EXIT_CODE:-0}" + ;; + run) + if [[ " $* " != *" --rm --no-deps -T worker python -c "* \ + || "${8:-}" != *"database_utc_now"* ]]; then + exit 96 + fi + printf 'watermark\n' >>"$DOCKER_CALLS_FILE" + printf 'LLMBENCHLAB_SCAN_WATERMARK=2026-08-30T08:00:00+00:00\n' + ;; + exec) + if [[ "${2:-}" != "-T" || "${4:-}" != "python" || "${5:-}" != "-c" \ + || "${7:-}" != "${LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES}" ]]; then + exit 93 + fi + expected="${LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES}" + if [[ "${3:-}" == "worker" && "${6:-}" == *"WorkerProcess"* \ + && "${6:-}" == *"last_scan_at"* && "${6:-}" == *"last_seen_at >= cutoff"* \ + && "${6:-}" == *"last_scan_at >= watermark"* \ + && "${9:-}" =~ ^[0-9]+$ ]]; then + printf 'scan\n' >>"$DOCKER_CALLS_FILE" + new_required="${9:-0}" + scan_attempt=0 + if [[ -f "$SCAN_ATTEMPT_FILE" ]]; then + scan_attempt="$(<"$SCAN_ATTEMPT_FILE")" + fi + scan_attempt=$((scan_attempt + 1)) + printf '%s\n' "$scan_attempt" >"$SCAN_ATTEMPT_FILE" + case "${FAKE_SCAN_MODE:-ready}" in + ready) + printf '%s/%s\n' "$expected" "$new_required" + exit 0 + ;; + transition) + if (( scan_attempt == 1 )); then + printf '0/0\n' + exit 1 + fi + printf '%s/%s\n' "$expected" "$new_required" + exit 0 + ;; + never) + printf '0/0\n' + exit 1 + ;; + stale) + printf '%s/0\n' "$expected" + exit 1 + ;; + *) + exit 91 + ;; + esac + fi + if [[ "${3:-}" == "api" && "${6:-}" == *"/api/v1/tasks/metrics"* \ + && "${6:-}" == *"worker_expected_processes"* \ + && "${6:-}" == *"worker_registered_processes"* \ + && "${6:-}" == *"worker_live_processes"* \ + && "${6:-}" == *"worker_stalled_processes"* \ + && "${6:-}" == *"worker_shortfall_processes"* ]]; then + printf 'metrics\n' >>"$DOCKER_CALLS_FILE" + attempt=0 + if [[ -f "$METRICS_ATTEMPT_FILE" ]]; then + attempt="$(<"$METRICS_ATTEMPT_FILE")" + fi + attempt=$((attempt + 1)) + printf '%s\n' "$attempt" >"$METRICS_ATTEMPT_FILE" + case "${FAKE_METRICS_MODE:-ready}" in + ready) + printf '%s/%s/%s/0/0\n' "$expected" "$expected" "$expected" + exit 0 + ;; + transition) + if (( attempt == 1 )); then + printf '%s/%s/0/0/%s\n' "$expected" "$expected" "$expected" + exit 1 + fi + printf '%s/%s/%s/0/0\n' "$expected" "$expected" "$expected" + exit 0 + ;; + never) + printf '%s/%s/0/0/%s\n' "$expected" "$expected" "$expected" + exit 1 + ;; + stale) + printf '%s/%s/%s/1/0\n' "$expected" "$((expected + 1))" "$expected" + exit 1 + ;; + *) + exit 91 + ;; + esac + fi + exit 93 + ;; + *) + printf 'unexpected:%s\n' "$*" >>"$DOCKER_CALLS_FILE" + exit 92 + ;; +esac +""", + ) + _write_executable( + fake_bin / "sleep", + """#!/usr/bin/env bash +exit 0 +""", + ) + return project, fake_bin + + +def _run_compose_up( + project: Path, + fake_bin: Path, + *, + worker_processes: str | None = None, + metrics_mode: str = "ready", + scan_mode: str = "ready", + current_workers: int = 0, + all_workers: int | None = None, + extra_environment: dict[str, str] | None = None, +) -> tuple[subprocess.CompletedProcess[str], Path, Path, Path]: + calls_file = project / "docker-calls.log" + expected_file = project / "expected-processes.log" + attempt_file = project / "metrics-attempts.log" + scan_attempt_file = project / "scan-attempts.log" + environment = os.environ.copy() + environment.update( + { + "PATH": f"{fake_bin}{os.pathsep}{environment['PATH']}", + "DOCKER_CALLS_FILE": str(calls_file), + "EXPECTED_PROCESSES_FILE": str(expected_file), + "METRICS_ATTEMPT_FILE": str(attempt_file), + "SCAN_ATTEMPT_FILE": str(scan_attempt_file), + "FAKE_METRICS_MODE": metrics_mode, + "FAKE_SCAN_MODE": scan_mode, + "FAKE_RUNNING_WORKERS": str(current_workers), + "FAKE_ALL_WORKERS": str(current_workers if all_workers is None else all_workers), + } + ) + environment.pop("LLMBENCHLAB_COMPOSE_WORKER_PROCESSES", None) + environment.pop("LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES", None) + if worker_processes is not None: + environment["LLMBENCHLAB_COMPOSE_WORKER_PROCESSES"] = worker_processes + if extra_environment is not None: + environment.update(extra_environment) + result = subprocess.run( + ["bash", str(project / "scripts" / "compose_up.sh")], + cwd=project, + env=environment, + capture_output=True, + text=True, + timeout=10, + check=False, + ) + return result, calls_file, expected_file, attempt_file + + +def test_compose_launcher_defaults_to_two_workers(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, expected_file, attempt_file = _run_compose_up(project, fake_bin) + + assert result.returncode == 0 + assert calls_file.read_text(encoding="utf-8").splitlines() == [ + "ps:all", + "ps:running", + "build", + "up:up --wait --wait-timeout 180 --remove-orphans postgres redis migrate", + "watermark", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps --scale worker=2 worker", + "scan", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps --force-recreate api", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps frontend", + "metrics", + ] + assert expected_file.read_text(encoding="utf-8") == "2\n" + assert attempt_file.read_text(encoding="utf-8") == "1\n" + assert "Worker expected/registered/live/stalled/shortfall=2/2/2/0/0" in result.stdout + + +def test_compose_launcher_uses_explicit_worker_count(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, expected_file, _ = _run_compose_up( + project, + fake_bin, + worker_processes="4", + ) + + assert result.returncode == 0 + assert "--scale worker=4" in calls_file.read_text(encoding="utf-8") + assert expected_file.read_text(encoding="utf-8") == "4\n" + assert "Worker expected/registered/live/stalled/shortfall=4/4/4/0/0" in result.stdout + + +def test_compose_launcher_reads_worker_count_from_dotenv(tmp_path: Path) -> None: + project, fake_bin = _prepare_project( + tmp_path, + dotenv='export LLMBENCHLAB_COMPOSE_WORKER_PROCESSES="6" # local scale\n', + ) + + result, calls_file, expected_file, _ = _run_compose_up(project, fake_bin) + + assert result.returncode == 0 + assert "--scale worker=6" in calls_file.read_text(encoding="utf-8") + assert expected_file.read_text(encoding="utf-8") == "6\n" + + +def test_explicit_worker_count_overrides_dotenv(tmp_path: Path) -> None: + project, fake_bin = _prepare_project( + tmp_path, + dotenv="LLMBENCHLAB_COMPOSE_WORKER_PROCESSES=7\n", + ) + + result, calls_file, expected_file, _ = _run_compose_up( + project, + fake_bin, + worker_processes="3", + ) + + assert result.returncode == 0 + assert "--scale worker=3" in calls_file.read_text(encoding="utf-8") + assert expected_file.read_text(encoding="utf-8") == "3\n" + + +def test_dotenv_does_not_override_unrelated_explicit_compose_environment( + tmp_path: Path, +) -> None: + project, fake_bin = _prepare_project( + tmp_path, + dotenv=( + "LLMBENCHLAB_COMPOSE_WORKER_PROCESSES=2\n" + "LLMBENCHLAB_COMPOSE_DATABASE_URL=postgresql://dotenv-secret@postgres/db\n" + ), + ) + + result, _, _, _ = _run_compose_up( + project, + fake_bin, + extra_environment={ + "LLMBENCHLAB_COMPOSE_DATABASE_URL": "postgresql://explicit-secret@postgres/db", + "FAKE_EXPECTED_DATABASE_URL": "postgresql://explicit-secret@postgres/db", + }, + ) + + assert result.returncode == 0 + assert "dotenv-secret" not in result.stdout + result.stderr + assert "explicit-secret" not in result.stdout + result.stderr + + +def test_dotenv_is_parsed_without_executing_shell_content(tmp_path: Path) -> None: + marker = tmp_path / "dotenv-command-ran" + project, fake_bin = _prepare_project( + tmp_path, + dotenv=(f"UNRELATED=$(touch {marker})\nLLMBENCHLAB_COMPOSE_WORKER_PROCESSES=2\n"), + ) + + result, _, _, _ = _run_compose_up(project, fake_bin) + + assert result.returncode == 0 + assert not marker.exists() + + +@pytest.mark.parametrize( + "worker_processes", + ["", "0", "02", "33", "-1", "2.5", "two", " 2", "18446744073709551618"], +) +def test_invalid_worker_count_fails_before_calling_docker( + tmp_path: Path, + worker_processes: str, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, expected_file, attempt_file = _run_compose_up( + project, + fake_bin, + worker_processes=worker_processes, + ) + + assert result.returncode == 2 + assert "must be an integer from 1 through 32" in result.stderr + assert not calls_file.exists() + assert not expected_file.exists() + assert not attempt_file.exists() + + +def test_metrics_poll_retries_until_all_workers_are_live(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, _, attempt_file = _run_compose_up( + project, + fake_bin, + metrics_mode="transition", + ) + + assert result.returncode == 0 + assert calls_file.read_text(encoding="utf-8").splitlines() == [ + "ps:all", + "ps:running", + "build", + "up:up --wait --wait-timeout 180 --remove-orphans postgres redis migrate", + "watermark", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps --scale worker=2 worker", + "scan", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps --force-recreate api", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps frontend", + "metrics", + "metrics", + ] + assert attempt_file.read_text(encoding="utf-8") == "2\n" + assert "Worker expected/registered/live/stalled/shortfall=2/2/2/0/0" in result.stdout + + +def test_scale_up_waits_for_worker_scans_before_recreating_api(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, _, _ = _run_compose_up( + project, + fake_bin, + worker_processes="2", + current_workers=1, + scan_mode="transition", + ) + + assert result.returncode == 0 + calls = calls_file.read_text(encoding="utf-8").splitlines() + worker_up = next(index for index, call in enumerate(calls) if "--scale worker=2" in call) + api_up = next(index for index, call in enumerate(calls) if call.endswith(" api")) + assert calls[worker_up + 1 : api_up] == ["scan", "scan"] + assert worker_up < api_up + + +def test_scale_down_recreates_api_before_graceful_worker_scale(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, _, _ = _run_compose_up( + project, + fake_bin, + worker_processes="1", + current_workers=2, + ) + + assert result.returncode == 0 + calls = calls_file.read_text(encoding="utf-8").splitlines() + api_up = next(index for index, call in enumerate(calls) if call.endswith(" api")) + worker_up = next(index for index, call in enumerate(calls) if "--scale worker=1" in call) + assert api_up < worker_up + assert calls[worker_up + 1] == "scan" + + +def test_exited_replica_is_counted_when_selecting_scale_down_order(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, _, _ = _run_compose_up( + project, + fake_bin, + worker_processes="1", + current_workers=1, + all_workers=2, + ) + + assert result.returncode == 0 + calls = calls_file.read_text(encoding="utf-8").splitlines() + assert calls[:2] == ["ps:all", "ps:running"] + api_up = next(index for index, call in enumerate(calls) if call.endswith(" api")) + worker_up = next(index for index, call in enumerate(calls) if "--scale worker=1" in call) + assert api_up < worker_up + + +def test_stale_generation_cannot_satisfy_post_scale_scan_gate(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, expected_file, attempt_file = _run_compose_up( + project, + fake_bin, + worker_processes="2", + current_workers=1, + all_workers=2, + scan_mode="stale", + ) + + assert result.returncode == 1 + calls = calls_file.read_text(encoding="utf-8").splitlines() + assert "watermark" in calls + assert calls[-30:] == ["scan"] * 30 + assert not any(call.endswith(" api") for call in calls) + assert not expected_file.exists() + assert not attempt_file.exists() + assert "last observed live/new=2/0" in result.stderr + + +def test_worker_scan_timeout_stops_before_api_recreate_and_leaves_stack( + tmp_path: Path, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, expected_file, attempt_file = _run_compose_up( + project, + fake_bin, + scan_mode="never", + ) + + assert result.returncode == 1 + calls = calls_file.read_text(encoding="utf-8").splitlines() + assert calls[-30:] == ["scan"] * 30 + assert not expected_file.exists() + assert not attempt_file.exists() + assert "last observed live/new=0/0" in result.stderr + assert "left running for inspection" in result.stderr + assert all("down" not in call for call in calls) + + +def test_stalled_or_extra_registered_worker_never_passes_ready_gate( + tmp_path: Path, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, _, attempt_file = _run_compose_up( + project, + fake_bin, + metrics_mode="stale", + ) + + assert result.returncode == 1 + calls = calls_file.read_text(encoding="utf-8").splitlines() + assert calls[-30:] == ["metrics"] * 30 + assert attempt_file.read_text(encoding="utf-8") == "30\n" + assert "last observed=2/3/2/1/0" in result.stderr + + +def test_compose_up_failure_does_not_enter_worker_or_metrics_polling(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, expected_file, attempt_file = _run_compose_up( + project, + fake_bin, + extra_environment={"FAKE_UP_EXIT_CODE": "17"}, + ) + + assert result.returncode == 17 + assert calls_file.read_text(encoding="utf-8").splitlines() == [ + "ps:all", + "ps:running", + "build", + "up:up --wait --wait-timeout 180 --remove-orphans postgres redis migrate", + ] + assert not expected_file.exists() + assert not attempt_file.exists() + + +def test_metrics_timeout_is_nonzero_and_leaves_stack_running(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + + result, calls_file, _, attempt_file = _run_compose_up( + project, + fake_bin, + metrics_mode="never", + ) + + assert result.returncode == 1 + calls = calls_file.read_text(encoding="utf-8").splitlines() + assert calls[:7] == [ + "ps:all", + "ps:running", + "build", + "up:up --wait --wait-timeout 180 --remove-orphans postgres redis migrate", + "watermark", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps --scale worker=2 worker", + "scan", + ] + assert calls[7:9] == [ + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps --force-recreate api", + "up:up --wait --wait-timeout 180 --remove-orphans --no-deps frontend", + ] + assert calls[9:] == ["metrics"] * 30 + assert attempt_file.read_text(encoding="utf-8") == "30\n" + assert "last observed=2/2/0/0/2" in result.stderr + assert "left running for inspection" in result.stderr + assert all("down" not in call for call in calls) + + +@pytest.mark.parametrize( + ("target", "make_variable", "launcher_variable"), + [ + ("dev", "DEV_WORKERS", "LLMBENCHLAB_DEV_WORKER_PROCESSES"), + ("docker-up", "WORKERS", "LLMBENCHLAB_COMPOSE_WORKER_PROCESSES"), + ], +) +def test_make_forwards_explicit_empty_worker_count_for_launcher_rejection( + target: str, + make_variable: str, + launcher_variable: str, +) -> None: + result = subprocess.run( + ["make", "--no-print-directory", "-n", target, f"{make_variable}="], + cwd=REPOSITORY_ROOT, + capture_output=True, + text=True, + timeout=10, + check=False, + ) + + assert result.returncode == 0 + assert f'{launcher_variable}=""' in result.stdout diff --git a/backend/tests/test_dev_script.py b/backend/tests/test_dev_script.py index 81845a8..66205f7 100644 --- a/backend/tests/test_dev_script.py +++ b/backend/tests/test_dev_script.py @@ -4,8 +4,10 @@ import os import re +import signal import stat import subprocess +import time from pathlib import Path import pytest @@ -46,11 +48,13 @@ def _prepare_project(tmp_path: Path, *, bootstrap_exit_code: int = 0) -> tuple[P service="$1" printf '%s child stdout\\n' "$service" printf '%s child stderr\\n' "$service" >&2 +printf '%s expected=%s\\n' "$service" "${LLMBENCHLAB_WORKER_EXPECTED_PROCESSES:-unset}" if [[ "${EXIT_SERVICE:-}" == "$service" ]]; then sleep "${EXIT_DELAY_SECONDS:-0.2}" exit "${EXIT_CODE:-0}" fi -trap 'printf "%s:terminated\\n" "$service" >>"$EVENT_FILE"; exit 0' TERM INT +trap '' INT +trap 'printf "%s:terminated\\n" "$service" >>"$EVENT_FILE"; exit 0' TERM while true; do sleep 0.1 done @@ -64,6 +68,9 @@ def _prepare_project(tmp_path: Path, *, bootstrap_exit_code: int = 0) -> tuple[P exec "$FAKE_SERVICE_SCRIPT" api fi if [[ " $* " == *" app.worker "* ]]; then + if [[ -n "${LLMBENCHLAB_DEV_WORKER_INDEX:-}" ]]; then + exec "$FAKE_SERVICE_SCRIPT" "worker-${LLMBENCHLAB_DEV_WORKER_INDEX}" + fi exec "$FAKE_SERVICE_SCRIPT" worker fi echo 'unexpected fake uv invocation' >&2 @@ -87,18 +94,17 @@ def _run_dev( *, exit_service: str, exit_code: int, + worker_processes: str = "1", + database_url: str | None = None, ) -> subprocess.CompletedProcess[str]: - event_file = project / "service-events.log" - environment = os.environ.copy() - environment.update( - { - "PATH": f"{fake_bin}{os.pathsep}{environment['PATH']}", - "FAKE_SERVICE_SCRIPT": str(fake_bin / "fake-service"), - "EVENT_FILE": str(event_file), - "EXIT_SERVICE": exit_service, - "EXIT_CODE": str(exit_code), - "LLMBENCHLAB_DEV_LOG_DIR": str(log_dir), - } + environment = _dev_environment( + project, + fake_bin, + log_dir, + exit_service=exit_service, + exit_code=exit_code, + worker_processes=worker_processes, + database_url=database_url, ) return subprocess.run( ["bash", str(project / "scripts" / "dev.sh")], @@ -111,6 +117,40 @@ def _run_dev( ) +def _dev_environment( + project: Path, + fake_bin: Path, + log_dir: Path, + *, + exit_service: str = "", + exit_code: int = 0, + worker_processes: str = "1", + database_url: str | None = None, +) -> dict[str, str]: + environment = os.environ.copy() + for variable_name in ( + "DATABASE_URL", + "LLMBENCHLAB_DATABASE_URL", + "LLMBENCHLAB_DEV_WORKER_PROCESSES", + "LLMBENCHLAB_WORKER_EXPECTED_PROCESSES", + ): + environment.pop(variable_name, None) + environment.update( + { + "PATH": f"{fake_bin}{os.pathsep}{environment['PATH']}", + "FAKE_SERVICE_SCRIPT": str(fake_bin / "fake-service"), + "EVENT_FILE": str(project / "service-events.log"), + "EXIT_SERVICE": exit_service, + "EXIT_CODE": str(exit_code), + "LLMBENCHLAB_DEV_LOG_DIR": str(log_dir), + "LLMBENCHLAB_DEV_WORKER_PROCESSES": worker_processes, + } + ) + if database_url is not None: + environment["LLMBENCHLAB_DATABASE_URL"] = database_url + return environment + + def test_dev_launcher_redirects_all_service_output_to_append_only_logs(tmp_path: Path) -> None: project, fake_bin = _prepare_project(tmp_path) log_dir = tmp_path / "custom-dev-logs" @@ -139,6 +179,7 @@ def test_dev_launcher_redirects_all_service_output_to_append_only_logs(tmp_path: ) assert f"{service} child stdout" in contents assert f"{service} child stderr" in contents + assert f"{service} expected=1" in contents assert stat.S_IMODE((log_dir / f"{service}.log").stat().st_mode) == 0o600 assert stat.S_IMODE(log_dir.stat().st_mode) == 0o700 @@ -183,3 +224,267 @@ def test_dev_launcher_suppresses_keyring_success_but_preserves_failure( assert "keyring success status" not in result.stderr assert result.stderr.strip() == "keyring bootstrap failed safely" assert not log_dir.exists() + + +def test_dev_launcher_runs_multiple_workers_with_private_logs_and_shared_expected_count( + tmp_path: Path, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + log_dir = tmp_path / "multi-worker-logs" + + result = _run_dev( + project, + fake_bin, + log_dir, + exit_service="worker-2", + exit_code=19, + worker_processes="3", + database_url="postgresql+psycopg://user:secret@db.example/bench", + ) + + assert result.returncode == 19 + console = result.stdout + result.stderr + assert "Error: Worker 2 exited with status 19." in result.stderr + assert "secret" not in console + assert "child stdout" not in console + assert "child stderr" not in console + assert "Workers" in result.stdout + + expected_logs = { + "api": log_dir / "api.log", + "worker-1": log_dir / "worker-1.log", + "worker-2": log_dir / "worker-2.log", + "worker-3": log_dir / "worker-3.log", + "frontend": log_dir / "frontend.log", + } + for service, log_path in expected_logs.items(): + contents = log_path.read_text(encoding="utf-8") + assert f"{service} child stdout" in contents + assert f"{service} child stderr" in contents + assert f"{service} expected=3" in contents + assert stat.S_IMODE(log_path.stat().st_mode) == 0o600 + + events = (project / "service-events.log").read_text(encoding="utf-8").splitlines() + assert set(events) == { + "api:terminated", + "worker-1:terminated", + "worker-3:terminated", + "frontend:terminated", + } + + +def test_explicit_worker_count_overrides_dotenv_launcher_default(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + log_dir = tmp_path / "dotenv-worker-override-logs" + (project / ".env").write_text( + "LLMBENCHLAB_DEV_WORKER_PROCESSES=1\n" + "LLMBENCHLAB_DATABASE_URL=postgresql://dotenv-user:dotenv-secret@db/bench\n", + encoding="utf-8", + ) + + result = _run_dev( + project, + fake_bin, + log_dir, + exit_service="worker-2", + exit_code=21, + worker_processes="2", + ) + + assert result.returncode == 21 + assert "Error: Worker 2 exited with status 21." in result.stderr + assert (log_dir / "worker-1.log").is_file() + assert (log_dir / "worker-2.log").is_file() + assert not (log_dir / "worker.log").exists() + assert "dotenv-secret" not in result.stdout + result.stderr + + +@pytest.mark.parametrize("worker_processes", ["0", "33", "two", "01", "2.5"]) +def test_dev_launcher_rejects_invalid_worker_process_count_before_logs_or_services( + tmp_path: Path, + worker_processes: str, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + log_dir = tmp_path / "invalid-count-logs" + + result = _run_dev( + project, + fake_bin, + log_dir, + exit_service="", + exit_code=0, + worker_processes=worker_processes, + database_url="postgresql://user:secret@db.example/bench", + ) + + assert result.returncode == 1 + assert result.stdout == "" + assert result.stderr.strip() == ( + "Error: LLMBENCHLAB_DEV_WORKER_PROCESSES must be an integer from 1 through 32." + ) + assert "secret" not in result.stderr + assert not log_dir.exists() + assert not (project / "service-events.log").exists() + + +@pytest.mark.parametrize( + "database_url", + [ + "sqlite:///./data/local.db", + "mysql+pymysql://user:do-not-print@db.example/bench", + "not-a-database-url-with-do-not-print", + ], +) +def test_dev_launcher_rejects_non_postgresql_multi_worker_before_logs_or_services( + tmp_path: Path, + database_url: str, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + log_dir = tmp_path / "wrong-database-logs" + + result = _run_dev( + project, + fake_bin, + log_dir, + exit_service="", + exit_code=0, + worker_processes="2", + database_url=database_url, + ) + + assert result.returncode == 1 + assert result.stdout == "" + assert result.stderr.strip() == ( + "Error: multiple development Workers require a PostgreSQL database URL." + ) + assert database_url not in result.stderr + assert "do-not-print" not in result.stderr + assert not log_dir.exists() + assert not (project / "service-events.log").exists() + + +def test_dev_launcher_uses_final_database_url_precedence_for_multi_worker_guard( + tmp_path: Path, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + log_dir = tmp_path / "precedence-logs" + (project / ".env").write_text( + "DATABASE_URL=postgresql://user:env-secret@db.example/bench\n" + "LLMBENCHLAB_DATABASE_URL=sqlite:///./data/winner.db\n", + encoding="utf-8", + ) + + result = _run_dev( + project, + fake_bin, + log_dir, + exit_service="", + exit_code=0, + worker_processes="2", + database_url="postgresql://user:process-secret@db.example/bench", + ) + + assert result.returncode == 1 + assert "PostgreSQL database URL" in result.stderr + assert "env-secret" not in result.stderr + assert "process-secret" not in result.stderr + assert not log_dir.exists() + assert not (project / "service-events.log").exists() + + +def test_dev_launcher_forwards_termination_and_cleans_up_every_child(tmp_path: Path) -> None: + project, fake_bin = _prepare_project(tmp_path) + log_dir = tmp_path / "signal-logs" + environment = _dev_environment( + project, + fake_bin, + log_dir, + worker_processes="2", + database_url="postgresql://user:secret@db.example/bench", + ) + process = subprocess.Popen( + ["bash", str(project / "scripts" / "dev.sh")], + cwd=project, + env=environment, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + expected_logs = [ + log_dir / "api.log", + log_dir / "worker-1.log", + log_dir / "worker-2.log", + log_dir / "frontend.log", + ] + deadline = time.monotonic() + 5 + while time.monotonic() < deadline: + if all(path.is_file() and "child stdout" in path.read_text() for path in expected_logs): + break + time.sleep(0.05) + else: + process.kill() + process.communicate(timeout=5) + pytest.fail("development services did not all start before the signal test deadline") + + process.send_signal(signal.SIGTERM) + stdout, stderr = process.communicate(timeout=10) + + assert process.returncode == 143 + assert "child stdout" not in stdout + stderr + events = (project / "service-events.log").read_text(encoding="utf-8").splitlines() + assert set(events) == { + "api:terminated", + "worker-1:terminated", + "worker-2:terminated", + "frontend:terminated", + } + + +def test_dev_launcher_converts_interrupt_to_term_cleanup_for_ignoring_children( + tmp_path: Path, +) -> None: + project, fake_bin = _prepare_project(tmp_path) + log_dir = tmp_path / "interrupt-logs" + environment = _dev_environment( + project, + fake_bin, + log_dir, + worker_processes="2", + database_url="postgresql://user:secret@db.example/bench", + ) + process = subprocess.Popen( + ["bash", str(project / "scripts" / "dev.sh")], + cwd=project, + env=environment, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + expected_logs = [ + log_dir / "api.log", + log_dir / "worker-1.log", + log_dir / "worker-2.log", + log_dir / "frontend.log", + ] + deadline = time.monotonic() + 5 + while time.monotonic() < deadline: + if all(path.is_file() and "child stdout" in path.read_text() for path in expected_logs): + break + time.sleep(0.05) + else: + process.kill() + process.communicate(timeout=5) + pytest.fail("development services did not all start before the interrupt test deadline") + + process.send_signal(signal.SIGINT) + stdout, stderr = process.communicate(timeout=10) + + assert process.returncode == 130 + assert "child stdout" not in stdout + stderr + events = (project / "service-events.log").read_text(encoding="utf-8").splitlines() + assert set(events) == { + "api:terminated", + "worker-1:terminated", + "worker-2:terminated", + "frontend:terminated", + } diff --git a/backend/tests/test_evaluation_cli.py b/backend/tests/test_evaluation_cli.py index 23d76e2..f9c562f 100644 --- a/backend/tests/test_evaluation_cli.py +++ b/backend/tests/test_evaluation_cli.py @@ -16,7 +16,7 @@ import app.cli.evaluate as evaluation_cli from app.core.constants import MAX_GENERATION_TOKENS from app.core.time import utc_now -from app.models import RunStatus +from app.models import ProviderType, RunStatus from app.providers import CanaryResult, ModelDiscoveryResult @@ -99,6 +99,90 @@ def test_help_exposes_only_api_key_environment_name(capsys: pytest.CaptureFixtur assert "--api-key" not in _all_option_strings(parser) +@pytest.mark.parametrize( + "provider_type", + [ProviderType.OPENAI_RESPONSES, ProviderType.ANTHROPIC_MESSAGES], +) +def test_new_provider_protocol_defaults_omit_seed(provider_type: ProviderType) -> None: + args = _parse_run("--provider-type", provider_type.value) + + defaults = evaluation_cli._provider_defaults(args) + + assert defaults == { + "temperature": None, + "top_p": None, + "max_tokens": 4000, + "seed": None, + "concurrency": 1, + } + + +@pytest.mark.parametrize( + "provider_type", + [ProviderType.OPENAI_RESPONSES, ProviderType.ANTHROPIC_MESSAGES], +) +def test_new_provider_protocol_preserves_explicit_sampling_parameters( + provider_type: ProviderType, +) -> None: + args = _parse_run( + "--provider-type", + provider_type.value, + "--temperature", + "0.5", + "--top-p", + "0.75", + ) + + defaults = evaluation_cli._provider_defaults(args) + + assert defaults["temperature"] == 0.5 + assert defaults["top_p"] == 0.75 + + +@pytest.mark.parametrize( + "provider_type", + [ProviderType.OPENAI_RESPONSES, ProviderType.ANTHROPIC_MESSAGES], +) +def test_new_provider_protocol_rejects_explicit_cli_seed( + provider_type: ProviderType, +) -> None: + args = _parse_run( + "--provider-type", + provider_type.value, + "--generation-seed", + "7", + ) + + with pytest.raises(evaluation_cli.EvaluationCLIError, match="does not support"): + evaluation_cli._provider_defaults(args) + + +def test_messages_protocol_rejects_temperature_above_one_in_cli() -> None: + args = _parse_run( + "--provider-type", + ProviderType.ANTHROPIC_MESSAGES.value, + "--temperature", + "1.5", + ) + + with pytest.raises(evaluation_cli.EvaluationCLIError, match="between 0 and 1"): + evaluation_cli._provider_defaults(args) + + +def test_model_payload_persists_explicit_provider_protocol() -> None: + payload = evaluation_cli._model_payload( + base_url="https://provider.example/zen/go/v1/responses", + remote_model="remote-model", + api_key_env="CLI_PROVIDER_KEY", + display_name=None, + input_price=None, + output_price=None, + provider_type=ProviderType.OPENAI_RESPONSES, + ) + + assert payload.provider_type == ProviderType.OPENAI_RESPONSES + + @pytest.mark.parametrize( "argv", [ @@ -490,9 +574,14 @@ async def test_resolve_and_canary_uses_discovery_but_passes_only_env_name_to_ada seen: dict[str, Any] = {} events: list[str] = [] - async def fake_discovery(base_url: str, api_key: str) -> ModelDiscoveryResult: + async def fake_discovery( + base_url: str, + api_key: str, + *, + provider_type: ProviderType, + ) -> ModelDiscoveryResult: events.append("discovery") - seen["discovery"] = (base_url, api_key) + seen["discovery"] = (base_url, api_key, provider_type) return ModelDiscoveryResult(models=("only-model",), request_id="fixture-discovery") async def fake_canary( @@ -529,7 +618,11 @@ def before_canary(remote_model: str) -> None: assert discovery is not None assert canary == _canary() assert events == ["discovery", "local-validation", "canary"] - assert seen["discovery"] == ("https://provider.example/v1", "offline-secret") + assert seen["discovery"] == ( + "https://provider.example/v1", + "offline-secret", + ProviderType.OPENAI_COMPATIBLE, + ) assert seen["canary"] == ( "https://provider.example/v1", "only-model", @@ -538,6 +631,88 @@ def before_canary(remote_model: str) -> None: ) +@pytest.mark.asyncio +async def test_resolve_and_canary_uses_explicit_non_chat_protocol( + monkeypatch: pytest.MonkeyPatch, +) -> None: + seen: dict[str, Any] = {} + + async def unexpected_chat_canary(*_args: Any, **_kwargs: Any) -> CanaryResult: + pytest.fail("a Responses Run must not use the Chat Completions canary") + + async def fake_provider_canary( + provider_type: ProviderType, + base_url: str, + remote_model: str, + api_key_env: str, + generation: dict[str, Any], + ) -> CanaryResult: + seen["canary"] = ( + provider_type, + base_url, + remote_model, + api_key_env, + generation, + ) + return _canary() + + monkeypatch.setattr(evaluation_cli, "run_chat_canary", unexpected_chat_canary) + monkeypatch.setattr(evaluation_cli, "run_provider_canary", fake_provider_canary) + + remote_model, discovery, canary = await evaluation_cli._resolve_and_canary( + base_url="https://provider.example/zen/go/v1/responses", + requested_model="remote-model", + api_key="offline-secret", + api_key_env="CLI_PROVIDER_KEY", + generation={"temperature": 0, "top_p": 1, "max_tokens": 64, "seed": None}, + skip_discovery=True, + question_count=2, + run_attempts=3, + yes=True, + provider_type=ProviderType.OPENAI_RESPONSES, + ) + + assert remote_model == "remote-model" + assert discovery is None + assert canary == _canary() + assert seen["canary"] == ( + ProviderType.OPENAI_RESPONSES, + "https://provider.example/zen/go/v1/responses", + "remote-model", + "CLI_PROVIDER_KEY", + {"temperature": 0, "top_p": 1, "max_tokens": 64, "seed": None}, + ) + + +def test_resume_reads_explicit_provider_protocol_snapshot() -> None: + run = SimpleNamespace( + model_parameters_snapshot={ + "model": { + "adapter_type": "anthropic_messages", + "base_url": "https://provider.example/zen/go/v1/messages", + "remote_model_name": "remote-model", + "api_key_env": "CLI_PROVIDER_KEY", + }, + "generation": { + "temperature": 0, + "top_p": 1, + "max_tokens": 64, + "seed": None, + }, + } + ) + + provider_type, base_url, remote_model, api_key_env, generation = ( + evaluation_cli._run_provider_configuration(run) + ) + + assert provider_type == ProviderType.ANTHROPIC_MESSAGES + assert base_url.endswith("/messages") + assert remote_model == "remote-model" + assert api_key_env == "CLI_PROVIDER_KEY" + assert generation["seed"] is None + + @pytest.mark.asyncio async def test_run_refuses_active_database_run_before_dataset_or_provider_access( monkeypatch: pytest.MonkeyPatch, @@ -667,7 +842,8 @@ async def test_resume_temporarily_copies_source_key_and_evaluates_only_missing_q status=RunStatus.PENDING, total_questions=10, completed_questions=3, - attempt_count=1, + attempt_count=3, + failed_attempt_count=0, max_attempts=3, model_parameters_snapshot={ "model": { @@ -707,7 +883,7 @@ async def fake_preflight(**kwargs: Any) -> tuple[str, None, CanaryResult]: assert kwargs["api_key"] == "offline-resume-secret" assert kwargs["api_key_env"] == "CLI_FROZEN_TARGET_KEY" assert kwargs["question_count"] == 7 - assert kwargs["run_attempts"] == 2 + assert kwargs["run_attempts"] == 3 assert evaluation_cli.os.environ["CLI_FROZEN_TARGET_KEY"] == ("offline-resume-secret") return "remote-model", None, _canary() @@ -745,6 +921,7 @@ async def test_resume_refuses_exhausted_run_before_accessing_key_or_provider( total_questions=10, completed_questions=5, attempt_count=3, + failed_attempt_count=3, max_attempts=3, model_parameters_snapshot={ "model": { diff --git a/backend/tests/test_migrations.py b/backend/tests/test_migrations.py index b02ad63..749a06b 100644 --- a/backend/tests/test_migrations.py +++ b/backend/tests/test_migrations.py @@ -18,6 +18,7 @@ GOVERNANCE_REVISION, INDEX_REPAIR_REVISION, LEGACY_REVISION, + OBSERVATIONAL_OVERDRAW_REVISION, PHASE_1_REVISION, WORKER_PROGRESS_REVISION, SchemaPreparationError, @@ -29,7 +30,7 @@ from app.db.session import create_database_engine BACKEND_ROOT = Path(__file__).resolve().parents[1] -HEAD_REVISION = "20260830_0007" +HEAD_REVISION = "20260830_0008" WEB_CREDENTIAL_REVISION = CREDENTIAL_REVISION RELIABILITY_REVISION = "20260825_0002" @@ -418,6 +419,84 @@ def test_clean_migration_round_trip(tmp_path: Path) -> None: assert _read_heads(database_path) == (HEAD_REVISION,) +def test_provider_protocol_migration_preserves_chat_and_guards_lossy_downgrade( + tmp_path: Path, +) -> None: + database_path = tmp_path / "provider-protocols.db" + _run_alembic(database_path, "upgrade", LEGACY_REVISION) + _insert_legacy_rows(database_path) + _run_alembic(database_path, "upgrade", OBSERVATIONAL_OVERDRAW_REVISION) + + _run_alembic(database_path, "upgrade", "head") + + with sqlite3.connect(database_path) as connection: + provider_column = next( + row + for row in connection.execute("PRAGMA table_info(models)") + if row[1] == "provider_type" + ) + assert provider_column[2].upper() == "VARCHAR(18)" + for model_id, provider_type in ( + ("model-responses", "openai_responses"), + ("model-messages", "anthropic_messages"), + ): + connection.execute( + "INSERT INTO models (" + "id, name, provider_type, base_url, remote_model_name, api_key_env, " + "credential_source, enabled, input_price_per_million, " + "output_price_per_million, default_parameters, created_at, updated_at" + ") VALUES (?, ?, ?, 'https://provider.example/v1', 'remote-model', " + "'PROVIDER_KEY', 'environment', 1, NULL, NULL, '{}', " + "CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)", + (model_id, provider_type, provider_type), + ) + assert connection.execute("SELECT provider_type FROM models ORDER BY id").fetchall() == [ + ("mock",), + ("anthropic_messages",), + ("openai_responses",), + ] + + failed = _invoke_alembic( + database_path, + "downgrade", + OBSERVATIONAL_OVERDRAW_REVISION, + ) + assert failed.returncode != 0 + assert "Cannot downgrade while openai_responses or anthropic_messages Models exist" in ( + failed.stderr + ) + assert _read_heads(database_path) == (HEAD_REVISION,) + + with sqlite3.connect(database_path) as connection: + assert ( + connection.execute( + "SELECT COUNT(*) FROM models WHERE provider_type IN " + "('openai_responses', 'anthropic_messages')" + ).fetchone()[0] + == 2 + ) + connection.execute( + "DELETE FROM models WHERE provider_type IN ('openai_responses', 'anthropic_messages')" + ) + + _run_alembic(database_path, "downgrade", OBSERVATIONAL_OVERDRAW_REVISION) + assert _read_heads(database_path) == (OBSERVATIONAL_OVERDRAW_REVISION,) + with sqlite3.connect(database_path) as connection: + assert connection.execute("SELECT id, provider_type FROM models").fetchall() == [ + ("model-1", "mock") + ] + with pytest.raises(sqlite3.IntegrityError): + connection.execute( + "INSERT INTO models (" + "id, name, provider_type, base_url, remote_model_name, api_key_env, " + "credential_source, enabled, input_price_per_million, " + "output_price_per_million, default_parameters, created_at, updated_at" + ") VALUES ('model-rejected', 'Rejected', 'openai_responses', " + "'https://provider.example/v1', 'remote-model', 'PROVIDER_KEY', " + "'environment', 1, NULL, NULL, '{}', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)" + ) + + def test_prepare_adopts_legacy_schema_and_preserves_rows(tmp_path: Path) -> None: database_path = tmp_path / "legacy.db" _run_alembic(database_path, "upgrade", LEGACY_REVISION) @@ -899,7 +978,7 @@ def test_overdraw_repair_preserves_explicit_hard_bound_overdraw( ("start_revision", "direction", "target_revision"), [ (INDEX_REPAIR_REVISION, "upgrade", "head"), - (HEAD_REVISION, "downgrade", INDEX_REPAIR_REVISION), + (OBSERVATIONAL_OVERDRAW_REVISION, "downgrade", INDEX_REPAIR_REVISION), ], ) def test_overdraw_repair_refuses_active_reservations_before_mutation( @@ -952,11 +1031,14 @@ def test_overdraw_repair_refuses_active_reservations_before_mutation( ("differences", "expected_row_guard_calls"), [ ( - (), + ("modify_type:models.provider_type:VARCHAR(17)->VARCHAR(18)",), 0, ), ( - ("add_index:evaluation_runs.ix_evaluation_runs_started_at_id",), + ( + "add_index:evaluation_runs.ix_evaluation_runs_started_at_id", + "modify_type:models.provider_type:VARCHAR(17)->VARCHAR(18)", + ), 1, ), ( @@ -964,6 +1046,7 @@ def test_overdraw_repair_refuses_active_reservations_before_mutation( "add_index:evaluation_runs.ix_evaluation_runs_finished_at_id", "add_index:evaluation_runs.ix_evaluation_runs_started_at_id", "add_index:governance_policies.uq_governance_policies_single_active", + "modify_type:models.provider_type:VARCHAR(17)->VARCHAR(18)", ), 1, ), @@ -988,6 +1071,40 @@ def test_prepare_postgresql_0005_accepts_only_canonical_or_known_index_gap( engine.dispose.assert_called_once_with() +def test_provider_type_difference_fingerprint_includes_exact_source_and_target_width() -> None: + difference = ( + "modify_type", + None, + "models", + "provider_type", + {}, + prepare_migrations_module.sa.String(length=16), + prepare_migrations_module.sa.String(length=18), + ) + + assert prepare_migrations_module._difference_fingerprint(difference) == ( + "modify_type:models.provider_type:VARCHAR(16)->VARCHAR(18)" + ) + + +def test_prepare_postgresql_rejects_unexpected_provider_type_width( + monkeypatch: pytest.MonkeyPatch, +) -> None: + engine, connection, metadata_calls = _mock_postgresql_historical_preflight( + monkeypatch, + differences=("modify_type:models.provider_type:VARCHAR(16)->VARCHAR(18)",), + source_revision=OBSERVATIONAL_OVERDRAW_REVISION, + ) + + with pytest.raises(SchemaPreparationError, match="historical database"): + prepare_database("postgresql+psycopg://localhost/llmbenchlab_test") + + assert metadata_calls == [connection] + connection.execute.assert_not_called() + connection.close.assert_called_once_with() + engine.dispose.assert_called_once_with() + + def test_prepare_postgresql_0005_rejects_unknown_drift_with_known_index_gap( monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -996,6 +1113,7 @@ def test_prepare_postgresql_0005_rejects_unknown_drift_with_known_index_gap( differences=( "add_index:evaluation_runs.ix_evaluation_runs_started_at_id", "add_index:models.ix_models_enabled", + "modify_type:models.provider_type:VARCHAR(17)->VARCHAR(18)", ), ) @@ -1013,7 +1131,7 @@ def test_prepare_postgresql_0006_checks_canonical_metadata( ) -> None: engine, connection, metadata_calls = _mock_postgresql_historical_preflight( monkeypatch, - differences=(), + differences=("modify_type:models.provider_type:VARCHAR(17)->VARCHAR(18)",), source_revision=INDEX_REPAIR_REVISION, ) @@ -1031,7 +1149,10 @@ def test_prepare_postgresql_0006_rejects_metadata_drift( ) -> None: engine, connection, metadata_calls = _mock_postgresql_historical_preflight( monkeypatch, - differences=("add_index:models.ix_models_enabled",), + differences=( + "add_index:models.ix_models_enabled", + "modify_type:models.provider_type:VARCHAR(17)->VARCHAR(18)", + ), source_revision=INDEX_REPAIR_REVISION, ) diff --git a/backend/tests/test_phase2_acceptance_script.py b/backend/tests/test_phase2_acceptance_script.py index ad2694e..b75cae0 100644 --- a/backend/tests/test_phase2_acceptance_script.py +++ b/backend/tests/test_phase2_acceptance_script.py @@ -34,7 +34,7 @@ def test_migration_acceptance_retains_both_populated_guards_and_empty_round_trip assert script.PRE_GOVERNANCE_REVISION == "20260827_0003" assert script.GOVERNANCE_REVISION == "20260827_0004" assert script.WORKER_PROGRESS_REVISION == "20260828_0005" - assert script.DATABASE_HEAD_REVISION == "20260830_0007" + assert script.DATABASE_HEAD_REVISION == "20260830_0008" assert "postgres_populated_0005_and_0004_downgrade_guards_with_empty_round_trips" in source assert "Cannot downgrade Worker progress schema" in source assert "Cannot downgrade governance schema" in source diff --git a/backend/tests/test_phase2_capacity_script.py b/backend/tests/test_phase2_capacity_script.py index c88d003..6e99a31 100644 --- a/backend/tests/test_phase2_capacity_script.py +++ b/backend/tests/test_phase2_capacity_script.py @@ -1976,7 +1976,7 @@ def test_acceptance_requires_both_populated_guards_and_empty_round_trips() -> No assert 'PRE_GOVERNANCE_REVISION = "20260827_0003"' in source assert 'GOVERNANCE_REVISION = "20260827_0004"' in source assert 'WORKER_PROGRESS_REVISION = "20260828_0005"' in source - assert 'DATABASE_HEAD_REVISION = "20260830_0007"' in source + assert 'DATABASE_HEAD_REVISION = "20260830_0008"' in source assert "application_worker_progress_0005_to_0004_guard" in source assert "isolated_governance_0004_to_0003_guard" in source assert "worker_progress_0005_to_0004_round_trip" in source diff --git a/backend/tests/test_provider_preflight.py b/backend/tests/test_provider_preflight.py index e126881..4467c45 100644 --- a/backend/tests/test_provider_preflight.py +++ b/backend/tests/test_provider_preflight.py @@ -6,12 +6,14 @@ import pytest from pydantic import ValidationError +from app.adapters import ModelGenerationResult from app.models import ProviderType from app.providers import ( ProviderPreflightError, discover_models, models_url, run_chat_canary, + run_provider_canary, select_remote_model, ) from app.schemas.model import ModelCreate @@ -40,6 +42,14 @@ async def aclose(self) -> None: "https://provider.example/openai/v1/chat/completions", "https://provider.example/openai/v1/models", ), + ( + "https://provider.example/zen/go/v1/responses", + "https://provider.example/zen/go/v1/models", + ), + ( + "https://provider.example/zen/go/v1/messages", + "https://provider.example/zen/go/v1/models", + ), ("http://127.0.0.1:11434/v1/", "http://127.0.0.1:11434/v1/models"), ("http://localhost:11434/v1", "http://localhost:11434/v1/models"), ("http://[::1]:11434/v1", "http://[::1]:11434/v1/models"), @@ -145,6 +155,123 @@ def handler(request: httpx.Request) -> httpx.Response: assert result.request_id == "discovery-1" +@pytest.mark.asyncio +async def test_messages_model_discovery_uses_native_headers_and_pagination() -> None: + seen: list[dict[str, str | None]] = [] + + def handler(request: httpx.Request) -> httpx.Response: + seen.append( + { + "url": str(request.url), + "authorization": request.headers.get("authorization"), + "x_api_key": request.headers.get("x-api-key"), + "anthropic_version": request.headers.get("anthropic-version"), + "accept_encoding": request.headers.get("accept-encoding"), + } + ) + if request.url.params.get("after_id") is None: + return httpx.Response( + 200, + headers={"request-id": "messages-discovery-1"}, + json={ + "data": [{"id": "message-model-b"}], + "has_more": True, + "last_id": "message-model-b", + }, + ) + return httpx.Response( + 200, + json={ + "data": [{"id": "message-model-a"}], + "has_more": False, + "last_id": "message-model-a", + }, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + result = await discover_models( + "https://provider.example/v1/messages", + "messages-secret", + provider_type=ProviderType.ANTHROPIC_MESSAGES, + client=client, + ) + + assert seen == [ + { + "url": "https://provider.example/v1/models", + "authorization": None, + "x_api_key": "messages-secret", + "anthropic_version": "2023-06-01", + "accept_encoding": "identity", + }, + { + "url": "https://provider.example/v1/models?after_id=message-model-b", + "authorization": None, + "x_api_key": "messages-secret", + "anthropic_version": "2023-06-01", + "accept_encoding": "identity", + }, + ] + assert result.models == ("message-model-a", "message-model-b") + assert result.request_id == "messages-discovery-1" + + +@pytest.mark.asyncio +async def test_messages_model_discovery_stops_unbounded_cursor_pagination( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr("app.providers.preflight.MAX_DISCOVERY_PAGES", 2) + calls = 0 + + def handler(_request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + return httpx.Response( + 200, + json={"data": [], "has_more": True, "last_id": f"cursor-{calls}"}, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + with pytest.raises(ProviderPreflightError) as caught: + await discover_models( + "https://provider.example/v1", + "messages-secret", + provider_type=ProviderType.ANTHROPIC_MESSAGES, + client=client, + ) + + assert calls == 2 + assert caught.value.code == "model_discovery_too_many_pages" + + +@pytest.mark.asyncio +async def test_messages_model_discovery_enforces_total_wall_clock_deadline( + monkeypatch: pytest.MonkeyPatch, +) -> None: + timestamps = iter((100.0, 161.0)) + + class FakeClock: + @staticmethod + def monotonic() -> float: + return next(timestamps) + + monkeypatch.setattr("app.providers.preflight.time", FakeClock()) + + def unexpected_request(_request: httpx.Request) -> httpx.Response: + pytest.fail("expired discovery must fail before another request") + + async with httpx.AsyncClient(transport=httpx.MockTransport(unexpected_request)) as client: + with pytest.raises(ProviderPreflightError) as caught: + await discover_models( + "https://provider.example/v1", + "messages-secret", + provider_type=ProviderType.ANTHROPIC_MESSAGES, + client=client, + ) + + assert caught.value.code == "model_discovery_deadline_exceeded" + + @pytest.mark.asyncio async def test_discover_models_stops_reading_oversized_response( monkeypatch: pytest.MonkeyPatch, @@ -307,6 +434,64 @@ def handler(request: httpx.Request) -> httpx.Response: assert result.attempts == 1 +@pytest.mark.asyncio +async def test_provider_canary_builds_the_explicit_protocol_adapter( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setenv("CANARY_KEY", "secret") + seen: dict[str, object] = {} + + class FakeAdapter: + async def generate(self, messages, generation_config): + seen["messages"] = messages + seen["generation"] = generation_config + return ModelGenerationResult( + text="A", + input_tokens=2, + output_tokens=1, + latency_ms=1.5, + provider_request_id="provider-canary", + metadata={ + "returned_model": "remote-model", + "finish_reason": "end_turn", + "attempts": 1, + }, + ) + + async def aclose(self) -> None: + seen["closed"] = True + + def fake_build_adapter(provider_type: str, **options): + seen["provider_type"] = provider_type + seen["options"] = options + return FakeAdapter() + + monkeypatch.setattr("app.providers.preflight.build_adapter", fake_build_adapter) + + result = await run_provider_canary( + ProviderType.OPENAI_RESPONSES, + "https://provider.example/zen/go/v1/responses", + "remote-model", + "CANARY_KEY", + {"temperature": 0, "top_p": 1, "max_tokens": None, "seed": None}, + ) + + assert seen["provider_type"] == "openai_responses" + assert seen["options"] == { + "base_url": "https://provider.example/zen/go/v1/responses", + "remote_model_name": "remote-model", + "api_key_env": "CANARY_KEY", + "client": None, + } + assert seen["messages"] == [ + {"role": "user", "content": "Compatibility check: reply with exactly A."} + ] + assert seen["generation"] == {"temperature": 0, "top_p": 1, "max_tokens": 16} + assert seen["closed"] is True + assert result.provider_request_id == "provider-canary" + assert result.finish_reason == "end_turn" + + @pytest.mark.asyncio async def test_chat_canary_redacts_key_reflected_in_finish_reason( monkeypatch: pytest.MonkeyPatch, diff --git a/backend/tests/test_provider_protocol_adapters.py b/backend/tests/test_provider_protocol_adapters.py new file mode 100644 index 0000000..82f0809 --- /dev/null +++ b/backend/tests/test_provider_protocol_adapters.py @@ -0,0 +1,840 @@ +from __future__ import annotations + +import json +import traceback + +import httpx +import pytest + +from app.adapters import ( + AdapterError, + OpenAICompatibleAdapter, + ProviderAttemptContext, + ProviderAttemptDisposition, + ProviderAttemptOutcome, + ProviderAttemptPermit, +) +from app.adapters.provider_protocols import ( + AnthropicMessagesAdapter, + OpenAIResponsesAdapter, +) + + +def _typed_sse(event_type: str, body: dict[str, object]) -> bytes: + return ( + f"event: {event_type}\n".encode() + + b"data: " + + json.dumps(body, ensure_ascii=False).encode() + + b"\n\n" + ) + + +class _AttemptController: + def __init__(self) -> None: + self.events: list[tuple[object, ...]] = [] + + async def reserve( + self, + context: ProviderAttemptContext, + *, + provider_attempt: int, + ) -> ProviderAttemptPermit: + self.events.append(("reserve", provider_attempt, context.question_id)) + return ProviderAttemptPermit( + reservation_id=f"reservation-{provider_attempt}", + provider_attempt=provider_attempt, + ) + + async def mark_send_started(self, permit: ProviderAttemptPermit) -> None: + self.events.append(("mark", permit.provider_attempt)) + + async def finish( + self, + permit: ProviderAttemptPermit, + *, + disposition: ProviderAttemptDisposition, + outcome: ProviderAttemptOutcome, + input_tokens: int | None = None, + output_tokens: int | None = None, + ) -> None: + self.events.append( + ( + "finish", + permit.provider_attempt, + disposition, + outcome, + input_tokens, + output_tokens, + ) + ) + + +def _attempt_context() -> ProviderAttemptContext: + return ProviderAttemptContext( + run_id="run-1", + question_id="question-1", + model_id="model-1", + provider_scope="provider-1", + lease_token=3, + execution_generation=2, + next_provider_attempt=1, + reserved_input_tokens=16, + reserved_output_tokens=8, + ) + + +@pytest.mark.parametrize( + ("adapter_type", "root_url", "full_url"), + [ + ( + OpenAIResponsesAdapter, + "https://provider.example/v1", + "https://provider.example/v1/responses", + ), + ( + AnthropicMessagesAdapter, + "https://provider.example/v1", + "https://provider.example/v1/messages", + ), + ], +) +def test_explicit_protocol_adapters_accept_root_or_matching_full_endpoint( + adapter_type: type[OpenAICompatibleAdapter], + root_url: str, + full_url: str, +) -> None: + assert adapter_type(root_url, "model", api_key="key").endpoint_url == full_url + assert adapter_type(full_url, "model", api_key="key").endpoint_url == full_url + + +@pytest.mark.parametrize( + ("adapter_type", "wrong_url"), + [ + (OpenAICompatibleAdapter, "https://provider.example/v1/responses"), + (OpenAICompatibleAdapter, "https://provider.example/v1/messages"), + (OpenAIResponsesAdapter, "https://provider.example/v1/chat/completions"), + (OpenAIResponsesAdapter, "https://provider.example/v1/messages"), + (AnthropicMessagesAdapter, "https://provider.example/v1/chat/completions"), + (AnthropicMessagesAdapter, "https://provider.example/v1/responses"), + ], +) +def test_explicit_protocol_adapters_reject_known_wrong_endpoint_suffix( + adapter_type: type[OpenAICompatibleAdapter], + wrong_url: str, +) -> None: + with pytest.raises(ValueError, match="does not match the selected"): + adapter_type(wrong_url, "model", api_key="key") + + +@pytest.mark.parametrize("adapter_type", [OpenAIResponsesAdapter, AnthropicMessagesAdapter]) +@pytest.mark.asyncio +async def test_protocol_json_parse_errors_do_not_retain_provider_secrets( + adapter_type: type[OpenAICompatibleAdapter], +) -> None: + secret = "malformed-json-provider-secret" + transport = httpx.MockTransport( + lambda request: httpx.Response( + 200, + content=f'{{"reflected":"{secret}"'.encode(), + request=request, + ) + ) + + async with httpx.AsyncClient(transport=transport) as client: + with pytest.raises(AdapterError) as caught: + await adapter_type( + "https://provider.example/v1", + "model", + api_key=secret, + client=client, + ).generate( + [{"role": "user", "content": "question"}], + {"max_tokens": 16}, + ) + + assert caught.value.error_type == "invalid_provider_response" + assert caught.value.__cause__ is None + assert caught.value.__context__ is None + formatted = "".join(traceback.format_exception(caught.type, caught.value, caught.tb)) + assert secret not in formatted + + +@pytest.mark.parametrize( + ("adapter_type", "event_type"), + [ + (OpenAIResponsesAdapter, "response.output_text.delta"), + (AnthropicMessagesAdapter, "content_block_delta"), + ], +) +@pytest.mark.asyncio +async def test_protocol_sse_parse_errors_do_not_retain_provider_secrets( + adapter_type: type[OpenAICompatibleAdapter], + event_type: str, +) -> None: + secret = "malformed-sse-provider-secret" + payload = f'event: {event_type}\ndata: {{"reflected":"{secret}"\n\n'.encode() + transport = httpx.MockTransport( + lambda request: httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=payload, + request=request, + ) + ) + + async with httpx.AsyncClient(transport=transport) as client: + with pytest.raises(AdapterError) as caught: + await adapter_type( + "https://provider.example/v1", + "model", + api_key=secret, + client=client, + ).generate( + [{"role": "user", "content": "question"}], + {"max_tokens": 16}, + ) + + assert caught.value.error_type == "invalid_provider_stream" + assert caught.value.__cause__ is None + assert caught.value.__context__ is None + formatted = "".join(traceback.format_exception(caught.type, caught.value, caught.tb)) + assert secret not in formatted + + +@pytest.mark.asyncio +async def test_responses_json_request_and_result_contract() -> None: + seen: dict[str, object] = {} + + def handler(request: httpx.Request) -> httpx.Response: + seen["url"] = str(request.url) + seen["authorization"] = request.headers.get("authorization") + seen["accept_encoding"] = request.headers.get("accept-encoding") + seen["payload"] = json.loads(request.content) + return httpx.Response( + 200, + json={ + "id": "resp-1", + "model": "resolved-model", + "status": "completed", + "output": [ + { + "type": "message", + "role": "assistant", + "content": [ + {"type": "output_text", "text": "first "}, + {"type": "refusal", "refusal": "ignored"}, + {"type": "output_text", "text": "answer"}, + ], + } + ], + "usage": {"input_tokens": 11, "output_tokens": 3, "total_tokens": 14}, + }, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + result = await OpenAIResponsesAdapter( + "https://provider.example/v1", + "remote-model", + api_key="responses-key", + client=client, + ).generate( + [{"role": "user", "content": "question"}], + { + "system_prompt": "Be concise.", + "temperature": 0.2, + "top_p": 0.9, + "max_tokens": 64, + }, + ) + + assert seen == { + "url": "https://provider.example/v1/responses", + "authorization": "Bearer responses-key", + "accept_encoding": "identity", + "payload": { + "model": "remote-model", + "input": [ + {"role": "system", "content": "Be concise."}, + {"role": "user", "content": "question"}, + ], + "stream": True, + "temperature": 0.2, + "top_p": 0.9, + "max_output_tokens": 64, + }, + } + assert result.text == "first answer" + assert result.input_tokens == 11 + assert result.output_tokens == 3 + assert result.provider_request_id == "resp-1" + assert result.metadata == { + "adapter": "openai_responses", + "attempts": 1, + "response_mode": "json", + "finish_reason": "completed", + "returned_model": "resolved-model", + } + + +@pytest.mark.asyncio +async def test_responses_typed_sse_requires_completed_and_uses_terminal_usage() -> None: + stream = b"".join( + [ + _typed_sse( + "response.created", + { + "type": "response.created", + "response": {"id": "resp-sse", "model": "model", "status": "in_progress"}, + }, + ), + _typed_sse( + "response.output_text.delta", + {"type": "response.output_text.delta", "delta": "stream "}, + ), + _typed_sse( + "response.output_text.delta", + {"type": "response.output_text.delta", "delta": "answer"}, + ), + _typed_sse( + "response.completed", + { + "type": "response.completed", + "response": { + "id": "resp-sse", + "model": "model", + "status": "completed", + "output": [], + "usage": {"input_tokens": 9, "output_tokens": 2}, + }, + }, + ), + ] + ) + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + result = await OpenAIResponsesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ).generate([{"role": "user", "content": "question"}], {}) + + assert result.text == "stream answer" + assert result.input_tokens == 9 + assert result.output_tokens == 2 + assert result.metadata["adapter"] == "openai_responses" + assert result.metadata["response_mode"] == "sse" + + +@pytest.mark.asyncio +async def test_responses_stream_eof_without_completed_fails_closed() -> None: + stream = _typed_sse( + "response.output_text.delta", + {"type": "response.output_text.delta", "delta": "partial"}, + ) + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + adapter = OpenAIResponsesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ) + with pytest.raises(AdapterError) as caught: + await adapter.generate([{"role": "user", "content": "question"}], {}) + + assert caught.value.error_type == "incomplete_provider_stream" + assert "response.completed" in caught.value.error_message + + +@pytest.mark.asyncio +async def test_responses_rejects_seed_before_sending_or_retrying() -> None: + calls = 0 + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + return httpx.Response(500, request=request) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + adapter = OpenAIResponsesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ) + with pytest.raises(AdapterError) as caught: + await adapter.generate( + [{"role": "user", "content": "question"}], + {"seed": 0}, + ) + + assert caught.value.error_type == "invalid_request" + assert calls == 0 + + +@pytest.mark.asyncio +async def test_responses_reuses_retry_and_attempt_settlement_contract() -> None: + calls = 0 + delays: list[float] = [] + controller = _AttemptController() + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + if calls == 1: + return httpx.Response(429, json={"error": {"message": "retry"}}, request=request) + return httpx.Response( + 200, + json={ + "id": "resp-retried", + "status": "completed", + "output": [ + { + "type": "message", + "content": [{"type": "output_text", "text": "answer"}], + } + ], + "usage": {"input_tokens": 3, "output_tokens": 1}, + }, + request=request, + ) + + async def fake_sleep(delay: float) -> None: + delays.append(delay) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + result = await OpenAIResponsesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + max_retries=1, + sleep=fake_sleep, + attempt_controller=controller, + ).generate( + [{"role": "user", "content": "question"}], + {}, + attempt_context=_attempt_context(), + ) + + assert result.text == "answer" + assert result.metadata["attempts"] == 2 + assert delays == [0.25] + assert controller.events == [ + ("reserve", 1, "question-1"), + ("mark", 1), + ( + "finish", + 1, + ProviderAttemptDisposition.SETTLED_CONSERVATIVE, + ProviderAttemptOutcome.HTTP_ERROR, + None, + None, + ), + ("reserve", 2, "question-1"), + ("mark", 2), + ( + "finish", + 2, + ProviderAttemptDisposition.SETTLED_ACTUAL, + ProviderAttemptOutcome.SUCCEEDED, + 3, + 1, + ), + ] + + +@pytest.mark.asyncio +async def test_responses_reuses_body_limit_and_does_not_follow_redirects() -> None: + redirect_calls = 0 + + def redirect_handler(request: httpx.Request) -> httpx.Response: + nonlocal redirect_calls + redirect_calls += 1 + return httpx.Response( + 307, + headers={"location": "https://other.example/v1/responses"}, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(redirect_handler)) as client: + adapter = OpenAIResponsesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ) + with pytest.raises(AdapterError) as redirect_error: + await adapter.generate([{"role": "user", "content": "question"}], {}) + + assert redirect_error.value.error_type == "provider_http_error" + assert redirect_calls == 1 + + def oversized_handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + headers={"content-length": str(4 * 1024 * 1024 + 1)}, + content=b"{}", + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(oversized_handler)) as client: + adapter = OpenAIResponsesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ) + with pytest.raises(AdapterError) as oversized_error: + await adapter.generate([{"role": "user", "content": "question"}], {}) + + assert oversized_error.value.error_type == "provider_response_too_large" + + +@pytest.mark.asyncio +async def test_messages_json_request_and_result_contract() -> None: + seen: dict[str, object] = {} + + def handler(request: httpx.Request) -> httpx.Response: + seen["url"] = str(request.url) + seen["x_api_key"] = request.headers.get("x-api-key") + seen["authorization"] = request.headers.get("authorization") + seen["anthropic_version"] = request.headers.get("anthropic-version") + seen["accept_encoding"] = request.headers.get("accept-encoding") + seen["payload"] = json.loads(request.content) + return httpx.Response( + 200, + json={ + "id": "msg-1", + "type": "message", + "model": "resolved-model", + "content": [ + {"type": "text", "text": "message "}, + {"type": "thinking", "thinking": "ignored"}, + {"type": "text", "text": "answer"}, + ], + "stop_reason": "end_turn", + "usage": {"input_tokens": 13, "output_tokens": 4}, + }, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + result = await AnthropicMessagesAdapter( + "https://provider.example/v1/messages", + "remote-model", + api_key="messages-key", + client=client, + ).generate( + [{"role": "user", "content": "question"}], + { + "system_prompt": "Be concise.", + "temperature": 0.1, + "top_p": 0.8, + "max_tokens": 72, + }, + ) + + assert seen == { + "url": "https://provider.example/v1/messages", + "x_api_key": "messages-key", + "authorization": None, + "anthropic_version": "2023-06-01", + "accept_encoding": "identity", + "payload": { + "model": "remote-model", + "messages": [{"role": "user", "content": "question"}], + "stream": True, + "max_tokens": 72, + "system": "Be concise.", + "temperature": 0.1, + "top_p": 0.8, + }, + } + assert result.text == "message answer" + assert result.input_tokens == 13 + assert result.output_tokens == 4 + assert result.provider_request_id == "msg-1" + assert result.metadata == { + "adapter": "anthropic_messages", + "attempts": 1, + "response_mode": "json", + "finish_reason": "end_turn", + "returned_model": "resolved-model", + } + + +@pytest.mark.asyncio +async def test_messages_rejects_temperature_above_protocol_limit_before_sending() -> None: + calls = 0 + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + return httpx.Response(500, request=request) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + adapter = AnthropicMessagesAdapter( + "https://provider.example/v1", + "messages-model", + api_key="fake-key", + client=client, + ) + with pytest.raises(AdapterError, match="temperature") as caught: + await adapter.generate( + [{"role": "user", "content": "question"}], + {"max_tokens": 16, "temperature": 1.5}, + ) + + assert caught.value.error_type == "invalid_request" + assert calls == 0 + + +@pytest.mark.asyncio +async def test_messages_empty_max_tokens_response_is_output_truncated() -> None: + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + json={ + "id": "msg-truncated", + "type": "message", + "model": "model", + "content": [], + "stop_reason": "max_tokens", + "usage": {"input_tokens": 4, "output_tokens": 8}, + }, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + adapter = AnthropicMessagesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ) + with pytest.raises(AdapterError) as caught: + await adapter.generate( + [{"role": "user", "content": "question"}], + {"max_tokens": 8}, + ) + + assert caught.value.error_type == "output_truncated" + + +@pytest.mark.asyncio +async def test_messages_typed_sse_merges_cumulative_usage_without_summing() -> None: + stream = b"".join( + [ + _typed_sse( + "message_start", + { + "type": "message_start", + "message": { + "id": "msg-sse", + "model": "model", + "content": [], + "stop_reason": None, + "usage": {"input_tokens": 15, "output_tokens": 1}, + }, + }, + ), + _typed_sse( + "content_block_delta", + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "text_delta", "text": "stream answer"}, + }, + ), + _typed_sse( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 5}, + }, + ), + _typed_sse("message_stop", {"type": "message_stop"}), + ] + ) + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + result = await AnthropicMessagesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ).generate( + [{"role": "user", "content": "question"}], + {"max_tokens": 64}, + ) + + assert result.text == "stream answer" + assert result.input_tokens == 15 + assert result.output_tokens == 5 + assert result.raw_usage == {"input_tokens": 15, "output_tokens": 5} + assert result.metadata["adapter"] == "anthropic_messages" + assert result.metadata["response_mode"] == "sse" + + +@pytest.mark.asyncio +async def test_messages_stream_eof_without_message_stop_fails_closed() -> None: + stream = _typed_sse( + "message_start", + { + "type": "message_start", + "message": { + "id": "msg-sse", + "model": "model", + "content": [], + "usage": {"input_tokens": 2, "output_tokens": 0}, + }, + }, + ) + _typed_sse( + "content_block_delta", + { + "type": "content_block_delta", + "delta": {"type": "text_delta", "text": "partial"}, + }, + ) + + def handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + adapter = AnthropicMessagesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ) + with pytest.raises(AdapterError) as caught: + await adapter.generate( + [{"role": "user", "content": "question"}], + {"max_tokens": 64}, + ) + + assert caught.value.error_type == "incomplete_provider_stream" + assert "message_stop" in caught.value.error_message + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "generation_config", + [ + {"max_tokens": None}, + {"max_tokens": 64, "seed": 1}, + ], +) +async def test_messages_rejects_invalid_parameters_before_sending( + generation_config: dict[str, object], +) -> None: + calls = 0 + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + return httpx.Response(500, request=request) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + adapter = AnthropicMessagesAdapter( + "https://provider.example/v1", + "model", + api_key="key", + client=client, + ) + with pytest.raises(AdapterError): + await adapter.generate( + [{"role": "user", "content": "question"}], + generation_config, + ) + + assert calls == 0 + + +@pytest.mark.asyncio +async def test_new_protocols_redact_current_key_from_results_and_http_errors() -> None: + secret = "current-provider-key" + + def responses_handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 200, + json={ + "id": f"id-{secret}", + "model": f"model-{secret}", + "status": "completed", + "output": [ + { + "type": "message", + "content": [{"type": "output_text", "text": f"answer {secret}"}], + } + ], + "usage": {"input_tokens": 1, "output_tokens": 1, "echo": secret}, + }, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(responses_handler)) as client: + result = await OpenAIResponsesAdapter( + "https://provider.example/v1", + "model", + api_key=secret, + client=client, + ).generate([{"role": "user", "content": "question"}], {}) + + rendered = repr(result) + assert secret not in rendered + assert "[REDACTED]" in rendered + + def messages_handler(request: httpx.Request) -> httpx.Response: + return httpx.Response( + 400, + json={"error": {"message": f"bad key {secret}"}}, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(messages_handler)) as client: + adapter = AnthropicMessagesAdapter( + "https://provider.example/v1", + "model", + api_key=secret, + client=client, + ) + with pytest.raises(AdapterError) as caught: + await adapter.generate( + [{"role": "user", "content": "question"}], + {"max_tokens": 16}, + ) + + assert secret not in caught.value.error_message + assert "[REDACTED]" in caught.value.error_message diff --git a/backend/tests/test_provider_protocol_plumbing.py b/backend/tests/test_provider_protocol_plumbing.py new file mode 100644 index 0000000..d3580e3 --- /dev/null +++ b/backend/tests/test_provider_protocol_plumbing.py @@ -0,0 +1,345 @@ +"""Registry-level coverage for explicit remote Provider protocols.""" + +from __future__ import annotations + +import json + +import httpx +import pytest + +from app.adapters import ( + AdapterError, + AnthropicMessagesAdapter, + OpenAIResponsesAdapter, + ProviderAttemptContext, + ProviderAttemptDisposition, + ProviderAttemptOutcome, + ProviderAttemptPermit, + build_adapter, +) + + +def _typed_sse(event_type: str, body: dict[str, object]) -> bytes: + return f"event: {event_type}\n".encode() + b"data: " + json.dumps(body).encode() + b"\n\n" + + +class _AttemptController: + def __init__(self) -> None: + self.finishes: list[tuple[object, ...]] = [] + + async def reserve( + self, + context: ProviderAttemptContext, + *, + provider_attempt: int, + ) -> ProviderAttemptPermit: + del context + return ProviderAttemptPermit( + reservation_id=f"reservation-{provider_attempt}", + provider_attempt=provider_attempt, + ) + + async def mark_send_started(self, permit: ProviderAttemptPermit) -> None: + del permit + + async def finish( + self, + permit: ProviderAttemptPermit, + *, + disposition: ProviderAttemptDisposition, + outcome: ProviderAttemptOutcome, + input_tokens: int | None = None, + output_tokens: int | None = None, + ) -> None: + self.finishes.append( + ( + permit.provider_attempt, + disposition, + outcome, + input_tokens, + output_tokens, + ) + ) + + +def _attempt_context() -> ProviderAttemptContext: + return ProviderAttemptContext( + run_id="run-1", + question_id="question-1", + model_id="model-1", + provider_scope="provider-scope", + lease_token=1, + execution_generation=1, + next_provider_attempt=1, + ) + + +@pytest.mark.parametrize( + ("provider_type", "adapter_class", "expected_url"), + [ + ( + "openai_responses", + OpenAIResponsesAdapter, + "https://provider.example/zen/go/v1/responses", + ), + ( + "anthropic_messages", + AnthropicMessagesAdapter, + "https://provider.example/zen/go/v1/messages", + ), + ], +) +def test_registry_builds_explicit_protocol_adapter( + provider_type: str, + adapter_class: type, + expected_url: str, +) -> None: + adapter = build_adapter( + provider_type, + base_url="https://provider.example/zen/go/v1", + remote_model_name="remote-model", + api_key_env="PROTOCOL_PROVIDER_KEY", + ) + + assert isinstance(adapter, adapter_class) + assert adapter.endpoint_url == expected_url + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "failure_kind", + [ + "responses_rate_limit", + "responses_server_error", + "responses_json_rate_limit", + "responses_json_server_error", + "messages_overloaded", + "messages_rate_limit", + "messages_api_error", + "messages_timeout_error", + "messages_http_529", + ], +) +async def test_typed_transient_failures_retry_and_settle_each_attempt( + failure_kind: str, +) -> None: + calls = 0 + controller = _AttemptController() + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + if calls == 1: + if failure_kind == "responses_json_rate_limit": + return httpx.Response( + 200, + json={ + "status": "failed", + "error": {"code": "rate_limit_exceeded", "message": "retry later"}, + }, + request=request, + ) + if failure_kind == "responses_json_server_error": + return httpx.Response( + 200, + json={ + "status": "failed", + "error": {"code": "server_error", "message": "retry later"}, + }, + request=request, + ) + if failure_kind == "responses_rate_limit": + stream = _typed_sse( + "error", + { + "type": "error", + "code": "rate_limit_exceeded", + "message": "retry later", + }, + ) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + if failure_kind == "responses_server_error": + stream = _typed_sse( + "response.failed", + { + "type": "response.failed", + "response": { + "status": "failed", + "error": {"code": "server_error", "message": "retry later"}, + }, + }, + ) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + if failure_kind == "messages_overloaded": + stream = _typed_sse( + "error", + { + "type": "error", + "error": {"type": "overloaded_error", "message": "retry later"}, + }, + ) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + if failure_kind in { + "messages_rate_limit", + "messages_api_error", + "messages_timeout_error", + }: + error_type = { + "messages_rate_limit": "rate_limit_error", + "messages_api_error": "api_error", + "messages_timeout_error": "timeout_error", + }[failure_kind] + stream = _typed_sse( + "error", + { + "type": "error", + "error": {"type": error_type, "message": "retry later"}, + }, + ) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + return httpx.Response( + 529, + json={"type": "error", "error": {"type": "overloaded_error"}}, + request=request, + ) + + if failure_kind.startswith("responses_"): + return httpx.Response( + 200, + json={ + "id": "response-retried", + "status": "completed", + "output": [ + { + "type": "message", + "content": [{"type": "output_text", "text": "A"}], + } + ], + "usage": {"input_tokens": 2, "output_tokens": 1}, + }, + request=request, + ) + return httpx.Response( + 200, + json={ + "id": "message-retried", + "type": "message", + "content": [{"type": "text", "text": "A"}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 2, "output_tokens": 1}, + }, + request=request, + ) + + async def no_sleep(_delay: float) -> None: + return None + + adapter_class = ( + OpenAIResponsesAdapter + if failure_kind.startswith("responses_") + else AnthropicMessagesAdapter + ) + generation = {} if failure_kind.startswith("responses_") else {"max_tokens": 16} + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + result = await adapter_class( + "https://provider.example/v1", + "remote-model", + api_key="key", + client=client, + max_retries=1, + sleep=no_sleep, + attempt_controller=controller, + ).generate( + [{"role": "user", "content": "question"}], + generation, + attempt_context=_attempt_context(), + ) + + assert calls == 2 + assert result.metadata["attempts"] == 2 + assert controller.finishes == [ + ( + 1, + ProviderAttemptDisposition.SETTLED_CONSERVATIVE, + ProviderAttemptOutcome.HTTP_ERROR, + None, + None, + ), + ( + 2, + ProviderAttemptDisposition.SETTLED_ACTUAL, + ProviderAttemptOutcome.SUCCEEDED, + 2, + 1, + ), + ] + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("adapter_class", "generation"), + [ + (OpenAIResponsesAdapter, {}), + (AnthropicMessagesAdapter, {"max_tokens": 16}), + ], +) +async def test_unknown_typed_stream_error_fails_closed_without_retry( + adapter_class: type, + generation: dict[str, object], +) -> None: + calls = 0 + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + stream = _typed_sse( + "error", + { + "type": "error", + "error": {"type": "unknown_error", "message": "do not retry"}, + }, + ) + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=stream, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + adapter = adapter_class( + "https://provider.example/v1", + "remote-model", + api_key="key", + client=client, + max_retries=1, + ) + with pytest.raises(AdapterError) as caught: + await adapter.generate( + [{"role": "user", "content": "question"}], + generation, + ) + + assert calls == 1 + assert caught.value.error_type == "provider_stream_error" + assert caught.value.retryable is False diff --git a/backend/tests/test_response_metadata_api.py b/backend/tests/test_response_metadata_api.py index b0a4ad9..dccb88f 100644 --- a/backend/tests/test_response_metadata_api.py +++ b/backend/tests/test_response_metadata_api.py @@ -123,3 +123,108 @@ def test_response_api_exposes_safe_provider_metadata_and_nulls_unsafe_values( } assert items[1]["http_attempt_count"] == 1 assert "sk-secretvalue123" not in response.text + + +def test_response_api_reports_partial_usage_totals_independently_of_pagination( + client, + db_session, +) -> None: + model = Model(id="usage-model", name="Usage Mock", provider_type=ProviderType.MOCK) + benchmark = Benchmark( + id="usage-benchmark", + slug="usage-benchmark", + name="Usage benchmark", + version="1.0.0", + description="fixture", + dimension="general", + language="en", + license="MIT", + source="local", + evaluator_type="exact_match", + evaluator_config={}, + prompt_template={}, + dataset_hash="usage-hash", + question_count=4, + ) + questions = [ + Question( + id=f"usage-question-{position}", + benchmark_id=benchmark.id, + external_id=f"usage-q{position}", + position=position, + question_type=QuestionType.EXACT_MATCH, + prompt=f"Question {position}", + reference_answer="one", + ) + for position in range(4) + ] + run = EvaluationRun( + id="usage-run", + model_id=model.id, + benchmark_id=benchmark.id, + status=RunStatus.COMPLETED, + model_parameters_snapshot={}, + benchmark_hash_snapshot=benchmark.dataset_hash, + prompt_template_snapshot={}, + total_questions=4, + completed_questions=4, + correct_questions=4, + input_tokens=None, + output_tokens=None, + ) + responses = [ + EvaluationResponse( + run_id=run.id, + question_id=question.id, + raw_response="one", + parsed_answer="one", + reference_answer_snapshot="one", + score=1, + evaluator_name="exact_match_v1", + input_tokens=input_tokens, + output_tokens=output_tokens, + ) + for question, input_tokens, output_tokens in zip( + questions, + (10, 30, None, 0), + (20, None, 40, None), + strict=True, + ) + ] + db_session.add_all([model, benchmark, *questions, run, *responses]) + db_session.commit() + + first_page = client.get(f"/api/v1/runs/{run.id}/responses?offset=0&limit=1") + last_page = client.get(f"/api/v1/runs/{run.id}/responses?offset=3&limit=1") + + assert first_page.status_code == last_page.status_code == 200 + expected_summary = { + "total": 4, + "known_input_tokens": 40, + "known_output_tokens": 60, + "input_token_reported_responses": 3, + "output_token_reported_responses": 2, + } + for payload in (first_page.json(), last_page.json()): + assert {key: payload[key] for key in expected_summary} == expected_summary + assert first_page.json()["items"][0]["question_external_id"] == "usage-q0" + assert last_page.json()["items"][0]["question_external_id"] == "usage-q3" + + run_payload = client.get(f"/api/v1/runs/{run.id}").json() + assert run_payload["input_tokens"] is None + assert run_payload["output_tokens"] is None + + +def test_response_usage_summary_openapi_fields_are_required_and_non_negative( + client, +) -> None: + schema = client.get("/openapi.json").json()["components"]["schemas"]["EvaluationResponseList"] + fields = { + "known_input_tokens", + "known_output_tokens", + "input_token_reported_responses", + "output_token_reported_responses", + } + + assert fields <= set(schema["required"]) + assert all(schema["properties"][field]["minimum"] == 0 for field in fields) diff --git a/backend/tests/test_run_progress_api.py b/backend/tests/test_run_progress_api.py new file mode 100644 index 0000000..7d42c17 --- /dev/null +++ b/backend/tests/test_run_progress_api.py @@ -0,0 +1,504 @@ +"""Fixed-block Run progress projection and canonical live evidence metrics.""" + +from datetime import UTC, datetime +from decimal import Decimal + +import pytest + +from app.core.logging import normalize_request_path +from app.models import ( + Benchmark, + EvaluationResponse, + EvaluationRun, + Model, + ProviderType, + Question, + QuestionType, + RunStatus, +) +from app.schemas.evaluation_progress import PROGRESS_BLOCK_SIZE +from app.services.run_progress import progress_block_count + + +@pytest.mark.parametrize( + ("total_questions", "expected_blocks"), + [(0, 0), (1, 1), (12_032, 24), (20_000, 40)], +) +def test_progress_block_count_covers_supported_run_sizes( + total_questions: int, + expected_blocks: int, +) -> None: + assert progress_block_count(total_questions) == expected_blocks + + +def _progress_fixture(db_session) -> tuple[EvaluationRun, list[Question]]: + model = Model(id="progress-model", name="Progress Mock", provider_type=ProviderType.MOCK) + benchmark = Benchmark( + id="progress-benchmark", + slug="progress-benchmark", + name="Progress benchmark", + version="1.0.0", + description="fixture", + dimension="general", + language="en", + license="MIT", + source="local", + evaluator_type="exact_match", + evaluator_config={}, + prompt_template={}, + dataset_hash="progress-hash", + question_count=515, + ) + questions = [ + Question( + id=f"progress-question-{position}", + benchmark_id=benchmark.id, + external_id=f"progress-q{position + 1}", + position=position, + question_type=QuestionType.EXACT_MATCH, + prompt=f"Question {position + 1}", + reference_answer="one", + ) + for position in range(515) + ] + run = EvaluationRun( + id="progress-run", + model_id=model.id, + benchmark_id=benchmark.id, + status=RunStatus.COMPLETED, + model_parameters_snapshot={}, + benchmark_hash_snapshot=benchmark.dataset_hash, + prompt_template_snapshot={}, + total_questions=515, + completed_questions=4, + correct_questions=0, + error_questions=0, + score=0, + completion_rate=0, + answered_accuracy=None, + average_latency_ms=None, + input_tokens=None, + output_tokens=None, + estimated_cost=None, + ) + created_at = datetime(2026, 8, 30, 1, 0, tzinfo=UTC) + responses = [ + EvaluationResponse( + id="progress-response-pass", + run_id=run.id, + question_id=questions[511].id, + raw_response="one", + parsed_answer="one", + reference_answer_snapshot="one", + score=1, + evaluator_name="exact_match_v1", + latency_ms=100, + input_tokens=10, + output_tokens=5, + estimated_cost=Decimal("0.001"), + created_at=created_at, + ), + EvaluationResponse( + id="progress-response-wrong", + run_id=run.id, + question_id=questions[0].id, + raw_response="two", + parsed_answer="two", + reference_answer_snapshot="one", + score=0, + evaluator_name="exact_match_v1", + latency_ms=200, + input_tokens=20, + output_tokens=10, + estimated_cost=None, + created_at=created_at, + ), + EvaluationResponse( + id="progress-response-error", + run_id=run.id, + question_id=questions[512].id, + raw_response=None, + parsed_answer=None, + reference_answer_snapshot="one", + # Deliberately inconsistent evidence proves error_type has outcome priority. + score=1, + evaluator_name="exact_match_v1", + latency_ms=300, + input_tokens=None, + output_tokens=None, + estimated_cost=None, + error_type="provider_error", + error_message="sensitive upstream detail omitted from compact progress", + created_at=created_at, + ), + EvaluationResponse( + id="progress-response-empty", + run_id=run.id, + question_id=questions[514].id, + raw_response="", + parsed_answer=None, + reference_answer_snapshot="one", + score=0, + evaluator_name="exact_match_v1", + latency_ms=None, + input_tokens=0, + output_tokens=0, + estimated_cost=Decimal("0"), + created_at=created_at, + ), + ] + db_session.add_all([model, benchmark, *questions, run, *responses]) + db_session.commit() + return run, questions + + +def test_progress_index_returns_one_snapshot_of_canonical_metrics_and_block_counts( + client, + db_session, +) -> None: + run, _ = _progress_fixture(db_session) + + response = client.get(f"/api/v1/runs/{run.id}/progress") + + assert response.status_code == 200 + assert response.headers["cache-control"] == "no-store" + progress = response.json() + assert set(progress) == { + "block_size", + "total_questions", + "completed_questions", + "correct_questions", + "error_questions", + "score", + "completion_rate", + "answered_accuracy", + "average_latency_ms", + "known_input_tokens", + "known_output_tokens", + "input_token_reported_responses", + "output_token_reported_responses", + "known_estimated_cost", + "estimated_cost_reported_responses", + "blocks", + } + assert progress["block_size"] == PROGRESS_BLOCK_SIZE + assert progress["total_questions"] == 515 + assert progress["completed_questions"] == 4 + # Canonical protocol-v1 aggregation rounds score_sum, even for malformed + # score=1/error evidence; the block outcome still gives error priority. + assert progress["correct_questions"] == 2 + assert progress["error_questions"] == 1 + assert progress["score"] == pytest.approx(2 / 515 * 100) + assert progress["completion_rate"] == pytest.approx(2 / 515 * 100) + assert progress["answered_accuracy"] == 100 + assert progress["average_latency_ms"] == 200 + assert progress["known_input_tokens"] == 30 + assert progress["known_output_tokens"] == 15 + assert progress["input_token_reported_responses"] == 3 + assert progress["output_token_reported_responses"] == 3 + assert progress["known_estimated_cost"] == pytest.approx(0.001) + assert progress["estimated_cost_reported_responses"] == 2 + assert progress["blocks"] == [ + {"block_index": 0, "response_count": 2}, + {"block_index": 1, "response_count": 2}, + ] + + db_session.expire_all() + persisted = db_session.get(EvaluationRun, run.id) + assert persisted is not None + assert persisted.input_tokens is None + assert persisted.output_tokens is None + assert persisted.estimated_cost is None + + +def test_progress_blocks_return_only_allowlisted_cells_in_absolute_position_order( + client, + db_session, +) -> None: + run, _ = _progress_fixture(db_session) + + first_response = client.get(f"/api/v1/runs/{run.id}/progress/blocks/0") + second_response = client.get(f"/api/v1/runs/{run.id}/progress/blocks/1") + + assert first_response.status_code == second_response.status_code == 200 + assert first_response.headers["cache-control"] == "no-store" + assert second_response.headers["cache-control"] == "no-store" + first = first_response.json() + second = second_response.json() + assert [item["position"] for item in first["items"]] == [0, 511] + assert [item["outcome"] for item in first["items"]] == ["wrong", "passed"] + assert [item["position"] for item in second["items"]] == [512, 514] + assert [item["outcome"] for item in second["items"]] == ["error", "wrong"] + assert second["items"][0]["score"] == 1 + assert second["items"][0]["error_type"] == "provider_error" + assert second["items"][1]["input_tokens"] == 0 + assert second["items"][1]["output_tokens"] == 0 + assert second["items"][1]["estimated_cost"] == 0 + + required_cell_fields = { + "position", + "outcome", + "score", + "latency_ms", + "input_tokens", + "output_tokens", + "estimated_cost", + "error_type", + } + assert all(set(item) == required_cell_fields for item in first["items"] + second["items"]) + forbidden = { + "id", + "run_id", + "question_id", + "question_external_id", + "prompt", + "choices", + "raw_response", + "parsed_answer", + "reference_answer_snapshot", + "error_message", + "provider_request_id", + "returned_model", + "system_fingerprint", + "finish_reason", + } + assert all(forbidden.isdisjoint(item) for item in first["items"] + second["items"]) + assert "sensitive upstream detail" not in second_response.text + + +def test_progress_block_can_lead_an_older_index_and_the_next_index_converges( + client, + db_session, +) -> None: + run, questions = _progress_fixture(db_session) + first_index = client.get(f"/api/v1/runs/{run.id}/progress").json() + + db_session.add( + EvaluationResponse( + id="progress-response-concurrent", + run_id=run.id, + question_id=questions[513].id, + raw_response="one", + parsed_answer="one", + reference_answer_snapshot="one", + score=1, + evaluator_name="exact_match_v1", + latency_ms=50, + input_tokens=1, + output_tokens=1, + estimated_cost=Decimal("0"), + ) + ) + db_session.commit() + + newer_block = client.get(f"/api/v1/runs/{run.id}/progress/blocks/1").json() + converged_index = client.get(f"/api/v1/runs/{run.id}/progress").json() + + assert first_index["blocks"][1]["response_count"] == 2 + assert [item["position"] for item in newer_block["items"]] == [512, 513, 514] + assert converged_index["blocks"][1]["response_count"] == 3 + assert converged_index["completed_questions"] == 5 + + +def test_progress_index_and_block_represent_an_empty_run_without_inventing_facts( + client, + db_session, +) -> None: + model = Model(id="empty-progress-model", name="Empty Mock", provider_type=ProviderType.MOCK) + benchmark = Benchmark( + id="empty-progress-benchmark", + slug="empty-progress-benchmark", + name="Empty progress benchmark", + version="1.0.0", + description="fixture", + dimension="general", + language="en", + license="MIT", + source="local", + evaluator_type="exact_match", + evaluator_config={}, + prompt_template={}, + dataset_hash="empty-progress-hash", + question_count=1, + ) + question = Question( + id="empty-progress-question", + benchmark_id=benchmark.id, + external_id="empty-progress-q1", + position=0, + question_type=QuestionType.EXACT_MATCH, + prompt="Question", + reference_answer="one", + ) + run = EvaluationRun( + id="empty-progress-run", + model_id=model.id, + benchmark_id=benchmark.id, + status=RunStatus.PENDING, + model_parameters_snapshot={}, + benchmark_hash_snapshot=benchmark.dataset_hash, + prompt_template_snapshot={}, + total_questions=1, + ) + db_session.add_all([model, benchmark, question, run]) + db_session.commit() + + index = client.get(f"/api/v1/runs/{run.id}/progress").json() + block = client.get(f"/api/v1/runs/{run.id}/progress/blocks/0").json() + + assert index["completed_questions"] == 0 + assert index["correct_questions"] == 0 + assert index["error_questions"] == 0 + assert index["score"] == 0 + assert index["completion_rate"] == 0 + assert index["answered_accuracy"] is None + assert index["average_latency_ms"] is None + assert index["known_input_tokens"] == 0 + assert index["known_output_tokens"] == 0 + assert index["input_token_reported_responses"] == 0 + assert index["output_token_reported_responses"] == 0 + assert index["known_estimated_cost"] == 0 + assert index["estimated_cost_reported_responses"] == 0 + assert index["blocks"] == [{"block_index": 0, "response_count": 0}] + assert block == {"block_index": 0, "items": []} + + +def test_progress_block_bounds_and_missing_run_are_typed(client, db_session) -> None: + run, _ = _progress_fixture(db_session) + + negative = client.get(f"/api/v1/runs/{run.id}/progress/blocks/-1") + too_large = client.get(f"/api/v1/runs/{run.id}/progress/blocks/2") + missing = client.get("/api/v1/runs/missing/progress") + + assert negative.status_code == 422 + assert too_large.status_code == 422 + assert too_large.json()["detail"]["code"] == "progress_block_out_of_range" + assert missing.status_code == 404 + assert missing.json()["detail"]["code"] == "run_not_found" + + +@pytest.mark.parametrize("invalid_position", [-1, 515]) +def test_progress_index_fails_closed_for_response_outside_frozen_plan( + client, + db_session, + invalid_position: int, +) -> None: + run, _ = _progress_fixture(db_session) + # -1 specifically covers SQLite's integer-division truncation toward block zero. + question = Question( + id=f"corrupt-progress-question-{invalid_position}", + benchmark_id=run.benchmark_id, + external_id="corrupt-progress-q", + position=invalid_position, + question_type=QuestionType.EXACT_MATCH, + prompt="Question", + reference_answer="one", + ) + evidence = EvaluationResponse( + id=f"corrupt-progress-response-{invalid_position}", + run_id=run.id, + question_id=question.id, + raw_response="one", + parsed_answer="one", + reference_answer_snapshot="one", + score=1, + evaluator_name="exact_match_v1", + ) + db_session.add_all([question, evidence]) + db_session.commit() + + response = client.get(f"/api/v1/runs/{run.id}/progress") + + assert response.status_code == 500 + assert response.json()["detail"]["code"] == "run_progress_integrity_error" + + +def test_progress_index_fails_closed_for_cross_benchmark_response(client, db_session) -> None: + run, _ = _progress_fixture(db_session) + other_benchmark = Benchmark( + id="cross-progress-benchmark", + slug="cross-progress-benchmark", + name="Cross progress benchmark", + version="1.0.0", + description="fixture", + dimension="general", + language="en", + license="MIT", + source="local", + evaluator_type="exact_match", + evaluator_config={}, + prompt_template={}, + dataset_hash="cross-progress-hash", + question_count=1, + ) + question = Question( + id="cross-progress-question", + benchmark_id=other_benchmark.id, + external_id="cross-progress-q", + position=1, + question_type=QuestionType.EXACT_MATCH, + prompt="Question", + reference_answer="one", + ) + evidence = EvaluationResponse( + id="cross-progress-response", + run_id=run.id, + question_id=question.id, + raw_response="one", + parsed_answer="one", + reference_answer_snapshot="one", + score=1, + evaluator_name="exact_match_v1", + ) + db_session.add_all([other_benchmark, question, evidence]) + db_session.commit() + + response = client.get(f"/api/v1/runs/{run.id}/progress") + + assert response.status_code == 500 + assert response.json()["detail"]["code"] == "run_progress_integrity_error" + + +def test_progress_openapi_and_logging_contract_are_explicit(client) -> None: + openapi = client.get("/openapi.json").json() + index_schema = openapi["components"]["schemas"]["EvaluationProgressIndex"] + block_schema = openapi["components"]["schemas"]["EvaluationProgressBlock"] + cell_schema = openapi["components"]["schemas"]["EvaluationProgressCell"] + + assert set(index_schema["required"]) == { + "block_size", + "total_questions", + "completed_questions", + "correct_questions", + "error_questions", + "score", + "completion_rate", + "answered_accuracy", + "average_latency_ms", + "known_input_tokens", + "known_output_tokens", + "input_token_reported_responses", + "output_token_reported_responses", + "known_estimated_cost", + "estimated_cost_reported_responses", + "blocks", + } + assert set(block_schema["required"]) == {"block_index", "items"} + assert set(cell_schema["required"]) == { + "position", + "outcome", + "score", + "latency_ms", + "input_tokens", + "output_tokens", + "estimated_cost", + "error_type", + } + assert cell_schema["properties"]["position"]["minimum"] == 0 + assert "/api/v1/runs/{run_id}/progress" in openapi["paths"] + assert "/api/v1/runs/{run_id}/progress/blocks/{block_index}" in openapi["paths"] + assert normalize_request_path("/runs/{run_id}/progress") == "/runs/{run_id}/progress" + assert ( + normalize_request_path("/runs/{run_id}/progress/blocks/{block_index}") + == "/runs/{run_id}/progress/blocks/{block_index}" + ) diff --git a/backend/tests/test_smoke.py b/backend/tests/test_smoke.py index 5813f23..6aa7a21 100644 --- a/backend/tests/test_smoke.py +++ b/backend/tests/test_smoke.py @@ -66,7 +66,23 @@ def _wait_for_terminal(client, run_id: str) -> dict: assert initial_payload["status"] == "pending" assert initial_payload["attempt_count"] == 0 assert initial_payload["lease_owner"] is None - assert client.get(f"/api/v1/runs/{run_id}/responses").json()["total"] == 0 + initial_evidence = client.get(f"/api/v1/runs/{run_id}/responses").json() + assert { + key: initial_evidence[key] + for key in ( + "total", + "known_input_tokens", + "known_output_tokens", + "input_token_reported_responses", + "output_token_reported_responses", + ) + } == { + "total": 0, + "known_input_tokens": 0, + "known_output_tokens": 0, + "input_token_reported_responses": 0, + "output_token_reported_responses": 0, + } assert not hasattr(client.app.state, "task_manager") worker = WorkerService( SessionLocal, @@ -163,13 +179,14 @@ def test_run_lifecycle_timestamps_ignore_worker_host_clock_skew(client, monkeypa def test_parse_error_evidence_identifies_nonempty_truncated_output() -> None: - assert EvaluationRunner._parse_error_evidence( - "choice_not_found", {"finish_reason": "length"} - ) == ( - "output_truncated", - "Provider stopped at the output token limit before a valid final answer was parsed " - "(choice_not_found).", - ) + for finish_reason in ("length", "max_tokens"): + assert EvaluationRunner._parse_error_evidence( + "choice_not_found", {"finish_reason": finish_reason} + ) == ( + "output_truncated", + "Provider stopped at the output token limit before a valid final answer was parsed " + "(choice_not_found).", + ) assert EvaluationRunner._parse_error_evidence( "choice_not_found", {"finish_reason": "stop"} ) == ("parse_error", "choice_not_found") @@ -227,6 +244,10 @@ def test_offline_mock_vertical_slice(client, db_session) -> None: assert responses.status_code == 200 evidence = responses.json() assert evidence["total"] == 15 + assert evidence["known_input_tokens"] == 120 + assert evidence["known_output_tokens"] == 30 + assert evidence["input_token_reported_responses"] == 15 + assert evidence["output_token_reported_responses"] == 15 assert len(evidence["items"]) == 15 assert {item["question_type"] for item in evidence["items"]} == { "exact_match", @@ -278,8 +299,12 @@ def test_single_question_error_does_not_fail_run(client, db_session) -> None: assert run["output_tokens"] is None assert run["estimated_cost"] is None - responses = client.get(f"/api/v1/runs/{run['id']}/responses?limit=100").json()["items"] - failed = [item for item in responses if item["error_type"] == "injected_failure"] + evidence = client.get(f"/api/v1/runs/{run['id']}/responses?limit=100").json() + assert evidence["known_input_tokens"] == 112 + assert evidence["known_output_tokens"] == 28 + assert evidence["input_token_reported_responses"] == 14 + assert evidence["output_token_reported_responses"] == 14 + failed = [item for item in evidence["items"] if item["error_type"] == "injected_failure"] assert len(failed) == 1 assert failed[0]["score"] == 0 @@ -324,9 +349,13 @@ def get_evaluator_with_fault(question_type: str): assert run["output_tokens"] == 30 assert run["estimated_cost"] == 0 - responses = client.get(f"/api/v1/runs/{run['id']}/responses?limit=100").json()["items"] + evidence = client.get(f"/api/v1/runs/{run['id']}/responses?limit=100").json() + assert evidence["known_input_tokens"] == 120 + assert evidence["known_output_tokens"] == 30 + assert evidence["input_token_reported_responses"] == 15 + assert evidence["output_token_reported_responses"] == 15 evaluator_errors = [ - item for item in responses if item["error_type"] == "evaluator_internal_error" + item for item in evidence["items"] if item["error_type"] == "evaluator_internal_error" ] assert len(evaluator_errors) == 5 assert all(item["raw_response"] for item in evaluator_errors) diff --git a/backend/tests/test_web_credentials.py b/backend/tests/test_web_credentials.py index f5f516e..9c12e00 100644 --- a/backend/tests/test_web_credentials.py +++ b/backend/tests/test_web_credentials.py @@ -383,6 +383,7 @@ def test_stored_credential_crud_origin_guard_active_run_lock_and_cleanup(client) {"base_url": "https://third-provider.example/v1"}, {"api_key": CANARY}, {"api_key_env": "LEGACY_PROVIDER_KEY"}, + {"provider_type": "openai_responses"}, {"provider_type": "mock"}, ] for update in sensitive_updates: diff --git a/compose.yaml b/compose.yaml index 8c3db5e..e99eeac 100644 --- a/compose.yaml +++ b/compose.yaml @@ -21,7 +21,6 @@ x-app-environment: &app-environment LLMBENCHLAB_WORKER_SHUTDOWN_GRACE_SECONDS: ${LLMBENCHLAB_COMPOSE_WORKER_SHUTDOWN_GRACE_SECONDS:-30} LLMBENCHLAB_WORKER_PROGRESS_FLUSH_SECONDS: ${LLMBENCHLAB_COMPOSE_WORKER_PROGRESS_FLUSH_SECONDS:-5} LLMBENCHLAB_WORKER_PROGRESS_STALE_SECONDS: ${LLMBENCHLAB_COMPOSE_WORKER_PROGRESS_STALE_SECONDS:-60} - LLMBENCHLAB_WORKER_EXPECTED_PROCESSES: ${LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES:-1} LLMBENCHLAB_WORKER_RECOVERY_ALERT_SECONDS: ${LLMBENCHLAB_COMPOSE_WORKER_RECOVERY_ALERT_SECONDS:-60} LLMBENCHLAB_REDIS_BLOCK_MILLISECONDS: ${LLMBENCHLAB_COMPOSE_REDIS_BLOCK_MILLISECONDS:-1000} LLMBENCHLAB_REDIS_OPERATION_TIMEOUT_SECONDS: ${LLMBENCHLAB_COMPOSE_REDIS_OPERATION_TIMEOUT_SECONDS:-1} @@ -84,6 +83,7 @@ services: environment: <<: *app-environment LLMBENCHLAB_CREDENTIAL_KEYS_FILE: /run/secrets/credential-keys + LLMBENCHLAB_WORKER_EXPECTED_PROCESSES: ${LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES:-1} secrets: - credential-keys command: diff --git a/docs/API.md b/docs/API.md index bcd1c32..bb2a0ab 100644 --- a/docs/API.md +++ b/docs/API.md @@ -139,6 +139,8 @@ Benchmark 校验错误还会提供文件、行、列或 JSON Pointer: | GET | `/runs/{run_id}` | 200 | 轮询 Run 状态与汇总 | | POST | `/runs/{run_id}/cancel` | 200 | 请求协作式取消 | | GET | `/runs/{run_id}/responses` | 200 | 分页读取逐题证据 | +| GET | `/runs/{run_id}/progress` | 200 | 读取固定 512 题 block 索引与同快照 live metrics | +| GET | `/runs/{run_id}/progress/blocks/{block_index}` | 200 | 读取一个 block 的轻量 absolute-position cells | | GET | `/runs/{run_id}/audit` | 200 | 按稳定时间顺序分页读取保留期内的类型化审计事件 | | GET | `/leaderboard` | 200 | 已完成 Run 的严格总分榜 | | GET | `/metrics/summary` | 200 | Dashboard 汇总 | @@ -391,7 +393,7 @@ curl -sS http://127.0.0.1:8000/api/v1/info "protocol_version": "llmbenchlab-protocol-v1", "environment": "development", "capabilities": { - "providers": ["mock", "openai_compatible"], + "providers": ["mock", "openai_compatible", "openai_responses", "anthropic_messages"], "question_types": ["exact_match", "multiple_choice", "numeric"], "runner": "independent_database_lease_worker" } @@ -444,7 +446,7 @@ curl -sS -X PUT http://127.0.0.1:8000/api/v1/governance/policy \ | 字段 | 类型/默认值 | 规则 | | --- | --- | --- | | `name` | string,必填 | 去首尾空白后 `1..160`,全库唯一 | -| `provider_type` | `mock` 或 `openai_compatible`,必填 | 首期封闭集合 | +| `provider_type` | `mock`、`openai_compatible`、`openai_responses` 或 `anthropic_messages`,必填 | 显式 Adapter 封闭集合;旧 `openai_compatible` 继续表示 Chat Completions | | `base_url` | string/null | 绝对 URL;远端只允许 HTTPS,明文 HTTP 仅允许 loopback;禁止 URL 内嵌账号密码、query 与 fragment | | `remote_model_name` | string/null | 最长 256 | | `api_key` | string/null | **仅写入**;8–8192 bytes、无首尾空白、只含可见 ASCII;OpenAPI 标记 `writeOnly`,所有响应均省略 | @@ -454,13 +456,13 @@ curl -sS -X PUT http://127.0.0.1:8000/api/v1/governance/policy \ | `output_price_per_million` | number/null,默认 `null` | 有限非负数;Mock 未填时规范化为明确的 `0` | | `default_parameters` | object,默认 `{}` | 只允许 `temperature`、`top_p`、`max_tokens`、`seed`;其中 `max_tokens` 可为 `null` 或 `1..131072`,其余字段使用与 Run 相同的类型/范围约束 | -`openai_compatible` 必须同时提供 `base_url`、`remote_model_name`,并在 `api_key` 与 `api_key_env` 中恰好选择一个;`mock` 的四个远端连接/凭据字段必须为空。Model Schema、Provider preflight 和 Adapter 都拒绝远端明文 HTTP,只有 `localhost` 或字面量 loopback IP 可使用 HTTP;HTTPS 私网、云元数据、DNS rebinding 和其他出站目标仍没有 allowlist,详见 [SECURITY.md](SECURITY.md)。 +三个远程类型都必须同时提供 `base_url`、`remote_model_name`,并在 `api_key` 与 `api_key_env` 中恰好选择一个;`mock` 的四个远端连接/凭据字段必须为空。协议必须显式选择:`openai_compatible`、`openai_responses`、`anthropic_messages` 分别使用 `/chat/completions`、`/responses`、`/messages`。根地址会追加所选后缀;与所选类型一致的完整 endpoint 保持不变;其他已知协议后缀在任何网络请求前拒绝。Model Schema、Provider preflight 和 Adapter 都拒绝远端明文 HTTP,只有 `localhost` 或字面量 loopback IP 可使用 HTTP;HTTPS 私网、云元数据、DNS rebinding 和其他出站目标仍没有 allowlist,详见 [SECURITY.md](SECURITY.md)。 读响应额外包含两个派生字段:`credential_source` 为 `none | environment | stored`;`has_api_key` 只表示该 Model 当前拥有应用加密保存的 Web Key。环境变量模式即使 Worker 环境中已有值也仍返回 `has_api_key=false`。`stored` 模式在独立 `model_credentials` 行中以 `model_id` 为主键保存 AES-GCM envelope,Model/Run/Response Schema 均不映射其内部列。 -本 API 的 `GET /models` 是 LLMBenchLab 本地模型注册表,不会代替操作者访问 Provider。可信本地 `llmbenchlab-evaluate` 才会调用上游 `/models` 与付费 canary:发现到的任一模型 ID 若包含当前 Key,预检立即失败;canary 成功体若明确返回不同于请求目标的模型名,也会失败。模型发现与正式 Chat 请求声明 `Accept-Encoding: identity` 并拒绝其他响应编码。Chat 内部传输显式请求 SSE 与流式 usage,持续消费到 `[DONE]`;这不是新的 LLMBenchLab 公开 SSE API。忽略流式参数的 Provider 可返回普通 JSON fallback。发现体上限为 2 MiB;Chat 普通 JSON 成功体上限为 4 MiB,SSE 累计 wire 上限为 64 MiB、单事件上限为 1 MiB、最终聚合 content 上限为 4 MiB,非 2xx 错误体上限为 64 KiB。成功内容、raw usage 的对象键/所有 JSON 标量、token/status 数值、request ID、返回模型名、system fingerprint 与 finish reason 中出现的当前 Key 会在进入持久化边界前按精确值替换为 `[REDACTED]`。SSE content 先完整聚合再替换,因此 Key 横跨多个 delta 也不会因分块而跳过该精确匹配。 +本 API 的 `GET /models` 是 LLMBenchLab 本地模型注册表,不会代替操作者访问 Provider。可信本地 `llmbenchlab-evaluate` 才会调用上游 `/models` 与付费 canary:三个已知 endpoint 后缀都回到同级 `/models`,discovery 按显式协议鉴权(Chat/Responses 使用 `Authorization: Bearer`,Messages 使用 `x-api-key` 与 `anthropic-version`);Messages 的 `has_more/last_id` 通过 `after_id` 分页,并受累计 100 页、60 秒 wall-clock、10,000 项、2 MiB 与缺失/重复 cursor 门禁保护。发现到的任一模型 ID 若包含当前 Key,预检立即失败;canary 成功体若明确返回不同于请求目标的模型名,也会失败。模型发现与正式请求声明 `Accept-Encoding: identity` 并拒绝其他响应编码。Chat、Responses、Messages 的内部流式传输分别以 `[DONE]`、`response.completed`、`message_stop` 为成功终止证据;这不是新的 LLMBenchLab 公开 SSE API。各协议普通 JSON 成功响应继续兼容。Provider 普通 JSON 成功体上限为 4 MiB,SSE 累计 wire 上限为 64 MiB、单事件上限为 1 MiB、最终聚合 content 上限为 4 MiB,非 2xx 错误体上限为 64 KiB。成功内容、raw usage 的对象键/所有 JSON 标量、token/status 数值、request ID、返回模型名、system fingerprint 与 finish reason 中出现的当前 Key 会在进入持久化边界前按精确值替换为 `[REDACTED]`。SSE content 先完整聚合再替换,因此 Key 横跨多个 delta 也不会因分块而跳过该精确匹配。 -Phase 1 的 Model 默认参数只覆盖上述四个实际由 Adapter 转发的生成字段。创建 Run 时,显式请求值优先;某字段未出现在请求 JSON 中时才使用 Model 默认值,否则使用协议默认值。`max_tokens:null` 是一个显式值:OpenAI-compatible Adapter 不发送 `max_tokens` 字段,由 Provider 决定其默认输出预算;这不表示无限输出,也不保证不同 Provider 使用相同上限。Run 的 `generation` 快照保存最终有效值,包括这个 `null`。 +Model 默认参数只覆盖上述四个生成字段。创建 Run 时,显式请求值优先;某字段未出现在请求 JSON 中时才使用 Model 默认值。为了兼容旧客户端,通用请求 Schema 仍显示 Chat 的 `temperature=0`、`top_p=1`、`seed=42` 默认;但 Responses/Messages 在请求与 Model 默认都没有显式提供这些字段时,会把三者归一化为 `null` 并从 Provider payload 省略,避免向不支持采样字段的模型发送参数。Chat Completions 把 `max_tokens` 原样发送;Responses 把它映射为 `max_output_tokens`;二者的 `null` 都表示省略该字段,由 Provider 决定默认输出预算,这不表示无限输出。Messages 必须使用有限 `max_tokens`,且显式 `temperature` 只能为 `0..1`。Responses/Messages 都不接受当前项目的非空 `seed`;非法组合在外发前稳定拒绝,不会静默忽略。Run 的 `generation` 快照保存最终有效值和显式 Adapter 类型。 Model 响应示例(后续接口引用为 `ModelRead`): @@ -485,7 +487,7 @@ Model 响应示例(后续接口引用为 `ModelRead`): ### 4.2 `GET /models` -筛选参数:`provider_type=mock|openai_compatible`、`enabled=true|false`,并支持通用分页。 +筛选参数:`provider_type=mock|openai_compatible|openai_responses|anthropic_messages`、`enabled=true|false`,并支持通用分页。 ```bash curl -sS 'http://127.0.0.1:8000/api/v1/models?provider_type=mock&enabled=true&offset=0&limit=20' @@ -531,14 +533,14 @@ curl -sS -X POST http://127.0.0.1:8000/api/v1/models \ -d '{"name":"Offline Mock","provider_type":"mock","enabled":true}' ``` -注册供 Web 使用的 OpenAI-compatible 配置。推荐在 Models 页面粘贴 Key;`api_key` 只出现在这次写请求中,成功响应不会返回它,数据库也不会保存其明文。下面只是请求结构,不是可直接填入真实 Key 的 shell 命令: +注册供 Web 使用的远程配置。推荐在 Models 页面粘贴 Key;`api_key` 只出现在这次写请求中,成功响应不会返回它,数据库也不会保存其明文。下面是 Responses 类型的请求结构,不是可直接填入真实 Key 的 shell 命令: ```json { - "name": "Local Compatible", - "provider_type": "openai_compatible", + "name": "Local Responses", + "provider_type": "openai_responses", "base_url": "https://llm-gateway.invalid/v1", - "remote_model_name": "example-chat-model", + "remote_model_name": "example-responses-model", "api_key": "", "enabled": true, "input_price_per_million": null, @@ -797,10 +799,10 @@ curl -sS -X POST http://127.0.0.1:8000/api/v1/benchmarks/reload-demo | --- | ---: | --- | | `model_id` | 必填 | 非空,最长 36 | | `benchmark_id` | 必填 | 非空,最长 36 | -| `temperature` | `0.0` | `0..2` | -| `top_p` | `1.0` | `>0` 且 `<=1` | +| `temperature` | `0.0` | `null` 或 `0..2`;Messages 非空值上限为 `1` | +| `top_p` | `1.0` | `null` 或 `>0` 且 `<=1` | | `max_tokens` | `256` | `null` 或整数 `1..131072`;`null` 表示不发送该字段,由 Provider 决定默认值,并非无限输出 | -| `seed` | `42` | 32 位有符号整数或 `null` | +| `seed` | `42` | 32 位有符号整数或 `null`;Responses/Messages 只接受 `null` | | `system_prompt` | `null` | 最长 4000;提供时覆盖 Benchmark system prompt | | `concurrency` | `1` | `1..4`;快照值即实际执行并发度 | | `input_token_reservation` | `null` | `null` 或严格整数 `1..10000000`;hard TPM/Token/费用启用时必须提供可证明的每题输入预留上界。`null` 时实现不会把 UTF-8/tokenizer 估算写成 hard reservation 或用它触发 input/cost overdraw | @@ -809,7 +811,7 @@ curl -sS -X POST http://127.0.0.1:8000/api/v1/benchmarks/reload-demo | `lifetime_cost_budget_usd` | `null` | `null` 或 `0..10000000.00000000` USD,最多 8 位小数;请求接受 JSON number/十进制 string,`RunRead` 的非空响应始终是 JSON string;费用硬边界还要求显式 Token 上界和冻结价格 | | `read_timeout_seconds` | `60` | 有限数字 `1..1800`;冻结为等待 Provider 下一批响应字节的空闲读取超时,不是请求总墙钟时限 | -保留 `max_tokens=256` 作为通用 API/protocol-v1 默认值是兼容要求;当 Model 没有对应默认且用户尚未手动修改时,Web 的新建评测表单会根据已知 Benchmark 预填更适合长推理的显式建议值。它们是可编辑的客户端起点,不会改变省略字段时的 API 默认: +保留 `max_tokens=256` 以及表中 Chat 采样值作为通用 API/protocol-v1 Schema 默认是兼容要求。创建 Responses/Messages Run 时,省略且没有 Model 默认的 `temperature`、`top_p`、`seed` 会冻结为 `null` 并从上游 payload 省略;这和客户端显式发送采样值不同。Web 在新协议下把 `temperature`/`top_p` 留空,seed 禁用;当 Model 没有输出默认且用户尚未手动修改时,新建评测表单仍根据已知 Benchmark 预填更适合长推理的显式输出建议。它们是可编辑的客户端起点,不会改变省略字段时的 API 默认: | Web Benchmark | 建议 `max_tokens` | 建议 `read_timeout_seconds` | | --- | ---: | ---: | @@ -818,7 +820,7 @@ curl -sS -X POST http://127.0.0.1:8000/api/v1/benchmarks/reload-demo | MMLU-Pro `official_cot` | `4000` | `300` | | GPQA-Diamond | `8192` | `600` | -Web 还允许显式选择“由 Provider 决定”,此时提交 `max_tokens:null`。更高数字、Provider 托管或更长读取超时都不是 Token/金额预算,也不证明模型支持对应输出长度。 +Chat/Responses 的 Web 表单还允许显式选择“由 Provider 决定”,此时提交 `max_tokens:null`;Messages 必须保留有限正整数输出上限。更高数字、Provider 托管或更长读取超时都不是 Token/金额预算,也不证明模型支持对应输出长度。 Run 响应示例(后续接口引用为 `RunRead`): @@ -1073,8 +1075,7 @@ Run 一旦提交,不会因随后 Redis 故障、并发/RPM/TPM 饱和而删除 curl -sS http://127.0.0.1:8000/api/v1/runs/44444444-4444-4444-8444-444444444444 ``` -`200 OK` 返回 `RunRead`。运行期间汇总字段可能为 `null`,进度由 -`completed_questions / total_questions` 表示。不存在返回: +`200 OK` 返回 `RunRead`。运行期间持久化 Run 汇总字段可能尚未追上逐题证据;动态成绩、完成率、错误数、延迟与 usage/cost 已知覆盖应读取 6.7 节的 progress index。Run 的精确 `input_tokens`、`output_tokens` 与 `estimated_cost` 仍遵守 all-or-nothing nullable 语义,不会由 progress 已知小计回填。不存在返回: ```json { @@ -1138,18 +1139,121 @@ curl -sS 'http://127.0.0.1:8000/api/v1/runs/44444444-4444-4444-8444-444444444444 ], "total": 15, "offset": 0, - "limit": 100 + "limit": 100, + "known_input_tokens": 120, + "known_output_tokens": 30, + "input_token_reported_responses": 15, + "output_token_reported_responses": 15 } ``` +四个 usage 汇总字段都针对该 Run 的**全部** Response,不受当前 `offset`/`limit` 影响: + +- `known_input_tokens` / `known_output_tokens` 分别汇总对应列中所有非 `null` 值;没有已知值时返回 `0`。 +- `input_token_reported_responses` / `output_token_reported_responses` 分别统计对应 Token 字段非 `null` 的 Response 数;合法的 `0` Token 仍计为已上报。两项计数彼此独立,相等不代表必然来自同一批 Response。 +- 这些字段是可审计的已知小计与覆盖证据,不是精确 Provider 账单。`RunRead.input_tokens/output_tokens` 继续遵守 protocol-v1 的 all-or-nothing 语义:任一逐题 usage 缺失时,精确 Run Token 保持 `null`,不会由部分小计回填。 + 请求失败、空回答或解析失败的记录仍会出现,`score=0`,并填写 `error_type` 与 -`error_message`;上游 usage 缺失时 Token 和费用为 `null`。非法 SSE UTF-8/JSON/字段映射为 `invalid_provider_stream`,200 SSE 内的上游 error 映射为 `provider_stream_error`,HTTP 干净结束却缺少 `[DONE]` 映射为 `incomplete_provider_stream`;已收到的部分 content 不会作为成功答案持久化。真实 transport 异常仍按 Run 快照的 protocol-v1 有限策略重试,因此仍可能重复上游计算或计费。若 Provider 返回 `finish_reason="length"`,空输出以及未能解析出有效最终答案的非空输出都会归类为 `output_truncated`,而不是泛化成 `empty_response` 或 `parse_error`;非空输出仍保存在 `raw_response`。成功内容若精确反射当前 Key,会在写入 `raw_response` 前替换为 `[REDACTED]`。 +`error_message`;上游 usage 缺失时 Token 和费用为 `null`。非法 SSE UTF-8/JSON/字段映射为 `invalid_provider_stream`,200 SSE 内的上游 error 映射为 `provider_stream_error`,HTTP 干净结束却缺少所选协议的终止证据(Chat `[DONE]`、Responses `response.completed`、Messages `message_stop`)映射为 `incomplete_provider_stream`;已收到的部分 content 不会作为成功答案持久化。除既有 retryable HTTP/transport 分类外,Responses 的 rate-limit/server typed error,以及 Messages 的 `rate_limit_error`、`api_error`、`overloaded_error`、`timeout_error` 会按 Run 快照有限重试;Messages 快照的 HTTP retryable status 另含 `529`。未知流内错误 fail closed,每次重试独立进入 attempt ledger,因此仍可能重复上游计算或计费。若 Provider 返回 `finish_reason="length"`,空输出以及未能解析出有效最终答案的非空输出都会归类为 `output_truncated`,而不是泛化成 `empty_response` 或 `parse_error`;非空输出仍保存在 `raw_response`。成功内容若精确反射当前 Key,会在写入 `raw_response` 前替换为 `[REDACTED]`。 逐题 transport 证据只保存并返回固定字段:Provider request ID、返回模型名、system fingerprint、finish reason 与实际 HTTP attempt 数;不会保存或返回任意 raw usage 对象。四个字符串必须是短、无空白的安全 opaque token,任何超长、控制字符、凭据形态或脱敏占位值都会 fail closed 为 `null`。这些值用于关联和诊断,不改变评分,也不把本地 Response 幂等扩展为 Provider exactly-once。可信本地报告的 `responses.jsonl` 导出同一组固定字段,并再次执行报告级 secret scrub 与安全字符边界。 -Web Run Detail 固定以 `limit=100` 请求一页逐题证据,并用 `offset` 提供上一页/下一页导航;它不会把大型正式 Benchmark 截止在前 100 条。Run 不存在返回 `404 run_not_found`;分页非法返回 `422`。 +Web Run Detail 固定以 `limit=100` 请求一页逐题证据,并用 `offset` 提供上一页/下一页导航;它不会把大型正式 Benchmark 截止在前 100 条。当前页仍分别显示未得分与执行异常。全 Run 动态指标与热力图改用下一节的 progress index/block,不从当前证据页推断。精确 Run Token 与 Response 汇总来自同一可核对快照时显示精确值;否则显示“已知小计”、输入/输出各自覆盖率和“完整总量未知”。Run 不存在返回 `404 run_not_found`;分页非法返回 `422`。 + +### 6.7 `GET /runs/{run_id}/progress` 与 `/progress/blocks/{block_index}` + +这两个只读接口为大型 Run Detail 提供轻量热力图事实,不使用通用 offset pagination,也没有 cursor。block 大小固定为 `512`,`block_index` 从 `0` 开始,absolute position 范围由 `block_index * block_size` 派生。 + +索引请求: + +```bash +curl -sS 'http://127.0.0.1:8000/api/v1/runs/44444444-4444-4444-8444-444444444444/progress' +``` + +`200 OK`: + +```json +{ + "block_size": 512, + "total_questions": 5, + "completed_questions": 3, + "correct_questions": 1, + "error_questions": 1, + "score": 20.0, + "completion_rate": 40.0, + "answered_accuracy": 50.0, + "average_latency_ms": 200.0, + "known_input_tokens": 30, + "known_output_tokens": 15, + "input_token_reported_responses": 2, + "output_token_reported_responses": 2, + "known_estimated_cost": 0.001, + "estimated_cost_reported_responses": 1, + "blocks": [ + {"block_index": 0, "response_count": 3} + ] +} +``` + +示例为字段结构演示;实际 `blocks` 必须覆盖该 Run 的全部计划 block,并按 `block_index` 升序返回,空 block 也保留 `response_count=0`。`completed_questions`、正确/异常数、三项成绩、平均延迟、known usage/cost 及其 reported coverage 与这些 block counts 从同一数据库读取快照派生,运行中每次请求都可变化。指标公式与 `llmbenchlab-protocol-v1` 相同;前端不得从尚未完全同步的 cells 子集另算主指标。 + +读取一个 block: + +```bash +curl -sS 'http://127.0.0.1:8000/api/v1/runs/44444444-4444-4444-8444-444444444444/progress/blocks/0' +``` + +`200 OK`: + +```json +{ + "block_index": 0, + "items": [ + { + "position": 0, + "outcome": "wrong", + "score": 0.0, + "latency_ms": 200.0, + "input_tokens": 20, + "output_tokens": 10, + "estimated_cost": null, + "error_type": null + }, + { + "position": 2, + "outcome": "passed", + "score": 1.0, + "latency_ms": 100.0, + "input_tokens": 10, + "output_tokens": 5, + "estimated_cost": 0.001, + "error_type": null + }, + { + "position": 3, + "outcome": "error", + "score": 0.0, + "latency_ms": 300.0, + "input_tokens": null, + "output_tokens": null, + "estimated_cost": null, + "error_type": "provider_error" + } + ] +} +``` + +`items` 只含该 block 已持久化的 Response,并按 absolute `position` 升序;没有返回的计划 position 隐式为 `not_run`。`outcome` 的互斥判定优先级固定为:`error_type != null` 时 `error`;否则 `score == 1` 时 `passed`;其余为 `wrong`。这保证执行异常不会同时被展示为普通答错。 + +cell 是严格白名单,只能包含 `position`、`outcome`、`score`、`latency_ms`、`input_tokens`、`output_tokens`、`estimated_cost`、`error_type`。接口不返回 Question/Response ID、external ID、prompt、choices、raw/parsed/reference answer、error message、Provider request/model/fingerprint/finish reason、任意 raw usage 或其他 Provider metadata。两个响应都带 `Cache-Control: no-store`。 + +客户端每秒读取小型 index,只拉取 `response_count` 非零且尚未同步、或 count 相对本地发生变化的 block。全部目标 block hydrate 完成前应显示“同步中”,不能把尚未加载的格子伪装成 `not_run`。Response 是 append-only 唯一事实;若新提交发生在 index 与 block 请求之间,block 可能已比旧 index 更新,客户端保留较新 items 并由下一次 index 收敛。Run 进入终态不代表最后一个 block 请求已经返回,客户端应在目标 counts 追齐后再停止 progress 轮询。 + +Run 不存在返回 `404 run_not_found`。负 `block_index` 由路径参数校验返回 `422`;`block_index >= block_count` 返回 typed `422 progress_block_out_of_range`。若已持久化 Response 的 Question/position 不属于 Run 冻结计划,index 整次 fail closed 为 `500 run_progress_integrity_error`,不返回部分指标或格子。索引响应不重复 Run `status`,客户端继续从 `GET /runs/{run_id}` 读取状态。 + +`known_input_tokens`、`known_output_tokens` 与 `known_estimated_cost` 始终只是非 `null` 证据的小计,三个 reported-response 计数分别给出覆盖范围;合法的零值仍算已上报。它们不改变 Run 精确字段:任一必要 usage 或价格缺失时,`RunRead.input_tokens/output_tokens/estimated_cost` 仍可为 `null`,不得把 known subtotal 冒充完整账单。 -### 6.7 `GET /runs/{run_id}/audit` +### 6.8 `GET /runs/{run_id}/audit` 返回该 Run 已保留的类型化审计事件。排序固定为 `(occurred_at, id)` 升序;`offset`/`limit` 使用通用 `0..` / `1..100` 分页规则。事件表由执行状态转换事务追加,分页读取不会修改事件、Run 或治理账本。 diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 590f0c9..0f450c6 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -1,6 +1,6 @@ # LLMBenchLab 架构 -本文描述 LLMBenchLab 当前 Phase 1 产品边界、Phase 2 可靠执行/治理/可观测性工作树,以及可信本地 MMLU-Pro/GPQA-Diamond 真实评测垂直切片。前端仍通过 REST API 操作同一组领域对象,但 API 不执行评测:PostgreSQL/数据库是任务、四层治理、逐 Provider attempt ledger、typed audit、Worker progress 与评分证据的唯一事实来源,Redis Streams 是非权威的 at-least-once 通知层,独立 Worker 或受信本地 CLI 用数据库租约、心跳与 fencing 执行同一 Runner。SQLite 保留为单 Worker 本地兼容路径。Exporter、告警和普通文件 archive 都是数据库事实的受控投影,不是第二状态机。Phase 2/3 总状态没有因此完成,这不是公网、HA、生产架构或 SLA 声明。 +本文描述 LLMBenchLab 当前 Phase 1 产品边界、Phase 2 可靠执行/治理/可观测性工作树,以及可信本地 MMLU-Pro/GPQA-Diamond 真实评测垂直切片。前端仍通过 REST API 操作同一组领域对象,但 API 不执行评测:PostgreSQL/数据库是任务、四层治理、逐 Provider attempt ledger、typed audit、Worker progress 与评分证据的唯一事实来源,Redis Streams 是非权威的 at-least-once 通知层,多个独立 Worker replica 或受信本地 CLI 用数据库租约、心跳与 fencing 执行同一 Runner。每个 Worker 同时只持有一个 Run,因此一个长期 Provider 请求只占用一个 Worker;其他 Worker仍可领取不同 Run。SQLite 保留为单 Worker 本地兼容路径。Exporter、告警和普通文件 archive 都是数据库事实的受控投影,不是第二状态机。Phase 2/3 总状态没有因此完成,这不是公网、HA、生产架构或 SLA 声明。 ## 架构目标与原则 @@ -30,7 +30,7 @@ flowchart LR Reports[本地完整报告] Env[进程环境变量] Keyring[部署 AES keyring] - Upstream[OpenAI-compatible API] + Upstream[Chat / Responses / Messages API] Git[本地 Git 元数据] User --> Browser @@ -63,7 +63,7 @@ flowchart TB subgraph FE[React 单页应用] Pages[Dashboard、Models、Benchmarks、Runs、New Run、Run Detail、Leaderboard] Client[集中式 API Client] - Poller[Run 状态轮询] + Poller[Run、证据页与 progress block 轮询] Pages --> Client Pages --> Poller end @@ -106,7 +106,7 @@ flowchart TB ReportFiles[summary / groups / responses] Client -->|REST JSON| Routes - Poller -->|GET Run 与 Responses| Routes + Poller -->|GET Run、当前 Responses 页、progress index/变化 blocks| Routes Importer -->|受限读取| Files APIRepo --> DB Governance --> DB @@ -117,7 +117,7 @@ flowchart TB AttemptLedger --> DB Runner --> DB Adapters -->|Mock: 无网络| Runner - Adapters -->|OpenAI-compatible: HTTPS remote / HTTP loopback| Provider + Adapters -->|三类显式远程协议: HTTPS remote / HTTP loopback| Provider PinnedSources -->|固定 revision / SHA| LocalCLI LocalCLI -->|复用 Dataset、Run、Lease、Runner| DB LocalCLI -->|发现 / canary / generate| Provider @@ -133,7 +133,7 @@ flowchart TB | Dataset Loader | 限制文件、解析、逐字段校验、稳定 Hash | 下载远程数据或执行数据集代码 | | Standard Dataset Converters | 固定下载/缓存 MMLU-Pro、GPQA,校验源 Hash 并生成 dataset-v1 ZIP | 接受任意 URL、解释许可或执行题目代码 | | Trusted-local CLI | prepare、Provider preflight/确认、创建/恢复 `legacy_unmanaged` Run、驱动 Runner、导出报告 | 提供公网 API、保存明文 Key、继承 Web/API governance 或多租户调度 | -| Provider Preflight | 推导 `/models`、拒绝 Key 反射、确定模型、执行最小可解析且返回模型一致的 Chat canary | 猜测多个付费目标或证明供应商完全兼容 | +| Provider Preflight | 推导 `/models`、拒绝 Key 反射、确定模型、按显式 Chat Completions / Responses / Messages 协议执行最小可解析且返回模型一致的 canary | 猜测多个付费目标、跨协议 fallback 或证明供应商完全兼容 | | Report Exporter | 分页读取终态证据,从计划题与 Responses 派生唯一指标集并原子写出三文件 | 覆盖已有报告、充当访问控制或修改 Run | | Redis queue | 提供低延迟、at-least-once 通知和 ACK/PEL | 保存权威状态、租约、取消或结果 | | WorkerService | 数据库对账、消费/确认通知、优雅停机 | 改变评分协议或用 Redis 裁决状态 | @@ -142,7 +142,7 @@ flowchart TB | ModelAdapter | 把统一生成请求映射到具体模型 | 评分 | | Evaluator | 安全解析答案并给出 0/1 分 | 调用模型或数据库 | | Repository / ORM | 事务与持久化 | API 序列化和业务展示 | -| React UI | 用户操作、write-only Key 表单、轮询和可视化 | 持久化/读回 Key、读取 keyring 或直接调用模型供应商 | +| React UI | 用户操作、write-only Key 表单、独立轮询、虚拟化可访问热力图和可视化 | 持久化/读回 Key、读取 keyring、直接调用模型供应商或从部分 cells 重算权威指标 | ## 关键数据流 @@ -217,7 +217,7 @@ sequenceDiagram R->>M: generate(messages, config, request-local governance context) M->>D: reserve:四层 scope/bucket/ledger 原子 admission M->>D: mark send_started(成功后才允许外发) - M->>M: 打开 Provider SSE,持续消费 token/心跳/usage 至 [DONE] + M->>M: 打开 Provider SSE,消费至 Chat [DONE] / Responses response.completed / Messages message_stop alt 生成成功且非空 M-->>R: ModelGenerationResult R->>E: evaluate(raw_response, reference, config) @@ -257,7 +257,7 @@ Redis 通知和 ACK 都可重复,所以系统是 at-least-once;数据库中 1. `prepare` 下载固定源、校验并转换;不读取 Key 或连接 Provider。 2. `run` 强制选择 `--limit` 或 `--full`,从环境变量/隐藏输入取得 Key,使用兼容根路径调用 `GET /models`;远端只允许 HTTPS,HTTP 仅允许 loopback,发现请求只接受 identity 编码且正文上限 2 MiB,多模型时不猜测目标,任何模型 ID 反射当前 Key 都使预检失败。 -3. CLI 输出 host、模型、题数、剩余 Run attempts 和最大 Chat HTTP 尝试数并要求确认;上界按 `(缺失题数 × 剩余 Run attempts + 1 个 canary) × 3` 包含 HTTP retries,再执行一个最小可解析、可能计费的 canary。若成功体明确返回其他模型名,canary 失败。 +3. CLI 输出 host、显式协议、模型、题数、剩余 Run failed-attempt 预算和最大 Provider HTTP 尝试数并要求确认;这里的剩余预算严格为 `max_attempts - failed_attempt_count`,不把 cooperative yield 或仍有效的成功领取误算为失败。请求上界按 `(缺失题数 × 剩余 failed-attempt 预算 + 1 个 canary) × 3` 包含 HTTP retries,再通过所选 Adapter 的 endpoint/headers/parser 执行一个最小可解析、可能计费的 canary。若成功体明确返回其他模型名,canary 失败。 4. 通过 preflight 后才持久化 Benchmark、Model 和带脱敏 preflight/完整配置快照的 pending Run,并在同一进程用独立 lease owner 驱动现有 Runner。 5. Runner 先启动租约心跳,再通过工作线程加载和物化数据库快照,避免大题集同步加载阻塞事件循环;随后以固定 `min(concurrency, question_count)` 个消费者协程从迭代器取题,不为 12,032 题一次性创建 task。Response 仍逐题短事务、fencing 与幂等。 6. `resume` 重新确认/canary;若旧的未完成租约已经过期,本地 Runner 会执行 fenced reclaim,而不是等待已经停止的外部 Worker。恢复时跳过已有 Response,只执行缺失题。`report` 仅读取终态数据库事实,不接触 Provider。 @@ -308,6 +308,12 @@ classDiagram class OpenAICompatibleAdapter { +generate(messages, generation_config) ModelGenerationResult } + class OpenAIResponsesAdapter { + +generate(messages, generation_config) ModelGenerationResult + } + class AnthropicMessagesAdapter { + +generate(messages, generation_config) ModelGenerationResult + } class ModelGenerationResult { +text: string +input_tokens: integer or null @@ -320,6 +326,8 @@ classDiagram ModelAdapter <|.. MockModelAdapter ModelAdapter <|.. OpenAICompatibleAdapter + ModelAdapter <|.. OpenAIResponsesAdapter + ModelAdapter <|.. AnthropicMessagesAdapter ModelAdapter --> ModelGenerationResult ``` @@ -329,16 +337,21 @@ classDiagram generate(messages, generation_config, *, attempt_context=None) -> ModelGenerationResult ``` -`messages` 是已经应用 Run 快照 Prompt 的消息数组;`generation_config` 至少承载 `temperature`、`top_p`、`max_tokens` 和 `seed`。`max_tokens` 为 `1..131072` 的数字时由 Adapter 原样发送,为 `null` 时不发送该字段并由 Provider 选择默认预算;这不等于无限输出。Prompt 渲染只在 manifest `system` 含非空白内容时插入 system message;空 system 被省略。当前 GPQA-Diamond 固定 `zero-shot-cot-answer-line-v1`,要求模型推理后在末行输出 `Answer: X`;MMLU-Pro 与 GPQA 的标准 system 都为空。上述 profile/省略规则属于模型输入身份,必须随模板与 Dataset Hash 一起比较。 +`messages` 是已经应用 Run 快照 Prompt 的消息数组;`generation_config` 承载 nullable `temperature`、`top_p`、`max_tokens` 和 `seed`。Chat 把有限 `max_tokens` 原样发送,Responses 映射为 `max_output_tokens`,二者的 `null` 表示省略字段并由 Provider 选择默认预算;Messages 则要求有限值。Responses/Messages 在请求和 Model 默认都没有显式采样值时,把 `temperature/top_p/seed` 冻结为 `null` 并从 Provider payload 省略;两者拒绝非空 `seed`,Messages 的非空 `temperature` 还限制为 `0..1`。Prompt 渲染只在 manifest `system` 含非空白内容时插入 system message;Messages Adapter 把这条初始 system instruction 提升为顶层 `system`,其他消息保持有序。空 system 被省略。当前 GPQA-Diamond 固定 `zero-shot-cot-answer-line-v1`,要求模型推理后在末行输出 `Answer: X`;MMLU-Pro 与 GPQA 的标准 system 都为空。上述 profile/省略规则属于模型输入身份,必须随模板与 Dataset Hash 一起比较。 `attempt_context` 是 request-local、只含 run/question/model/provider opaque scope、lease token、execution generation、attempt ordinal 与数值预留的非秘密对象;可选 controller 在 Adapter 内围住每一个真实 HTTP retry,保证一次 `generate()` 的多个 attempt 不会被漏记。controller 不接触 Key、header、Prompt 或响应正文。 Adapter Registry 根据 `provider_type` 选择实现: - `mock`:使用输入中稳定标识或 Demo 题目映射产生可预测回答,不读取密钥且不得发起网络请求;latency、Token、usage shape 与模拟错误等全部确定性本地配置先验证,非法配置不会调用 reserve/mark/finish hook,成功与模拟 Provider 错误仍走同一三阶段治理语义。 -- `openai_compatible`:校验 `base_url` 和 `remote_model_name`,并要求恰好一种凭据输入:environment Run 在调用前按 `api_key_env` 读取,stored Run 由 Worker 传入已解密的 `SecretStr`。Adapter 不读取数据库或 keyring。远端 URL 必须为 HTTPS,HTTP 仅允许 loopback;请求声明 `Accept-Encoding: identity` 并拒绝压缩响应。Chat payload 显式发送 `stream:true` 与 `stream_options.include_usage:true`,request-local parser 支持任意字节/UTF-8 分块、SSE comment 心跳、多 `data` 行、role/null delta、usage-only 尾块和 `[DONE]`,并只聚合 `delta.content`;推理扩展字段不混入评测答案。看到 finish reason 不会提前返回,HTTP 干净 EOF 却缺少 `[DONE]` 不作成功;忽略 stream 请求而返回普通 JSON 的 Provider 继续兼容。普通 JSON 成功体上限 4 MiB,SSE wire/单事件/聚合 content 上限分别为 64 MiB/1 MiB/4 MiB,非 2xx 错误体上限 64 KiB。对 429、部分 5xx 和暂时性 transport 错误仍执行快照内有上限的指数退避;明显的 4xx 配置错误不重试。 +- 三类远程 Adapter 共用 `base_url`/模型/凭据、HTTPS(HTTP 仅 loopback)、identity-only、禁 redirect、有界读取、有限 retry、attempt ledger 和聚合后当前-Key 脱敏边界。根地址分别追加 `/chat/completions`、`/responses`、`/messages`;匹配的完整 endpoint 原样使用,其他已知协议后缀在外发前拒绝。environment Run 在调用前按 `api_key_env` 读取,stored Run 由 Worker 传入已解密的 `SecretStr`;Adapter 不读取数据库或 keyring。 +- `openai_compatible`:保留 Chat Completions 向后兼容值。payload 显式发送 `stream:true` 与 `stream_options.include_usage:true`;request-local parser 支持任意字节/UTF-8 分块、SSE comment、多 `data` 行、role/null delta、usage-only 尾块和 `[DONE]`,只聚合 `delta.content`。 +- `openai_responses`:发送 Responses `input` 与 `max_output_tokens`,按 `response.output_text.delta` 聚合文本,以 `response.completed` 为成功终止证据;普通 JSON 从 `output` 文本项提取答案并归一化 `input_tokens/output_tokens`。 +- `anthropic_messages`:使用 `x-api-key`、`anthropic-version` 和 Messages payload,按 `content_block_delta` 聚合文本,以 `message_stop` 为成功终止证据;普通 JSON从 `content[].text` 提取答案,usage 从 `message_start`/`message_delta` 或最终对象归一化且不重复求和。 -Adapter 将供应商错误映射为稳定的内部分类,例如 `authentication_error`、`rate_limited`、`provider_4xx`、`provider_5xx`、`connect_timeout`、`read_timeout`、`network_error`、`invalid_provider_stream`、`provider_stream_error`、`incomplete_provider_stream`、`empty_response` 和 `output_truncated`。Provider 以 `finish_reason="length"` 返回空内容时,Adapter 直接产生 `output_truncated`;返回非空内容但 Evaluator 无法解析有效最终答案时,Runner 根据同一 finish reason 把普通 parse error 提升为 `output_truncated`,并保留已有 raw response、usage、延迟与成本证据。最终 `httpx.TransportError` 转成安全 `AdapterError` 时不保留可达 request/Authorization 的 `__cause__` 或 `__context__`。日志和持久化错误不得包含 Authorization、密钥值、原始 SSE 行/事件或完整敏感响应头。SSE content 先聚合再执行当前 Key 的精确替换,避免 Key 横跨 delta 时泄漏。Provider 返回证据会递归检查成功内容、raw usage 的对象键和 JSON 标量,以及 provider request ID、返回模型名、system fingerprint 和 finish reason;其中出现的当前 Key 会先做精确替换。这不是对任意敏感内容的通用 DLP,也不扫描与 Provider 响应无关的固定数据。 +看到 finish reason 不会提前返回,HTTP 干净 EOF 却缺少各协议终止事件不作成功。普通 JSON 成功体上限 4 MiB,SSE wire/单事件/聚合 content 上限分别为 64 MiB/1 MiB/4 MiB,非 2xx 错误体上限 64 KiB。对 429、部分 5xx 和暂时性 transport 错误仍执行快照内有上限的指数退避;明显的 4xx 配置错误不重试。协议不根据 URL/模型名猜测,也不在失败后 fallback,Run snapshot 的 Adapter 类型决定恢复语义。 + +Adapter 将供应商错误映射为稳定的内部分类,例如 `authentication_error`、`rate_limited`、`provider_4xx`、`provider_5xx`、`connect_timeout`、`read_timeout`、`network_error`、`invalid_provider_stream`、`provider_stream_error`、`incomplete_provider_stream`、`empty_response` 和 `output_truncated`。除既有 retryable HTTP/transport 分类外,Responses 的 rate-limit/server typed error,以及 Messages 的 `rate_limit_error`、`api_error`、`overloaded_error`、`timeout_error` 才允许重试;Messages 的 HTTP `529` 也写入冻结 retry status。未知流内错误 fail closed,每个重试都使用独立 attempt ledger 行。Provider 以 `finish_reason="length"` 返回空内容时,Adapter 直接产生 `output_truncated`;返回非空内容但 Evaluator 无法解析有效最终答案时,Runner 根据同一 finish reason 把普通 parse error 提升为 `output_truncated`,并保留已有 raw response、usage、延迟与成本证据。最终 `httpx.TransportError` 转成安全 `AdapterError` 时不保留可达 request/`Authorization`/`x-api-key` 的 `__cause__` 或 `__context__`。日志和持久化错误不得包含秘密 header、密钥值、原始 SSE 行/事件或完整敏感响应头。SSE content 先聚合再执行当前 Key 的精确替换,避免 Key 横跨 delta 时泄漏。Provider 返回证据会递归检查成功内容、raw usage 的对象键和 JSON 标量,以及 provider request ID、返回模型名、system fingerprint 和 finish reason;其中出现的当前 Key 会先做精确替换。这不是对任意敏感内容的通用 DLP,也不扫描与 Provider 响应无关的固定数据。 ## Evaluator 架构 @@ -534,9 +547,13 @@ Runner 从 Run 快照读取模型连接配置、凭据来源、价格、生成 每条 EvaluationResponse 除 raw/parsed/reference/score/error/usage 外,可保存经过长度、字符与当前 Key 精确反射检查的 `provider_request_id`、`returned_model`、`system_fingerprint`、`finish_reason` 和 `http_attempt_count`;任一不安全字符串归一化为 `null`,报告只导出这些 typed 字段,不持久化任意 raw usage object。 -Model 的 `default_parameters` 在 Phase 1 只接受 Adapter 实际转发的 `temperature`、`top_p`、`max_tokens`、`seed`。创建 Run 时显式字段覆盖 Model 默认,省略字段才使用 Model 默认;显式 `max_tokens:null` 也是覆盖值,不会回退。`generation` 块因此只包含实际执行值,不把未转发的 Provider 扩展伪装成有效参数。通用 API 保留 protocol-v1 的 `max_tokens=256` 和读取超时 `60s` 默认;没有对应 Model 默认且用户尚未手动修改时,Web 根据已知 Benchmark 提交可编辑的显式建议:Demo `256/60s`、MMLU-Pro direct `1024/180s`、official CoT `4000/300s`、GPQA-Diamond `8192/600s`。 +Model 的 `default_parameters` 在 Phase 1 只接受 Adapter 实际转发的 `temperature`、`top_p`、`max_tokens`、`seed`。创建 Run 时显式字段覆盖 Model 默认,省略字段才使用 Model 默认;显式 `max_tokens:null` 也是覆盖值,不会回退。为兼容旧 Chat 客户端,请求 Schema 继续以 `temperature=0/top_p=1/seed=42` 为默认;Responses/Messages 若请求和 Model 默认都未提供这些采样字段,则 Run 构建时归一化为 `null` 并从上游 payload 省略。`generation` 块因此只包含实际执行值,不把未转发的 Provider 扩展伪装成有效参数。通用 API 保留 protocol-v1 的 `max_tokens=256` 和读取超时 `60s` 默认;没有对应 Model 默认且用户尚未手动修改时,Web 根据已知 Benchmark 提交可编辑的显式输出建议:Demo `256/60s`、MMLU-Pro direct `1024/180s`、official CoT `4000/300s`、GPQA-Diamond `8192/600s`。Messages 始终使用有限输出上限。 + +React 主导航包含独立的 Runs 列表页。该页通过 `GET /runs` 以 20 条为一页显示所有状态,支持状态筛选,并在当前页存在 active Run 时轮询;列表和 Run Detail 都以持久化 Run ID 建立链接。Run Detail 对 `GET /runs/{id}/responses` 使用每页 100 条的 offset 分页,而不是只加载大型正式 Benchmark 的前 100 条;详情页页码与全 Run 进度读取相互独立。Run Detail 还显式展示 `managed/delayed/exhausted/legacy_unmanaged` 治理状态,只将封闭的稳定 reason 映射为人类可读文案,`governance_not_before` 以 UTC 显示,未知值不原样反射。 + +全 Run 热力图采用固定 512 题 absolute-position blocks,而不是 Response offset 页或时间 cursor。`GET /runs/{id}/progress` 在一个数据库读取快照中返回全部计划 block 的 `response_count` 与 evidence-derived live metrics;`GET /runs/{id}/progress/blocks/{block_index}` 只返回该范围内按 position 排序的已持久化 cell 白名单。Response 对 Run/Question 唯一且追加,因而 block count 单调:客户端每秒只比较 index,hydrate 非空或 count 变化的 block;index→block 之间的新提交可使 payload 比旧 index 更新,但不会永久漏失,下一 index 会收敛。没有 Response 的计划 position 才是 `not_run`,非空 block 初始 hydrate 完成前为“同步中”。页面切 Run、旧请求返回、hidden/visible 和终态先到都由独立 progress reducer/poller 处理;终态需等目标 counts 追齐后才停止该 poller。 -React 主导航包含独立的 Runs 列表页。该页通过 `GET /runs` 以 20 条为一页显示所有状态,支持状态筛选,并在当前页存在 active Run 时轮询;列表和 Run Detail 都以持久化 Run ID 建立链接。Run Detail 对 `GET /runs/{id}/responses` 使用每页 100 条的 offset 分页,而不是只加载大型正式 Benchmark 的前 100 条;active Run 继续轮询当前证据页,进入终态后停止。Run Detail 还显式展示 `managed/delayed/exhausted/legacy_unmanaged` 治理状态,只将封闭的稳定 reason 映射为人类可读文案,`governance_not_before` 以 UTC 显示,未知值不原样反射。 +四态 outcome 的后端优先级为 `error_type != null -> error`、否则 `score == 1 -> passed`、否则 `wrong`;未持久化为 `not_run`。index 的 score/completion/answered accuracy/error/latency 与 known Token/cost coverage 复用 protocol-v1 聚合定义,前端不从部分 block Map 重算。known subtotal 只是覆盖证据,不能写回或替代 Run `input_tokens/output_tokens/estimated_cost` 的 all-or-nothing nullable 真值。固定白名单排除 Question/Response ID、正文/答案、error message 与 Provider transport metadata;两个 progress 响应均为 `no-store`。该设计复用现有唯一约束和 `Question.position`,不增加表、revision 或 migration。 写入约束: @@ -560,7 +577,7 @@ React 主导航包含独立的 Runs 列表页。该页通过 `GET /runs` 以 20 带凭据的目标 DSN 必须通过 `--target-env` 从受控环境读取;`--target` 拒绝 URL password 和 password query。COMMIT 未获得 PostgreSQL 确认时退出 `4`/`commit_outcome_unknown`;由于事务原子性,目标可能为空,也可能是完整的 precommit 快照。COMMIT 已确认但连接收尾、postcommit 快照/对账或报告失败时退出 `3`/`committed_but_verification_failed`;这时目标已提交完整 precommit 快照,不会自动回滚。两种结果都禁止盲目重试,必须保持目标离线,按已输出摘要独立检查目标是空还是完整提交。非空目标会拒绝再次导入,工具也不提供 PostgreSQL 到 SQLite 的反向同步。 -Alembic `20260827_0004` 引入治理/审计表与 Run/Response 字段;`20260828_0005` 增加 `worker_processes` 与 audit retention/exporter 扫描索引;schema-equivalent `20260829_0006` 只条件补齐早期 `0004` 变体缺少的三个 canonical 索引;当前 data-only head `20260830_0007` 不改 schema、ledger 或 actual usage,只按显式 hard reservation 语义重算 `governance_scopes.overdrawn`。`0007` upgrade/downgrade 均在任何更新前拒绝 `reserved/send_started` active reservation;downgrade 只恢复旧派生谓词。兼容 preflight 仍以精确 fingerprint、integrity/FK、索引定义和 single-active 数据约束 fail closed。`0006 → 0005` 不删除 canonical 对象;后续两个 downgrade guard 都在第一条有损 DDL 前拒绝可能丢失的事实:0005 拒绝任意 Worker generation,0004 拒绝 policy/scope/bucket/question-execution/ledger/audit 或新 Run/Response 证据。隔离空数据库分别验证 `0005 ↔ 0004` 与 `0004 ↔ 0003`;已使用环境应优先向前修复,或恢复经核验的旧备份并单独保留新 schema 证据。完整流程见 [`OPERATIONS.md`](./OPERATIONS.md)。 +Alembic `20260827_0004` 引入治理/审计表与 Run/Response 字段;`20260828_0005` 增加 `worker_processes` 与 audit retention/exporter 扫描索引;schema-equivalent `20260829_0006` 只条件补齐早期 `0004` 变体缺少的三个 canonical 索引;data-only `20260830_0007` 不改 schema、ledger 或 actual usage,只按显式 hard reservation 语义重算 `governance_scopes.overdrawn`;当前 head `20260830_0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并同时替换 Provider 类型 check 与远程配置 check,旧 Model 不改写。`0008 → 0007` 在新协议 Model 存在时于 DDL 前拒绝;`0007` upgrade/downgrade 均在任何更新前拒绝 `reserved/send_started` active reservation,且 downgrade 只恢复旧派生谓词。兼容 preflight 仍以精确 fingerprint、integrity/FK、索引定义和 single-active 数据约束 fail closed;无版本但 metadata-current 的 SQLite 仍先 stamp `0006`,使 data-only `0007` 真实执行后再进入 `0008`。`0006 → 0005` 不删除 canonical 对象;后续两个 downgrade guard 都在第一条有损 DDL 前拒绝可能丢失的事实:0005 拒绝任意 Worker generation,0004 拒绝 policy/scope/bucket/question-execution/ledger/audit 或新 Run/Response 证据。隔离空数据库分别验证 `0005 ↔ 0004` 与 `0004 ↔ 0003`;已使用环境应优先向前修复,或恢复经核验的旧备份并单独保留新 schema 证据。完整流程见 [`OPERATIONS.md`](./OPERATIONS.md)。 ## 错误处理与可观察性 @@ -590,7 +607,7 @@ Alembic `20260827_0004` 引入治理/审计表与 Run/Response 字段;`2026082 ## 部署拓扑与安全边界 -本地 Make 模式启动 API、独立 Worker 和 Vite,默认 SQLite 且 Redis 可选;SQLite 只支持一个 Worker。`make setup` 与其他相关启动入口通过 `uv run --script` 显式选择满足 `>=3.11` 的独立 CPython,再由 bootstrap 为 Web credential 生成 Git 忽略、权限为 `0600` 的 `.secrets/credential-keys.json`,API 与 Worker 读取同一文件;这避免 `PATH` 中其他 Python 实现破坏安全原子安装语义,也不会为 Docker-only 入口同步宿主后端依赖。可信本地 CLI 是第三条运维入口:它直接复用当前数据库与 Runner,因此运行时必须停止连接同库的常规 API/Worker 并独占数据库,默认把下载、转换 ZIP 与报告放入 Git 忽略的 `artifacts/`。Compose 包含六个 service:长运行的 PostgreSQL、Redis、API、Worker、frontend,以及一次性 migrate;同一只读 Compose secret 只挂载到 API/Worker。PostgreSQL/Redis 各自使用 named volume,Redis 启用 AOF;API/frontend host port 明确绑定 loopback,DB/Redis 无 host port。CORS 只允许配置的前端 Origin。 +本地 Make 模式启动 API、独立 Worker 和 Vite,默认 SQLite 且 Redis 可选;SQLite 只支持一个 Worker。连接 PostgreSQL 时,`make dev DEV_WORKERS=N` 可由同一启动器管理多个独立 Worker 进程,并把每个进程写入独立私有日志;非 PostgreSQL 请求 `N>1` 会在启动前拒绝。`make setup` 与其他相关启动入口通过 `uv run --script` 显式选择满足 `>=3.11` 的独立 CPython,再由 bootstrap 为 Web credential 生成 Git 忽略、权限为 `0600` 的 `.secrets/credential-keys.json`,API 与 Worker 读取同一文件;这避免 `PATH` 中其他 Python 实现破坏安全原子安装语义,也不会为 Docker-only 入口同步宿主后端依赖。可信本地 CLI 是第三条运维入口:它直接复用当前数据库与 Runner,因此运行时必须停止连接同库的常规 API/Worker 并独占数据库,默认把下载、转换 ZIP 与报告放入 Git 忽略的 `artifacts/`。Compose 包含六类 service:长运行的 PostgreSQL、Redis、API、可复制 Worker、frontend,以及一次性 migrate;`make dev-multi` / `make docker-up` 默认把 Worker scale 与 API 的 expected minimum 同时设为 2,并在返回前核对数据库 gauges。同一只读 Compose secret 只挂载到 API/Worker。PostgreSQL/Redis 各自使用 named volume,Redis 启用 AOF;API/frontend host port 明确绑定 loopback,DB/Redis 无 host port。CORS 只允许配置的前端 Origin。 当前 Compose 只是本地开发/故障验收拓扑,示例数据库密码不是生产秘密管理。前端 Nginx 位于浏览器→API 路径,不代表 Worker→Provider 路径上的 Cloudflare、Caddy 或其他 Gateway;真 SSE 必须在这条上游链路上保留正确 Content-Type 并持续 flush,且仍受每层独立的缓冲/超时限制。虽然远端 Provider 已强制 HTTPS、明文 HTTP 只允许 loopback,`base_url` 的允许范围仍未达到公网多租户要求;有效的 HTTPS URL 仍可能指向私网/云元数据或发生 DNS rebinding。本版本仅供受信任的本地操作者使用,不应直接暴露公网。后续公开部署必须增加鉴权、TLS、URL allowlist、DNS/IP 重绑定防护、出站网络策略、上传隔离、权限拆分、备份/PITR 和资源配额。当前不声称生产、HA 或灾备 SLA。 @@ -604,7 +621,7 @@ v2 aggregate schema 为 `llmbenchlab-phase2-slo-evidence-v2`。aggregate 与 raw ## 当前限制 -- PostgreSQL 租约已支持受限多 Worker 协调;SQLite 仍只适合单 Worker、单机低并发。这不是无限水平扩展或 HA 保证。 +- PostgreSQL 租约已支持受限多 Worker 协调;普通 Compose 入口默认两个 replica,按扩/缩方向同步 API expected,并在启动时验证 expected/registered/live/stalled/shortfall。SQLite 仍只适合单 Worker、单机低并发。这不是无限水平扩展或 HA 保证;当前容量资格不覆盖三个以上 Worker。 - managed Web/API Run 已实现数据库权威四层限流/预算/背压与公平调度;默认 policy 可关闭限制,`legacy_unmanaged` CLI 不覆盖,当前 global scope 锁会串行化新 admission,Mock 容量基线也不是生产调优结论。 - typed audit/history/archive、低基数 exporter、八条规则、Worker DB-time progress 与逐题 Provider metadata 已实现,但 append-only/普通文件 hash 不是 WORM;尚无告警发送器、tracing/认证监控面板、`resume` canary 独立事件或公网对象级访问控制。 - at-least-once 不能防止 `send_started` 后崩溃留下 Provider 幽灵请求,或 Provider 响应到本地 COMMIT 之间的重复远程调用/费用;本地 ledger 幂等和保守结算都不是远端 exactly-once/账单真值。 diff --git a/docs/BENCHMARK_PROTOCOL.md b/docs/BENCHMARK_PROTOCOL.md index bd9ad5a..96c5a20 100644 --- a/docs/BENCHMARK_PROTOCOL.md +++ b/docs/BENCHMARK_PROTOCOL.md @@ -100,7 +100,7 @@ MMLU-Pro 的标准转换器把题目选项以及 profile 指令直接固化在 每道计划题恰好产生一条最终 EvaluationResponse。过程如下: 1. 使用 Run 快照构造消息和 generation config。 -2. 调用选定 ModelAdapter;临时错误按快照策略有限重试。OpenAI-compatible 远端只允许 HTTPS,明文 HTTP 仅允许 loopback;Chat 请求声明且响应只接受 identity encoding,显式发送 `stream:true` 与流式 usage 选项,持续聚合 `delta.content`。看到 finish 后不提前返回;若有 usage-only 尾块则继续读取,并验证 `[DONE]`,usage 缺失时保持未知。返回普通 JSON 的 Provider 使用 fallback。普通 JSON 成功体最多 4 MiB,SSE wire/单事件/聚合 content 最多 64 MiB/1 MiB/4 MiB,错误体最多 64 KiB。数字 `max_tokens` 原样发送,快照值为 `null` 时省略该请求字段;每次等待下一批字节使用 Run 冻结的空闲读取超时。 +2. 调用 Run 快照选定的 ModelAdapter;临时错误按快照策略有限重试。三类远程 Adapter 都只允许 HTTPS,明文 HTTP 仅允许 loopback,并声明且只接受 identity encoding。Chat、Responses、Messages 分别向 `/chat/completions`、`/responses`、`/messages` 发送各自 payload,持续聚合对应文本 delta,并以 `[DONE]`、`response.completed`、`message_stop` 验证完整终止;usage 缺失时保持未知,各协议普通 JSON 成功体作为 fallback。普通 JSON 成功体最多 4 MiB,SSE wire/单事件/聚合 content 最多 64 MiB/1 MiB/4 MiB,错误体最多 64 KiB;malformed/oversized 错误不链回原始 Provider 内容。Chat 的数字 `max_tokens` 原样发送,Responses 映射为 `max_output_tokens`,二者的 `null` 省略该字段;Messages 必须使用有限正整数。Responses/Messages 在请求与 Model 默认都未提供时省略 `temperature/top_p/seed`,并拒绝非空 seed;每次等待下一批字节使用 Run 冻结的空闲读取超时。 3. 成功时保存未经答案规范化的 `raw_response`、input/output Token 和总延迟。若成功内容、raw usage 的对象键/字符串值、request ID、返回模型名或 system fingerprint 精确包含当前 Key,Adapter 会先替换为 `[REDACTED]`。Adapter 结果仍只在调用期携带 request ID/raw usage/返回模型/fingerprint;Phase 1 的 Response Schema 不持久化这些 transport 扩展字段。 4. 根据 manifest 的题型映射选择版本化 Evaluator。 5. 保存 `parsed_answer`、标准答案快照、0/1 分、Evaluator 名称和解析元数据。 @@ -197,9 +197,9 @@ answered\_accuracy = 100 \times \frac{\sum_{i=1}^{N}s_i}{\sum_{i=1}^{N}a_i} - 任一必要 usage 或价格缺失时,该题成本为 `null`;Run 成本只有在所有已完成模型调用均可计算时才给出总值。Mock 的明确零单价可计算为 0。 - 价格必须在 Run 创建时快照。估算成本不等同供应商账单,不包含缓存、阶梯价、税费或重试计费差异。 -可信本地 CLI 在创建新 Run 或恢复缺失题前先执行一个最小 Chat Completions canary。模型发现若有任何模型 ID 反射当前 Key,预检立即失败;canary 必须可解析为预期答案,且成功体明确返回模型名时必须与请求目标完全一致。新 Run 保存脱敏的初次模型发现/canary 状态、返回模型名、request ID、usage、延迟和尝试次数。确认界面的 HTTP 请求上界包含每次调用最多 3 次 HTTP attempts,以及新 Run 的全部或恢复 Run 的剩余 execution attempts;当前公式为 `(缺失计分题数 × 剩余 Run attempts + 1 个 canary) × 3`。canary 不是计分题,也不进入 Run 的三项成绩;它和失败重试可能产生的费用也不保证被 Run `estimated_cost` 完整覆盖,最终应以 Provider 账单为准。 +可信本地 CLI 在创建新 Run 或恢复缺失题前,按显式 Adapter 执行一个最小 canary:Chat、Responses、Messages 分别调用 `/chat/completions`、`/responses`、`/messages`,并以 `[DONE]`、`response.completed`、`message_stop` 作为流式成功终止证据。模型发现也按显式协议鉴权;若任何模型 ID 反射当前 Key,预检立即失败。canary 必须可解析为预期答案,且成功体明确返回模型名时必须与请求目标完全一致。新 Run 保存脱敏的初次模型发现/canary 状态、返回模型名、request ID、usage、延迟和尝试次数。确认界面的 HTTP 请求上界包含每次调用最多 3 次 HTTP attempts,以及新 Run 或恢复 Run 剩余的 failed-attempt 预算;该预算严格为 `max_attempts - failed_attempt_count`,不把 cooperative yield 或仍有效的成功领取算作失败。当前公式为 `(缺失计分题数 × 剩余 failed-attempt 预算 + 1 个 canary) × 3`。canary 不是计分题,也不进入 Run 的三项成绩;它和失败重试可能产生的费用也不保证被 Run `estimated_cost` 完整覆盖,最终应以 Provider 账单为准。 -P2-06 的审计链尚未闭合:`resume` 会重新执行 canary,但不会把这次证据追加成独立审计事件;逐题 Provider request ID、返回模型名和 system fingerprint 也尚未持久化。因此初次 preflight 快照不能证明恢复期间或每一道题实际命中的远端版本。 +P2-06 的审计链尚未完全闭合:`resume` 会重新执行 canary,但不会把这次证据追加成独立审计事件。逐题 Provider request ID、返回模型名、system fingerprint、finish reason 与 HTTP attempt count 已按安全边界持久化;初次 preflight 快照仍不能单独证明恢复期间的 canary 结果或 Provider 账单真值。 ## 完整报告导出 diff --git a/docs/DEPLOYMENT.md b/docs/DEPLOYMENT.md index dad6547..dc86894 100644 --- a/docs/DEPLOYMENT.md +++ b/docs/DEPLOYMENT.md @@ -5,7 +5,7 @@ LLMBenchLab 当前有三条本地运行路径: - 本地开发兼容路径:SQLite、可选 Redis、FastAPI API、独立 Worker 和 Vite。它便于开发与离线 Mock 验收,但 SQLite 只支持单 Worker 低并发,不能替代 PostgreSQL 并发证据。 -- Phase 2 Compose 可靠执行路径:PostgreSQL 是唯一任务事实来源,Redis Streams 是 at-least-once 通知层,API 与 Worker 是独立进程,migrate 是唯一 Alembic upgrade owner,Nginx 提供前端。 +- Phase 2 Compose 可靠执行路径:PostgreSQL 是唯一任务事实来源,Redis Streams 是 at-least-once 通知层,API 与多个可复制 Worker 是独立进程,migrate 是唯一 Alembic upgrade owner,Nginx 提供前端。标准 Make 入口默认两个 Worker。 - 可信本地正式评测 CLI:直接连接已迁移数据库,固定下载/转换 MMLU-Pro 或 GPQA-Diamond,在同一进程做 Provider preflight、运行有界 Runner、恢复缺失题并导出全量报告;不要求浏览器、API、Redis 或常驻 Worker。 可靠执行基础已经覆盖租约、心跳、fencing、幂等 Response、取消、有限重试、数据库 reconciliation 和故障恢复;数据库权威的 global/provider/model/run 治理、逐 HTTP attempt ledger、背压、公平 slice、typed audit/archive、Worker progress、固定 exporter/规则、历史延迟及 Mock 容量基线也已进入 Phase 2 工作树,`llmbenchlab-protocol-v1` 评分含义没有改变。部署仍不等于生产高可用;容量边界见 [`PERFORMANCE.md`](PERFORMANCE.md),告警、故障、retention 与 0005/0004 安全回滚见 [`OPERATIONS.md`](OPERATIONS.md)。 @@ -18,7 +18,7 @@ LLMBenchLab 当前有三条本地运行路径: | --- | --- | --- | --- | --- | | 本地 Make | `http://127.0.0.1:5173` | `http://127.0.0.1:8000` | `backend/data/llmbenchlab.db`;Redis 可选 | `make dev` 启动 API、独立 Worker、frontend;默认 loopback | | 可信本地 CLI | 不需要 | 不需要 | 当前 `DATABASE_URL`;`artifacts/` 中缓存、Benchmark ZIP 与报告 | `llmbenchlab-evaluate` 自己运行 Runner;同一数据库不得有竞争 Worker | -| Docker Compose | `http://127.0.0.1:8080` | `http://127.0.0.1:8000` | `postgres-data`、`redis-data` named volumes | API/frontend 仅 loopback;PostgreSQL/Redis 无 host port | +| Docker Compose | `http://127.0.0.1:8080` | `http://127.0.0.1:8000` | `postgres-data`、`redis-data` named volumes | `make dev-multi` 默认两个 Worker;API/frontend 仅 loopback;PostgreSQL/Redis 无 host port | API 系统端点: @@ -85,6 +85,16 @@ make frontend SQLite 本地路径只允许单 Worker。不要启动多个 SQLite Worker,也不要用 SQLite kill/restart 代替真实 PostgreSQL 并发验收。 +当 `.env` 的有效 `DATABASE_URL` 已指向迁移到 head 的 PostgreSQL 时,同一 Vite 开发入口可以管理多个 Worker 进程: + +```bash +make dev DEV_WORKERS=2 +``` + +`DEV_WORKERS` 映射到 launcher-only 的 `LLMBENCHLAB_DEV_WORKER_PROCESSES`,范围为 1–32。单 Worker继续写 `worker.log`;多 Worker分别写 `worker-1.log`、`worker-2.log` 等,并向 API 注入同值的 `LLMBENCHLAB_WORKER_EXPECTED_PROCESSES`。任一 Worker、API 或 frontend 退出时,启动器向同一会话所有剩余进程发送 TERM、逐一等待并传播原退出码。请求 `N>1` 而有效 DSN 不是 PostgreSQL 时,会在创建日志或启动子进程前失败,错误不会回显 DSN。 + +已有 SQLite 数据不能仅靠改 DSN 出现在 PostgreSQL。需要保留 Model、Benchmark 和 Run 时,必须先让 `pending/running`、active reservation 与 live Worker generation 通过受支持的取消/租约/对账流程收敛,随后停写并使用第 7 节的 SQLite→空 PostgreSQL importer;启动器不会自动迁移或双写。 + ### 3.3 直接运行子项目 统一 Make 命令应是首选。排障时可显式运行: @@ -149,7 +159,7 @@ uv run llmbenchlab-evaluate prepare \ 不要在 shell 命令中拼接真实值。CLI 没有 `--api-key` 参数,非交互且变量为空时会停止。 -使用兼容根地址的限题评测: +使用兼容根地址的限题评测;`--provider-type` 必须与目标模型实际使用的协议一致: ```bash cd backend @@ -157,21 +167,27 @@ uv run llmbenchlab-evaluate run \ --dataset mmlu-pro \ --profile official_cot \ --limit 20 \ + --provider-type openai_responses \ --base-url https://provider.example.invalid/v1 \ --model replace-with-provider-model-id \ --concurrency 1 ``` -Base URL 支持以下两种形式: +Base URL 支持协议根地址或与显式协议匹配的完整 endpoint: + +- `--provider-type openai_compatible`:生成 endpoint 为 `/chat/completions`; +- `--provider-type openai_responses`:生成 endpoint 为 `/responses`; +- `--provider-type anthropic_messages`:生成 endpoint 为 `/messages`。 + +例如根地址 `https://host/v1` 会把发现请求发到 `GET https://host/v1/models`,再按所选协议生成对应 endpoint;完整地址 `https://host/v1/chat/completions`、`https://host/v1/responses` 或 `https://host/v1/messages` 则原样用于生成,发现请求仍推导为同级 `/models`。完整 endpoint 与 `--provider-type` 不匹配时会在发送 Key 前拒绝;不会根据模型名/URL 猜测协议,也不会在失败后跨协议 fallback。 -- 兼容根地址 `https://host/v1`:发现请求为 `GET https://host/v1/models`,生成请求为 `POST https://host/v1/chat/completions`; -- 完整 Chat 端点 `https://host/v1/chat/completions`:生成请求使用原地址,发现请求仍为同级 `https://host/v1/models`。 +远端 Provider 必须使用 HTTPS;明文 HTTP 仅允许 `localhost` 或字面量 loopback IP,用于操作者控制的本地推理服务。模型发现按 `--provider-type` 鉴权:Chat/Responses 使用 `Authorization: Bearer`,Messages 使用 `x-api-key` 与 `anthropic-version`;Messages 的 `has_more/last_id` 通过 `after_id` 分页,并受累计 100 页、60 秒 wall-clock、2 MiB、10,000 项与缺失/重复 cursor 门禁保护。默认模型发现只在聚合后返回唯一模型时自动选择;返回多个模型必须提供 `--model`。任何模型 ID 若包含当前 Key,预检立即失败且错误不会回显该值。如果 `/models` 返回 404/405,只有显式提供模型名才继续。已知 Provider 不实现发现时可用 `--no-model-discovery --model ...`,但付费 canary 仍会执行。 -远端 Provider 必须使用 HTTPS;明文 HTTP 仅允许 `localhost` 或字面量 loopback IP,用于操作者控制的本地推理服务。默认模型发现只在返回唯一模型时自动选择;返回多个模型必须提供 `--model`。任何模型 ID 若包含当前 Key,预检立即失败且错误不会回显该值。如果 `/models` 返回 404/405,只有显式提供模型名才继续。已知 Provider 不实现发现时可用 `--no-model-discovery --model ...`,但付费 canary 仍会执行。 +CLI 在发出 canary 前显示 Provider host、协议、模型、计分题数、剩余 failed-attempt 预算和最多 Provider HTTP 尝试数,交互要求精确输入 `RUN`。剩余预算严格为 `max_attempts - failed_attempt_count`,不把 cooperative yield 算作失败。canary 必须可解析为预期答案;若成功体明确返回不同于目标的模型名也会失败。上界包含每个逻辑调用最多 3 次 HTTP attempts:`(缺失计分题数 × 剩余 failed-attempt 预算 + 1 个 canary) × 3`。`--yes` 仅用于操作者明确批准的非交互环境。 -CLI 在发出 canary 前显示 Provider host、模型、计分题数、剩余 Run attempts 和最多 Chat Completion HTTP 尝试数,交互要求精确输入 `RUN`。canary 必须可解析为预期答案;若成功体明确返回不同于目标的模型名也会失败。上界包含每个逻辑调用最多 3 次 HTTP attempts,以及新 Run 的全部或恢复 Run 的剩余 execution attempts:`(缺失计分题数 × 剩余 Run attempts + 1 个 canary) × 3`。`--yes` 仅用于操作者明确批准的非交互环境。MMLU `official_cot` 默认 `temperature=0/max_tokens=4000`;其他配置默认 `temperature=0/max_tokens=1024`,共同默认 `top_p=1/seed=42/concurrency=1`,都可由命令参数覆盖并冻结到 Run。 +MMLU `official_cot` 的输出上限默认 `max_tokens=4000`,其他配置默认 `max_tokens=1024`,共同默认 `concurrency=1`。Chat Completions 另默认 `temperature=0/top_p=1/seed=42`;Responses 与 Messages 在未显式传入或由 Model 默认提供时省略 `temperature`、`top_p` 和 `seed`,避免向不支持采样字段的模型发送它们。Responses/Messages 不接受非空 seed;Messages 的 `temperature` 上限为 `1`,且 `max_tokens` 必须始终为有限正整数。允许的命令参数会冻结到 Run 快照。 -模型发现与 Chat 请求固定声明 `Accept-Encoding: identity` 并拒绝其他响应编码;发现体最多读取 2 MiB。Chat 显式请求 SSE 与流式 usage,持续消费到 `[DONE]`;普通 JSON 成功体上限 4 MiB,SSE wire/单事件/聚合 content 上限分别为 64 MiB/1 MiB/4 MiB,非 2xx 错误体上限 64 KiB。Provider 返回证据会递归检查成功内容、raw usage 的对象键/全部 JSON 标量、request ID、返回模型名、system fingerprint 和 finish reason;SSE content 先完整聚合,当前 Key 的精确回显再于进入 Runner/快照/Response 边界前替换为 `[REDACTED]`。这些控制不扫描无关 Benchmark/Question 内容,也不替代 Provider 账单检查、内容访问控制或通用 DLP。 +模型发现与三类远程请求固定声明 `Accept-Encoding: identity` 并拒绝其他响应编码;discovery 聚合最多 2 MiB/10,000 个模型 ID,Messages 分页另限制累计 100 页/60 秒 wall-clock,并拒绝 cursor 循环或 `has_more=true` 时缺失 `last_id`。Chat、Responses、Messages 的 SSE 必须分别消费到 `[DONE]`、`response.completed`、`message_stop`;普通 JSON 成功体上限 4 MiB,SSE wire/单事件/聚合 content 上限分别为 64 MiB/1 MiB/4 MiB,非 2xx 错误体上限 64 KiB。Provider 返回证据会递归检查成功内容、raw usage 的对象键/全部 JSON 标量、request ID、返回模型名、system fingerprint 和 finish reason;SSE content 先完整聚合,当前 Key 的精确回显再于进入 Runner/快照/Response 边界前替换为 `[REDACTED]`。这些控制不扫描无关 Benchmark/Question 内容,也不替代 Provider 账单检查、内容访问控制或通用 DLP。 API URL 与 Key 已足以在 `/models` 只返回一个可用 ID 时选择模型;若返回多个模型,仍需提供 `--model`,避免猜测付费目标。`--input-price-per-million`/`--output-price-per-million` 是可选成本估算输入;不提供价格或 Provider usage 时报告成本为 unknown,而不是虚假 `0`。 @@ -184,6 +200,7 @@ cd backend uv run llmbenchlab-evaluate run \ --dataset gpqa-diamond \ --full \ + --provider-type openai_compatible \ --base-url https://provider.example.invalid/v1/chat/completions \ --model replace-with-provider-model-id \ --api-key-env MY_PROVIDER_API_KEY \ @@ -250,6 +267,7 @@ Pydantic 应用设置优先读取 `LLMBENCHLAB_*`,并为数据库、Redis、CO | `LLMBENCHLAB_WORKER_SHUTDOWN_GRACE_SECONDS` | `30` | SIGTERM 后等待活动 Run 的应用 grace | | `LLMBENCHLAB_WORKER_PROGRESS_FLUSH_SECONDS` | `5` | 真实 scan/claim/progress/heartbeat bit 合并写入 DB 的最大间隔;timer 不生成 keepalive | | `LLMBENCHLAB_WORKER_PROGRESS_STALE_SECONDS` | `60` | DB UTC 下 active generation 的 stale cutoff | +| `LLMBENCHLAB_DEV_WORKER_PROCESSES` | `1` | 仅 `scripts/dev.sh` 使用的本地 Worker 进程数,范围 1–32;大于 1 要求 PostgreSQL | | `LLMBENCHLAB_WORKER_EXPECTED_PROCESSES` | `1` | 部署明确声明的最小 live Worker 数;扩缩容必须同步修改 | | `LLMBENCHLAB_WORKER_RECOVERY_ALERT_SECONDS` | `60` | expired lease age 告警比较阈值 | | `LLMBENCHLAB_MOCK_GENERATION_DELAY_SECONDS` | `0` | 只用于确定性 Mock 故障测试;不改变报告 latency 或协议评分 | @@ -287,7 +305,8 @@ Pydantic 应用设置优先读取 `LLMBENCHLAB_*`,并为数据库、Redis、CO | `LLMBENCHLAB_COMPOSE_WORKER_SHUTDOWN_GRACE_SECONDS` | `30` | 应用 grace;容器另有 45 秒 stop grace | | `LLMBENCHLAB_COMPOSE_WORKER_PROGRESS_FLUSH_SECONDS` | `5` | 映射到 Worker progress flush | | `LLMBENCHLAB_COMPOSE_WORKER_PROGRESS_STALE_SECONDS` | `60` | 映射到 Worker stale cutoff | -| `LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES` | `1` | Compose 目标 Worker 数;`--scale` 时必须同步设置 | +| `LLMBENCHLAB_COMPOSE_WORKER_PROCESSES` | `2` | 标准 Compose 包装器的 Worker replica 数,范围 1–32;当前只有 1–2 经过容量资格 | +| `LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES` | `1`(直接 Compose) | API 的低层 expected 声明;标准包装器从 Worker 数自动设置,直接 `docker compose` 时仍须自行同步 | | `LLMBENCHLAB_COMPOSE_WORKER_RECOVERY_ALERT_SECONDS` | `60` | exporter lease recovery 告警阈值 | | `LLMBENCHLAB_COMPOSE_REDIS_BLOCK_MILLISECONDS` | `1000` | 映射到 Worker Redis blocking read 上限 | | `LLMBENCHLAB_COMPOSE_REDIS_OPERATION_TIMEOUT_SECONDS` | `1` | 容器 Redis 操作 timeout | @@ -298,7 +317,7 @@ Pydantic 应用设置优先读取 `LLMBENCHLAB_*`,并为数据库、Redis、CO | `LLMBENCHLAB_COMPOSE_MOCK_GENERATION_DELAY_SECONDS` | `0` | 可靠性测试专用 Mock delay | | `LLMBENCHLAB_COMPOSE_CREDENTIAL_KEYS_FILE` | `.secrets/credential-keys.json` | Compose 只读挂载给 API/Worker 的宿主 keyring | -### 4.4 OpenAI-compatible Key +### 4.4 远程 Provider Key Web 服务路径让用户在 Models 表单直接输入真实 Provider Key。API 将它作为 write-only 字段接收,用共享 keyring 做 AES-256-GCM 加密,并只把认证密文写入 `model_credentials`;Worker 读取同一 keyring 解密。API 不把凭据流中的 Key 或 Provider 回显复制进 Model GET/list、Run 的 model snapshot、队列和报告证据,也不返回密文、nonce 或 key id;这不排除无关用户数据的独立字面巧合。keyring 必须与数据库分开备份;数据库与 keyring 同时泄漏时,Provider Key 仍可被解密。 @@ -339,6 +358,7 @@ Redis 开启 AOF (`appendfsync everysec`) 只改善通知持久性,不是备 - `20260828_0005`:Worker progress facts 与 audit retention/exporter 有界扫描索引。 - `20260829_0006`:不改变逻辑 schema 的兼容修复;仅为早期 `0004` 历史变体补齐三个 canonical governance 索引。 - `20260830_0007`:不改变 schema、ledger 或 Provider actual usage 的数据修复;仅按显式 input/output hard reservation 与由完整上界和价格派生的 reserved cost 语义重算 `governance_scopes.overdrawn`。 +- `20260830_0008`:将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,同时替换 Provider 类型 check 与远程配置 check,加入显式 `openai_responses` / `anthropic_messages` Adapter;旧 Model 不改写,存在新类型 Model 时 downgrade 先拒绝。 本地 SQLite 更新: @@ -346,7 +366,7 @@ Redis 开启 AOF (`appendfsync everysec`) 只改善通知持久性,不是备 make migrate ``` -命令先执行 `app.db.prepare_migrations`,再 `alembic upgrade head`。受支持的未版本化 SQLite 会在严格结构/integrity/FK 检查和一致性备份后 stamp;与 current metadata 一致的未版本化库只 stamp 到 `0006`,确保 data-only `0007` 仍实际执行而不会被跳过。已知早期 `0004/0005` 变体只有在 revision fingerprint 为 canonical,或仅缺这三个已知索引的非空子集且最多一条 active policy 时,才会备份并交给 `0006` 修复,因此修复 DDL 中断后可安全重入。新近成为 historical 的 PostgreSQL `0005/0006` 按各自规则先做 metadata drift 校验;未知 drift 在写 revision 或 repair DDL 前拒绝。`0007` 在更新 materialized flag 前拒绝任何 `reserved/send_started` reservation。普通 API/Worker 启动只检查 head,不运行 `create_all`、preflight 或 upgrade。 +命令先执行 `app.db.prepare_migrations`,再 `alembic upgrade head`。受支持的未版本化 SQLite 会在严格结构/integrity/FK 检查和一致性备份后 stamp;即使结构已与 current metadata 一致,也只 stamp 到 `0006`,确保 data-only `0007` 仍实际执行,然后再由 `0008` 扩展 `provider_type` 列宽并替换两个 Provider check。已知早期 `0004/0005` 变体只有在 revision fingerprint 为 canonical,或仅缺这三个已知索引的非空子集且最多一条 active policy 时,才会备份并交给 `0006` 修复,因此修复 DDL 中断后可安全重入。新近成为 historical 的 PostgreSQL `0005/0006/0007` 按各自规则先做 metadata drift 校验;未知 drift 在写 revision 或 repair DDL 前拒绝。`0007` 在更新 materialized flag 前拒绝任何 `reserved/send_started` reservation。普通 API/Worker 启动只检查 head,不运行 `create_all`、preflight 或 upgrade。 Compose 中只有一次性 `migrate` 服务执行: @@ -356,7 +376,7 @@ python -m app.db.prepare_migrations && alembic upgrade head && alembic check `api` 与 `worker` 必须等待 migrate exit 0,然后仅执行 head check。不要把 Alembic 命令加回 API/Worker entrypoint,也不要同时运行多个 migration owner。 -`0007 -> 0006` 只按旧语义重算 `governance_scopes.overdrawn`,保留 reservation、actual usage、Response、audit 与 Run 终态;若有 active reservation 会在更新前拒绝。`0006 -> 0005` 是 no-op downgrade,因为三个索引本来就属于 canonical `0004`;`0005 -> 0004` 在 `worker_processes` 有任意 generation fact 时于第一条 DDL 前拒绝;`0004 -> 0003` 在任何 policy/scope/bucket/question-execution/ledger/audit 行或新 Run/Response 证据存在时同样拒绝。正常使用后的数据库不能把它们当普通代码回滚。`0003 -> 0002` 只要 `model_credentials` 存在任意行也会拒绝,避免静默丢失 Provider Key。`0002 -> 0001` 在发现 `pending` 或 `running` Run 时拒绝;它会删除可靠性元数据但保留核心实体与协议证据。完整 0007/0006/0005/0004 回滚流程见 [OPERATIONS.md](OPERATIONS.md);schema downgrade 不是 PostgreSQL→SQLite 反向同步,也不恢复 Phase 1 进程内 Runner。 +`0008 -> 0007` 会把 `models.provider_type` 从 `VARCHAR(18)` 收回 `VARCHAR(17)`,并恢复旧 Provider 类型 check 与远程配置 check;若存在 `openai_responses` 或 `anthropic_messages` Model,它会在第一条 DDL 前拒绝。`0007 -> 0006` 只按旧语义重算 `governance_scopes.overdrawn`,保留 reservation、actual usage、Response、audit 与 Run 终态;若有 active reservation 会在更新前拒绝。`0006 -> 0005` 是 no-op downgrade,因为三个索引本来就属于 canonical `0004`;`0005 -> 0004` 在 `worker_processes` 有任意 generation fact 时于第一条 DDL 前拒绝;`0004 -> 0003` 在任何 policy/scope/bucket/question-execution/ledger/audit 行或新 Run/Response 证据存在时同样拒绝。正常使用后的数据库不能把它们当普通代码回滚。`0003 -> 0002` 只要 `model_credentials` 存在任意行也会拒绝,避免静默丢失 Provider Key。`0002 -> 0001` 在发现 `pending` 或 `running` Run 时拒绝;它会删除可靠性元数据但保留核心实体与协议证据。完整 0008/0007/0006/0005/0004 回滚流程见 [OPERATIONS.md](OPERATIONS.md);schema downgrade 不是 PostgreSQL→SQLite 反向同步,也不恢复 Phase 1 进程内 Runner。 ### 5.3 备份与恢复证据边界 @@ -374,16 +394,20 @@ Compose 定义六个 service,其中五个常驻,`migrate` 为一次性任务 | `redis` | Redis 7 Streams 通知层,AOF everysec | `redis-cli ping`;`redis-data`;无 host port | | `migrate` | 唯一 Alembic preflight/upgrade/check owner | 等 PostgreSQL healthy;成功后 exit 0,不常驻 | | `api` | FastAPI CRUD、write-only Key 加密、Run commit 与 best-effort publish | 等 migrate 成功;只读挂载 keyring;启动只 head check;`/ready` 为容器 health;loopback API port | -| `worker` | 独立租约 Worker、stored Key 解密、DB reconciliation、Redis consume/ACK、DB-time process progress | 等 migrate 成功;只读挂载同一 keyring;启动只 head check;dependency-only probe;容器 stop grace 45 秒 | +| `worker` | 可横向复制的独立租约 Worker、stored Key 解密、DB reconciliation、Redis consume/ACK、DB-time process progress | 等 migrate 成功;只读挂载同一 keyring;启动只 head check;dependency-only probe;容器 stop grace 45 秒 | | `frontend` | Nginx 静态站与 `/api/` 同源代理 | 等 API healthy;loopback frontend port;关闭 API request buffering,避免 Key body 被 Nginx 临时落盘 | ### 6.1 启动、检查和停止 ```bash make docker-up +# 等价的产品化名称 +make dev-multi +# 显式副本数 +make docker-up WORKERS=2 ``` -该命令执行 `docker compose up --build --wait --wait-timeout 180 --remove-orphans`。检查: +标准包装器校验副本数后,用同一个值导出 `LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES` 并分阶段执行 build、依赖/migrate、Worker、API、frontend。它以全部 Compose replica(包括 exited)判断是否缩容,并另算 running replica 数;扩容/重启时先增加 Worker,要求恰好 `N` 个 fresh active generation 已有真实数据库 scan,且至少 `N-running` 个 generation 在本轮 DB-time watermark 之后启动并完成 scan,再强制重建 API 提高 expected。缩容时先重建 API 降低 expected,再 graceful scale Worker。默认 `N=2`。最后从 API container 有界轮询 `/api/v1/tasks/metrics`,只有 `worker_expected_processes=N`、`worker_registered_processes=N`、`worker_live_processes=N`、`worker_stalled_processes=0`、`worker_shortfall_processes=0` 才返回成功;scan 或 gauges 超时会保留当前 stack 供诊断,不自动删除 volume 或容器。检查: ```bash docker compose ps -a @@ -616,10 +640,10 @@ operational/security audit 分别至少保留 90/365 天,清理不在请求链 | 领域 | 当前可靠执行基础 | 生产前仍需 | | --- | --- | --- | | 身份与权限 | 无鉴权,所有端点可读写 | 登录/API Token、RBAC、对象授权、管理员导入/Model 权限、审计 | -| 网络 | API/frontend loopback;PG/Redis 内部 Compose 网络;Provider Chat 真 SSE 客户端 | TLS 反向代理、可信 Host/代理、Worker→Provider 全链路 SSE flush/buffering/timeout 核对、网络策略、认证与安全 headers | +| 网络 | API/frontend loopback;PG/Redis 内部 Compose 网络;Provider Chat/Responses/Messages 真 SSE 客户端 | TLS 反向代理、可信 Host/代理、Worker→Provider 全链路 SSE flush/buffering/timeout 核对、网络策略、认证与安全 headers | | PostgreSQL | 单实例、named volume、迁移/故障测试 | 托管/HA、TLS、最小权限角色、加密、备份/PITR、RPO/RTO 与真实恢复演练 | | Redis | 单实例 AOF、非权威通知层 | 认证/TLS、HA/容量/保留策略、监控;继续保持 DB 事实来源 | -| Worker | 租约/心跳/fencing/重试/取消;数据库权威四层治理、attempt ledger、背压/公平 slice;DB-time scan/claim/progress/heartbeat aggregate;真实双 Worker Mock 基线 | 滚动排空自动化、更高 Worker 数/真实 Provider 的独立容量与成本规划 | +| Worker | 租约/心跳/fencing/重试/取消;数据库权威四层治理、attempt ledger、背压/公平 slice;DB-time scan/claim/progress/heartbeat aggregate;标准 Compose 默认双 Worker且按方向扩缩并校验 expected/registered/live/stalled/shortfall;真实双 Worker Mock 基线 | 滚动排空自动化、三个以上 Worker/真实 Provider 的独立容量与成本规划 | | Secrets | Web write-only;AES-GCM 密文入库;API/Worker 共享独立 keyring;legacy env 可用 | 身份/对象授权、KMS/HSM、短期凭据、批量重加密、轮换审计与每进程最小权限 | | SSRF/数据外发 | 远端 HTTPS、HTTP 仅 loopback、禁重定向与有界正文 | allowlist、DNS/IP 验证、出站代理、元数据阻断、外发审批 | | 可观测性 | 受控 JSON 日志源、组件健康、DB gauges、typed audit/history、固定低基数 Prometheus exporter、8 条规则、Worker progress、逐题安全 Provider metadata | 受认证 Dashboard、统一 traces、告警发送/值班集成、生产 SLO、数据库管理员级不可篡改审计 | diff --git a/docs/NEXT_TASK.md b/docs/NEXT_TASK.md index 08d3fc1..f3bd8e9 100644 --- a/docs/NEXT_TASK.md +++ b/docs/NEXT_TASK.md @@ -1,10 +1,11 @@ # 下一任务(当前修复完成后):实施 P2-07 最小恢复验证切片 > 状态:`planned`;P2-07 独立计划、工作日志与 ADR-0016 已建立,功能实现尚未开始 +> 当前前置:P3-06 Run Detail 热力图/live metrics 已完成实现、普通 push 与精确 SHA CI,状态为 `completed`;P2-07 恢复为下一项但仍是 `planned`,本文件不表示其已经开始 > 对应阶段:[Phase 2 — Reliability](phases/PHASE-2-RELIABILITY.md) > 当前计划:[Phase 2 恢复与运维闭环](plans/2026-08-28-phase-2-recovery-operations.md) > 当前日志:[2026-08-28 P2-07 工作日志](worklogs/2026-08-28-phase-2-recovery-operations.md) -> 当前决策:[ADR-0016](decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md),exact-head amendments [ADR-0017](decisions/ADR-0017-schema-equivalent-governance-index-repair.md) / [ADR-0018](decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md) +> 当前决策:[ADR-0016](decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md),exact-head amendments [ADR-0017](decisions/ADR-0017-schema-equivalent-governance-index-repair.md) / [ADR-0018](decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md) / [ADR-0019](decisions/ADR-0019-explicit-provider-api-protocol-adapters.md) > 决策基础:[ADR-0005](decisions/ADR-0005-durable-task-execution.md)、[ADR-0009](decisions/ADR-0009-database-governance-audit-fair-scheduling.md)、[ADR-0010](decisions/ADR-0010-phase-2-governance-delivery-boundaries.md)、[ADR-0011](decisions/ADR-0011-confirmed-pre-send-release-retry-generation.md)、[ADR-0015](decisions/ADR-0015-observability-worker-progress-audit-retention.md) ## 当前事实 @@ -13,7 +14,7 @@ P2-01 已完成,不再重复资格或“重跑碰绿”。P2-06 也已完成 当前 P2-06 已按 [ADR-0015](decisions/ADR-0015-observability-worker-progress-audit-retention.md) 实现: -- P2-06 的 `20260828_0005` 新增 `worker_processes` 和 bounded audit scan indexes;`20260829_0006` 仅修复早期 `0004` 三索引缺口。当前应用 head `20260830_0007` 只按显式 hard reservation 语义重算 scope overdrawn,不改变 13 表 schema/importer/archive 合同、never-delete ledger 或 actual usage。SQLite→PostgreSQL importer 继续做 13 表精确 digest,拒绝 live generation 并复制 stopped/stale facts;表非空时 `0005 -> 0004` 在 DDL 前拒绝。 +- P2-06 的 `20260828_0005` 新增 `worker_processes` 和 bounded audit scan indexes;`20260829_0006` 仅修复早期 `0004` 三索引缺口,`20260830_0007` 只按显式 hard reservation 语义重算 scope overdrawn。当前应用 head `20260830_0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并替换 Provider 类型 check 与远程配置 check;它不改写旧 Model,也不改变 13 表/importer/archive event 合同、never-delete ledger 或 actual usage。SQLite→PostgreSQL importer 继续做 13 表精确 digest,拒绝 live generation 并复制 stopped/stale facts;表非空时 `0005 -> 0004` 在 DDL 前拒绝。 - 长运行 Worker 注册唯一 generation,只在真实 scan/claim/lease-heartbeat/progress 后按 DB UTC 合并刷新;`/tasks/metrics` 和 exporter 只公开 expected/registered/live/stalled/shortfall 与聚合时间。dependency probe 固定声明不检查 main-loop progress。 - `GET /api/v1/metrics/prometheus` 输出固定 Prometheus text `0.0.4` gauge,使用一个 DB-time 快照、有界 15 分钟 audit 与 1 小时 latency 窗口、固定 enum label、整次 fail-closed 和每 API 进程 single-flight。 - `deploy/observability/` 提供固定八条告警规则、抓取示例与对应 Operations Runbook;仓库不部署 Prometheus、Alertmanager、OTel 或通知发送器。 @@ -26,10 +27,43 @@ P2-01 已完成,不再重复资格或“重跑碰绿”。P2-06 也已完成 同日又把 `artifacts/benchmarks/` 中已存在的 GPQA-Diamond 与两种 MMLU-Pro profile 通过正式导入 API 加载到默认个人 SQLite,当前为 `4` 个 Benchmark、`24,277` 题,原 Model/Run/Response 保持且没有 Provider 调用。记录 commit `0163b67c00eb59ae59db5f3adb679ad85c799142` 已 push,其精确 SHA run `33266167547` 4/4 成功;该本地数据操作不提交第三方题目、不改变产品实现或路线图,P2-07 的下一任务仍保持不变。 +随后又从固定来源 revision 准备并通过同一正式导入 API 加载 GSM8K、中文 MGSM、HellaSwag、WinoGrande、TruthfulQA Binary 五套 100 题 mini 子集,以及完整 100 题中文 XCOPA validation。默认个人 SQLite 因此现为 `10` 个 Benchmarks、`24,877` 题;第三方题目、转换器、provenance 与导入前备份仍在 Git 忽略目录。导入任务没有创建、取消或重置 Run;当时既有 12,032 题 Run 正在执行,导入完成后的取消请求和 MGSM mini Run 创建属于另一个并发客户端时间线,其 Response/Provider 流量不属于导入副作用。记录 commit `8faa2093b2c3308994d50e42a31063cdbf5264a6` 已 push,精确 SHA CI run `33296049611` 4/4 成功。该个人数据维护不实现 Plugin SDK、IFEval 或新的产品合同,P2-07 仍是下一项且保持 `planned`。 + [Observational Token overdraw 修复](plans/2026-08-30-fix-observational-token-overdraw.md) 已完成。ADR-0018 已把非显式 input 估算与 hard reservation 分离,并把 data-only head 定为 `20260830_0007`;该 revision 只重算 `governance_scopes.overdrawn`,upgrade/downgrade 均拒绝 active reservation,历史 ledger/actual/Response/Run 终态保持。本地完整验证、当前个人 SQLite 迁移、修正 SHA `cb00924…` 的 real-Compose 9/9 与 [exact-SHA CI run `33271095910`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33271095910) 4/4 均通过。P2-07 前置阻碍已解除,但本次任务到此停止。 +[Run Detail 错题与部分 Token 展示修复](plans/2026-08-30-fix-run-detail-metrics.md) 已完成:它只扩充 Responses 读取 API 和页面证据表达,不改变 protocol-v1 精确 Token、成绩、历史数据、migration 或 P2-07 范围。实现 SHA `0003e429…` 已普通 push,[PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5) 的精确 SHA [CI run `33286730109`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33286730109) 4/4 成功;下一独立切片仍是本文件定义的 P2-07 最小只读 recovery verifier。 + +[Run Detail 热力图与实时指标](plans/2026-08-30-run-progress-heatmap-live-metrics.md) 是已完成的 P3-06 切片。它冻结为固定 `512` 题 absolute-position block index/payload 和后端同快照 evidence-derived live metrics;不改变 `/api/v1`、`llmbenchlab-protocol-v1`、Run 精确 nullable Token/cost、数据库 schema 或 P2-07 范围。初版 cursor 的 4 个 red tests 因没有数据库单调提交序列已在实现前废弃。本地 fixed-block 验证为 backend/frontend target `37/32 passed`(Run Detail `20` + heatmap `12`),完整 backend `964 passed, 33 skipped`、frontend `64 passed`;lint、Mock smoke、build、Compose config 与目标 198 题 Run 的三档宽度/键盘/Tooltip/console 实页验收均通过,12,032/20,000 题仅为自动化虚拟化边界。实现 SHA [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) 已普通 push 到 `codex/complete-evaluation-workflow` 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5),精确 SHA [Actions run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) 的 backend、backend-integration、full-stack-reliability、frontend 四个 job 全部成功。执行入口现回到下面的 P2-07 verifier;P2-07 继续为 `planned`。 + P2-06 本地与 clean evidence 数值保持记录不变:合并定向、lint/test/smoke、双方言 migration、真实 PostgreSQL/Redis integration、frontend build、Compose config、Prometheus 规则、clean capacity/acceptance 与技术/安全终审均已通过;原始 evidence 仍不得公开。P2-06 当时未擅自迁移默认用户 SQLite;随后的兼容修复已在自动备份后将当时重建库前进到 `0006` 并通过 startup/check。 +## 已完成:日常多 Worker评测入口维护 + +本轮只把既有 PostgreSQL lease/fencing 能力暴露到日常启动入口,不改 P2-07 范围。 +`make dev DEV_WORKERS=N` 管理本地 PostgreSQL Worker进程,`make dev-multi` / +`make docker-up WORKERS=N` 默认两个 Compose Worker并按扩/缩方向切换 API expected, +最终校验 expected/registered/live/stalled/shortfall; +SQLite 多 Worker在服务启动前拒绝。离线启动器 `42 passed`,迁移到 head 的隔离 +PostgreSQL 16 跨 Benchmark/唯一 lease 回归 `2 passed`,隔离 Compose 的 `2→1→2` +扩缩与五 gauges/cleanup 通过;完整 lint/test/smoke/build/config 也已通过。实现 SHA +[`b06594c2df67d6e2a8b117651b193cd0fa409bf5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/b06594c2df67d6e2a8b117651b193cd0fa409bf5) +已普通 push,其 exact-SHA Actions [run `33299883513`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33299883513) +四个必需 job 全绿,因此本维护为 `completed`。本文件的下一独立任务仍是下面的 P2-07 +最小只读 verifier。 + +## 已完成:Provider API 三协议适配 + +[ADR-0019](decisions/ADR-0019-explicit-provider-api-protocol-adapters.md) 与 +[独立计划](plans/2026-08-30-provider-api-protocols.md) 已把旧 `openai_compatible` +保留为 Chat Completions,并新增显式 `openai_responses` / `anthropic_messages`。 +Adapter、Model/API/Run snapshot、Worker/CLI preflight、`0008` migration 与 Web 表单已实现; +本地 backend `1079 passed, 36 skipped`、frontend `72 passed`、lint、Mock smoke、build、 +Compose config 和隔离 PostgreSQL 16 migration 门禁均通过,未调用真实 Provider。 +完整本地门禁与隔离 PostgreSQL 16 迁移验证均通过;实现 SHA +[`6943aa29a154c82bdfbe5efb2578c916c3cbf632`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/6943aa29a154c82bdfbe5efb2578c916c3cbf632) +已普通 push,[exact-SHA Actions run `33304667092`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33304667092) +四个必需 job 全部成功。该维护现为 `completed`;下一独立任务回到下面的 P2-07 最小只读 verifier。 + ## 已完成:建立 P2-07 工作包 1. 已按 AGENTS/PLANS 新建独立执行计划与工作日志,记录目标、非目标、验收、风险和实施步骤。 @@ -41,7 +75,7 @@ P2-06 本地与 clean evidence 数值保持记录不变:合并定向、lint/te P2-07 当前为 `planned`、尚未实现。恢复实施时按当前计划从最小只读 verifier 切片开始,之后才依次完成: -- PostgreSQL backup → 空目标 restore → Alembic `20260830_0007` exact head → 13 表 count/PK/content fingerprint → managed Run/ledger/audit/Worker stopped-or-stale facts 可读。 +- PostgreSQL backup → 空目标 restore → Alembic `20260830_0008` exact head → 13 表 count/PK/content fingerprint → managed Run/ledger/audit/Worker stopped-or-stale facts 可读。 - 数据库与数据库外 keyring 独立备份/恢复:匹配 keyring 能解密,缺失/错误 keyring fail closed;日志/证据不得回显 Key 或 envelope。 - Redis 重建/consumer group 恢复、Worker 扩缩、八条告警响应、dead-letter、commit outcome unknown、governance integrity 与 remaining cancel/retry/lease/budget crash matrix 的真实 PostgreSQL/Redis 演练。 - audit archive 作为数据库恢复后的精确校验/补回工具参与演练,但不得把 archive 自身 restore 冒充整库、PITR、RPO/RTO 或 WORM 认证。 @@ -57,11 +91,13 @@ P2-07 当前为 `planned`、尚未实现。恢复实施时按当前计划从最 ## Definition of Done - P2-06 已完成,不再重复其资格或证据门禁。 +- P3-06 热力图/live metrics 的 fixed-block 目标回归、完整本地门禁、目标 Run 浏览器验收、12,032/20,000 自动化虚拟化边界、实现 commit/push 与精确 SHA CI 已全部完成;P2-07 现在是下一项但仍须在实际实施开始时才标成 `in_progress`。 +- 日常多 Worker入口维护的完整本地门禁、实现 commit/push 和 exact-SHA CI 已全绿;这不等于 3+ Worker、真实 Provider、HA 或 SLA 资格。 - P2-07:backup/restore、Redis 重建、告警处置与约定故障矩阵必须有隔离真实 PostgreSQL/Redis、Mock-only Compose、秘密审查、独立 commit/push 和精确 SHA CI 证据,才能标记 completed。 - Phase 2:只有 P2-07 也完成后才可评估 `completed`;在此之前保持 `in_progress`,不得宣称生产 HA、灾难恢复 SLA、无限横向扩展、WORM 或 Provider exactly-once。 ## 可直接复制给 Codex 的任务指令 ```text -确认 observational Token overdraw 修复已完成本地/远程门禁且当前应用 head 为 20260830_0007 后,再继续执行 docs/NEXT_TASK.md。P2-07 的 ADR-0016、独立计划和工作日志已建立,不要重复设计或扩大范围。先只实现计划步骤 2 的最小只读 recovery verifier 及其目标测试;完成、复核并记录后再决定是否进入 Redis/Worker 或 rules/harness。自动化只用 Mock/Stub,所有 destructive 操作只针对隔离、精确目标;Phase 2 在 P2-07 完成前保持 in_progress。 +P3-06 Run Detail 热力图/live metrics 已按固定 512 题 block 合同完成本地/远程门禁并标为 completed。继续执行 docs/NEXT_TASK.md:P2-07 的 ADR-0016、独立计划和工作日志已建立,不要重复设计或扩大范围。先只实现计划步骤 2 的最小只读 recovery verifier 及其目标测试;完成、复核并记录后再决定是否进入 Redis/Worker 或 rules/harness。自动化只用 Mock/Stub,所有 destructive 操作只针对隔离、精确目标;Phase 2 在 P2-07 完成前保持 in_progress。 ``` diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md index ed8cd0f..c7276ea 100644 --- a/docs/OPERATIONS.md +++ b/docs/OPERATIONS.md @@ -253,21 +253,46 @@ Run 终态、defer 或 exhaust 转换会先在短事务中提交,再做 lease - 增加 Worker 不会自动提高治理 limit;也不能消除 Provider fixed-minute 或 lifetime budget。 - 当前实测只覆盖最多 2 个 Worker;更高数量必须重新测量。 -Compose 扩到两个 Worker: +标准 Compose 入口默认启动两个 Worker,并在返回成功前核对 API 暴露的 +`expected/registered/live/stalled/shortfall=2/2/2/0/0`: ```bash -LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES=2 \ - docker compose up -d --no-deps --scale worker=2 worker +make dev-multi +# 等价入口,也可显式声明规模 +make docker-up WORKERS=2 ``` -缩到一个 Worker: +缩到一个 Worker时也必须走同一包装器,让 API expected 声明与实际 scale +一起变化: ```bash -LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES=1 \ - docker compose up -d --no-deps --scale worker=1 worker +make docker-up WORKERS=1 ``` -`LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES` 是部署声明,不从历史进程行猜测;规模与 expected 不一致会产生有意的 shortfall。缩容依赖 SIGTERM grace。先观察活动 Run,等待被停止 Worker 排空;若 grace 耗尽,未完成 lease 留到数据库自然过期,由 peer 以递增 token 接管。缩容后检查 Worker health、DB-time registered/live/stalled、`running/expired_running/retry_scheduled`、active reservations 和 Response 唯一性。 +`WORKERS` 映射到 launcher-only 的 +`LLMBENCHLAB_COMPOSE_WORKER_PROCESSES`,包装器再用同一值设置低层 +`LLMBENCHLAB_COMPOSE_WORKER_EXPECTED_PROCESSES` 和 `--scale worker=N`。包装器把 exited +replica 也纳入扩缩方向判断;扩容/重启先启动 Worker,要求 fresh active generation 数 +精确为 `N`,并用应用数据库时钟 watermark 证明至少 `N-running` 个新 generation 已完成 +本轮 scan,随后才重建 API 提高 expected;缩容先重建 API 降低 expected,再让 Compose +graceful scale Worker。 +不要只对 `worker` service 执行 `--no-deps --scale`:那不会重建 API,旧 expected +会与实际规模漂移。直接使用低层 `docker compose` 时,操作者必须自行同步两者并核对 +gauges;标准包装器最多轮询 30 秒,超时非零且保留栈供诊断。 + +连接已经迁移到 head 的 PostgreSQL 时,本地 API/Vite 开发会话也可启动多个独立 +Worker 进程: + +```bash +make dev DEV_WORKERS=2 +``` + +每个 Worker 同时持有一个 Run,Run 内仍可有 1–4 个 Provider 请求并发。因此未被 +governance 限制时,潜在外发并发上界约为 `Worker 数 × Run concurrency`;增加 +Worker 会放大 Provider 限流和费用风险。一个长时间 SSE/HTTP 请求只占住其所在 +Worker,peer 仍能领取其他数据集的 Run;若所有 Worker 都被长请求占满,队列仍会等待。 + +缩容依赖 SIGTERM grace。先观察活动 Run,等待被停止 Worker 排空;若 grace 耗尽,未完成 lease 留到数据库自然过期,由 peer 以递增 token 接管。缩容后检查 Worker health、DB-time registered/live/stalled、`running/expired_running/retry_scheduled`、active reservations 和 Response 唯一性。 ### 6.2 Worker crash/lease expiry @@ -372,25 +397,29 @@ llmbenchlab-audit-retention restore \ 退出码 `0` 表示已确认成功,`2` 表示在提交前安全失败,`3` 表示提交已成功但后验核验失败,`4` 表示 commit outcome unknown。遇到 `3` 或 `4` 不得盲目重跑 mutation:保留 archive/count/digest,先执行只读 `reconcile` 判定数据库与 archive 的精确关系,再由操作者决定恢复或删除。Archive 与数据库/keyring 备份是不同资产;archive 不含 credential ciphertext/nonce/keyring,也不能单独恢复 stored Provider Key。 -## 10. 0007、0006、0005 与 0004 安全回滚 +## 10. 0008、0007、0006、0005 与 0004 安全回滚 + +### 10.1 Provider API 协议约束 0008 + +当前 head `20260830_0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并同时替换 Provider 类型 check 与远程配置 check,使 `openai_responses`、`anthropic_messages` 成为显式 Adapter 值;既有 `mock`/`openai_compatible` 行不改写,13 表、Run/Response、ledger、audit 与 archive-v1 字段不变。`0008 -> 0007` 会先锁定 Model 表并检查新类型;只要存在任一新协议 Model 就在第一条 DDL 前拒绝,否则恢复 `VARCHAR(17)` 与两个旧 check。安全回退必须先停止 API/Worker/CLI writer,确认无 active Run,并在 0008 应用中显式删除或转换这些配置;不要直接改 check、删历史 Run 或绕过凭据/active-Run 门禁。 -### 10.1 Observational overdraw 数据修复 0007 +### 10.2 Observational overdraw 数据修复 0007 -当前 head `20260830_0007` 不新增表、字段或索引,也不修改 never-delete reservation、Provider actual usage、Response、audit 或 Run 终态。它只关联 managed reservation 对应的 `evaluation_runs.input_token_reservation`,按以下规则重算每个 `governance_scopes.overdrawn`:input/cost 维度只有在 Run 冻结了显式 input reservation 时参与;显式 `max_tokens` 形成的 output reservation 独立参与;没有关联 Run 的内部 synthetic reservation 继续把调用者提供的值视为显式。 +`20260830_0007` 不新增表、字段或索引,也不修改 never-delete reservation、Provider actual usage、Response、audit 或 Run 终态。它只关联 managed reservation 对应的 `evaluation_runs.input_token_reservation`,按以下规则重算每个 `governance_scopes.overdrawn`:input/cost 维度只有在 Run 冻结了显式 input reservation 时参与;显式 `max_tokens` 形成的 output reservation 独立参与;没有关联 Run 的内部 synthetic reservation 继续把调用者提供的值视为显式。 `0006 -> 0007` 和 `0007 -> 0006` 都会在任何 materialized 更新前拒绝仍为 `reserved` 或 `send_started` 的 attempt。标准维护流程是停止 API/Worker/CLI writer,确认 active reservation 为零并创建一致性备份,再由唯一 migration owner 执行 `make migrate`。直接 `UPDATE governance_scopes`、删除 ledger 或裁剪 actual usage 均不受支持。降级到 `0006` 只按旧谓词重算 flag,让旧应用看到与旧 runtime 一致的 projection;它不会恢复旧代码、删除事实或把历史失败 Run 改成成功。 -### 10.2 Governance index compatibility repair 0006 +### 10.3 Governance index compatibility repair 0006 revision `20260829_0006` 不增加新的业务字段或表,只为曾执行早期 `0004` 变体的数据库条件补齐 `ix_evaluation_runs_started_at_id`、`ix_evaluation_runs_finished_at_id` 与 `uq_governance_policies_single_active`。`make migrate` 只在 revision fingerprint 为 canonical,或缺失项是这三个索引的非空子集时放行,并先创建 SQLite 一致性备份;这允许在 SQLite repair DDL 部分完成但 marker 仍为 `0005` 时安全重入。PostgreSQL `0005` 也必须通过 metadata drift 校验。若有多条 active policy、错误的同名索引或任何额外 drift,必须人工核对,工具不会自动停用 policy。`0006 -> 0005` 保留这三个 canonical `0004` 索引,因此是有意的 no-op downgrade。 -### 10.3 Worker progress / retention revision 0005 +### 10.4 Worker progress / retention revision 0005 revision `20260828_0005` 增加 `worker_processes` 和 audit retention/exporter 扫描索引。`0005 -> 0004` 在第一条 DDL 前检查 Worker process facts;只要表中存在任何 generation 行就拒绝,以免静默丢失注册、进展或 graceful-stop 证据。安全降级必须先停止 Worker、归档需要保留的 process facts,并由明确的数据生命周期审批清空;正常代码回滚应保留 0005 schema 并优先向前修复。 正式迁移验收同时保留两层独立门禁:populated 0005 数据库拒绝跨过 Worker progress revision且 revision/核心事实不变;隔离空库允许 `0005 -> 0004 -> 0005`。这不会替代下节已有的 populated 0004 governance 拒绝与空库 `0004 -> 0003 -> 0004` 往返。 -### 10.4 Governance/audit revision 0004 +### 10.5 Governance/audit revision 0004 revision `20260827_0004` 增加 policy、scope、minute bucket、question execution、Provider attempt ledger、typed audit、Run governance/fairness 字段及 Response Provider metadata。downgrade guard 在第一条 DDL 前检查数据损失风险。 diff --git a/docs/PROJECT_STATUS.md b/docs/PROJECT_STATUS.md index 7239f6e..f166990 100644 --- a/docs/PROJECT_STATUS.md +++ b/docs/PROJECT_STATUS.md @@ -7,7 +7,7 @@ - Phase 0 — 项目治理和架构:`completed`(2026-08-24) - Phase 1 — MVP 垂直链路:`completed`(2026-08-25) - Phase 2 — 可靠性与任务执行:`in_progress`(可靠基础、治理/审计、P2-01 单机资格与 P2-06 已完整交付;P2-07 工作包已建立,状态为 `planned`,功能尚未实现) -- Phase 3 — 标准 Benchmark 与代码评测:`in_progress`(仅可信本地 MMLU-Pro/GPQA-Diamond 客观题提前切片) +- Phase 3 — 标准 Benchmark 与代码评测:`in_progress`(可信本地 MMLU-Pro/GPQA-Diamond 与 P3-06 Run Detail 热力图/live metrics 切片已交付;IFEval、沙箱与完整插件体系仍未完成) - Phase 4–6:`planned` ## 当前版本与远程边界 @@ -16,14 +16,42 @@ 公开仓库:[`CWNU-Open-Source-Community/LLMBenchLab`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab),当前开发分支为 `codex/complete-evaluation-workflow`。P2-06 实现 SHA [`9a20676dcf545040782f04c166205d0043345753`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/9a20676dcf545040782f04c166205d0043345753) 已普通 push 并进入 [PR #3](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/3),其精确 SHA 的 GitHub Actions [run `33164609388`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33164609388) 四个必需 job 全部成功;绑定该 clean SHA 的 capacity 与 9/9 acceptance 也已通过。Evidence closeout 文档 commit [`ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6) 已 push,其精确 SHA 的 GitHub Actions [run `33165775037`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33165775037) 四个必需 job 全部成功,因此 P2-06 已完成仓库级收尾并标记为 `completed`。[ADR-0017](decisions/ADR-0017-schema-equivalent-governance-index-repair.md) / `20260829_0006` 数据库兼容修复实现 SHA [`8fb51b690ae6335b8ef93b3cbe54e039781fb173`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/8fb51b690ae6335b8ef93b3cbe54e039781fb173) 已普通 push,其精确 SHA 的 GitHub Actions [run `33263405214`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33263405214) 四个必需 job 全部成功,因此该维护任务为 `completed`。[ADR-0018](decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md) / data-only `20260830_0007` 已修复 observational input estimate 误触发 overdraw;本地完整门禁、当前个人 SQLite 数据验真与最终 SHA [`cb00924ea3ba3d01ce5bc322b7eabdae1345baf3`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/cb00924ea3ba3d01ce5bc322b7eabdae1345baf3) 的 [run `33271095910`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33271095910) 4/4 全部通过,该维护为 `completed`。Phase 2 仍为 `in_progress`;P2-07 已建立 ADR-0016、独立计划和工作日志,状态为 `planned`,功能实现尚未开始。历史 P2-01 位于 [PR #2](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/2):实现 SHA `b6a35fef1dd069ebb54b69955058915c722aa34d` 的 [run `33146681285`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33146681285) 4/4 成功,证据文档 commit `875f13a253c40b7573d45c6287385e60f2bb8f04` 的 [run `33150080341`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33150080341) 也已 4/4 成功。 +Run Detail 指标维护实现 SHA [`0003e4291769a851005ba46c7e59b156a6b789eb`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/0003e4291769a851005ba46c7e59b156a6b789eb) 已普通 push 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5);其精确 SHA 的 [GitHub Actions run `33286730109`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33286730109) 对 backend、真实 PostgreSQL/Redis integration、real-Compose acceptance 和 frontend 四个 job 全部成功,因此该维护为 `completed`。它不改变 Phase 2/3 或 P2-07 状态。 + +P3-06 的 [Run Detail 热力图/live metrics 计划](plans/2026-08-30-run-progress-heatmap-live-metrics.md) 状态为 `completed`。公共合同已从初版无可靠提交序的 cursor 改为固定 `512` 题 absolute-position blocks:progress index 在同一数据库读取快照返回 evidence-derived live metrics 与所有 block counts,block payload 只返回 position/outcome/score/latency/usage/cost/error type 白名单。该切片保持 `/api/v1` 与 `llmbenchlab-protocol-v1`,无 migration、ADR 或 SECURITY 边界修改;初版 cursor 失败先行套件记录为 `4 failed` 后已废弃。实现 SHA [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) 已普通 push 到 `codex/complete-evaluation-workflow` 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5);精确 SHA 的 [GitHub Actions run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) 对 backend、backend-integration、full-stack-reliability、frontend 四个必需 job 全部成功。Phase 3 整体仍为 `in_progress`,P2-07 恢复为下一项且仍为 `planned`。 + +## 2026-08-30 Provider API 三协议适配(`completed`) + +- [ADR-0019](decisions/ADR-0019-explicit-provider-api-protocol-adapters.md) 保留 `openai_compatible` 作为 Chat Completions 兼容值,并新增 `openai_responses` 与 `anthropic_messages`。Model API、Run snapshot、Worker/CLI preflight 和 Web 表单都显式携带该类型;不会从模型名/URL 猜测,也不会在失败后跨协议 fallback。 +- 三类 Adapter 各自构造 endpoint、payload、认证 header、普通 JSON 与 typed SSE,并要求 `[DONE]`、`response.completed` 或 `message_stop` 终止。Responses/Messages 不支持的非空 seed、Messages 的空 `max_tokens` 和已知错误 suffix 都在外发前拒绝;typed transient error 才进入既有逐-attempt retry/ledger。 +- Alembic head 为 `20260830_0008`,将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)` 并替换 Provider 类型 check 与远程配置 check;数据库中存在新类型时 downgrade 在 DDL 前 fail closed。隔离 PostgreSQL 16 已通过 upgrade/check、populated downgrade 拒绝和清空后 downgrade/upgrade/check;没有迁移或重启用户当前数据库/服务。 +- 本地门禁已通过:backend `1079 passed, 36 skipped`、frontend `72 passed`、`make lint`、Mock smoke `1 passed, 7 deselected`、frontend build、Compose config 与目标协议/迁移回归均全绿。实现 SHA [`6943aa29a154c82bdfbe5efb2578c916c3cbf632`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/6943aa29a154c82bdfbe5efb2578c916c3cbf632) 已普通 push,[exact-SHA Actions run `33304667092`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33304667092) 四个必需 job 全部成功。自动化仅使用 Mock/MockTransport,未调用真实 Provider;该维护为 `completed`。 + ## 已交付基线 - Phase 0/1 的治理、架构、协议、数据格式、ADR、FastAPI/SQLAlchemy/Alembic、React/TypeScript、Mock 垂直链路、三类 Evaluator、Demo 数据、API/UI、离线测试和开源流程。 - PostgreSQL/Redis 可靠执行基础:数据库事实来源、Redis at-least-once 通知、独立 Worker、DB scan、租约/heartbeat/fencing、逐题幂等、有限 retry/backoff、取消、租约接管、dead-letter 和终态 Response 重算。 -- OpenAI-compatible SSE、严格 `[DONE]`、JSON fallback、identity-only、wire/event/content/error 上限、idle read timeout、bounded error 与精确当前-Key 脱敏。 +- 显式 Chat Completions / OpenAI Responses / Anthropic Messages Adapter;分别要求 `[DONE]` / `response.completed` / `message_stop`,支持普通 JSON 与 typed SSE,并共享 identity-only、wire/event/content/error 上限、idle read timeout、bounded error 与精确当前-Key 脱敏边界。 - Web write-only `api_key`、AES-256-GCM `model_credentials`、数据库外 API/Worker 共享 keyring、legacy `api_key_env`、origin/active-Run 门禁和 fail-closed repair/remove 路径。 - MMLU-Pro test 与 GPQA-Diamond 固定 revision/SHA 转换、可信本地 `llmbenchlab-evaluate prepare/run/resume/report`、请求上界确认和原子终态报告。该 CLI 仍要求独占数据库,未受 Phase 2 managed budget 保护。 -- React 中文界面覆盖 Dashboard、Models、Benchmarks、Evaluation Runs、New Run、Run Detail、Leaderboard;Run 列表全状态筛选/分页/活动轮询,详情逐题分页,关键桌面/平板/移动布局已修复。 +- React 中文界面覆盖 Dashboard、Models、Benchmarks、Evaluation Runs、New Run、Run Detail、Leaderboard;Run 列表全状态筛选/分页/活动轮询,详情逐题分页,关键桌面/平板/移动布局已修复。Run Detail 现区分未得分、普通答错与执行异常,并在精确 Token 未知时显示 Run-wide 已知小计、输入/输出覆盖率和“不完整”提示。 + +## P3-06 Run Detail 热力图/live metrics(`completed`) + +- 用户 Run `a3de7e4d-40b2-4d8c-994b-c713047393ae` 的只读证据对账为 total/completed/correct/error=`198/198/179/2`;198 条 Response 中 `score < 1` 为 19、`error_type` 非空为 2,因此互斥四态应为通过 179、普通答错 17、执行异常 2、未执行 0。旧卡片显示“错误题 2”实际只反映执行异常,已复现原问题。 +- 同一 Run 的已知 input/output Token 为 `45,509 / 4,561,625`,各自覆盖 `196/198`;平均延迟为 `181,454.235 ms`。Run 精确 input/output/cost 仍为 `null`,新 UI 只能显示 known subtotal + reported coverage,不能把两条缺失 usage 当 0 或回填账单真值。这里未记录任何 Response 正文。 +- 后端合同固定为 `GET /runs/{id}/progress` index 与 `GET /runs/{id}/progress/blocks/{block_index}` payload,`block_size=512`。outcome 优先级为 execution error、passed、wrong;没有 Response 的计划 position 才是 `not_run`。两个响应均 `no-store`,不包含 ID、题目/回答正文、error message 或 Provider metadata。 +- 前端已实现虚拟化 ARIA grid、非仅颜色图例、hover/focus/tap 等价详情和独立 block reducer/poller;非空 block 追齐前显示“同步中”,terminal 先到仍追齐,当前 Responses 页码与 progress 更新互不重置。 +- 本地证据:backend target `37 passed`;frontend target `32 passed`(Run Detail `20` + heatmap `12`);完整 backend `964 passed, 33 skipped`、frontend `64 passed`;`make lint`、Mock smoke `1 passed, 7 deselected`、frontend build 与 `docker compose config --quiet` 通过。终态且 progress 已 reconciled 时只做一次最终 Run/当前 evidence 页刷新;同路由切换 `runId` 会把 evidence offset 重置为 0。12,032/20,000 题是自动化虚拟化边界,不是大型真实 Run 的手工 DevTools 性能测量。 +- 目标 Run 实页显示通过 179、普通答错 17、执行异常 2、未执行 0,Token `45,509 / 4,561,625`、输入/输出覆盖均为 `196/198`;desktop/768/375 无横向溢出,console 无 warning/error,键盘与 Tooltip 验收通过。 +- 实现 SHA `99791964621165c9cc7ec36b4b2d27fe04e6acd5` 已普通 push,PR #5 的 exact-SHA Actions run `33289522923` 四个必需 job 全部成功,因此本切片为 `completed`。Phase 3 整体仍为 `in_progress`;P2-07 恢复为下一独立任务并保持 `planned`。 + +## 2026-08-30 多 Worker 日常评测入口(`completed`) + +- 现场只读诊断确认当前不是 Worker 进程死亡:唯一 Worker 仍持续续租,但一个 Run 使用 4 路题目并发、300 秒 read timeout 和最多 3 次 HTTP attempt;长时间未返回的 Provider 请求可以让该 Worker 约 15 分钟不产生新题证据,其余 3 个 due Run 因没有空闲 Worker而排队。诊断未读取 Key/正文、未取消或改写 Run。 +- 核心租约/fencing 不变。`make dev DEV_WORKERS=N` 现可在 PostgreSQL 下管理 1–32 个独立 Worker进程和私有日志,SQLite 请求 `N>1` 会在 keyring、日志和服务创建前固定失败;`make dev-multi` / `make docker-up WORKERS=N` 默认 2,用同一参数驱动 Compose scale 与 API expected,扩容先证明 active Worker scan、缩容先降低 API expected,并在返回前校验 `expected/registered/live/stalled/shortfall=N/N/N/0/0`。 +- 新增的真实 PostgreSQL 16 回归证明两个 owner 可并发领取属于不同 Benchmark 的两个 due Run,且另一个有效 lease 不阻塞不同 Run、也不能重复领取同一 Run。启动器目标回归为 `42 passed`;终审后又用 DB 时钟 watermark 证明本轮新增 generation 已真实 scan,并用 all/running 两类 Compose replica 正确判定含 exited container 的缩容方向。第一次临时 PG 测试因空库未迁移在 fixture setup 报两项 `UndefinedTable`,按真实部署顺序迁移到 `20260830_0007` 后相同两项 `2 passed`。修复后的隔离 Compose 冷启动得到 `2/2/2/0/0`,随后 `2→1→2` 的 gauges 依次为 `1/1/1/0/0`、`2/2/2/0/0`,最终 container/volume/network/本轮 image tag 均为 0。所有临时资源按精确 project/container/image 名清理,未触碰用户 SQLite、默认 Compose volume 或共享 image tag。 +- 终审修复后的完整本地门禁已通过:`make lint`(Ruff/format 160 files、ESLint、TypeScript)、`make test`(backend `1003 passed, 35 skipped`、frontend `64 passed`)、Mock smoke `1 passed, 7 deselected`、frontend build、Compose config 与 `git diff --check`。实现 SHA [`b06594c2df67d6e2a8b117651b193cd0fa409bf5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/b06594c2df67d6e2a8b117651b193cd0fa409bf5) 已普通 push,其精确 SHA 的 GitHub Actions [run `33299883513`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33299883513) 四个必需 job 全部成功,因此该维护为 `completed`。它不迁移当前个人 SQLite、不改变 schema/API/protocol/governance,也不把允许配置 3–32 个 Worker写成容量资格;当前资格仍仅覆盖 1–2 个 Worker,P2-07 仍为下一独立切片。 ## 已通过候选门禁的 Phase 2 切片 @@ -46,7 +74,7 @@ - [ADR-0015](decisions/ADR-0015-observability-worker-progress-audit-retention.md) 已接受;实现 SHA `9a20676dcf545040782f04c166205d0043345753` 将 Alembic head 扩展到 `20260828_0005`。`worker_processes` 保存 generation 级 DB UTC `started/seen/scan/claim/progress/lease-heartbeat/stop`,主循环只在真实事件后合并刷新;JSON metrics 公开 expected/registered/live/stalled/shortfall 与最近时间,不公开 Worker/generation ID。dependency probe 固定声明 `main_loop_progress=not_checked`。 - `GET /api/v1/metrics/prometheus` 已实现固定 Prometheus text `0.0.4` gauge:一个 DB-time 读快照、15 分钟 typed-audit 窗口、1 小时 Run latency、硬读取上限、固定 enum label、整次 fail-closed 与每 API 进程 single-flight。`deploy/observability/` 提供固定八条规则和安全抓取示例;仓库不部署 Prometheus、Alertmanager 或通知发送器。 - `llmbenchlab-audit-retention archive|verify|reconcile|restore|delete` 已实现 canonical JSONL v1、严格权限/大小/行/schema/hash/rollup 校验、离线 verify、精确 digest 绑定、默认不删除、双方言事务与 commit outcome 分类。Archive 是敏感运维文件,hash 只用于完整性/绑定,不是签名或 WORM,也不替代 P2-07 的数据库+keyring 备份。 -- P2-06 的 `0005` 将 importer 逻辑合同扩展为 13 表精确 count/PK/content digest;source/target 必须位于唯一 current head,现为 data-only `0007`,13 表 schema/内容合同不变。live generation 在源 preflight 被拒绝,stopped/stale facts 可复制,终审又补强 committed target canonical integrity postverify。`0005 -> 0004` 在 `worker_processes` 非空时于 DDL 前拒绝,原有 `0004` governance/audit downgrade guard 继续保留。 +- P2-06 的 `0005` 将 importer 逻辑合同扩展为 13 表精确 count/PK/content digest;source/target 必须位于唯一 current head,现为 `0008`(data-only `0007` 后扩展 `provider_type` 列宽并替换 Provider 类型/远程配置两个 check),13 表内容合同不变。live generation 在源 preflight 被拒绝,stopped/stale facts 可复制,终审又补强 committed target canonical integrity postverify。`0005 -> 0004` 在 `worker_processes` 非空时于 DDL 前拒绝,原有 `0004` governance/audit downgrade guard 继续保留。 - 生产日志源已统一治理:应用日志消息必须是无格式参数字面量,结构化字段按白名单和有限数值输出,第三方动态消息固定化且不能通过 allowlisted extra 注入,raw Uvicorn access handler 关闭。Archive 终审补充了 FIFO/非普通文件拒绝及 decode 前行数上限;retention 零行 mutation 仍须 postverify,PostgreSQL mutation 保持 advisory/row lock。 - 上述实现的全部实现门禁已完成:合并定向套件、`make lint`(Ruff 152 files、ESLint、TypeScript)、`make test`(后端 `916 passed, 33 skipped`、前端 `38 passed`)、Mock smoke(`1 passed, 7 deselected`)、临时 PostgreSQL 16/Redis 7 migration/check 与真实 integration(`33 passed, 0 skipped`)、隔离 SQLite migration/check、frontend build、Compose config、八规则 `promtool` 和修复后 76-file staged 技术/安全终审均通过;实现 SHA 已 push,精确 SHA run `33164609388` 4/4 成功。Clean acceptance `.pytest_cache/artifacts/phase2-acceptance/llmbenchlab-p2-92e173eeee28/evidence.json` 的 SHA-256 为 `e4ffb8668fd3fa62d59b5d83f5c29eede35b327d88e6099345acd5950670fc47`,9/9 通过,Worker expected/registered/live/stalled/shortfall=`2/2/2/0/0`,cleanup C/V/N 全空。Clean capacity `.pytest_cache/artifacts/phase2-capacity/llmbenchlab-p2-ca5673061b0f/evidence.json` 的 SHA-256 为 `2382f9138f09028f269d76c341b236dd4089d678c8a2323582045fac2b4f5039`;1W/2W/burst QPS=`7.267474/12.962228/9.333604`、wall=`8.255963/4.628834/6.428385s`,最终 18 Runs/270 Responses/270 question executions/271 reservations/1230 audit,0 question error/drift/duplicate/PEL/lag,Worker expected=2、shortfall=0,cleanup C/V/N/image 全零且 image counters=`1/1/0/0`。两份 evidence 均为 `dirty=false` 并绑定 `9a20676…`;这是 Mock-only、非 SLO。此前 dirty acceptance/capacity 继续作为历史证据保留。Evidence closeout 文档 commit `ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6` 已 push,精确 SHA run `33165775037` 4/4 成功,P2-06 仓库级收尾完成。P2-06 当时默认用户 SQLite 尚未在 head,直接 `alembic check` 失败后按保护原则未擅自迁移。 @@ -71,6 +99,12 @@ - 该次导入完成时,默认个人 SQLite 有 `4` 个 Benchmark、`24,277` 道题;逐集持久化题数与 manifest 一致。当时原有 `1` 个 Model、`1` 个 completed Run 和 `15` 条 Response 不变,active Run 为 `0`;`quick_check=ok`、外键错误 `0`、Alembic head=`20260829_0006`,API 列表/逐集 total 也已对账。后续 Run 与 `0007` 维护事实见下节,不回写本条历史快照。 - 目标 Loader/标准转换器离线测试 `40 passed`;未调用真实 Provider,未修改 Schema/API/协议或产品代码。仓库记录 commit [`0163b67c00eb59ae59db5f3adb679ad85c799142`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/0163b67c00eb59ae59db5f3adb679ad85c799142) 已 push,其精确 SHA 的 [run `33266167547`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33266167547) 四个必需 job 全部成功;本地加载维护为 `completed`,不改变 Phase 3/P2-07 状态。 +## 2026-08-30 小型评测集本地加载(`completed`) + +- 从固定官方/维护者 revision 准备 6 套各 100 题的 Git 忽略 dataset-v1 ZIP:GSM8K、中文 MGSM、HellaSwag、WinoGrande、TruthfulQA Binary 五套稳定 mini 子集,加上完整 100 题的中文 XCOPA validation。题型仅使用当前可自动评分的 numeric/multiple-choice;seed 42 的稳定选择/转换、源/archive/Dataset Hash、许可、源路径/行数、所选源行和答案分布均记录在本地 provenance 清单。TruthfulQA 固定 CSV 没有官方 split;本地 Best Answer 对 Best Incorrect Answer 的 binary mini 不能与官方 MC1/MC2 全量分数混用。 +- 导入前 SQLite online backup 冻结为 Models/Benchmarks/Questions/Runs/Responses=`2/4/24,277/5/1,000`,并保留当时唯一活动 Run 的 `765/12,032` 进度。六个正式导入请求均返回 `201`;默认个人 SQLite 现有 `10` 个 Benchmarks、`24,877` 道 Questions,六个新 Benchmark 的 API/manifest/数据库题数与 Dataset Hash 一致。 +- 导入任务没有创建、取消、重置或修改 Run;当时既有 12,032 题 Run 继续推进,Response 数量按预期增长。导入完成后的并发客户端活动随后请求取消该大 Run,并创建了 MGSM mini Run;这些变化有独立 Run/audit 时间线,不是导入副作用。本任务没有直接触发 Provider。导入后 `quick_check=ok`、外键错误 `0`、head=`20260830_0007`;用金标做评分器格式自检时 600/600 reference answers 被接受(不是模型 600/600 成绩),工程目标测试 `40 passed`。记录 commit [`8faa2093b2c3308994d50e42a31063cdbf5264a6`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/8faa2093b2c3308994d50e42a31063cdbf5264a6) 已 push,其精确 SHA 的 [run `33296049611`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33296049611) 四个必需 job 全部成功。本维护不改变产品代码、Schema、API、协议、Phase 3 或 P2-07 状态。 + ## 2026-08-30 observational Token overdraw 修复(`completed`) - 只读核查 Run `2181503c-eab2-4699-bede-db48bd078f95` 发现:15 题完成 7 题后以 `failed/exhausted`、`governance_global_overdrawn` 终止;7 个 Provider attempt 均为 `settled_actual/succeeded`,没有 conservative settlement、429 或 HTTP retry。第七次非 hard 输入估算/reservation 为 59,而 Provider actual 为 75;冻结 policy 与 Run override 的 request/Token/cost hard limit 全为 `null`,因此这是本地语义缺陷,不是 OpenCode Go 套餐额度结算。 @@ -78,14 +112,24 @@ - 应用 Alembic head 已前进到 data-only `20260830_0007`。该 migration 不改 schema、ledger、actual usage、Response、audit 或 Run 终态,只重算 `governance_scopes.overdrawn`;upgrade/downgrade 在任何更新前拒绝 active reservation。前端 overdrawn 文案改为“实际用量曾被判定超过预留”,既适用于新 hard overdraw,也不会误述升级前保留的历史终态。 - 本地验证已完成:backend `946 passed, 33 skipped`,真实 PostgreSQL+Redis integration `33 passed`,双方言 migration upgrade/downgrade/upgrade/check、`make lint`、frontend `39 passed`/build、Mock smoke `1 passed`、real-Compose `9/9` 与 Compose config 均通过。当前个人 SQLite 已到 `0007`;四层 scope `overdrawn` 从 4 降为 0,7 Responses/7 ledger、407 input/599 output、13 张业务表行数均保留,`quick_check=ok`、FK=0。未调用真实 Provider。首次实现 SHA 的 acceptance-only `float(None)` 失败已保留,最终修正 SHA `cb00924…` 的 run `33271095910` 4/4 成功,因此本维护为 `completed`。 +## 2026-08-30 Run Detail 错题与部分 Token 展示修复(`completed`) + +- 目标 Run `a3de7e4d-40b2-4d8c-994b-c713047393ae` 的 198 条证据实际为正确 179、普通答错 17、执行异常 2;旧页面把只统计异常的 `error_questions=2` 标成“错误题”。页面现显示未得分 19,并明确拆分三类数量;当前页也分别统计未得分与执行异常。 +- 196/198 条 Response 有 usage,已知输入 45,509、输出 4,561,625。protocol-v1 精确 Run Token 继续因两条缺失而保持 `null`;Responses API 追加分页无关的输入/输出已知小计和独立覆盖数,页面显示“已知小计”和“完整总量未知”,不会修改历史数据或冒充 Provider 账单。 +- API/UI、OpenAPI、零/全/部分/非对称 usage、合法零 Token、分页、并行快照竞态与页内错题拆分的目标测试、完整本地门禁和目标实页核对均通过。实现 commit `0003e429…` 已普通 push;PR #5 的精确 SHA CI run `33286730109` 4/4 成功。本维护不改变 Phase 2/3 或 P2-07 状态。 + ## 状态与后续 - P2-06:状态为 `completed`;实现、clean-SHA Compose evidence、实现 commit 与 evidence closeout 文档 commit 的 push 和精确 SHA CI 均已完成。 - 0004 历史索引兼容修复:状态为 `completed`;实现 commit `8fb51b690ae6335b8ef93b3cbe54e039781fb173` 已 push,精确 SHA run `33263405214` 4/4 成功。 - 本地数据恢复与静默启动:状态为 `completed`;实现 commit `5075bdb5e9b53f527a43e5aff7b7d2c7b48c5c9b` 已 push,精确 SHA run `33265171953` 4/4 成功。 - 已下载标准评测集本地加载:状态为 `completed`;三个现有正式 ZIP 已导入并完成本地数据库/API/目标测试验证,`0163b67…` 的 run `33266167547` 4/4 成功。 +- 小型评测集本地加载:状态为 `completed`;五套 100 题 mini 子集与完整 100 题 XCOPA validation 已生成并导入,Loader/Evaluator、API/数据库/Hash 与完整性验证通过;本导入任务没有创建或取消 Run,后续并发 Run 操作另行留有审计时间线;`8faa209…` 的 exact-SHA run `33296049611` 4/4 成功。 - observational Token overdraw 修复:状态为 `completed`;目标 head `0007`、本地完整验证、当前库迁移、最终修正 commit/push 与 exact-SHA CI 4/4 均完成。 -- P2-07:状态为 `planned`,已建立 [ADR-0016](decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md)、exact-head amendments [ADR-0017](decisions/ADR-0017-schema-equivalent-governance-index-repair.md) / [ADR-0018](decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md)、[独立计划](plans/2026-08-28-phase-2-recovery-operations.md) 和 [工作日志](worklogs/2026-08-28-phase-2-recovery-operations.md)。PostgreSQL backup/restore、数据库与 keyring 配对恢复、Redis 重建、Worker 扩缩/告警处置和剩余故障矩阵的功能实现尚未开始;P2-06 的 audit archive 自身 restore 不能替代整库恢复认证。P2-07 recovery-manifest-v1 的尚未实施 exact head 现为 `20260830_0007`。 +- Run Detail 错题与部分 Token 展示修复:状态为 `completed`;实现 SHA `0003e429…` 已 push,PR #5 的精确 SHA CI run `33286730109` 4/4 成功。 +- P3-06 Run Detail 热力图/live metrics:状态为 `completed`;实现 SHA `99791964621165c9cc7ec36b4b2d27fe04e6acd5` 已普通 push,PR #5 的精确 SHA CI run `33289522923` 4/4 成功;不改变 protocol-v1 或 Phase 3 整体 `in_progress` 状态。 +- Provider API 三协议适配:状态为 `completed`;代码、本地全量门禁、隔离 PostgreSQL 16 迁移验证、实现 SHA `6943aa29…` 普通 push 与 exact-SHA run `33304667092` 4/4 均已完成。 +- P2-07:状态为 `planned`,已建立 [ADR-0016](decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md)、exact-head amendments [ADR-0017](decisions/ADR-0017-schema-equivalent-governance-index-repair.md) / [ADR-0018](decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md) / [ADR-0019](decisions/ADR-0019-explicit-provider-api-protocol-adapters.md)、[独立计划](plans/2026-08-28-phase-2-recovery-operations.md) 和 [工作日志](worklogs/2026-08-28-phase-2-recovery-operations.md)。PostgreSQL backup/restore、数据库与 keyring 配对恢复、Redis 重建、Worker 扩缩/告警处置和剩余故障矩阵的功能实现尚未开始;P2-06 的 audit archive 自身 restore 不能替代整库恢复认证。P2-07 recovery-manifest-v1 的尚未实施 exact head 现为 `20260830_0008`。 - Phase 3:IFEval、通用 Dataset Plugin SDK、代码题 schema/隔离沙箱、完整分组 UI 和安全红队;Phase 4–6 尚未开始。 ## 已知边界与风险 @@ -127,7 +171,12 @@ | P2-01 证据文档收尾 CI | `875f13a…` run `33150080341` 4/4 | 精确文档 SHA 全绿;P2-01 仓库级收尾完成 | | 2026-08-30 本地恢复/静默启动 | 启动器 `3 passed`;完整 backend `930 passed, 33 skipped`、frontend `38 passed`;lint/build/smoke/config、恢复库 digest/quick/FK/head 与真实 API/Web 读取通过 | 默认库恢复 `1/1/15/1/15`;`5075bdb…` 的 run `33265171953` 4/4 成功,不改变 P2-07 | | 2026-08-30 标准评测集本地加载 | 三个 ZIP Loader 校验通过;API 导入 `201/201/201`;数据库/API 对账为 `4` Benchmarks、`24,277` Questions;`quick_check=ok`、FK `0`、head `0006`;目标测试 `40 passed` | 本地加载完成,原 Model/Run/Response 保持;`0163b67…` 的 run `33266167547` 4/4 成功,无 Provider 调用 | +| 2026-08-30 小型评测集本地加载 | 六个 ZIP Loader 校验、重复生成与 archive/Dataset Hash 对账通过;API 导入 `201` × 6;金标评分器格式自检 600/600(非模型成绩);目标测试 `40 passed`;数据库为 `10` Benchmarks、`24,877` Questions,`quick_check=ok`、FK `0`、head `0007` | 六套均为 100 题且可直接选择;导入任务未创建/取消 Run,既有大 Run 与随后创建的 MGSM mini Run 的 Response/Provider 流量属于并发客户端操作,不归因于导入;`8faa209…` 的 run `33296049611` 4/4 成功 | | 2026-08-30 observational overdraw 修复 | backend `946 passed, 33 skipped`;真实 PG+Redis integration `33 passed`;双方言 migration 往返/check、`make lint`、frontend `39 passed`/build、Mock smoke `1 passed`、real-Compose `9/9`、Compose config 与当前库 backup/migrate/check 通过 | 当前 SQLite head=`20260830_0007`,scope `4→0`,7 Responses/7 ledger/407 input/599 output、13 表行数、quick/FK 保持;无真实 Provider;`cb00924…` 的 run `33271095910` 4/4 成功 | +| 2026-08-30 Run Detail 指标修复 | backend API/Smoke 目标 `11 passed`、frontend Run Detail/format `20 passed`;完整 backend `951 passed, 33 skipped`、frontend `47 passed`;lint/build/Mock smoke/Compose config/实页验收通过 | 目标 Run 19/17/2、已知 Token `45,509/4,561,625` 和 `196/198` 覆盖可见;无真实 Provider;`0003e429…` 的 run `33286730109` 4/4 成功 | +| P3-06 Run Detail 热力图/live metrics | 初版 cursor 后端 red suite `4 failed`(预期且合同已废弃);fixed-block backend/frontend target `37/32 passed`(20 Run Detail + 12 heatmap);完整 backend `964 passed, 33 skipped`、frontend `64 passed`;lint/smoke/build/config/目标 Run 浏览器验收通过 | `99791964621165c9cc7ec36b4b2d27fe04e6acd5` 已 push;PR #5 exact-SHA run `33289522923` 4/4;切片 `completed`,Phase 3 仍 `in_progress`,P2-07 为下一任务 | +| 2026-08-30 多 Worker评测入口 | 启动器/fake Compose `42 passed`;隔离 PostgreSQL 16 首次因未迁移在 fixture setup 报 `UndefinedTable`,迁移到 `0007` 后跨 Benchmark/唯一 lease `2 passed`;终审后的 fresh/watermark 与 exited-replica 回归、真实 Compose `2→1→2` gauges/cleanup 通过 | 完整 backend `1003 passed, 35 skipped`、frontend `64 passed`,lint/Mock smoke/build/config/diff check 全绿;`b06594c…` 的 exact-SHA run `33299883513` 4/4,维护 `completed` | +| 2026-08-30 Provider API 三协议适配 | 目标协议/API/Runner/CLI/迁移回归通过;完整 backend `1079 passed, 36 skipped`、frontend `72 passed`;lint、Mock smoke、build、config 全绿;隔离 PostgreSQL 16 `0008` 往返与 populated downgrade guard 通过 | 自动化只用 Mock/MockTransport,未调用真实 Provider;实现 `6943aa29…` 已 push,exact-SHA run `33304667092` 4/4,维护 `completed` | | 最新本地 `make lint` | Ruff/format、ESLint、TypeScript 通过 | 本地冻结树通过 | | P2-01 冻结树 `make test` | 后端 `829 passed, 29 skipped`;前端 `38 passed` | v2 实现历史冻结树通过;当前 P2-06 全量见上方独立行 | | P2-01 真实 PostgreSQL/Redis integration | `29/29 passed` | v2 实现历史冻结树通过;当前 P2-06 integration 见上方独立行 | @@ -153,8 +202,13 @@ - [Phase 2 可观测性与审计保留](worklogs/2026-08-28-phase-2-observability-retention.md) - [本地数据恢复与静默启动](worklogs/2026-08-30-restore-data-quiet-startup.md) - [加载已下载的标准评测集](worklogs/2026-08-30-load-downloaded-benchmarks.md) +- [准备并加载小型模型评测数据集](worklogs/2026-08-30-small-benchmark-datasets.md) - [修复 observational Token overdraw](worklogs/2026-08-30-fix-observational-token-overdraw.md) +- [修复 Run Detail 错题与部分 Token 展示](worklogs/2026-08-30-fix-run-detail-metrics.md) +- [Run Detail 热力图与实时指标](worklogs/2026-08-30-run-progress-heatmap-live-metrics.md) +- [多 Worker并行评测](worklogs/2026-08-30-multi-worker-evaluation.md) +- [Provider API 三协议适配](worklogs/2026-08-30-provider-api-protocols.md) ## 当前任务入口 -[NEXT_TASK.md](NEXT_TASK.md) 提供后续任务入口。observational Token overdraw 修复已完成本地验证、个人 SQLite 受控迁移、commit/push 和 exact-SHA CI 4/4,仓库级闭环完成。后续入口恢复为 P2-07 最小只读 recovery verifier,但本次不继续实施。P2-06 已完成仓库级收尾,P2-07 仍为 `planned`;Phase 2 与 Phase 3 继续保持 `in_progress`。 +[Provider API 三协议适配计划](plans/2026-08-30-provider-api-protocols.md) 已完成本地门禁、实现提交/push 与 exact-SHA CI。当前入口回到 [NEXT_TASK.md](NEXT_TASK.md) 的 P2-07 最小只读 recovery verifier。P2-07 保持 `planned`,Phase 2 与 Phase 3 继续保持 `in_progress`。 diff --git a/docs/REQUIREMENTS.md b/docs/REQUIREMENTS.md index 3925bca..8b70244 100644 --- a/docs/REQUIREMENTS.md +++ b/docs/REQUIREMENTS.md @@ -7,7 +7,7 @@ ## 1. 范围与术语 -MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark,创建 Run,在后台逐题调用、解析和评分,持久化结果,并在前端查看进度、逐题结果及排行榜。OpenAI-compatible 接入属于 MVP 配置能力,但自动化验收不得访问真实服务。 +MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark,创建 Run,在后台逐题调用、解析和评分,持久化结果,并在前端查看进度、逐题结果及排行榜。Chat Completions、OpenAI Responses 与 Anthropic Messages 接入属于可选远程配置能力,但自动化验收不得访问真实服务。 - **Model**:一个可调用的模型配置;公开表示不包含真实密钥值,Provider 凭据由独立的加密记录或兼容环境变量提供。 - **Benchmark**:有版本、题目集合、评分器和稳定 Hash 的数据集。 @@ -21,16 +21,16 @@ MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark ### FR-MOD:模型注册与调用 - **FR-MOD-01** 系统必须支持创建、读取、更新、删除和分页列出 Model。 -- **FR-MOD-02** `provider_type` 首期只允许 `mock` 与 `openai_compatible`。 +- **FR-MOD-02** `provider_type` 只允许 `mock`、`openai_compatible`、`openai_responses` 与 `anthropic_messages`;协议必须显式选择,旧 `openai_compatible` 继续表示 Chat Completions。 - **FR-MOD-03** Model 必须包含:`id`、`name`、`provider_type`、`base_url`、`remote_model_name`、`credential_source`、兼容字段 `api_key_env`、`enabled`、每百万输入/输出 Token 价格、默认参数、`created_at`、`updated_at`。 -- **FR-MOD-04** `openai_compatible` 必须校验并要求 `base_url`、`remote_model_name`,并选择 `stored` 或 `environment` 凭据来源;`mock` 的远端连接字段必须为空且凭据来源必须为 `none`。 +- **FR-MOD-04** 三个远程 Provider 类型都必须校验并要求 `base_url`、`remote_model_name`,并选择 `stored` 或 `environment` 凭据来源;`mock` 的远端连接字段必须为空且凭据来源必须为 `none`。根地址或完整 endpoint 必须与所选协议一致,已知错误协议后缀必须在外发前拒绝。 - **FR-MOD-05** Web/REST 必须支持只写的 `api_key` 输入:明文只用于当前请求,服务端使用独立 keyring 的 AES-256-GCM 加密后保存到 `model_credentials`,不得把凭据数据流中的明文/Authorization 或 Provider 对 Key 的回显复制到 Model、Run-model snapshot、Response、Leaderboard、报告、日志或错误,也不得公开 nonce、ciphertext、key id 或 keyring material。该控制不要求扫描无关 Benchmark/Question 数据的独立字面巧合;`api_key_env` 仅作为 CLI 与既有部署的兼容路径。 - **FR-MOD-06** 必须定义 `ModelAdapter.generate(messages, generation_config)`,返回文本、Token、延迟、provider request id、原始 usage 和元数据。 - **FR-MOD-07** Mock Adapter 必须完全离线、输出可预测,可完成 Demo 中预定义的部分或全部问题,并可用于单元测试、CI 和 Smoke Test。 -- **FR-MOD-08** OpenAI-compatible Adapter 必须使用 Chat Completions 风格请求,支持 system prompt、temperature、top_p、max_tokens、seed、可配置 base URL/模型名及连接/读取超时。 -- **FR-MOD-09** 对 429、部分 5xx 和暂时网络错误必须有限次指数退避;明显的 4xx 配置错误不得无限重试。 +- **FR-MOD-08** `openai_compatible` 必须实现 Chat Completions,`openai_responses` 必须实现 Responses,`anthropic_messages` 必须实现 Messages;每类必须具有独立 endpoint、headers、payload、普通 JSON 与 typed SSE 解析及明确的流终止证据,并把文本、usage、finish reason、request id 与返回模型归一化。Chat 支持 `seed`;Responses/Messages 在请求与 Model 默认都未提供时必须省略 `temperature/top_p/seed`,并拒绝非空 `seed`;Messages 还必须限制 `temperature<=1` 并拒绝 `max_tokens:null`。 +- **FR-MOD-09** 对 429、部分 5xx 和暂时网络错误必须有限次指数退避;Responses rate-limit/server typed error、Messages `rate_limit_error`/`api_error`/`overloaded_error`/`timeout_error` 与 HTTP `529` 只按显式白名单重试,未知 typed error fail closed;明显的 4xx 配置错误不得无限重试。 - **FR-MOD-10** 上游没有 Token Usage 时相关值允许为 `null`;错误应保存稳定类型与经过脱敏的可读信息。 -- **FR-MOD-11** Phase 1 的 `default_parameters` 只允许 `temperature`、`top_p`、`max_tokens`、`seed` 及其 Run 级约束;创建 Run 时显式请求值优先,最终有效值必须进入快照。 +- **FR-MOD-11** Phase 1 的 `default_parameters` 只允许 `temperature`、`top_p`、`max_tokens`、`seed` 及其 Run 级约束;创建 Run 时显式请求值优先,最终有效值必须进入快照;Responses/Messages 未显式提供的采样字段必须以 `null` 冻结而不是继承 Chat Schema 默认。 ### FR-BEN:Benchmark 与导入 @@ -71,7 +71,7 @@ MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark - **FR-REP-01** protocol version 初始为 `llmbenchlab-protocol-v1`。 - **FR-REP-02** Run 必须快照:Benchmark ID/version/SHA-256、Evaluator 名称及版本、Prompt template、system prompt、temperature、top_p、max_tokens、seed、模型名、Adapter 类型、模型参数、Git commit SHA(不可得时为 `null`)、开始/结束时间、并发度和重试策略。 -- **FR-REP-03** 默认公平参数为 `temperature=0`、`top_p=1`、固定 max tokens、并发度 1 或较小安全值;上游支持时传 seed。 +- **FR-REP-03** Chat 的默认公平参数为 `temperature=0`、`top_p=1`、固定 max tokens、并发度 1 或较小安全值,并在上游支持时传 seed;Responses/Messages 默认省略未配置的采样字段,不能假称跨协议同一采样参数必然等价。 - **FR-REP-04** 不同 protocol version、Benchmark version 或 dataset hash 的结果不得无提示混合比较。 ### FR-API:REST API @@ -80,21 +80,21 @@ MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark - **FR-API-02** 系统端点:`GET /health`、`GET /info`;健康检查不得依赖真实模型服务。 - **FR-API-03** 模型端点:`GET/POST /models`、`GET/PATCH/DELETE /models/{id}`。 - **FR-API-04** Benchmark 端点:`GET /benchmarks`、`GET /benchmarks/{id}`、`POST /benchmarks/import`、`POST /benchmarks/reload-demo`。 -- **FR-API-05** Run 端点:`GET/POST /runs`、`GET /runs/{id}`、`POST /runs/{id}/cancel`、`GET /runs/{id}/responses`。 +- **FR-API-05** Run 端点:`GET/POST /runs`、`GET /runs/{id}`、`POST /runs/{id}/cancel`、`GET /runs/{id}/responses`,以及固定 512 题 absolute-position block 的 `GET /runs/{id}/progress` index 和 `GET /runs/{id}/progress/blocks/{block_index}` payload。progress index 必须从同一读取快照给出 block counts 与 evidence-derived live metrics;block payload 只能给出热力图所需轻量字段。 - **FR-API-06** 汇总端点:`GET /leaderboard`、`GET /metrics/summary`。 - **FR-API-07** 列表必须支持基本分页;Leaderboard 必须支持 Benchmark/Model 筛选和得分排序。 -- **FR-API-08** 请求/响应必须使用明确 Schema、合理状态码和可读校验错误,任何响应不得泄漏秘密。 +- **FR-API-08** 请求/响应必须使用明确 Schema、合理状态码和可读校验错误,任何响应不得泄漏秘密。progress cell 必须使用固定字段白名单,禁止 Question/Response ID、正文/答案、error message 与 Provider metadata,并返回 `Cache-Control: no-store`。 - **FR-API-09** CORS 只能允许配置的前端来源。 ### FR-UI:用户界面 - **FR-UI-01** 中文 Dashboard 必须展示模型、Benchmark、Run、成功 Run 数,最近运行及得分、延迟、Token 概览。 -- **FR-UI-02** Models 页面必须支持列表、添加、编辑、删除和表单校验;OpenAI-compatible 模型由用户直接在密码输入框粘贴 API Key,提交后立即清空,后续只显示“已安全保存”等状态,绝不回显密钥。环境变量名称不得作为 Web 主流程输入项。 +- **FR-UI-02** Models 页面必须支持列表、添加、编辑、删除和表单校验;远程模型必须显式选择 Chat Completions、OpenAI Responses 或 Anthropic Messages,由用户直接在密码输入框粘贴 API Key,提交后立即清空,后续只显示“已安全保存”等状态,绝不回显密钥。环境变量名称不得作为 Web 主流程输入项。 - **FR-UI-03** Benchmarks 页面必须展示名称、版本、维度、语言、题数、Hash、详情、格式说明、Demo 标识及重新载入操作。 -- **FR-UI-04** New Run 必须允许选择 Model/Benchmark,设置 temperature、top_p、max_tokens、seed,创建后跳转详情。 -- **FR-UI-05** Run Detail 必须轮询状态并展示进度、三类得分/比率、正确/错误数、延迟、Token、成本、配置快照及逐题原始/解析/标准答案、得分与错误。 +- **FR-UI-04** New Run 必须允许选择 Model/Benchmark,按协议设置或省略 temperature、top_p、max_tokens、seed;Messages 禁止 Provider-managed 输出预算且 temperature 上限为 1,创建后跳转详情。 +- **FR-UI-05** Run Detail 必须轮询状态并展示进度、三类得分/比率、正确/普通答错/执行异常数、延迟、Token、成本、配置快照及逐题原始/解析/标准答案、得分与错误。全 Run 进度必须以通过、普通答错、执行异常、未执行四态热力图呈现;动态主指标取自后端同快照证据聚合,Token/cost 已知小计必须同时显示 reported coverage,不能回填或冒充 nullable 的精确 Run 总量。 - **FR-UI-06** Leaderboard 必须展示模型、Benchmark/version、protocol version、严格总分、回答准确率、完成率、延迟、Token、成本和运行时间,并支持筛选与排序。 -- **FR-UI-07** 所有页面必须有加载、空数据和错误状态,在常见桌面和移动宽度可用;不能是占位空壳。 +- **FR-UI-07** 所有页面必须有加载、空数据和错误状态,在常见桌面和移动宽度可用;不能是占位空壳。热力图状态不能只靠颜色,必须有文字图例、可访问名称和键盘导航;hover、focus 与移动端 tap 提供等价详情,block 尚未 hydrate 时必须显示“同步中”而非伪装为未执行。 ## 3. 数据要求 @@ -119,7 +119,7 @@ MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark ## 4. 用户故事 - **US-01** 作为首次使用者,我可以不配置 Key 注册 Mock 模型、载入 Demo 并完成一次 Run,以验证系统可用。 -- **US-02** 作为本地用户,我可以在 Web 中直接粘贴 OpenAI-compatible API Key;保存后界面和 API 不再返回明文,后台 Worker 可以直接用于真实模型评测。 +- **US-02** 作为本地用户,我可以在 Web 中显式选择远程 API 协议并直接粘贴 Provider API Key;保存后界面和 API 不再返回明文,后台 Worker 可以直接用于真实模型评测。 - **US-03** 作为研究人员,我可以导入经过校验的版本化小型 Benchmark,并看到题数与稳定 Hash。 - **US-04** 作为评测者,我创建 Run 后立即获得 ID,离开创建页面也能查看进度和终态。 - **US-05** 作为分析者,我能检查每题原始输出、解析答案、标准答案、评分、耗时、Token 和错误。 @@ -138,14 +138,14 @@ MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark ### 性能与可靠性 - **NFR-PERF-01** 在开发机、默认 SQLite、单个 12–20 题 Demo 上,Mock Run 应能在 30 秒内完成;该指标不适用于真实上游延迟。 -- **NFR-PERF-02** 创建 Run 应在 2 秒内返回;列表和详情在本地千级 Response 数据下目标响应时间为 1 秒内(不含首次启动和前端网络开销)。 -- **NFR-PERF-03** 列表必须分页,后台并发必须有上限,数据库会话和 HTTP 客户端必须正确释放。 -- **NFR-REL-01** 单题失败隔离,汇总指标可由持久化 Response 重算;进程重启遗留状态不得永远显示运行中。 +- **NFR-PERF-02** 创建 Run 应在 2 秒内返回;列表和详情在本地千级 Response 数据下目标响应时间为 1 秒内(不含首次启动和前端网络开销)。12,032–20,000 题热力图每秒只能读取有界 index 并重取计数变化的 512 题 block,不能每轮拉取全题正文或创建等量常驻 DOM 节点。 +- **NFR-PERF-03** 列表必须分页,热力图必须虚拟化,后台并发必须有上限,数据库会话和 HTTP 客户端必须正确释放。 +- **NFR-REL-01** 单题失败隔离,汇总指标可由持久化 Response 重算;进程重启遗留状态不得永远显示运行中。progress index 的 live metrics 与 block counts 必须来自同一读取快照;乱序完成、index→block 并发提交、终态先到及页面切换不得造成永久漏格或旧响应污染。 - **NFR-REL-02** MVP 明确不保证后台任务自动恢复、跨进程互斥或高可用;这些属于 Phase 2。 ### 安全与隐私 -- **NFR-SEC-01** 不得明文保存或返回真实 Key,不得记录 Authorization;Web 提交的 Key 只能以 AES-256-GCM 密文保存,错误信息、请求标识和日志必须经过 canary 测试证明不会反射密钥。 +- **NFR-SEC-01** 不得明文保存或返回真实 Key,不得记录 `Authorization` 或 `x-api-key`;Web 提交的 Key 只能以 AES-256-GCM 密文保存,错误信息、请求标识和日志必须经过 canary 测试证明不会反射密钥。 - **NFR-SEC-02** 导入限制大小、拒绝路径穿越和任意文件引用,不执行数据集代码;数值解析禁止 `eval`。 - **NFR-SEC-03** CORS 来源显式配置,API 校验输入;恶意 `base_url`/SSRF 是已记录风险。 - **NFR-SEC-04** `.env` 被 Git 忽略,CI secret 非必需;依赖应固定或有 lockfile 并接受自动化审计。 @@ -162,7 +162,7 @@ MVP 必须完成以下链路:注册 Mock 模型,载入内置 Demo Benchmark - **TST-01** 后端覆盖三类 Evaluator 的正常、边界、冲突、容差、boxed 与非法输入。 - **TST-02** 覆盖 Mock Adapter、manifest/JSONL 校验与行号、Hash 稳定性、Model API 脱敏、CRUD 和 Health。 -- **TST-03** 覆盖创建与完成 Mock Run、单题故障隔离、汇总分数和 Leaderboard 聚合。 +- **TST-03** 覆盖创建与完成 Mock Run、单题故障隔离、汇总分数和 Leaderboard 聚合;另覆盖 progress 四态优先级、绝对位置稀疏/乱序、block index/payload 一致性与并发收敛、live metrics、nullable usage/cost、字段白名单、12,032/20,000 边界,以及前端同步/虚拟化/键盘/轮询竞态。 - **TST-04** 前端覆盖格式化、Run 状态、得分/完成率、API 错误和至少一个主要页面,并通过 typecheck 与 production build。 - **TST-05** Smoke Test 使用临时 SQLite,完成注册 Mock、导入 Demo、Run、Response、Score 和 Leaderboard 断言,全程禁止网络。 - **TST-06** CI 在 PR 和 main push 上运行后端 lint/test、前端 lint/test/build,不要求 API Key。 diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index b2e523e..b84296e 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -31,7 +31,7 @@ | Phase 0 | 项目治理和架构 | 可执行的需求、架构、协议、ADR 与持续文档流程 | `completed` | [PHASE-0-GOVERNANCE.md](phases/PHASE-0-GOVERNANCE.md) | | Phase 1 | MVP 垂直链路 | Mock 模型到 Run、逐题结果与排行榜的离线闭环 | `completed` | [PHASE-1-MVP.md](phases/PHASE-1-MVP.md) | | Phase 2 | 可靠性与任务执行 | 可靠 Worker、治理/审计、P2-01、P2-06 与 observational overdraw 维护已交付;P2-07 工作包已建立但功能尚未实现 | `in_progress` | [PHASE-2-RELIABILITY.md](phases/PHASE-2-RELIABILITY.md) | -| Phase 3 | 标准 Benchmark 与代码评测 | 已有 MMLU-Pro/GPQA 可信本地切片;IFEval、沙箱与完整插件体系待完成 | `in_progress` | [PHASE-3-BENCHMARKS.md](phases/PHASE-3-BENCHMARKS.md) | +| Phase 3 | 标准 Benchmark 与代码评测 | 已有 MMLU-Pro/GPQA 可信本地切片;P3-06 Run Detail 热力图/live metrics 已完成实现、push 与精确 SHA CI,IFEval、沙箱与完整插件体系待完成 | `in_progress` | [PHASE-3-BENCHMARKS.md](phases/PHASE-3-BENCHMARKS.md) | | Phase 4 | Judge、Arena 与长上下文 | 可校准 Judge、Pairwise Judge、个人 Arena 和长上下文评测 | `planned` | [PHASE-4-JUDGE-ARENA.md](phases/PHASE-4-JUDGE-ARENA.md) | | Phase 5 | Agent、私有与 Live Benchmark | 工具调用轨迹、隔离私有集和持续更新的 Live Benchmark | `planned` | [PHASE-5-AGENT-LIVE.md](phases/PHASE-5-AGENT-LIVE.md) | | Phase 6 | 公共发布 | 多用户、鉴权、运维加固和正式版本发布 | `planned` | [PHASE-6-PUBLIC-RELEASE.md](phases/PHASE-6-PUBLIC-RELEASE.md) | @@ -181,7 +181,7 @@ - 独立 Worker、原子领取、数据库时间租约、心跳、单调 fencing token、逐题幂等、有限重试、取消、过期接管和 dead-letter;大 Run 快照加载移出事件循环并保持租约心跳,dead-letter 前从持久化 Response 重聚合证据。 - Alembic `0004` 的 policy/scope/minute bucket/question execution/Provider reservation/audit event 六类治理表及 `0005` 的 Worker process/progress 表与 bounded audit indexes;Run/Response 证据字段和 13 表 SQLite→PostgreSQL importer。 - managed Run 冻结 policy/hash 与显式 overrides;global/provider/model/run 四层 concurrency、固定窗口 RPM/TPM、global/run lifetime request/Token/USD budget 和逐 Provider HTTP attempt ledger。 -- 当前 data-only head `20260830_0007` 按 ADR-0018 将观测 input 估算与 hard reservation 分离:无显式 input bound 时不生成 input reservation/reserved cost,actual usage 仍保存;显式 input/output 上界及由完整上界和价格派生的 reserved cost 超额继续 fail closed。本地完整门禁、当前 SQLite 迁移、修正 SHA `cb00924…` 的 real-Compose 9/9 与远程 CI 4/4 均通过。 +- data-only `20260830_0007` 按 ADR-0018 将观测 input 估算与 hard reservation 分离:无显式 input bound 时不生成 input reservation/reserved cost,actual usage 仍保存;显式 input/output 上界及由完整上界和价格派生的 reserved cost 超额继续 fail closed。本地完整门禁、当前 SQLite 迁移、修正 SHA `cb00924…` 的 real-Compose 9/9 与远程 CI 4/4 均通过。当前 head `20260830_0008` 按 ADR-0019 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并替换 Provider 类型 check 与远程配置 check,以加入显式 Chat Completions / Responses / Messages Adapter;不改评分或 protocol-v1。 - materialized counter 只作 ledger 投影;counter、policy/hash 或 Run override 漂移 fail closed。confirmed pre-send release 按 ADR-0011 不消耗未发送 HTTP retry。 - 有限 backlog、typed `429`、database not-before、question quantum、dispatch/failure 分离和跨 Model 公平排序。 - typed audit、Run audit、task history/latency、Provider metadata、credential 非秘密事件和前端治理状态;P2-06 实现 SHA `9a20676…` 另交付固定低基数 Prometheus exporter/八条规则、canonical retention archive/verify/reconcile/restore/delete、Worker DB-time progress/liveness 聚合与全日志源治理。 @@ -208,6 +208,7 @@ | P2-04 生命周期可靠性 | 可靠基础已交付 | retry/取消/恢复/dead-letter/终态重算及三个确定性 DB crash-seam 场景已通过完整 Compose acceptance;外部调用仍不保证 exactly-once | | P2-05 并发治理 | 切片已实现 | 四层 concurrency/RPM/TPM/lifetime budget、per-attempt ledger、backpressure、finite quantum、公平排序和完整性 fail-closed 已实现;真实 PG/capacity/acceptance/精确 SHA CI 候选门禁已通过 | | P2-05 observational overdraw 维护 | `completed` | ADR-0018 与 data-only `0007` 只重算 overdrawn 并保留 ledger/actual,active reservation 时拒绝;当前库迁移和本地门禁通过,最终 SHA `cb00924…` 的 run `33271095910` 4/4 成功 | +| Provider API 三协议维护 | `in_progress` | ADR-0019、Chat / Responses / Messages 显式 Adapter、`0008`、CLI/Web 与本地完整门禁已完成;等待 commit/push/exact-SHA CI,不改变 protocol-v1 或 P2-07 | | P2-06 可观测性 | `completed` | exporter/八规则、retention CLI、Worker DB-time progress、`0005`/13 表 importer 与全日志源治理已实现;`9a20676…` 已 push、PR #3、实现 run `33164609388` 4/4,clean capacity/9/9 acceptance 全绿;evidence-doc commit `ec29596…` 已 push且精确 SHA run `33165775037` 4/4 | | P2-07 验证与运维 | `planned` | ADR-0016、独立计划与工作日志已建立;功能尚未实现,后续从最小只读 verifier 开始,再开展数据库+keyring restore、Redis 重建、告警响应与完整失败矩阵 | @@ -285,7 +286,7 @@ 3. 实现离线缓存、Hash 清单、分片和验证失败诊断。 4. 构建无网络、最小权限、CPU/内存/时间/输出受限的代码沙箱。 5. 实现编译、运行、测试用例隔离和结构化错误分类。 -6. 增加分组指标、UI 筛选、测试夹具与端到端验证。 +6. 增加分组指标、UI 筛选、Run Detail 大规模进度热力图/live metrics、测试夹具与端到端验证(P3-06)。 7. 完成威胁建模、渗透测试范围和操作手册。 ### 验收标准 @@ -296,6 +297,7 @@ - 不可信代码在隔离环境运行,默认无网络,并强制资源与输出上限。 - 沙箱逃逸防线、超时、fork bomb、磁盘耗尽等关键威胁有自动化验证。 - 指标按任务/子集正确聚合,协议不兼容时不混排。 +- 12,032–20,000 题 Run Detail 以固定 512 题 block 和虚拟化可访问四态 grid 展示;后端同快照动态指标、乱序/并发/终态收敛及 nullable usage/cost 语义有回归。 ### 风险 @@ -312,8 +314,15 @@ ### 状态 `in_progress`(2026-08-27 开始)。MMLU-Pro 与 GPQA-Diamond 已有固定 revision/SHA、 -可复现转换、受限 real-Provider CLI 和证据派生分组报告切片;IFEval、正式 Plugin SDK、代码题模型、 -安全沙箱、分组 UI、完整数据卡/红队及全部阶段验收仍未完成。 +可复现转换、受限 real-Provider CLI 和证据派生分组报告切片。P3-06 的固定 512 题 absolute-position +block progress API、后端同快照 live metrics 与虚拟化四态 Run Detail 已完成;定向 backend/frontend +`37/32 passed`(Run Detail `20` + heatmap `12`),完整 backend `964 passed, 33 skipped`、frontend +`64 passed`,lint、Mock smoke、build、Compose config 和目标 198 题 Run 的三档宽度/键盘/Tooltip/console +实页验收通过。12,032/20,000 题边界为自动化虚拟化验证。初版 cursor 合同因没有数据库单调提交序列 +而在实现前废弃;最终实现 SHA [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) +已普通 push 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5),精确 SHA [Actions run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) +四个必需 job 全部成功,因此 P3-06 切片为 `completed`。IFEval、正式 Plugin SDK、代码题模型、安全沙箱、 +分组 UI、完整数据卡/红队及全部阶段验收仍未完成,所以 Phase 3 整体保持 `in_progress`。 ## 7. Phase 4:Judge、Arena 与长上下文 diff --git a/docs/SECURITY.md b/docs/SECURITY.md index 1b5045e..9b1f3f5 100644 --- a/docs/SECURITY.md +++ b/docs/SECURITY.md @@ -28,7 +28,7 @@ LLMBenchLab 当前开发基线是供个人开发者在受信任机器上使用 | API 客户端/浏览器 | 仅来自显式配置的本地前端;Web 表单会短暂持有待提交 Key | 无鉴权接口被滥用、Key/数据被读取或篡改、付费 Run 被启动 | | FastAPI API | 受信任的写入边界;接收 write-only Key 并持有部署 keyring | 可在加密前读取 Key,或替换 endpoint/envelope、泄漏 keyring | | Benchmark 作者 | 文件内容不可信,来源需人工判断 | 资源消耗、提示注入、答案污染、敏感数据外发 | -| OpenAI-compatible Provider | 地址和服务由用户信任 | 收集提示、伪造响应、消耗预算、返回恶意文本 | +| Chat / Responses / Messages Provider | 地址、协议选择和服务由用户信任 | 收集提示、伪造响应、消耗预算、返回恶意文本 | | 独立 Worker | 受信任的执行边界,持有数据库/Redis 凭据与部署 keyring,并按 source 解密或读取 Provider Key | 凭据/keyring 泄漏、越权读写全部评测事实、重复外部调用 | | PostgreSQL / Redis | 只在受控内部网络可达,数据卷由受信操作者保护 | 数据事实被篡改、任务被干扰、运行元数据泄漏或服务不可用 | | GitHub/CI/依赖源 | 外部供应链 | 依赖投毒、Actions 权限滥用、秘密泄漏 | @@ -90,7 +90,7 @@ Web stored 流程如下: 4. Model 读响应只返回凭据状态 `credential_source`/`has_api_key`,以及 environment 兼容模式所需的变量名称;`has_api_key=true` 仅表示 stored row 存在,environment 模式不会因进程里恰好有变量值而返回 true。API/Run/Response/OpenAPI 读 schema 都不含 `api_key`、ciphertext、nonce 或 keyring 数据。 5. Run 没有 credential reference/envelope 列。快照只保存 source、Model ID、远端模型和 endpoint;environment 模式另保存变量名。stored 模式的 Worker 按 `run.model_id` 读取 row,并用 `run.model_id + Run snapshot base_url origin` 做 AAD 解密,绝不以当前 Model Base URL 替代快照目标。 6. keyring 缺失/无效、未知 key id、row 缺失、密文篡改、跨 Model 或错误 origin 都会 fail closed,并在 Adapter 构造和任何 Provider 网络请求之前结束该 Run attempt。environment 变量缺失则产生安全的 `missing_api_key`,同样不会无凭据联网。 -7. 解密值只在内存中用于 `Authorization: Bearer ...`。Provider 响应在离开 Adapter 前递归脱敏:成功内容、raw usage 的对象键和所有 JSON 标量(包括数值/布尔/null)、token/status 数值、request ID、返回模型名、system fingerprint 和 finish reason 若精确包含当前 Key,都会替换为 `[REDACTED]`。 +7. 解密值只在内存中用于与显式协议匹配的秘密 header:Chat/Responses 使用 `Authorization: Bearer ...`,Messages 使用 `x-api-key`。Provider 响应在离开 Adapter 前递归脱敏:成功内容、raw usage 的对象键和所有 JSON 标量(包括数值/布尔/null)、token/status 数值、request ID、返回模型名、system fingerprint 和 finish reason 若精确包含当前 Key,都会替换为 `[REDACTED]`。 8. credential create/replace/delete/source switch 记录 `credential_changed`,origin/active-Run/恢复边界拒绝记录 `credential_rejected`,key ID/认证解密失败记录 `credential_decrypt_failed`。security audit 以数据库 UTC 记时,只保存 Model、source/action/reason 和安全 key ID,不保存 Key、origin 或 envelope;被回滚的拒绝以 server request ID 在独立短事务中幂等写入,但进程在响应前崩溃仍可能缺事件,因此不声称跨进程 exactly-once。 这里的保证限定于**不把 Key 或 Provider 对 Key 的回显从凭据数据流复制进公开字段或持久化证据**。create/PATCH 精确扫描 `ModelRead` 的所有字段和 Run snapshot 的 `model` 子投影;Adapter 递归扫描 Provider 返回证据。它不扫描整个 Run 中与 Model 无关的 Benchmark/Question 固定内容,也不声称用户数据与 Key 永远不会发生独立的字面巧合。 @@ -115,21 +115,21 @@ Web stored 流程如下: - 给每个环境和用途使用独立、最小权限 Key;能设置预算、允许模型或来源 IP 时应开启。怀疑泄漏时先在 Provider 侧吊销/轮换,再清理仓库历史、数据库副本、日志和 keyring 备份。 - 本地 `make dev` 的 API/Worker 可能继承同一 `.env`,但只有 environment 执行路径解析已登记变量;Vite 不会自动把非 `VITE_*` 变量打入客户端。更严格部署应分别注入:API/Worker 共享 keyring,legacy Provider env 只给 Worker,frontend/migrate 两者都不给。 - Worker 同时需要数据库与 Redis 连接能力。生产设计应拆分 API/Worker/迁移数据库角色,限制 Worker 只能访问所需表和操作,并用独立 Redis ACL、短期凭据及轮换流程代替共享本地凭据。 -- API/Worker 日志、崩溃转储、进程列表和诊断端点都属于秘密边界。不得记录环境、DSN、Authorization header、请求体、解密值或原始 Provider 请求。 +- API/Worker 日志、崩溃转储、进程列表和诊断端点都属于秘密边界。不得记录环境、DSN、`Authorization`/`x-api-key` header、请求体、解密值或原始 Provider 请求。 - at-least-once 只保证任务最终可再次处理;逐 attempt ledger 防止同一本地逻辑 attempt 重复结算,但 `send_started` 后崩溃仍可能留下远端幽灵请求。若 Provider 已响应而本地 Response 提交前崩溃,接管 Worker 可能重复上游调用和计费;本地 `(run_id, question_id)` 唯一约束、保守 consumed 或释放 permit 都不能消除这一外部副作用或替代真实账单。 ### 3.4 可信本地 CLI 秘密边界 - CLI 只适合受信任的交互式机器。它将读取到的 Key 临时放入 Run 快照所引用的进程环境变量,使现有 Adapter 能复用同一秘密接口;上下文结束时恢复原值或删除临时值。拥有同一用户权限、调试/进程转储能力的程序仍可能读取内存或环境。 -- 模型发现 `GET /models`、最小 Chat canary 和正式题目请求都使用同一个 Key。任何发现到的模型 ID 若反射该 Key,预检立即失败;canary 若明确返回不同于请求目标的模型名也失败。CLI 只保存脱敏 preflight 元数据,不保存请求 header 或 Key;完整 Provider access log 不受本应用控制。 -- 在 canary 前,CLI 打印 Provider host、目标模型、题数、剩余 Run attempts 和最大 Chat HTTP 尝试数,并要求输入 `RUN`;上界按 `(缺失题数 × 剩余 Run attempts + 1 个 canary) × 3` 包含 HTTP retries,但仍不是 Token/金额预算。`--yes` 只用于操作者明确授权的非交互运行,不是预算、速率限制或安全审批。该路径创建 `legacy_unmanaged` Run,不继承 managed Web/API policy。 +- 模型发现 `GET /models`、按显式 Chat Completions / Responses / Messages 协议构造的最小 canary 和正式题目请求都使用同一个 Key。discovery 同样按显式协议鉴权:Chat/Responses 使用 `Authorization: Bearer`,Messages 使用 `x-api-key` 与 `anthropic-version`;Messages 的 `has_more/last_id` 只通过 `after_id` 跟进,并受累计 100 页、60 秒 wall-clock、2 MiB、10,000 个模型 ID 与重复 cursor 门禁保护。任何发现到的模型 ID 若反射该 Key,预检立即失败;canary 若明确返回不同于请求目标的模型名也失败。CLI 只保存脱敏 preflight 元数据,不保存请求 header 或 Key;完整 Provider access log 不受本应用控制。 +- 在 canary 前,CLI 打印 Provider host、显式协议、目标模型、题数、剩余 failed-attempt 预算和最大 Provider HTTP 尝试数,并要求输入 `RUN`;剩余预算严格为 `max_attempts - failed_attempt_count`,不把 cooperative yield 算作失败,上界按 `(缺失题数 × 剩余 failed-attempt 预算 + 1 个 canary) × 3` 包含 HTTP retries,但仍不是 Token/金额预算。`--yes` 只用于操作者明确授权的非交互运行,不是预算、速率限制或安全审批。该路径创建 `legacy_unmanaged` Run,不继承 managed Web/API policy。 - `resume` 会再次读取 Key、确认并发送 canary,然后只处理本地缺失 Response。初次 canary 会固化到新 Run 快照,但 resume canary 当前不会追加为独立 audit event;逐题 request ID、返回模型名、system fingerprint、finish reason 与 HTTP attempt count 已安全归一化持久化。远端调用不是 exactly-once,恢复前应同时检查 Provider 账单与是否仍允许继续。 - `report` 和 `prepare` 不需要 Provider Key;不得为了方便给这两条命令或 CI 注入真实凭据。 - 正式 CLI 必须在常规 API/Worker 停止后独占数据库。代码只能拒绝已有 `running` Run,不能识别空闲 Worker;若空闲 Worker 抢到新 `pending` Run,可能在错误的 Key/进程边界发起调用。 ## 4. 日志与错误脱敏 -当前应用日志和 OpenAI-compatible Adapter 的错误处理会: +当前应用日志和三类远程 Adapter 的错误处理会: - 只有显式登记的 LLMBenchLab 应用 logger 可输出 literal message 与 structured extra;字段除了 allowlist,还分别受固定 event/result/error/component 枚举、canonical UUID、Redis stream ID、HTTP method/route template 和有限数值合同约束。非法身份字段被省略,未知 method/path/code 只映射到固定 `unsupported`,异常类名只保留固定错误族,不反射原值。API 不记录原始查询串或请求体。 - `configure_logging` 把已知进程内 `Uvicorn`、`httpx`/`httpcore`、SQLAlchemy 与 Redis client logger 统一路由到同一 sanitizer:第三方动态 logger 名、message、exception 与伪造 structured extra 不进入 JSON;原始 `uvicorn.access` handler 被禁用,由应用 middleware 的 route-template 事件替代。 @@ -138,11 +138,11 @@ Web stored 流程如下: - 用 `[REDACTED]` 替换当前 API Key 的精确值。 - 识别常见 Bearer、Authorization、API key、token 和 secret 表达形式。 - 把上游错误折叠为单行并截断到 500 字符。 -- 不保存请求头、完整上游请求或响应对象;最终 `httpx.TransportError` 在 `except` 边界外转成安全 `AdapterError`,其 `__cause__`、`__context__` 和格式化 traceback 不保留可达 request/Authorization;Run 只持久化分类后的错误类型与可读消息。 +- 不保存请求头、完整上游请求或响应对象;最终 `httpx.TransportError`,以及三协议 malformed JSON/SSE 或 oversized 响应,都在捕获原始异常的 `except` 边界外转成安全 `AdapterError`,其 `__cause__`、`__context__` 和格式化 traceback 不保留可达 request、秘密 header 或原始 Provider bytes;Run 只持久化分类后的错误类型与可读消息。 - 不记录原始 SSE 行、事件或单个 delta;只在收到完整终止信号后使用聚合结果。 -- 对 Chat 成功内容、raw usage 的对象键和全部 JSON 标量,以及 provider request ID、返回模型名、system fingerprint、finish reason 递归执行当前 Key 的精确替换,再允许其进入后续 Runner/preflight 边界。 +- 对 Chat/Responses/Messages 成功内容、raw usage 的对象键和全部 JSON 标量,以及 provider request ID、返回模型名、system fingerprint、finish reason 递归执行当前 Key 的精确替换,再允许其进入后续 Runner/preflight 边界。 - EvaluationResponse/API/report 只保存经过长度、字符和 Key 反射检查的 provider request ID、returned model、system fingerprint、finish reason 与整数 HTTP attempt count;不保存任意 raw usage object,被拒字符串归一化为 `null`。 -- SSE `delta.content` 先完整聚合、再执行当前 Key 的精确替换,覆盖 Key 被分到多个 delta 的情况;这仍不是通用 DLP。 +- 三类 SSE 的文本 delta 先按各自 typed event 完整聚合、再执行当前 Key 的精确替换,覆盖 Key 被分到多个 delta 的情况;这仍不是通用 DLP。 - 把 `finish_reason="length"` 的空输出或无法解析最终答案的输出归类为稳定的 `output_truncated`;这只是安全的诊断分类,不回显完整 Provider 响应对象或请求头。 仍需遵守以下规则: @@ -171,18 +171,19 @@ allowlist 只减少聚合面,不等于 artifact 无敏感性。raw child 仍 ## 5. `base_url` 与 SSRF -`openai_compatible` 允许用户配置 `base_url`,Adapter 会调用其 Chat Completions 路径。这是当前最高优先级的公开部署阻断项。 +三个远程 Provider 类型都允许用户配置 `base_url`,Adapter 会根据显式类型调用 Chat Completions、OpenAI Responses 或 Anthropic Messages 路径。这是当前最高优先级的公开部署阻断项。 ### 5.1 已实现的有限校验 - URL 必须是含 hostname 的绝对地址;远端 Provider 必须使用 `https://`,只有 `localhost` 或字面量 loopback IP 可使用 `http://`。 - 禁止内嵌 username/password、query 与 fragment,并去除末尾 `/`。 - Adapter 使用有限连接/读取超时及有限重试;普通配置型 4xx 不会无限重试。 -- Provider 请求禁用 HTTP redirect。CLI 从兼容根地址推导同路径的 `/models` 和 `/chat/completions`;若传入完整 `/chat/completions`,模型发现回到其同级 `/models`。 -- 模型发现与 Chat 请求都发送 `Accept-Encoding: identity`,并在读取正文前拒绝压缩响应;发现体流式限制为 2 MiB,Chat 普通 JSON 成功体限制为 4 MiB,SSE 累计 wire/单事件/最终 content 分别限制为 64 MiB/1 MiB/4 MiB,非 2xx 错误体限制为 64 KiB,超限即中止而不保留整段正文。 +- 除既有 retryable HTTP/transport 分类外,只重试显式白名单的 typed transient error:Responses 的 rate-limit/server error,以及 Messages 的 `rate_limit_error`、`api_error`、`overloaded_error`、`timeout_error`;Messages 的 HTTP `529` 也进入冻结的 retryable status 集。未知流内错误 fail closed,每个重试 HTTP attempt 都独立结算 ledger。 +- Provider 请求禁用 HTTP redirect。CLI 从根地址或三个已知完整 endpoint 推导同级 `/models`;discovery 与生成都使用显式 Adapter 对应的认证 header,生成 endpoint 也由该类型决定。匹配的完整 endpoint 原样使用,其他已知协议后缀在发送 Key 前拒绝。系统不按模型名/URL猜测协议,也不在失败后跨协议 fallback。 +- 模型发现与三类远程请求都发送 `Accept-Encoding: identity`,并在读取正文前拒绝压缩响应;discovery 聚合限制为 2 MiB/10,000 个模型 ID,Messages 分页另限制累计 100 页/60 秒 wall-clock,并拒绝重复或 `has_more=true` 时缺失 `last_id` 的继续游标。普通 JSON 成功体限制为 4 MiB,SSE 累计 wire/单事件/最终 content 分别限制为 64 MiB/1 MiB/4 MiB,非 2xx 错误体限制为 64 KiB,超限即中止而不保留整段正文。Chat、Responses、Messages 分别要求 `[DONE]`、`response.completed`、`message_stop` 终止事件,截断流不会保存部分答案。 - stored credential 的 AES-GCM AAD 绑定规范化 origin;改变 scheme/host/非默认 port 时必须重输 Key,active Run 期间禁止 endpoint/credential 更新,Worker 解密只接受 Run snapshot origin。这阻止 Key 被静默换绑到另一 origin,但不判断首次配置的 HTTPS 目标是否可信。 -这些校验降低凭据经远端明文 HTTP 外泄和大/压缩响应耗尽内存的风险,但不保证目标安全。loopback HTTP 是明确支持的本地推理路径;RFC 1918 私网、IPv6 link-local、云元数据和其他敏感目标仍可能通过 HTTPS 被访问,DNS rebinding 也未防御。CORS 对服务端 SSRF 没有帮助。模型发现另有 10,000 个模型 ID 上限和 2 MiB 正文上限;Chat 普通 JSON 成功体为 4 MiB,SSE wire/单事件/聚合 content 为 64 MiB/1 MiB/4 MiB,错误体为 64 KiB。 +这些校验降低凭据经远端明文 HTTP 外泄和大/压缩响应耗尽内存的风险,但不保证目标安全。loopback HTTP 是明确支持的本地推理路径;RFC 1918 私网、IPv6 link-local、云元数据和其他敏感目标仍可能通过 HTTPS 被访问,DNS rebinding 也未防御。CORS 对服务端 SSRF 没有帮助。模型发现另有 10,000 个模型 ID 上限和 2 MiB 正文上限;远程协议普通 JSON 成功体为 4 MiB,SSE wire/单事件/聚合 content 为 64 MiB/1 MiB/4 MiB,错误体为 64 KiB。Anthropic Messages 的 `x-api-key` 与 Chat/Responses 的 `Authorization` 同属秘密 header,不得进入日志、异常或证据。 ### 5.2 MVP 使用要求 @@ -191,7 +192,7 @@ allowlist 只减少聚合面,不等于 artifact 无敏感性。raw child 仍 - 远端只使用 HTTPS 并验证证书;HTTP 只用于确认由本机操作者控制的 loopback 推理服务,不使用把凭据写进 URL 的反向代理。 - 在主机防火墙、容器网络或出站代理层拒绝云元数据、loopback、link-local 与内网网段;若必须访问本地推理服务,应为它建立精确的目标例外。 - 创建 Run 前确认 Benchmark 内容允许发送给目标 Provider。 -- 把 Web 的 Demo `256/60s`、MMLU-Pro direct `1024/180s`、official CoT `4000/300s`、GPQA-Diamond `8192/600s` 只当作可编辑建议。更高输出预算和更长读取超时可能增加费用;“由 Provider 决定”提交 `max_tokens:null`,只表示请求中省略该字段,并不取消 Provider 自身限制或平台费用风险。 +- 把 Web 的 Demo `256/60s`、MMLU-Pro direct `1024/180s`、official CoT `4000/300s`、GPQA-Diamond `8192/600s` 只当作可编辑建议。更高输出预算和更长读取超时可能增加费用;Chat/Responses 的“由 Provider 决定”提交 `max_tokens:null`,只表示请求中省略对应输出字段,并不取消 Provider 自身限制或平台费用风险。Messages 不支持该选择,必须提交有限正整数 `max_tokens`。 - 把 `read_timeout_seconds` 理解为 LLMBenchLab 等待下一批 Provider 字节的空闲窗口,不是模型生成总时限;它不会配置 Worker→Provider 链路上的 Cloudflare、Caddy 或其他 Gateway。真实评测前必须另行核对代理的 SSE flush、buffering、空闲与绝对总时长。 - 真实 CLI 先使用小额 `--limit` 验证;只有确认模型发现/canary、响应解析、失败率和账单后才考虑 `--full`。不要把 CLI 的请求尝试上界误当作 Token 或金额硬上限。 @@ -289,7 +290,7 @@ Redis 仅可置于受控内部网络。当前本地 Compose 使用 AOF、无 ACL - 退出码 `3` 表示 `COMMIT` 已确认、但提交后验证或报告失败;导入已经提交。**禁止盲目重试或把它描述为回滚**,应先只读核验目标,必要时从已验证备份执行人工恢复。 - 工具是单向导入,不提供 PostgreSQL→SQLite 自动回迁。schema downgrade 也不等于数据平台回滚。 -当前 head `20260830_0007` 是 data-only 修复:它不改 schema、never-delete ledger、Provider actual usage、Response、audit 或 Run 终态,只依据冻结的 `evaluation_runs.input_token_reservation` 重算 `governance_scopes.overdrawn`。upgrade/downgrade 都会在任何更新前拒绝 `reserved/send_started` active reservation;downgrade 仅恢复旧派生谓词。`20260829_0006` 仍只补齐 canonical `0004` 的三个索引;兼容入口只接受 canonical schema,或缺失项为这三个索引的非空子集,以便 SQLite 非事务 DDL 中断后安全重入。新近成为 historical 的 PostgreSQL `0005/0006` 必须按各自规则通过 metadata drift 校验;任何额外差异仍 fail closed。SQLite preflight 与 repair migration 都会在恢复 single-active 唯一索引前拒绝多条 active policy。`0006 → 0005` 不删除对象;`20260828_0005 → 0004` 会移除 Worker progress,guard 在第一条 DDL 前拒绝任何 generation 行;`20260827_0004 → 0003` 会移除治理/ledger/audit/Provider metadata并拒绝任何相关事实。正常运行后的数据库不可原地 downgrade;应优先向前修复,或恢复经核验的旧备份并保留只读证据。只有隔离空数据库用于双方向往返验证;详见 [OPERATIONS.md](OPERATIONS.md)。 +当前 head `20260830_0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并同时替换 Provider 类型 check 与远程配置 check;它不改写旧 `mock`/`openai_compatible` 行。数据库中存在 `openai_responses` 或 `anthropic_messages` Model 时,`0008 → 0007` 在第一条 DDL 前拒绝,避免静默丢失新配置。其上游 `20260830_0007` 是 data-only 修复:它不改 schema、never-delete ledger、Provider actual usage、Response、audit 或 Run 终态,只依据冻结的 `evaluation_runs.input_token_reservation` 重算 `governance_scopes.overdrawn`。`0007` upgrade/downgrade 都会在任何更新前拒绝 `reserved/send_started` active reservation;downgrade 仅恢复旧派生谓词。`20260829_0006` 仍只补齐 canonical `0004` 的三个索引;兼容入口只接受 canonical schema,或缺失项为这三个索引的非空子集,以便 SQLite 非事务 DDL 中断后安全重入。新近成为 historical 的 PostgreSQL `0005/0006/0007` 必须按各自规则通过 metadata drift 校验;任何额外差异仍 fail closed。SQLite preflight 与 repair migration 都会在恢复 single-active 唯一索引前拒绝多条 active policy。`0006 → 0005` 不删除对象;`20260828_0005 → 0004` 会移除 Worker progress,guard 在第一条 DDL 前拒绝任何 generation 行;`20260827_0004 → 0003` 会移除治理/ledger/audit/Provider metadata并拒绝任何相关事实。正常运行后的数据库不可原地 downgrade;应优先向前修复,或恢复经核验的旧备份并保留只读证据。只有隔离空数据库用于双方向往返验证;详见 [OPERATIONS.md](OPERATIONS.md)。 ## 10. 依赖与构建供应链 diff --git a/docs/TESTING.md b/docs/TESTING.md index 075569a..32756e4 100644 --- a/docs/TESTING.md +++ b/docs/TESTING.md @@ -2,7 +2,7 @@ ## 1. 测试原则 -LLMBenchLab 的默认验收路径必须完全离线、可重复且不产生模型费用。自动化测试和 CI 只能使用 `MockModelAdapter`,或使用进程内 `httpx.MockTransport` 验证 OpenAI-compatible 协议;不得访问真实 Provider,不要求 API Key,不把“本机碰巧可用”当作通过证据。 +LLMBenchLab 的默认验收路径必须完全离线、可重复且不产生模型费用。自动化测试和 CI 只能使用 `MockModelAdapter`,或使用进程内 `httpx.MockTransport` 验证 Chat Completions、OpenAI Responses 与 Anthropic Messages 协议;不得访问真实 Provider,不要求 API Key,不把“本机碰巧可用”当作通过证据。 每次报告测试结果时必须写明实际命令、通过/失败、测试数量、失败原因和未运行项。没有执行的 Docker、浏览器或联网验证不能写成通过。 @@ -15,9 +15,9 @@ LLMBenchLab 的默认验收路径必须完全离线、可重复且不产生模 | API 与进程边界测试 | FastAPI Schema、状态码、秘密安全、Run 提交与 API/Worker 分离 | `backend/tests/test_api.py`、`test_run_dispatch.py`、`test_process_boundaries.py` | 禁止 | | Governance / audit 测试 | policy/ledger 完整性、typed history/audit、Provider/credential evidence,以及 canonical archive/离线 verify/精确 reconcile/restore/delete | `test_governance.py`、`test_audit_api.py`、`test_audit_archive.py`、`test_audit_retention.py`、`test_audit_retention_cli.py` | 禁止;SQLite/Mock/fixture;真实 PG retention 只连测试库 | | 租约与 Worker 测试 | 条件领取、fencing、心跳、取消、幂等 Response、重试/恢复、队列 ACK,以及 generation 级 DB-time scan/claim/lease-heartbeat/progress/stale | `test_run_leases.py`、`test_evaluation_runner_reliability.py`、`test_worker.py`、`test_worker_progress.py`、`test_worker_probe.py` | 禁止;SQLite/假队列 | -| Metrics / alert / logging 测试 | 固定 Prometheus exposition、snapshot/hard cap/single-flight/取消竞态、精确八规则/Runbook、全部生产 logger source/第三方 handler 治理,以及组合开发启动器的日志分流/退出传播 | `test_prometheus_exporter.py`、`test_prometheus_alert_rules.py`、`test_logging.py`、`test_logging_sources.py`、`test_dev_script.py` | 禁止;SQLite/假队列/标准库 JSON/假子进程 | +| Metrics / alert / logging / 启动器测试 | 固定 Prometheus exposition、snapshot/hard cap/single-flight/取消竞态、精确八规则/Runbook、全部生产 logger source/第三方 handler 治理,以及本地/Compose 多 Worker启动、日志分流、环境优先级、gauges 轮询与退出传播 | `test_prometheus_exporter.py`、`test_prometheus_alert_rules.py`、`test_logging.py`、`test_logging_sources.py`、`test_dev_script.py`、`test_compose_up_script.py` | 禁止;SQLite/假队列/标准库 JSON/假子进程/假 Docker | | 迁移与导入回归 | SQLite/真实 PostgreSQL migration、`0005` populated downgrade refusal/空库往返,以及 13 表 SQLite→PostgreSQL 原子导入 | `test_migrations.py`、`test_sqlite_postgres_import.py` | 导入/本地部分禁止;真实 PostgreSQL 用 `integration` marker | -| 真实基础设施集成 | PostgreSQL 并发领取/取消竞态、Redis Streams PEL/ACK/重复投递 | `backend/tests/integration/` 与 importer 的 `integration` 用例 | 只连接显式测试 PostgreSQL/Redis;禁止 Provider | +| 真实基础设施集成 | PostgreSQL 并发领取/取消竞态、不同 Benchmark Run 被不同 owner 领取且同一 Run lease 唯一、Redis Streams PEL/ACK/重复投递 | `backend/tests/integration/` 与 importer 的 `integration` 用例 | 只连接显式测试 PostgreSQL/Redis;禁止 Provider | | Mock 端到端 Smoke | API 提交 pending Run → 独立 WorkerService → Responses → Leaderboard/Metrics | `backend/tests/test_smoke.py`,marker 为 `smoke` | 禁止 | | Compose 故障验收 | 六服务拓扑、双 Worker、进程/Redis/lease 故障、治理/ledger、取消、重复消息、Worker expected count 与 `0005` 安全回滚 | `scripts/phase2_acceptance.py` / `make phase2-acceptance` | 只拉取/构建基础镜像;模型执行始终为离线 Mock | | Mock 容量基线 | 真实 PostgreSQL 16/Redis 7、全有限 policy、1/2 Worker、精确 `202/429` backlog、cooperative quantum、跨 Model 公平性、lease/Redis/重复通知故障及 DB/queue/ledger/audit 对账 | `scripts/phase2_capacity.py` / `make phase2-capacity` | 只拉取/构建基础镜像;模型执行始终为离线 Mock | @@ -62,12 +62,26 @@ npm ci make test # 后端 pytest + 前端 Vitest make lint # Ruff + ESLint + TypeScript 类型检查 make smoke # 只跑完全离线的后端垂直切片 +make dev-multi # PostgreSQL/Redis Compose 默认双 Worker,并验证 gauges make phase2-acceptance # 隔离的真实 Compose 九场景可靠性验收 make phase2-capacity # PostgreSQL 16/Redis 7/1→2 Worker 的 Mock 容量基线 make phase2-slo # clean commit 上固定 v2 四-cell、1 warm-up + 5 measured 的单机资格 make format # 运行项目约定的格式化器 ``` +多 Worker启动层的纯离线目标回归为: + +```bash +cd backend +uv run pytest tests/test_dev_script.py tests/test_compose_up_script.py +``` + +它用假子进程和假 Docker 覆盖 `1..32` 校验、SQLite fail-fast、独立日志、信号清理、 +Compose scale/expected 同步与 30 次有界 gauges 超时,不会启动真实 Provider。租约并发 +断言还必须在已迁移的显式 PostgreSQL 测试库上运行 +`test_postgres_leases.py`;没有 `LLMBENCHLAB_TEST_POSTGRES_URL` 时该层按设计跳过,不能 +把 skip 写成真实 PostgreSQL 通过。 + 格式化会修改文件;仅检查时使用下面的直接命令。提交 PR 前还应执行前端 production build。 ### 4.1 后端 @@ -148,7 +162,7 @@ npm run build docker compose config --quiet ``` -这只验证配置渲染,不证明镜像已构建、migration 已完成或服务健康。普通启动验证使用 `make docker-up`、检查 `/api/v1/live`、`/health`、`/ready` 和 Worker probe,最后执行 `make docker-down`。真实故障验收使用 `make phase2-acceptance`,容量基线使用 `make phase2-capacity`;两者都创建唯一 Compose project、随机 loopback 端口和隔离卷,失败路径也执行精确 `down -v` 并检查无项目残留。不得对日常项目名手工套用其清理命令。 +这只验证配置渲染,不证明镜像已构建、migration 已完成或服务健康。普通启动验证使用 `make docker-up` / `make dev-multi`;标准包装器默认两个 Worker,先拒绝 stale generation 冒充本轮新增 Worker scan,并把 exited replica 纳入缩容方向判断;只有 `/api/v1/tasks/metrics` 收敛到 `expected/registered/live/stalled/shortfall=2/2/2/0/0` 才成功。随后检查 `/api/v1/live`、`/health`、`/ready` 和 Worker probe,最后执行 `make docker-down`。真实故障验收使用 `make phase2-acceptance`,容量基线使用 `make phase2-capacity`;两者都创建唯一 Compose project、随机 loopback 端口和隔离卷,失败路径也执行精确 `down -v` 并检查无项目残留。不得对日常项目名手工套用其清理命令。 ## 5. 后端测试覆盖 @@ -165,22 +179,24 @@ docker compose config --quiet ### 5.2 Adapter 单元测试 -`test_adapters.py` 覆盖: +`test_adapters.py` 覆盖 Mock 与既有 Chat transport;`test_provider_protocol_adapters.py` 覆盖三协议 request/response 合同,`test_provider_protocol_plumbing.py` 覆盖 API/Run/Runner/CLI、typed retry 与 attempt ledger 联动: - Mock 输出、Token、延迟、request ID 可预测且不进行 I/O。 - Mock 可注入分类错误,用于验证单题故障隔离;latency/Token/usage shape 等确定性本地配置在 reserve 前全部验证,非法配置断言治理 hook 零调用,成功与模拟 Provider error 断言 reserve→mark→actual/conservative finish 顺序。 -- OpenAI-compatible 的 Chat Completions URL、messages、`Accept: text/event-stream`、`stream:true`、`stream_options.include_usage:true` 与 `temperature/top_p/max_tokens/seed`;数字 `max_tokens` 原样发送,`null` 时请求体完全省略该字段。 +- Chat Completions 的 URL、messages、`Accept: text/event-stream`、`stream:true`、`stream_options.include_usage:true` 与 `temperature/top_p/max_tokens/seed`;数字 `max_tokens` 原样发送,`null` 时请求体完全省略该字段。 +- Responses 的 `/responses`、Bearer header、`input`、`max_output_tokens`、普通 JSON `output` 与 typed SSE `response.output_text.delta`/`response.completed`;Messages 的 `/messages`、`x-api-key`/`anthropic-version`、顶层 system、有限 `max_tokens`、普通 JSON `content[].text` 与 typed SSE `content_block_delta`/`message_stop`。 +- 根 URL 与匹配完整 endpoint 均正确解析,已知跨协议后缀、Responses/Messages 非空 seed、Messages `temperature>1` 和 `max_tokens:null` 在 transport 零调用时稳定拒绝;请求/Model 默认均省略时,Responses/Messages 的 `temperature/top_p/seed` 快照为 `null` 且 Provider payload 不包含这些字段。 - usage 缺失时 Token 字段为 `null`。 -- 429、选定 5xx、网络超时的有限指数退避,以及普通 4xx 不重试。 -- 每个 OpenAI-compatible HTTP retry 都经过 request-local reserve→mark-send-started→finish hook;pre-send 失败 release,usage 完整 actual,transport/usage 缺失 conservative,settlement unknown 停止后续外发。 +- 429、选定 5xx、网络超时的有限指数退避,以及普通 4xx 不重试;Responses 的 rate-limit/server typed error 与 Messages 的 `rate_limit_error`、`api_error`、`overloaded_error`、`timeout_error` 在 JSON/SSE 路径有限重试,Messages HTTP `529` 写入 Run retry snapshot,未知 typed error fail closed。 +- 每个远程协议 HTTP retry 都经过 request-local reserve→mark-send-started→finish hook;pre-send 失败 release,usage 完整 actual,transport/usage 缺失 conservative,settlement unknown 停止后续外发。 - 远端 HTTPS 强制、loopback HTTP 例外,以及在发送 Key 前拒绝远端明文 HTTP。 - 真 SSE 的任意网络/UTF-8 拆包与合包、LF/CRLF/独立 CR、BOM、comment ping、多 `data` 行、role/null delta、reasoning/timings 扩展忽略、finish 后继续读取可选 usage-only 块至 `[DONE]`,以及普通 JSON fallback。 -- 非法 UTF-8/JSON/字段、SSE 内 Provider error、缺失 `[DONE]`、transport 中断的有限重试,以及同一 Adapter 并发请求的 request-local 状态隔离。 -- 模型发现与 Chat 的 `Accept-Encoding: identity`、读取前拒绝压缩,以及发现 2 MiB、Chat JSON 4 MiB/错误 64 KiB、SSE wire 64 MiB/单事件 1 MiB/聚合 content 4 MiB 的上限。 +- 非法 UTF-8/JSON/字段、SSE 内 Provider error、分别缺失 `[DONE]`/`response.completed`/`message_stop`、transport 中断的有限重试,以及同一 Adapter 并发请求的 request-local 状态隔离。 +- 模型发现按显式协议鉴权:Chat/Responses 使用 `Authorization: Bearer`,Messages 使用 `x-api-key` 与 `anthropic-version`;Messages 的 `has_more/last_id` 通过有界 `after_id` 分页聚合,并断言累计 100 页、60 秒 wall-clock、10,000 项、2 MiB、缺失/重复 cursor 边界 fail closed。发现与三类生成请求都使用 `Accept-Encoding: identity`、读取前拒绝压缩,并覆盖 JSON 4 MiB/错误 64 KiB、SSE wire 64 MiB/单事件 1 MiB/聚合 content 4 MiB 的上限。 - legacy Key 环境变量缺失、write-only direct `SecretStr`、空 Provider 回答、非法配置与错误脱敏;`finish_reason="length"` 的空输出分类为 `output_truncated`,普通空输出仍为 `empty_response`;成功内容、raw usage 键/字符串值、request ID、返回模型名、fingerprint 和 finish reason 的当前 Key 精确替换,包括 Key 横跨 SSE delta 的聚合后替换。 -- 最终 `httpx.TransportError` 的安全 `AdapterError` 同时断言 `__cause__ is None`、`__context__ is None`,格式化 traceback 不含带 Authorization request 中的 canary Key。 +- 最终 transport、malformed JSON/SSE 或 oversized 响应的安全 `AdapterError` 同时断言 `__cause__ is None`、`__context__ is None`,格式化 traceback 不含带 `Authorization`/`x-api-key` request 中的 canary Key 或原始 Provider bytes。 -OpenAI-compatible 测试只给进程内 transport 使用虚构 token;不得把测试地址改为真实域名。 +远程协议测试只给进程内 transport 使用虚构 token;不得把测试地址改为真实域名。 ### 5.3 Dataset Loader 单元测试 @@ -202,9 +218,9 @@ OpenAI-compatible 测试只给进程内 transport 使用虚构 token;不得把 - GPQA-Diamond 内层 CSV Hash、198 行约束、逐 Record ID 确定性选项重排、seed/domain 筛选,以及不携带作者/解释字段; - 输出 ZIP 再由普通 dataset-v1 Loader round-trip 校验。 -`test_provider_preflight.py` 只使用 `httpx.MockTransport`,覆盖 `/v1` 与完整 `/chat/completions` 的 `/models` 推导、远端 HTTP 拒绝、identity-only/2 MiB 发现响应、压缩体读取前拒绝、认证错误脱敏、发现模型 ID 反射当前 Key 时失败、唯一/多模型选择、最小 Chat canary 使用同一流式 payload/响应 Adapter、finish reason 脱敏,以及 canary 明确返回不同模型时失败。它不能证明任何真实 Provider 兼容,也不应改为读取开发者环境 Key。 +`test_provider_preflight.py` 只使用 `httpx.MockTransport`,覆盖 `/v1` 与完整 `/chat/completions`、`/responses`、`/messages` 的 `/models` 推导、远端 HTTP 拒绝、identity-only/2 MiB 发现响应、压缩体读取前拒绝、认证错误脱敏、发现模型 ID 反射当前 Key 时失败、唯一/多模型选择、按显式协议执行最小 canary、finish reason 脱敏,以及 canary 明确返回不同模型时失败。它不能证明任何真实 Provider 兼容,也不应改为读取开发者环境 Key。 -`test_evaluation_cli.py` 只做离线编排,覆盖无 `--api-key`、环境/隐藏输入生命周期、确认口令、含 HTTP retries 与剩余 Run attempts 的请求上界、profile 默认值、active Run 早拒绝、Run 创建/恢复/报告顺序、过期 incomplete lease 的 fenced reclaim 和 Key 值不持久化。它验证的是本地控制流,不证明真实 Provider 或操作系统级独占;人工 runbook 仍必须先停常规 API/Worker。 +`test_evaluation_cli.py` 只做离线编排,覆盖无 `--api-key`、环境/隐藏输入生命周期、确认口令、含 HTTP retries 与 `max_attempts - failed_attempt_count` 剩余预算的请求上界、profile/协议默认值、active Run 早拒绝、Run 创建/恢复/报告顺序、过期 incomplete lease 的 fenced reclaim 和 Key 值不持久化。它验证的是本地控制流,不证明真实 Provider 或操作系统级独占;人工 runbook 仍必须先停常规 API/Worker。 `test_run_report.py` 使用临时数据库覆盖分页导出全部 Response、非重叠分组、三文件内容、目标拒绝覆盖、文件权限与秘密脱敏。回归还构造 failed Run 的陈旧汇总字段,验证 `summary.metrics` 从计划题与 Responses 派生、`metrics_provenance` 标出漂移,并与 groups/responses 保持同一口径。报告含题目和 raw response,因此测试 fixture 必须完全虚构。 @@ -253,17 +269,30 @@ SQLite 测试适合快速验证状态机和兼容路径;跨连接并发保证 - SQLite 并发测试断言 Model PATCH 与 Run create 在读取 Model 前以 `BEGIN IMMEDIATE` 串行化;PostgreSQL integration 断言两条路径共用 Model row `FOR UPDATE` 锁。SQLite 竞争期间允许请求短暂等待,这仍只是低并发本地模式;生产/并发评测门禁使用 PostgreSQL。 - SQLAlchemy 基本 CRUD 与外键/Schema 基线。 - Run 创建 `202`、取消、轮询、逐题证据、汇总和排行榜;API 提交不在进程内执行 Adapter。生成边界测试覆盖兼容默认 `max_tokens=256`、显式 `null`、数字上限 `131072`、读取超时默认 `60s`/上限 `1800s`,并断言最终 generation 与 `execution.timeouts_seconds.read` 快照。 +- Response 列表的 Run-wide usage summary 覆盖零 Response、完整 usage、部分/非对称 usage、合法零 Token 和跨页一致性;输入/输出已知小计与非 `null` 计数各自独立,部分 usage 不会回填 protocol-v1 的精确 Run Token。 +- P3-06 progress index/block 回归必须覆盖:固定 `block_size=512`;全部计划 block(含空 block)升序 count;block absolute position 的稀疏/乱序完成;`error_type` 优先于 score、`score==1` 通过、其余普通答错;没有 Response 的 planned position 由客户端视为 `not_run`。index 的 completed/correct/error、三项 protocol-v1 指标、平均延迟及 known Token/cost coverage 必须与 block counts 在同一读取快照派生。 +- progress 竞态回归必须在 index→block 之间插入新 Response,证明 payload 与下一 index 最终收敛、不漏不重;覆盖 0/1/12,032/20,000 题边界、运行中和 completed/cancelled/failed 部分证据、nullable latency/usage/cost、负 block FastAPI 422、越界 typed `422 progress_block_out_of_range`、Run 404、两个接口 `Cache-Control: no-store` 及显式 OpenAPI Schema。相同语义需在快速 SQLite 与真实 PostgreSQL 读取快照用例验证。 +- progress cell 白名单必须精确限制为 `position/outcome/score/latency_ms/input_tokens/output_tokens/estimated_cost/error_type`;marker 测试必须证明 Question/Response ID、external ID、prompt/choices/raw/parsed/reference、error message、Provider request/model/fingerprint/finish reason、raw usage 和其他 Provider metadata 均不可表示或反射。 - Web stored Key 纵向用例通过模拟 SSE 验证 API 写入→Worker 解密→Adapter 聚合→逐题/Run Token 持久化→报告脱敏;另一用例保留 JSON fallback,并断言 Run 的空闲读取超时和 stream payload 到达 Provider request。 - Response/API/report 纵向用例验证安全归一化的 provider request ID、returned model、system fingerprint、finish reason 和 HTTP attempt count;过长/控制字符/非标值 fail closed 为 `null`,raw usage object 不作为任意持久化字段暴露。 - Runner 诊断测试覆盖非空但无法解析且 `finish_reason="length"` 时的 `output_truncated`,并确认普通解析失败仍保持 `parse_error`;两者都不改变严格计零语义。 增加或修改路由时至少断言:成功状态码与 Schema、一项校验错误、404/409 等业务错误、分页/筛选(若适用),以及响应中不出现秘密值。API 行为改变必须同步更新 [API.md](API.md)。 +P3-06 后端定向命令: + +```bash +cd backend +uv run pytest tests/test_run_progress_api.py tests/test_response_metadata_api.py -q +``` + +初版 cursor 合同曾建立 4 个失败先行用例并得到 `4 failed`;随后发现 Response 没有数据库单调提交序列,因此已在生产实现前废弃。固定 512 题 block 合同的后端定向结果为 `37 passed`;旧 cursor red 结果不计入通过数。 + ### 6.2 Alembic 与遗留 SQLite `backend/tests/test_migrations.py` 使用独立临时 SQLite 和 Alembic 子进程验证: -- 空库 upgrade/check/downgrade/upgrade 往返,以及 `20260830_0007` 最终 revision、可靠性/凭据/治理/ledger/audit/Provider metadata、`worker_processes` 字段/约束、两个 bounded audit scan indexes,以及早期 `0004/0005` 三索引缺口、repair DDL 部分完成后的重入和 PostgreSQL `0005/0006` metadata 白名单/额外 drift 拒绝控制流(Mock)。 +- 空库 upgrade/check/downgrade/upgrade 往返,以及 `20260830_0008` 最终 revision、可靠性/凭据/治理/ledger/audit/Provider metadata、`worker_processes` 字段/约束、两个 bounded audit scan indexes,以及早期 `0004/0005` 三索引缺口、repair DDL 部分完成后的重入和 PostgreSQL `0005/0006/0007` metadata 白名单/额外 drift 拒绝控制流(Mock)。 - 有模型、Benchmark、题目、Run 与 Response 的 legacy schema 被一致性备份、严格识别并无损升级;题目按原插入顺序回填 0-based `position`。 - 与当前 metadata 一致但没有版本标记的库可安全收养,已有 head 重复 preflight 不生成多余备份。 - 部分表、server default/CHECK 内容或重名、PK/UNIQUE/FK/index/partial index、trigger、SQLite conflict policy/generated column、`STRICT`/`WITHOUT ROWID` 等未知 drift 在创建版本标记和备份前被拒绝;已在 head 的库同样验证。 @@ -273,6 +302,7 @@ SQLite 测试适合快速验证状态机和兼容路径;跨连接并发保证 - `0003 -> 0004` 把既有 Run 标为 `legacy_unmanaged` 并保留 protocol-v1 证据;任意 policy/scope/bucket/question execution/attempt ledger/audit、新 Run fairness/governance 字段或 Response Provider metadata 存在时,`0004 -> 0003` 必须在第一条 DDL 前拒绝。只有隔离空库用于 `0004 -> 0003 -> 0004` roundtrip。 - `0004 -> 0005` 不回填虚构 Worker generation;任意 `worker_processes` 行都使 `0005 -> 0004` 在第一条 DDL 前拒绝。只有显式清空 process facts 或隔离空库才能往返,进入 `0004` 后原 governance/audit downgrade guard 继续生效。 - `0006 -> 0007` 不改 schema 或历史 ledger/actual/Response,只重算 scope `overdrawn`:无显式 Run input bound 的 historical observational input/cost 超额被清除,显式 input/output 超额与由完整显式上界和价格派生的 cost 超额保留;`0007 -> 0006` 恢复旧派生结果。两个方向在任何更新前都拒绝 `reserved/send_started` active reservation。 +- `0007 -> 0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并同时替换 Provider 类型 check 与远程配置 check,旧 `mock`/`openai_compatible` 行值不改写;preflight 只接受精确 `17→18` type fingerprint,任意其他列宽/type drift 都拒绝。存在 `openai_responses` 或 `anthropic_messages` Model 时 downgrade 在第一条 DDL 前拒绝,隔离空库可往返。 - 应用启动 revision 门禁拒绝未迁移库;测试夹具中的 `create_all` 仅用于隔离临时库,并显式 stamp 到与 metadata 对应的 head,不是运行时建表路径。 目标化运行: @@ -282,7 +312,7 @@ cd backend uv run pytest tests/test_migrations.py ``` -真实 PostgreSQL `backend-integration` job 在空的专用 management database 上执行 migration 往返与 `alembic check`,验证 revision/DDL;它不提供已使用数据库可安全丢弃新事实的证明。带数据证据来自 Compose 验收:脚本完成 managed Mock baseline并停止 API/Worker 后,从 head `0007` 发起 downgrade;`0007 -> 0006` 只恢复旧 overdraw 派生谓词,`0006 -> 0005` 为 schema no-op,随后 populated `0005 -> 0004` 在任何有损 DDL 前拒绝。另建隔离空 PostgreSQL 跨过 `0005 -> 0004 -> 0005`,最终 `upgrade head` 回到 `0007` 并 check。历史 `0004 -> 0003` governance/audit guard 仍保留;schema downgrade 不是 PostgreSQL→SQLite 平台回迁。 +真实 PostgreSQL `backend-integration` job 在空的专用 management database 上执行 migration 往返与 `alembic check`,验证 revision/DDL;它不提供已使用数据库可安全丢弃新事实的证明。带数据证据来自 Compose 验收:脚本完成 managed Mock baseline并停止 API/Worker 后,从 head `0008` 发起 downgrade;因该基线没有新协议 Model,`0008 -> 0007` 会恢复 `VARCHAR(17)` 列宽、旧 Provider 类型 check 与旧远程配置 check,`0007 -> 0006` 只恢复旧 overdraw 派生谓词,`0006 -> 0005` 为 schema no-op,随后 populated `0005 -> 0004` 在任何有损 DDL 前拒绝。另建隔离空 PostgreSQL 跨过 `0005 -> 0004 -> 0005`,最终 `upgrade head` 回到 `0008` 并 check。历史 `0004 -> 0003` governance/audit guard 仍保留;schema downgrade 不是 PostgreSQL→SQLite 平台回迁。 ### 6.3 SQLite→PostgreSQL 导入 @@ -341,16 +371,16 @@ Smoke Test 证明 API 与 Worker 责任边界以及数据库驱动的最小离 自动化安全依赖多层约束: 1. 所有端到端测试注册的 Provider 都是 `mock`;`MockModelAdapter.generate` 不执行网络 I/O。 -2. OpenAI-compatible 协议测试向 Adapter 注入 `httpx.MockTransport`,响应在进程内生成。 +2. Chat Completions、OpenAI Responses 与 Anthropic Messages 协议测试都向 Adapter 注入 `httpx.MockTransport` 或内存字节流,响应在进程内生成。 3. 测试数据库和日志级别在应用导入前通过 fixture 环境变量设置,不读取开发 `.env`。 -4. CI 不配置任何真实 Provider Key;测试进程只生成独立临时 keyring 和明显虚构 canary。所有 OpenAI-compatible 路径必须注入 `MockTransport`,篡改/错误凭据路径在构造 Adapter 前失败。 +4. CI 不配置任何真实 Provider Key;测试进程只生成独立临时 keyring 和明显虚构 canary。所有三协议远程路径必须注入 `MockTransport`,篡改/错误凭据路径在构造 Adapter 前失败。 5. 测试数据中的 Key 与域名必须是明显无效占位符,不从开发者环境复制。 6. 标准数据测试必须注入内存 fetcher;CI 不运行在线 `llmbenchlab-evaluate prepare`,也不依赖本机已有 `artifacts/` 缓存。 7. 报告和正式流程组件测试必须使用临时目录/临时数据库,不能读取或覆盖操作者已有正式 Run。 当前测试套件没有操作系统级的“禁止所有出站网络”沙箱,因此最后一层仍是代码 Review:任何新增测试若构造真实 `httpx` client、读取开发 Key 或依赖在线服务,都必须被拒绝。公开 CI 加固可在后续增加 egress-disabled runner 或网络拦截 fixture;不能因为 CI runner 通常没有 Key 就认定任意网络访问安全。 -真实 OpenAI-compatible Provider 只允许作为用户主动执行、明确知晓费用和数据政策的可选手工验证;它不是 PR、CI、Smoke 或 Phase 2 可靠执行基础的完成条件。 +真实远程 Provider 只允许作为用户主动执行、显式选择 Chat/Responses/Messages 协议并明确知晓费用和数据政策的可选手工验证;它不是 PR、CI、Smoke 或 Phase 2 可靠执行基础的完成条件。 真实模型验收应在隔离的本地数据库上先运行 `--limit`,人工核对模型发现、付费 canary、请求上界、逐题错误、Provider 账单和报告三文件,再决定是否 `--full`。这项手工操作不得写入自动化测试结果或 CI 通过数;如果本次没有真实 API URL/Key,就应明确记录“未运行”,不能用 MockTransport 结果替代“真实 Provider 已验证”。 @@ -364,16 +394,22 @@ Smoke Test 证明 API 与 Worker 责任边界以及数据库驱动的最小离 - `pending/running/completed/failed/cancelled` 五种状态标签。 - Dashboard 主页面的 API 加载、严格总分、完成率、Run/模型/Benchmark 汇总与最近运行。 - 后端不可达时的结构化、可重试错误状态。 -- Models password input、创建必填、编辑留空保留、origin 变化重输、请求 pending/成功/失败/关闭/切 Mock/unmount 清空、AbortSignal、恶意错误回显脱敏,以及不写 storage/console。 -- New Run 的 GPQA `8192/600s` 初始建议、MMLU-Pro official/direct 切换建议、手动预算不被覆盖、“应用建议”恢复、`max_tokens:null` Provider 托管提交,以及数字 `131072`/超时 `1800s` DOM 上限。 +- Models password input、创建必填、编辑留空保留、origin 变化重输、请求 pending/成功/失败/关闭/切 Mock/unmount 清空、AbortSignal、恶意错误回显脱敏、不写 storage/console,以及三协议 selector 的值、说明、匹配完整 endpoint 与提交 payload。 +- New Run 的 GPQA `8192/600s` 初始建议、MMLU-Pro official/direct 切换建议、手动预算不被覆盖、“应用建议”恢复、Chat/Responses 的 `max_tokens:null` Provider 托管提交,以及数字 `131072`/超时 `1800s` DOM 上限;协议切换还覆盖 Responses/Messages 默认 `temperature/top_p/seed=null`、逐 Model seed 恢复、Messages `temperature` DOM 上限 `1` 与有限 `max_tokens` 回落。 - Runs 主导航、20 条 offset 分页、状态筛选、active Run 定时刷新、错误/空状态和详情链接。 - Run Detail 返回 Runs 列表、终态停止轮询、逐题每页 100 条的 offset 导航,以及跨页序号/总数显示。 +- Run Detail 把全局和当前页的未得分与执行异常分开显示;Token 测试覆盖精确 Run 总量、部分已知小计、输入/输出非对称覆盖、零 usage、零 Response,以及并行 Run/Response 快照不一致时的保守降级。 +- P3-06 纯函数/状态测试覆盖 position↔row/column 的 0、1、12,032、20,000 边界,四态计数、nullable usage 的 aria-label、block reducer 幂等、旧请求世代/旧 runId 响应丢弃与 reset;实现不保留 API cursor 合同代码路径。 +- 热力图组件测试必须覆盖非仅颜色的图例/形状或边框、hover/focus/tap 等价详情、方向键、Home/End、Ctrl+Home/End、PageUp/PageDown、`aria-rowcount/aria-colcount/aria-activedescendant`、virtual DOM 节点上限、移动端 pinned detail,以及 block 未 hydrate 时“同步中”而非虚假 `not_run`。 +- Run Detail 集成测试必须覆盖 live metrics 更新但不触发整页 loading、当前 Response page 2 保持、旧 Run/旧 block 返回丢弃、一个 tick 不重复 fetch、hidden 暂停/visible 恢复、terminal 先到仍追齐目标 counts、progress 失败不阻断 Run/当前证据页。terminal 且 progress reconciled 后必须恰好执行一次最终 Run/当前 evidence 页刷新,避免终态 `Promise.all` 交错留下旧证据;同一路由切换 `runId` 必须把 evidence offset 重置为 0。known subtotal/coverage 只能展示,不能写成精确 Run Token/cost。 - Run Detail 显示 `managed`、`delayed`、`exhausted`、`legacy_unmanaged` 治理状态,稳定 closed reason 使用受控文案,`not_before` 明确按 UTC 格式化;未知 reason 只显示安全兜底,不反射服务端原值。 -所有 fetch 与 Recharts 均在进程内 stub,没有真实网络;这些组件测试不会调用真实 Provider,也不会验证 Provider 实际接受某个输出长度或读取超时。Benchmark Demo 导入与完整 raw/parsed/reference/score/error 证据仍以离线后端 Smoke 和手工验收补充,不能把 DOM/API stub 结果描述成真实模型兼容性验证。 +所有 fetch 与 Recharts 均在进程内 stub,没有真实网络;这些组件测试不会调用真实 Provider,也不会验证 Provider 实际接受某个输出长度或读取超时。Benchmark Demo 导入与完整 raw/parsed/reference/score/error 证据仍以离线后端 Smoke 和手工验收补充,不能把 DOM/API stub 结果描述成真实模型兼容性验证。P3-06 定向命令 `cd frontend && npm test -- --run tests/run-detail-page.test.tsx tests/run-progress-heatmap.test.tsx` 为 `32 passed`(Run Detail `20` + heatmap `12`);完整 `make test` 为 backend `964 passed, 33 skipped`、frontend `64 passed`。`make lint`、Mock smoke `1 passed, 7 deselected`、frontend production build 与 `docker compose config --quiet` 也已通过。 本轮还在可信 loopback 的真实浏览器中手工核对 Models 表单:Key 控件实际为 password input,页面没有 `api_key_env` 控件,保存成功后表单/卡片/网络响应均不回显测试 Key,应用日志也没有该测试 Key。该检查使用无效测试值且没有触发真实 Provider;它补充 DOM stub 自动化,但不增加 Vitest 或后端测试计数。 +P3-06 可信 loopback 浏览器验收使用历史 Run `a3de7e4d-40b2-4d8c-994b-c713047393ae`:热力图/计数为 179 passed、17 wrong、2 error,known input/output Token 为 `45,509 / 4,561,625` 且两项覆盖均为 `196/198`。desktop、768px、375px 无横向溢出,console 无 warning/error,键盘导航与 Tooltip 通过。12,032/20,000 题仅由组件自动化验证虚拟 DOM 边界与导航,未进行大型真实 Run 的手工 DevTools 长任务/内存测量;不得把自动化结果扩写为该性能声明。实现 SHA [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) 已普通 push 到 `codex/complete-evaluation-workflow` 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5);精确 SHA 的 [GitHub Actions run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) 对 backend、backend-integration、full-stack-reliability、frontend 四个必需 job 全部成功,因此 P3-06 切片为 `completed`,但 Phase 3 整体仍为 `in_progress`。 + 测试应优先按可见文本、label 和 role 查询 DOM,不依赖内部 class 或实现细节。时间、ID 和 API 返回应固定;不要用长 sleep 消除竞态。 ## 10. CI @@ -387,7 +423,7 @@ GitHub Actions 对 `main` push 和 Pull Request 触发四类 job: | `full-stack-reliability` | `python3 scripts/phase2_acceptance.py` 的隔离 Compose 九场景 | 唯一项目/卷、随机 loopback 端口、Mock-only;总是上传已脱敏 evidence,脚本总是精确清理 | | `frontend` | ESLint、Vitest 组件测试、production build(`tsc -b` + Vite) | `npm ci` 锁定依赖;fetch/Recharts stub;`VITE_API_BASE_URL=/api/v1` | -CI 不配置 Provider Key、不调用真实模型,也不在线下载 MMLU-Pro/GPQA。PostgreSQL/Redis 是测试依赖,不是 Provider 网络;标准数据转换只使用 fixture fetcher。`P2-local-control-plane-v2` 的 validator、统计、四-cell 编排失败路径、ledger projection 和 exact-project cleanup 合同可以进入普通自动化,但 GitHub-hosted runner 不运行 `make phase2-slo` 的绝对吞吐/延迟门禁:共享 runner 的 CPU、内存和 Docker 调度不是稳定性能实验室。P2-01 正式实现 SHA `b6a35fef1dd069ebb54b69955058915c722aa34d` 的 [run 33146681285](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33146681285) 4/4 成功;它是同一 SHA 的正确性门禁,不替代本机 1+5。P2-06 实现 SHA `9a20676dcf545040782f04c166205d0043345753` 的 [run 33164609388](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33164609388) 与证据文档 SHA `ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6` 的 [run 33165775037](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33165775037) 均已 4/4 成功。所有必需 job 通过后才能合并;跳过用例、降低断言或使用 `continue-on-error` 都不算修复。具体分支和 Review 门槛见 [GITHUB_WORKFLOW.md](GITHUB_WORKFLOW.md)。 +CI 不配置 Provider Key、不调用真实模型,也不在线下载 MMLU-Pro/GPQA。PostgreSQL/Redis 是测试依赖,不是 Provider 网络;标准数据转换只使用 fixture fetcher。`P2-local-control-plane-v2` 的 validator、统计、四-cell 编排失败路径、ledger projection 和 exact-project cleanup 合同可以进入普通自动化,但 GitHub-hosted runner 不运行 `make phase2-slo` 的绝对吞吐/延迟门禁:共享 runner 的 CPU、内存和 Docker 调度不是稳定性能实验室。P2-01 正式实现 SHA `b6a35fef1dd069ebb54b69955058915c722aa34d` 的 [run 33146681285](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33146681285) 4/4 成功;它是同一 SHA 的正确性门禁,不替代本机 1+5。P2-06 实现 SHA `9a20676dcf545040782f04c166205d0043345753` 的 [run 33164609388](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33164609388) 与证据文档 SHA `ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6` 的 [run 33165775037](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33165775037) 均已 4/4 成功。P3-06 实现 SHA `99791964621165c9cc7ec36b4b2d27fe04e6acd5` 的 [run 33289522923](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) 也已 4/4 成功。所有必需 job 通过后才能合并;跳过用例、降低断言或使用 `continue-on-error` 都不算修复。具体分支和 Review 门槛见 [GITHUB_WORKFLOW.md](GITHUB_WORKFLOW.md)。 P2-06 实现 SHA 与提交前阶段性证据如下。Dirty 结果只作为历史过程记录,仓库收尾以 `9a20676dcf545040782f04c166205d0043345753` 绑定的 clean evidence 和实现精确 SHA CI、以及证据文档 SHA `ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6` 的精确 SHA CI 为准;两层门禁均已完成: @@ -531,23 +567,28 @@ make dev ```bash curl -sS 'http://127.0.0.1:8000/api/v1/runs//responses?limit=100' + curl -sS 'http://127.0.0.1:8000/api/v1/runs//progress' + curl -sS 'http://127.0.0.1:8000/api/v1/runs//progress/blocks/0' curl -sS 'http://127.0.0.1:8000/api/v1/runs//audit?limit=100' curl -sS 'http://127.0.0.1:8000/api/v1/tasks/history?window_hours=24' curl -sS 'http://127.0.0.1:8000/api/v1/leaderboard?benchmark_id=&order=score_desc' curl -sS http://127.0.0.1:8000/api/v1/metrics/summary ``` - 预期 Responses 共 15 条,Run audit 能稳定分页看到 admission/claim/question/terminal typed event,history counters/latency 包含本次 Run;排行榜包含该 Run 且 `is_demo=true`,汇总包含 1 个已完成 Run。 + 预期 Responses 共 15 条;progress index 为 `block_size=512`、一个 block/15 responses,block payload 为 15 个 absolute-position passed cells,且两个响应均 `no-store`。Run audit 能稳定分页看到 admission/claim/question/terminal typed event,history counters/latency 包含本次 Run;排行榜包含该 Run 且 `is_demo=true`,汇总包含 1 个已完成 Run。 ### 11.2 前端验收 打开 `http://127.0.0.1:5173`,按顺序检查: - Dashboard 的模型、Benchmark、Run、得分/延迟/Token 汇总与最近运行来自 API,而非固定假数据。 -- Models 能新增、编辑、删除 Mock;选择 OpenAI-compatible 后出现 masked API Key 输入框,用户直接粘贴真实 Key,保存后输入框清空且卡片只显示“已安全保存”。编辑留空保留,改变 Provider origin 必须重输。 +- Models 能新增、编辑、删除 Mock,并显式选择 Chat Completions、OpenAI Responses 或 Anthropic Messages;远程类型出现 masked API Key 输入框,用户直接粘贴真实 Key,保存后输入框清空且卡片只显示“已安全保存”。编辑留空保留,改变 Provider origin 必须重输。 +- New Run 在 Responses/Messages 下禁用并清空 seed;Messages 禁用 Provider-managed 输出预算并回落到有限 Benchmark 建议值;切回 Chat 时恢复适用控件。 - Benchmarks 能重载 Demo、显示版本/题数/Hash/许可证与醒目的 Demo 警告。 - New Run 能选择 Model 与 Benchmark,区分 protocol-v1 API 默认和 Web 建议,允许数字预算或显式 Provider 托管,并把读取超时随创建请求提交。 -- Runs 列表能从主导航进入,按状态/20 条分页显示并链接回详情;Run Detail 轮询进度并在终态停止,配置快照和逐题证据以每页 100 条查看,不把大型 Run 截断在第一页。 +- Runs 列表能从主导航进入,按状态/20 条分页显示并链接回详情;Run Detail 独立轮询 Run、当前证据页与 progress index/变化 blocks,终态先到时追齐热力图再停止。切到证据第 2 页、切换 Run、隐藏/恢复标签页或 progress 暂时失败都不应清空当前页或混入旧 block。 +- 在运行中、completed、cancelled 和 failed 的部分证据样本核对绿/红/黑/白四态、计数与后端 live metrics;初次非空 block 未 hydrate 时显示“同步中”。hover、键盘 focus/导航与移动端 tap 显示同一 score/Token/延迟/cost/error_type,未知 usage 不显示 0。 +- 在 320/375/768/desktop 宽度、200% zoom、仅键盘、VoiceOver 或 NVDA、forced-colors 和 reduced-motion 下检查可用性;以 12,032 和 20,000 题样本核对虚拟 DOM 上限、滚动、长任务与内存,没有全量正文轮询。 - Leaderboard 可按模型/Benchmark 筛选、按得分排序,协议、Hash、完成率和 Demo 标识可见。 - 刷新页面、空数据库、API 关闭和常见移动宽度下都有明确可操作状态。 @@ -577,7 +618,7 @@ make phase2-acceptance 6. Redis 完全 stop/start;`live`/`health` 保持可用、`ready` 降级,API 仍以 `202` 提交数据库事实,Worker 仅靠 DB reconciliation 完成;Redis 恢复后新消息正常 ACK。 7. Worker 停止时取消 pending Run;Worker 恢复消费旧通知后终态和 0 Response 不漂移。 8. 运行中取消并再次 XADD 同一 Run;Response 数在取消后冻结,重复投递被 ACK 且 canonical snapshot 不变。 -9. 停止 API/Worker 后从 current head `20260830_0007` 尝试 downgrade 到 `0004`:data-only `0007 -> 0006` 只按旧谓词重算 overdrawn,schema-no-op `0006 -> 0005` 后,Worker progress rows 存在时必须在 `0005 -> 0004` 第一条有损 DDL 前拒绝,13 表计数、Run/Response core protocol hash 与可靠性字段不变;另建独立空 PostgreSQL 完成 `0005 -> 0004 -> 0005`,最终回到 `0007` 并 check,随后重启 API/Worker。历史 `0004` governance/audit guard 继续由 migration 回归覆盖;schema downgrade 不是数据平台回迁。 +9. 停止 API/Worker 后从 current head `20260830_0008` 尝试 downgrade 到 `0004`:Mock baseline 没有新协议 Model,因此 `0008 -> 0007` 恢复 `VARCHAR(17)` 列宽、旧 Provider 类型 check 与旧远程配置 check;data-only `0007 -> 0006` 只按旧谓词重算 overdrawn,schema-no-op `0006 -> 0005` 后,Worker progress rows 存在时必须在 `0005 -> 0004` 第一条有损 DDL 前拒绝,13 表计数、Run/Response core protocol hash 与可靠性字段不变;另建独立空 PostgreSQL 完成 `0005 -> 0004 -> 0005`,最终回到 `0008` 并 check,随后重启 API/Worker。历史 `0004` governance/audit guard 继续由 migration 回归覆盖;schema downgrade 不是数据平台回迁。 任何一个场景失败、未运行、使用真实 Provider、最终 PEL/lag 非零或清理不完整,都不能把可靠执行基础写成通过。`--self-check-only` 只验证 Docker/Compose、隔离和清理 guard,不执行九场景,不能替代正式命令。精确 SHA `665244e…` 的最终本地运行已 9/9 通过;artifact 与 hash 见第 10.1 节。 diff --git a/docs/decisions/ADR-0006-local-real-provider-evaluation.md b/docs/decisions/ADR-0006-local-real-provider-evaluation.md index e31cfa7..211c635 100644 --- a/docs/decisions/ADR-0006-local-real-provider-evaluation.md +++ b/docs/decisions/ADR-0006-local-real-provider-evaluation.md @@ -5,11 +5,11 @@ - **Scope**: 标准数据集供应链、真实 OpenAI-compatible 接入、评测编排与报告 - **Related requirements**: FR-MOD-05–10、FR-BEN-01–08、FR-REP-01–04、NFR-SEC-01–05 - **Supersedes**: 无;补充 [ADR-0004](ADR-0004-secret-management.md) 与 [ADR-0005](ADR-0005-durable-task-execution.md) -- **Partially superseded by**: [ADR-0007](ADR-0007-web-provider-credentials.md) 已取代本文关于“REST/前端不得接收 Key”和“数据库不得保存加密凭据”的限制;[ADR-0008](ADR-0008-openai-compatible-sse-transport.md) 已取代本文把 Chat 成功响应统一限制为 4 MiB 的决定,改为 JSON 4 MiB、SSE wire/单事件/聚合 content 64 MiB/1 MiB/4 MiB;其余数据集供应链、可信本地 CLI、预检与报告决定继续有效 +- **Partially superseded by**: [ADR-0007](ADR-0007-web-provider-credentials.md) 已取代本文关于“REST/前端不得接收 Key”和“数据库不得保存加密凭据”的限制;[ADR-0008](ADR-0008-openai-compatible-sse-transport.md) 已取代本文把 Chat 成功响应统一限制为 4 MiB 的决定;[ADR-0019](ADR-0019-explicit-provider-api-protocol-adapters.md) 将本文 Chat-only 的 discovery/canary 扩展为显式 Chat Completions / OpenAI Responses / Anthropic Messages Adapter;其余数据集供应链、可信本地 CLI、预检与报告决定继续有效 ## Status -Accepted;Web/REST 凭据边界由 ADR-0007 部分取代,Chat transport/资源边界由 ADR-0008 部分取代。 +Accepted;Web/REST 凭据边界由 ADR-0007 部分取代,Chat transport/资源边界由 ADR-0008 部分取代,Chat-only discovery/canary 范围由 ADR-0019 扩展为三个显式协议。 ## Context @@ -45,12 +45,12 @@ MMLU-Pro 当前固定版本含 12,032 道 test 题,超过 dataset-v1 原 10,00 ### 真实 Provider 入口 - CLI 的 Base URL 是普通参数;API Key 只允许从指定环境变量读取或用 `getpass` 从终端安全输入。禁止 `--api-key` 明文参数。 -- 远程 Provider 必须使用 HTTPS;明文 HTTP 只允许 loopback 本地推理服务。发现与 Chat 都禁用 redirect、声明并只接受 identity encoding;本文接受时的发现正文上限为 2 MiB,Chat 成功/错误正文上限分别为 4 MiB/64 KiB。Chat 成功边界后来由 ADR-0008 部分取代。 -- CLI 先调用兼容的 `GET /models` 验证认证并发现模型。若调用方明确给出模型名,可在 Provider 不支持模型列表时继续;未给模型且无法唯一发现时必须停止并给出候选,不猜测付费目标。发现结果中任何模型 ID 反射当前 Key 都会安全失败。 -- 在创建正式 Run 前执行一次最小 Chat Completions canary,验证目标模型、请求形状和答案提取;成功体若明确返回不同模型则失败。预检结果只保存脱敏状态、返回模型名、request id、usage、finish reason 与延迟,不保存 headers 或 Key。 +- 远程 Provider 必须使用 HTTPS;明文 HTTP 只允许 loopback 本地推理服务。发现与三类远程生成都禁用 redirect、声明并只接受 identity encoding;本文接受时的发现正文上限为 2 MiB,Chat 成功/错误正文边界后来由 ADR-0008 部分取代,Responses/Messages 的同类边界由 ADR-0019 统一纳入。 +- CLI 先调用同级 `GET /models` 验证认证并发现模型,鉴权跟随显式协议:Chat/Responses 使用 `Authorization: Bearer`,Messages 使用 `x-api-key` 与 `anthropic-version`,并以受累计 100 页、60 秒 wall-clock、2 MiB、10,000 项和重复 cursor 门禁保护的 `after_id` 跟进 `has_more/last_id`。若调用方明确给出模型名,可在 Provider 不支持模型列表时继续;未给模型且无法唯一发现时必须停止并给出候选,不猜测付费目标。发现结果中任何模型 ID 反射当前 Key 都会安全失败。 +- 在创建正式 Run 前按显式协议执行一次最小 canary,分别调用 `/chat/completions`、`/responses` 或 `/messages`,并要求流式结果以 `[DONE]`、`response.completed` 或 `message_stop` 完整终止;它验证目标模型、请求形状和答案提取,成功体若明确返回不同模型则失败。预检结果只保存脱敏状态、返回模型名、request id、usage、finish reason 与延迟,不保存 headers 或 Key。 - 成功 content、raw usage 的字符串值/对象键、request ID、返回模型、system fingerprint 和 finish reason 若包含当前 Key,会在进入 Runner/快照/Response 边界前按精确值替换为 `[REDACTED]`。这是针对当前凭据的最后防线,不是通用 DLP。 - CLI 把 Key 临时注入由 `api_key_env` 指定、并冻结进 Run 的环境变量名;若该变量原来存在则退出上下文后恢复,否则删除。数据库 Model 与 Run 快照仍只保存变量名。ADR-0004 的 REST/前端禁收明文密钥规则不变。 -- CLI 默认要求显式确认预计题数和 Provider 请求保守上界;上界同时计算 Adapter HTTP retries、题数、canary 和剩余 Run attempts。非交互自动化只能通过 `--yes` 继续。价格未知时不得显示虚假零成本。 +- CLI 默认要求显式确认预计题数和 Provider 请求保守上界;上界同时计算 Adapter HTTP retries、题数、canary 和剩余 failed-attempt 预算 `max_attempts - failed_attempt_count`,不把 cooperative yield 算作失败。非交互自动化只能通过 `--yes` 继续。价格未知时不得显示虚假零成本。 ### 执行、恢复与报告 @@ -138,6 +138,7 @@ MMLU-Pro 当前固定版本含 12,032 道 test 题,超过 dataset-v1 原 10,00 - [GPQA official repository](https://github.com/idavidrein/gpqa)(访问 2026-08-27;数据内 `license.txt` 为 CC BY 4.0) - [ADR-0004 — 数据库仅保存密钥环境变量名](ADR-0004-secret-management.md) - [ADR-0005 — Durable task execution](ADR-0005-durable-task-execution.md) +- [ADR-0019 — 显式 Provider API 协议](ADR-0019-explicit-provider-api-protocol-adapters.md) ## Change history @@ -146,3 +147,4 @@ MMLU-Pro 当前固定版本含 12,032 道 test 题,超过 dataset-v1 原 10,00 | 2026-08-27 | Accepted | 用户明确要求优先形成可真实模型完整评测的可信本地流程 | | 2026-08-27 | Hardened | 终审后固定 HTTPS/loopback、identity/响应上限、Key 反射与返回模型拒绝、成功元数据脱敏及过期租约恢复边界 | | 2026-08-27 | Partially superseded | ADR-0008 以真 SSE 和独立 wire/event/content 上限取代统一 Chat 4 MiB 成功正文边界 | +| 2026-08-30 | Partially superseded | ADR-0019 将 Chat-only discovery/canary 扩展为三个显式协议,并保留本文的数据供应链与可信本地确认边界 | diff --git a/docs/decisions/ADR-0015-observability-worker-progress-audit-retention.md b/docs/decisions/ADR-0015-observability-worker-progress-audit-retention.md index 0cd0c3f..0c81873 100644 --- a/docs/decisions/ADR-0015-observability-worker-progress-audit-retention.md +++ b/docs/decisions/ADR-0015-observability-worker-progress-audit-retention.md @@ -5,7 +5,7 @@ - **Deciders**: LLMBenchLab maintainers - **Scope**: Phase 2 P2-06 metrics exporter、告警、Worker 主循环进展与 audit retention - **Amends**: [ADR-0005](ADR-0005-durable-task-execution.md) 的任务可观测性、[ADR-0009](ADR-0009-database-governance-audit-fair-scheduling.md) 的 typed audit 保留流程,以及 [ADR-0010](ADR-0010-phase-2-governance-delivery-boundaries.md) 的恢复指标边界 -- **Amended by**: [ADR-0017](ADR-0017-schema-equivalent-governance-index-repair.md) 将 schema-equivalent `0006` 加入 archive-v1 compatible-head allowlist;[ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 将 data-only `0007` 加入同一 allowlist +- **Amended by**: [ADR-0017](ADR-0017-schema-equivalent-governance-index-repair.md) 将 schema-equivalent `0006` 加入 archive-v1 compatible-head allowlist;[ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 将 data-only `0007` 加入同一 allowlist;[ADR-0019](ADR-0019-explicit-provider-api-protocol-adapters.md) 在不改 archive event/field 语义的前提下将 `0008` 加入 allowlist - **Preserves**: PostgreSQL/数据库事实来源、Redis 非权威通知、`llmbenchlab-protocol-v1`、write-only Provider Key、Mock-only 自动化和可信 loopback 部署边界 ## Context @@ -221,7 +221,7 @@ Dead-letter 与 integrity 是 15 分钟 rolling symptom;Prometheus 整个窗 - cutoff 不由用户任意传入;空集合也生成有效 archive。 - archive 默认绝不删除;运维顺序固定为 archive -> offline verify -> maintenance-window delete。 -单文件 canonical JSONL schema 为 `llmbenchlab-audit-archive-v1`:一行 header、零至一万行 `audit_event`、唯一末行 manifest。V1 的 event/payload/retention contract 与 compatible Alembic head allowlist 独立冻结;compatible heads 为 `20260828_0005`、schema-equivalent repair head `20260829_0006` 与 data-only repair head `20260830_0007`。`0006` 只恢复 canonical `0004` 的三个索引;`0007` 只按显式 hard reservation 语义重算 `governance_scopes.overdrawn`,两者都不改变 archive event、字段或保留语义。`source_alembic_head` 不是装饰性字符串:write/verify/restore/delete/reconcile 都拒绝未列入 allowlist 的旧版、分支或未来 head;未来 schema 只有在证明 V1 全字段语义仍兼容后才能显式扩充 allowlist,否则必须提升 archive schema 并保留 V1 reader。Event 保存完整恢复事实: +单文件 canonical JSONL schema 为 `llmbenchlab-audit-archive-v1`:一行 header、零至一万行 `audit_event`、唯一末行 manifest。V1 的 event/payload/retention contract 与 compatible Alembic head allowlist 独立冻结;compatible heads 为 `20260828_0005`、schema-equivalent repair head `20260829_0006`、data-only repair head `20260830_0007` 与 Provider-column/check head `20260830_0008`。`0006` 只恢复 canonical `0004` 的三个索引;`0007` 只按显式 hard reservation 语义重算 `governance_scopes.overdrawn`;`0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)` 并替换 Provider 类型 check 与远程配置 check;三者都不改变 archive event、字段或保留语义。`source_alembic_head` 不是装饰性字符串:write/verify/restore/delete/reconcile 都拒绝未列入 allowlist 的旧版、分支或未来 head;未来 schema 只有在证明 V1 全字段语义仍兼容后才能显式扩充 allowlist,否则必须提升 archive schema 并保留 V1 reader。Event 保存完整恢复事实: ```text id, event_key, event_type, payload_hash, payload, retention_class, @@ -279,7 +279,7 @@ Delete 永不执行宽泛 `DELETE WHERE expires_at < cutoff`: ### 11. Schema、migration 与 importer -Worker progress revision 为 `20260828_0005`,`20260829_0006` 是 schema-equivalent index repair,当前 Alembic head `20260830_0007` 是 observational-overdraw data repair。三者都是 archive-v1 compatible source/target head;`0007` 不新增表、字段、索引或 archive contract: +Worker progress revision 为 `20260828_0005`,`20260829_0006` 是 schema-equivalent index repair,`20260830_0007` 是 observational-overdraw data repair,当前 Alembic head `20260830_0008` 扩展 `models.provider_type` 列宽并替换 Provider 类型/远程配置两个 check。四者都是 archive-v1 compatible source/target head;`0007/0008` 不新增 archive 表、字段、索引或 contract: - 创建 `worker_processes` 及本文约束/索引; - 为 audit archive 扫描增加 `(expires_at,id)` 索引; diff --git a/docs/decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md b/docs/decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md index e23363e..7bd5154 100644 --- a/docs/decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md +++ b/docs/decisions/ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md @@ -5,7 +5,7 @@ - **Deciders**: LLMBenchLab maintainers - **Scope**: Phase 2 P2-07 备份/恢复验证、Redis rebuild、告警响应与故障矩阵 - **Amends**: [ADR-0005](ADR-0005-durable-task-execution.md) 的恢复运维边界、[ADR-0007](ADR-0007-web-provider-credentials.md) 的数据库外 keyring 备份边界,以及 [ADR-0015](ADR-0015-observability-worker-progress-audit-retention.md) 的告警与 archive 后续演练 -- **Amended by**: [ADR-0017](ADR-0017-schema-equivalent-governance-index-repair.md) 将尚未实施的 recovery-manifest-v1 exact head 显式更新为 schema-equivalent `20260829_0006`;[ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 再将 exact head 更新为 data-only `20260830_0007` +- **Amended by**: [ADR-0017](ADR-0017-schema-equivalent-governance-index-repair.md) 将尚未实施的 recovery-manifest-v1 exact head 显式更新为 schema-equivalent `20260829_0006`;[ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 再将 exact head 更新为 data-only `20260830_0007`;[ADR-0019](ADR-0019-explicit-provider-api-protocol-adapters.md) 将尚未实施的 exact head 更新为 Provider-column/check `20260830_0008` - **Preserves**: PostgreSQL 唯一任务事实来源、Redis 非权威通知、write-only Provider Key、`llmbenchlab-protocol-v1`、Mock-only 自动化和可信本地运维边界 ## Context @@ -39,7 +39,7 @@ Phase 2 已交付 PostgreSQL 任务/治理事实、Redis Streams 通知、独立 - 创建备份前停止 admission、API、Worker、CLI mutation 和所有 application/audit writer;确认 Worker 已 graceful stop 或进入可解释 stale 状态。数据库健康探针可以继续只读连接,但任何未知 writer 都使备份集不合格。 - `pg_dump` 与 source manifest snapshot 必须来自同一停写窗口。恢复后的 13 表摘要与 manifest 不一致即证明期间发生漂移或 artifact 不匹配,整组 fail closed。 - 目标必须由操作者预先创建且没有任何用户 schema/table/Alembic row。先运行只读 empty-target 检查;目标非空时拒绝,绝不由工具清空。 -- 按 [ADR-0017](ADR-0017-schema-equivalent-governance-index-repair.md) 与 [ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md),恢复后数据库必须已经精确位于 Alembic `20260830_0007`。`0006` 与 `0005` 逻辑 schema 等价,只修复早期 `0004` 三索引缺口;`0007` 不改 schema 或 ledger/actual,只按“仅显式 input reservation 构成 hard bound”的规则重算 `governance_scopes.overdrawn`。P2-07 尚未实施,因此 recovery-manifest-v1 在实现前直接固定当前 head。禁止先升级旧 dump 再把它报告为原备份恢复成功;若未来支持新 head,必须在新 ADR/manifest schema 中显式声明兼容。 +- 按 [ADR-0017](ADR-0017-schema-equivalent-governance-index-repair.md)、[ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 与 [ADR-0019](ADR-0019-explicit-provider-api-protocol-adapters.md),恢复后数据库必须已经精确位于 Alembic `20260830_0008`。`0006` 与 `0005` 逻辑 schema 等价,只修复早期 `0004` 三索引缺口;`0007` 不改 schema 或 ledger/actual,只按“仅显式 input reservation 构成 hard bound”的规则重算 `governance_scopes.overdrawn`;`0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并替换 Provider 类型 check 与远程配置 check。P2-07 尚未实施,因此 recovery-manifest-v1 在实现前直接固定当前 head。禁止先升级旧 dump 再把它报告为原备份恢复成功;若未来支持新 head,必须在新 ADR/manifest schema 中显式声明兼容。 - 13 张核心表固定为: ```text @@ -72,7 +72,7 @@ Manifest schema 固定为 `llmbenchlab-recovery-manifest-v1`,UTF-8、canonical "schema": "llmbenchlab-recovery-manifest-v1", "backup_set_id": "00000000-0000-4000-8000-000000000000", "created_at_utc": "2026-08-30T00:00:00.000000Z", - "source_alembic_head": "20260830_0007", + "source_alembic_head": "20260830_0008", "source_git_commit": null, "database_dump": { "format": "postgresql-custom", diff --git a/docs/decisions/ADR-0017-schema-equivalent-governance-index-repair.md b/docs/decisions/ADR-0017-schema-equivalent-governance-index-repair.md index 08339a5..e5bbb71 100644 --- a/docs/decisions/ADR-0017-schema-equivalent-governance-index-repair.md +++ b/docs/decisions/ADR-0017-schema-equivalent-governance-index-repair.md @@ -5,7 +5,7 @@ - **Deciders**: LLMBenchLab maintainers - **Scope**: 早期 `20260827_0004` 本地数据库兼容修复、audit archive head 兼容与 P2-07 recovery head - **Amends**: [ADR-0015](ADR-0015-observability-worker-progress-audit-retention.md) 的 archive-v1 compatible-head allowlist,以及 [ADR-0016](ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md) 的 P2-07 exact recovery head -- **Amended by**: [ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 将 data-only `0007` 加入 archive allowlist,并把尚未实施的 P2-07 exact recovery head 更新为 `20260830_0007` +- **Amended by**: [ADR-0018](ADR-0018-observational-token-estimates-are-not-hard-reservations.md) 将 data-only `0007` 加入 archive allowlist,并把尚未实施的 P2-07 exact recovery head 更新为 `20260830_0007`;[ADR-0019](ADR-0019-explicit-provider-api-protocol-adapters.md) 再将当前运行/recovery head 更新为 `20260830_0008` - **Preserves**: canonical 数据模型、13 表/importer、audit archive v1 字段语义、`llmbenchlab-protocol-v1`、未知 drift fail-closed 与数据库外 keyring ## Context @@ -35,7 +35,7 @@ - 当前应用启动仍只接受唯一 Alembic head。升级代码后必须先由唯一 migration owner 运行 `make migrate`;API/Worker 不自行迁移。 - 支持的 migration owner 入口始终先执行 preflight,再执行 Alembic。migration 内的 active-policy 与 reflection-visible 同名索引门禁是纵深保护,不替代 SQLite preflight 对 `DESC`/`COLLATE` 等深层 DDL modifier 的检查;裸 `alembic upgrade` 绕过 preflight 不属于受支持的旧库修复入口。 - 这不是新的产品 schema、API、Benchmark、评分协议、数据表或安全授权模型,也不推进 P2-07 功能实现。 -- 本 ADR 的 `0006` 索引修复事实与回滚边界保持不变;当前唯一 head 已由 ADR-0018 前进为 `0007`,不能把本 ADR 中“当前 `0006`”的历史措辞当作新的运行门禁。 +- 本 ADR 的 `0006` 索引修复事实与回滚边界保持不变;当前唯一 head 已经 ADR-0018 的 `0007` 并由 ADR-0019 前进到 `0008`,不能把本 ADR 中“当前 `0006`”的历史措辞当作新的运行门禁。 ## Rejected alternatives diff --git a/docs/decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md b/docs/decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md index ddea74a..a3f0369 100644 --- a/docs/decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md +++ b/docs/decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md @@ -6,6 +6,7 @@ - **Scope**: managed Run 的 Provider attempt reservation、overdraw 派生语义与历史物化修复 - **Amends**: [ADR-0009](ADR-0009-database-governance-audit-fair-scheduling.md) 第 3、4 节的 input reservation / overdraw 语义 - **Also amends**: [ADR-0015](ADR-0015-observability-worker-progress-audit-retention.md) 的 archive-v1 compatible-head allowlist,以及 [ADR-0016](ADR-0016-postgresql-keyring-recovery-and-redis-rebuild.md) 的 P2-07 exact recovery head +- **Amended by**: [ADR-0019](ADR-0019-explicit-provider-api-protocol-adapters.md) 将当前 Alembic/archive/recovery exact head 从 data-only `0007` 前进到扩展 `provider_type` 列宽并替换 Provider 类型/远程配置两个 check 的 `0008`;本 ADR 的 hard-reservation 语义不变 - **Preserves**: never-delete attempt ledger、Provider actual usage、显式 hard Token/cost fail-closed、四层 scope、`llmbenchlab-protocol-v1` ## Context diff --git a/docs/decisions/ADR-0019-explicit-provider-api-protocol-adapters.md b/docs/decisions/ADR-0019-explicit-provider-api-protocol-adapters.md new file mode 100644 index 0000000..d362806 --- /dev/null +++ b/docs/decisions/ADR-0019-explicit-provider-api-protocol-adapters.md @@ -0,0 +1,116 @@ +# ADR-0019:显式 Provider API 协议与独立适配器 + +- Status: Accepted +- Date: 2026-08-30 +- Deciders: LLMBenchLab maintainers(用户明确要求兼容 OpenCode Go 的三类 API 端点) +- Scope: Model 类型、Provider Adapter、可信本地预检、Run 快照、Web 模型配置、迁移与安全边界 +- Related requirements: FR-MOD-02–11、FR-RUN-02、FR-REP-01–04、NFR-REL-02、NFR-SEC-01–05 +- Supersedes: 部分扩展 [ADR-0006](ADR-0006-local-real-provider-evaluation.md) 与 [ADR-0008](ADR-0008-openai-compatible-sse-transport.md) 的 Chat-only Provider 范围;既有 Chat 语义继续有效 +- Superseded by: 无 + +## Context + +OpenCode Go 在同一 API 根地址下按模型暴露三种不兼容协议:OpenAI-compatible Chat Completions、OpenAI Responses 和 Anthropic Messages。现有 `openai_compatible` Adapter 只实现 Chat Completions:它固定追加 `/chat/completions`、发送 Chat payload,并按 `choices[0].message.content` 或 `choices[].delta.content + [DONE]` 解析。把完整 `/responses` 或 `/messages` 地址填入 `base_url` 会生成错误的嵌套路径;即使只修正路径,请求、认证、JSON、SSE、usage 与终止语义仍不兼容。 + +`provider_type` 在当前系统中实际承担“选择 Adapter”的职责,并已冻结进 Run 的 `adapter_type` 快照。因此,本次不新增可漂移的第二个协议列,而是扩展这个封闭 Adapter 类型集合。已有数据库行和 API 请求必须无修改地继续表示 Chat Completions。 + +## Decision drivers + +- 协议必须显式选择,不能依赖 URL 猜测或错误后的隐式 fallback;一次 Run 的请求/解析语义必须可从快照复现。 +- 既有 `openai_compatible` 数据、API 客户端、CLI 默认值与 Chat SSE 行为必须向后兼容。 +- 三种协议必须复用相同的 HTTPS/loopback、redirect、identity encoding、正文上限、秘密脱敏、有限 retry 和逐 attempt ledger 边界。 +- 自动化只能使用 Mock、MockTransport 或自定义内存字节流,不得调用真实 Provider。 +- 适配器必须把不同协议的文本、usage、finish reason、request id 和返回模型归一化为 `ModelGenerationResult`,而不能让协议细节泄漏到 Evaluator。 + +## Decision + +扩展 `ProviderType`/Adapter 类型为: + +- `openai_compatible`:保留现有 OpenAI-compatible Chat Completions 行为和默认值; +- `openai_responses`:OpenAI Responses 请求、JSON 与 typed SSE; +- `anthropic_messages`:Anthropic Messages 请求、JSON 与 typed SSE。 + +三类远程 Adapter 都接受兼容根地址或与所选协议匹配的完整 endpoint。根地址分别追加 `/chat/completions`、`/responses`、`/messages`;完整 endpoint 必须与显式类型一致,其他已知协议后缀在 Model 校验/Adapter 构造阶段被拒绝,不发送网络请求。`GET /models` 从三个已知后缀推导同级 `/models`,并按显式协议鉴权:Chat/Responses 使用 `Authorization: Bearer`,Messages 使用 `x-api-key` 与 `anthropic-version`。Messages discovery 对 `has_more/last_id` 使用有界 `after_id` 分页,受累计 100 页、60 秒 wall-clock、10,000 个模型 ID、2 MiB 与缺失/重复 cursor 门禁约束。 + +`openai_responses` 把渲染后的消息映射到 `input`,把 `max_tokens` 映射为 `max_output_tokens`,解析普通 JSON 的 `output` 文本项和 `input_tokens/output_tokens`,并以 `response.completed` 为成功流终止证据。失败、incomplete 或干净 EOF 缺终止事件不得保存部分答案。 + +`anthropic_messages` 把初始 system instruction 放入顶层 `system`,正文放入 `messages`,使用 Messages 所需认证/version headers,解析普通 JSON 的 `content[].text`、`stop_reason` 与 `input_tokens/output_tokens`,并以 `message_stop` 为成功流终止证据。`message_start`/`message_delta` usage 合并但不重复求和。 + +现有 Run 配置字段保留通用名称。Chat 可转发 `temperature/top_p/max_tokens/seed`。为保持旧客户端兼容,请求 Schema 继续暴露 Chat 的 `temperature=0/top_p=1/seed=42` 默认;Responses 与 Messages 在请求和 Model 默认都未显式提供采样字段时,将 `temperature/top_p/seed` 归一化为 `null` 并从 Provider payload 省略,避免某些模型拒绝不支持的采样字段。两类新协议都不支持当前项目的非空 `seed`;Messages 的非空 `temperature` 限制为 `0..1`,并且需要有限 `max_tokens`,显式 Provider 托管的 `null` 在外发前稳定拒绝。前端按 Adapter 类型留空/禁用不支持的字段并解释原因,REST/CLI 仍由后端做最终校验。 + +### 约束与不变量 + +- 不根据模型名称自动选择协议,也不在一次调用失败后切换协议或重复外发到另一 endpoint。 +- `openai_compatible` 的 URL、payload、SSE `[DONE]`、JSON fallback、重试和元数据合同保持兼容。 +- 新协议沿用现有 wire/event/content/error 上限、聚合后 Key 脱敏、非 2xx 分类和 attempt settlement;三协议的 malformed JSON/SSE 与 oversized 响应都生成无原始 Provider `__cause__`/`__context__` 的安全异常,原始 SSE/headers/bytes 不进入日志或持久化证据。 +- 只有显式白名单中的 typed transient 错误可重试:Responses 的 rate-limit/server error,Messages 的 `rate_limit_error`、`api_error`、`overloaded_error`、`timeout_error`,以及 Messages HTTP `529`;普通 JSON 和 SSE 使用同一分类,未知流内错误仍 fail closed,每次重试继续独立进入 attempt ledger。 +- Provider 类型、endpoint、远端模型或凭据在 active Run 期间继续不可修改;Run 快照的 `adapter_type` 是恢复时的权威选择。 +- 不把 OpenCode Go 的当前模型清单硬编码成后端 allowlist;模型与 endpoint 对应关系可能变化,操作者按官方文档选择显式协议。 + +## Alternatives + +### 方案 A:任意完整 URL 原样 POST + +- 优点:改动最小。 +- 缺点:Chat payload/headers/parser 仍会让 Responses/Messages 失败,并把错误从 404 推迟到 400 或响应解析阶段。 +- 未选择原因:不能形成真实协议兼容,且容易误发付费请求。 + +### 方案 B:根据 URL 后缀自动推断协议 + +- 优点:UI 少一个选择项。 +- 缺点:根地址无法可靠推断,路径别名/网关会产生歧义,历史 Run 也没有显式协议快照。 +- 未选择原因:不满足可复现和 fail-closed 要求。 + +### 方案 C:新增独立 `api_protocol` 数据库列 + +- 优点:概念上可把供应商与协议分开。 +- 缺点:当前 `provider_type` 本来就是 Adapter registry key;保留两个可表达同一事实的字段会引入非法组合与迁移复杂度。 +- 未选择原因:当前阶段没有独立 vendor 抽象,扩展既有封闭 Adapter 类型更小且不丢语义。 + +## Consequences + +### Positive + +- OpenCode Go 的 Responses 与 Messages 模型可以在同一评测链路中显式、安全地配置。 +- 旧 Chat 模型和历史 Run 不需数据重写,默认行为不变。 +- 失败会在协议边界得到稳定分类,不再构造 `/responses/chat/completions` 等错误 URL。 + +### Negative + +- Adapter/测试矩阵扩大,协议上游扩展字段仍可能需要后续兼容。 +- Responses/Anthropic 的采样参数与 Chat 不完全相同;调用方必须接受新协议的采样字段默认省略、seed 不可用、Messages `temperature<=1` 及有限输出上限。 +- Mock 门禁不能证明 OpenCode Go 当日网关实现、模型可用性或真实账单行为。 + +### Neutral / follow-up + +- 本决定不改变 `llmbenchlab-protocol-v1` 的题目、评分、分母或排行榜隔离;它只扩展 transport Adapter。 +- 未来新增 Gemini 等协议时继续增加显式 Adapter 类型并另行记录合同,不扩展成任意代理。 + +## Validation + +- 用 MockTransport 分别断言三类 endpoint、生成与 discovery headers、Messages bounded pagination、payload、JSON 成功、typed SSE 成功、usage/metadata、typed transient retry/529、非 2xx、EOF/终止、超限与当前 Key 脱敏。 +- 用 API/Runner/credential 测试断言迁移后的类型校验、active-Run 门禁、Run snapshot 与 Adapter registry。 +- 用前端组件测试断言类型选择、协议说明、字段禁用/清空和提交 payload。 +- 运行完整 lint/test、双方言 migration、Mock smoke、frontend build 和 Compose config;真实 Provider 有意不运行。 + +## Security and privacy impact + +三类协议都向用户选择的 Provider 外发相同评测内容,沿用既有 SSRF、数据外发和费用风险。Messages 会在进程内同时构造 `x-api-key` 与版本 header;这些 header 与 Chat/Responses 的 Authorization 一样不得记录、回显或进入异常对象。新增解析器必须在聚合完成后执行同一当前-Key 递归脱敏,并保持 identity-only、禁 redirect 和有界读取。 + +## Rollback or migration + +迁移 `20260830_0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并同时替换 Provider 类型 check 与远程配置 check;它不改写既有 `mock`/`openai_compatible` 值。回退前若存在 `openai_responses` 或 `anthropic_messages` Model,downgrade 必须在 DDL 前拒绝,要求操作者先在无 active Run 时显式删除或转换这些配置;随后 downgrade 恢复 `VARCHAR(17)` 与两个旧 check,历史 Run/Response 不自动删除。应用回滚期间应停止 API/Worker,避免新类型已提交而旧代码无法加载。由于 `0008` 不改 audit archive event/field 语义,archive-v1 compatible-head allowlist 显式加入 `0008`;P2-07 尚未实施的 recovery-manifest-v1 exact head 也从 `0007` 前进到 `0008`。 + +## References + +- [OpenCode Go API 端点](https://opencode.ai/docs/zh-cn/go#api-%E7%AB%AF%E7%82%B9)(访问 2026-08-30) +- [OpenAI Responses API quickstart](https://platform.openai.com/docs/quickstart/make-your-first-api-request)(访问 2026-08-30) +- [Anthropic Messages API examples](https://platform.claude.com/docs/en/build-with-claude/working-with-messages)(访问 2026-08-30) +- [ADR-0007 — Web Provider 凭据](ADR-0007-web-provider-credentials.md) +- [ADR-0008 — OpenAI-compatible SSE](ADR-0008-openai-compatible-sse-transport.md) + +## Change history + +| 日期 | 变化 | 原因 | +|---|---|---| +| 2026-08-30 | Accepted | 用户确认直接实现 OpenCode Go 三类端点兼容 | diff --git a/docs/phases/PHASE-2-RELIABILITY.md b/docs/phases/PHASE-2-RELIABILITY.md index cfe20aa..ae682fc 100644 --- a/docs/phases/PHASE-2-RELIABILITY.md +++ b/docs/phases/PHASE-2-RELIABILITY.md @@ -15,6 +15,7 @@ - 可观测性与审计保留:[ADR-0015](../decisions/ADR-0015-observability-worker-progress-audit-retention.md) - Schema-equivalent 索引修复:[ADR-0017](../decisions/ADR-0017-schema-equivalent-governance-index-repair.md) - Observational reservation 修正:[ADR-0018](../decisions/ADR-0018-observational-token-estimates-are-not-hard-reservations.md) +- Provider API 显式协议:[ADR-0019](../decisions/ADR-0019-explicit-provider-api-protocol-adapters.md) ## 阶段目标 @@ -22,7 +23,7 @@ ## 当前功能范围 -- PostgreSQL 多 Worker 目标、SQLite 单 Worker 兼容;P2-06 implementation SHA `9a20676…` 将双方言 Alembic 链扩展至 `20260828_0005`,`20260829_0006` 仅修复早期 `0004` 三索引缺口;当前 data-only head `20260830_0007` 只按显式 hard reservation 语义重算 scope overdrawn。 +- PostgreSQL 多 Worker 目标、SQLite 单 Worker 兼容;日常入口维护已把 `make dev DEV_WORKERS=N` 和默认双 Worker的 `make dev-multi` / `make docker-up WORKERS=N` 暴露为受保护入口,并按扩/缩方向同步 scale、API expected 与 expected/registered/live/stalled/shortfall。P2-06 implementation SHA `9a20676…` 将双方言 Alembic 链扩展至 `20260828_0005`,`20260829_0006` 仅修复早期 `0004` 三索引缺口,`20260830_0007` 只按显式 hard reservation 语义重算 scope overdrawn;当前 head `20260830_0008` 将 `models.provider_type` 从 `VARCHAR(17)` 扩为 `VARCHAR(18)`,并替换 Provider 类型 check 与远程配置 check,旧 Model 不改写。 - Redis at-least-once 通知;Run、取消、重试、租约、Response、终态、治理、attempt ledger 和 audit 全由数据库裁决。 - 原子 claim、数据库时间 lease/heartbeat、fencing、有限 retry/backoff、取消、过期接管、duplicate no-op 和 dead-letter。 - 停写只读 SQLite→空 PostgreSQL 的单向 importer;`0005` 按依赖顺序复制 13 张应用表并做 count/PK/content fingerprint,源有 live Worker generation 时拒绝,stopped/stale progress 可精确复制。keyring 仍在数据库之外。 @@ -57,7 +58,7 @@ - write-only `api_key`、AES-256-GCM `model_credentials`、数据库外共享 keyring、legacy `api_key_env`、origin/active-Run 门禁继续有效。Key、Authorization、ciphertext、nonce、keyring、Provider URL、题目/prompt/response正文不得进入 audit。 - Provider request ID/returned model/system fingerprint/finish reason 仅在固定字符、长度和凭据形态检查后保存;不安全值为 `null`,不生成含 Provider 控制文本的 redaction event。 - audit 是应用 append-only、event-key 幂等并有 hash/schema read validation,不是数据库管理员不可篡改的 WORM。 -- OpenAI-compatible SSE、严格 `[DONE]`、JSON fallback、wire/event/content 上限和聚合后 Key 脱敏保持不变。 +- Chat Completions 的严格 `[DONE]`,Responses 的 `response.completed`,Messages 的 `message_stop`,各自 JSON fallback、wire/event/content 上限和聚合后 Key 脱敏保持不变;协议由 Run snapshot 显式选择,不做失败 fallback。 ## 任务状态 @@ -66,6 +67,8 @@ | P2-01 一致性与容量设计 | `completed` | ADR-0012~0014、DB truth/lease/fencing/治理、v2 四 cell 多轮统计、恢复与连接模型已交付;clean SHA `b6a35fe…` 的 1+5 资格为 23/23、`qualified`;证据文档 commit `875f13a…` 已 push,精确 SHA CI 4/4 成功 | | P2-02 PostgreSQL 迁移 | `slice_delivered` | 历史 `0002`~`0004` 与 12 表 importer 已通过精确 SHA 远程门禁;`9a20676…` 增加 `0005` / 13 表 importer、live Worker preflight 和 populated downgrade guard,clean Compose 与远程实现门禁已通过 | | P2-03 Queue/Worker | `foundation_delivered` | Redis 通知、DB scan、claim、lease/heartbeat/fencing、ACK/no-op 已交付;`9a20676…` 增加 generation 级 DB-time scan/claim/lease-heartbeat/progress 与 stale 聚合,dependency probe 仍只表示 capability | +| P2-03 日常多 Worker入口维护 | `completed` | 本地 PostgreSQL 多进程与 Compose 默认双 Worker入口、SQLite fail-fast、fresh/watermark scan、all/running scale direction、五 gauges 与跨 Benchmark PG lease 回归已通过本地/远程门禁;不改变 P2-07 或 Phase 2 总状态 | +| Provider API 三协议维护 | `completed` | ADR-0019、`openai_responses` / `anthropic_messages` Adapter、`0008` 列宽与两个 check 变更、CLI/Web 显式协议与 MockTransport 合同已实现;本地完整门禁和隔离 PostgreSQL 16 往返通过,实现 `6943aa29…` 已 push 且 exact-SHA run `33304667092` 4/4;不改变 protocol-v1、P2-07 或 Phase 2 总状态 | | P2-04 生命周期可靠性 | `foundation_delivered` | retry/backoff、取消、恢复、dead-letter、Response 幂等和三个确定性 DB crash-seam 场景已通过完整 Compose acceptance;Provider 外部副作用仍为 at-least-once | | P2-05 并发治理 | `slice_delivered` | 四层 concurrency/RPM/TPM/lifetime budget、per-attempt ledger、backpressure、finite quantum、公平排序、counter 重算 fail-closed 与 ADR-0011 已实现;精确 SHA 的真实 PG/capacity/acceptance/CI 候选门禁已通过 | | P2-05 observational overdraw 维护 | `completed` | ADR-0018 与 data-only `0007` 只重算 overdrawn 并保留 ledger/actual/Response/Run,active reservation 时拒绝;本地完整验证、当前库迁移和最终 SHA `cb00924…` 的 CI 4/4 均通过 | @@ -78,6 +81,8 @@ 同日的 OpenCode Go `hy3` Run 又暴露 observational input estimate 被错误写成 hard reservation:7 个 attempt 全部 actual settlement,第七次 estimate/actual 为 59/75,而所有 hard policy/Run override 均为 `null`,四层 scope 却被标记 overdrawn。ADR-0018/`0007` 修正这一派生语义并把 UI 文案改为“实际用量曾被判定超过预留”;旧 ledger、actual usage、7 条 Response 和 failed/exhausted Run 终态不改写。本地完整门禁、当前 SQLite 迁移、最终 SHA `cb00924…` 的 real-Compose 9/9 与精确 SHA CI 4/4 均通过,仓库级闭环完成。 +同日的 Run Detail 维护又修正了两个只读展示缺口:`error_questions` 继续只表示执行异常,页面改用 `completed_questions - correct_questions` 显示全部未得分并拆出普通答错;Responses API 追加分页无关的输入/输出已知 Token 小计和独立覆盖数,使精确 Run Token 为 `null` 时仍可显示明确不完整的证据。该维护不改 protocol-v1、数据库 schema、历史 Response/Run、治理 ledger 或 P2-07 范围;并行 Run/Responses 快照不一致时页面保守显示已知小计,不把旧值标成当前完整总量。 + ## 验收标准与当前结论 - [x] 可靠执行基础:API/Worker restart、真实 lease-owner `SIGKILL`、Redis stop/start、duplicate delivery、pending/running cancel 和 lease takeover 有历史真实 PostgreSQL/Redis/Compose 证据。 @@ -91,6 +96,8 @@ - [x] **P2-06 仓库级闭环完成**:clean implementation commit `9a20676dcf545040782f04c166205d0043345753` 已 push,clean capacity/9/9 acceptance 与 [run `33164609388`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33164609388) 4/4 通过;evidence-doc commit `ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6` 的 [run `33165775037`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33165775037) 也精确 4/4 通过。 - [x] **0004 历史索引兼容修复完成**:schema-equivalent `20260829_0006`、仅允许三个已知索引缺失子集的可重入 SQLite preflight、PostgreSQL `0005` metadata 白名单控制流、重复 active/额外 drift 拒绝、真实失败备份副本升级及本地门禁均通过;实现 SHA [`8fb51b690ae6335b8ef93b3cbe54e039781fb173`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/8fb51b690ae6335b8ef93b3cbe54e039781fb173) 的 [run `33263405214`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33263405214) 4/4 成功。historical PG missing-index 分支仍仅有 Mock 回归,标准 CI 真实 PG 只覆盖 fresh canonical 分支。 - [x] **Observational overdraw 修复完成**:目标行为与 `0007` migration 已通过本地完整测试、双方言 migration 和当前个人 SQLite 验真;最终修正 SHA [`cb00924ea3ba3d01ce5bc322b7eabdae1345baf3`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/cb00924ea3ba3d01ce5bc322b7eabdae1345baf3) 的 [run `33271095910`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33271095910) 4/4 成功。 +- [x] **Run Detail 指标维护完成**:API/UI 与零/全/部分/非对称 usage、页内错题拆分及并行快照回归已实现;本地 lint/test/smoke/build/config 和目标实页核对通过。实现 SHA [`0003e4291769a851005ba46c7e59b156a6b789eb`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/0003e4291769a851005ba46c7e59b156a6b789eb) 已 push,[PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5) 的 [run `33286730109`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33286730109) 4/4 成功;不改变 P2-07 或 Phase 2 整体状态。 +- [x] **日常多 Worker入口维护闭环**:离线启动器 `42 passed`,隔离 PostgreSQL 16 迁移到 head 后跨 Benchmark/唯一 lease 回归 `2 passed`,终审后的 fresh/watermark 与 exited-replica 回归、真实 Compose `2→1→2` gauges/cleanup 和完整本地门禁通过;implementation SHA `b06594c…` 的 run `33299883513` 4/4 成功。 - [ ] **P2-07 正式闭环未通过**:没有数据库+keyring backup/restore 认证、完整故障矩阵和告警处置演练。 ## 已实际运行的中间证据 @@ -122,6 +129,7 @@ | P2-06 evidence-doc remote gate | [run `33165775037`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33165775037) | 精确 `ec2959680459a14aa308bd4d9ebcc6bb7bfcf3a6`,四个必需 job 全 success;P2-06 仓库级闭环完成 | | 2026-08-29 DB compatibility repair | migration `52 passed`;完整 backend `927 passed, 33 skipped`、frontend `38 passed`;lint/smoke/config、真实失败备份副本与当前库 startup/check 全绿 | `8fb51b6…` 的 exact-SHA run `33263405214` 4/4;不改变 P2-07 planned 状态 | | 2026-08-30 本地恢复/静默启动维护 | 启动器 `3 passed`;完整 backend `930 passed, 33 skipped`、frontend `38 passed`;lint/build/smoke/config、SQLite digest/quick/FK/head、真实 API/Web 读取通过 | 个人本地 Demo 数据恢复和开发 UX;`5075bdb…` 的 run `33265171953` 4/4 成功,且不是 P2-07 恢复认证 | +| 2026-08-30 多 Worker入口目标回归 | 启动器/fake Compose `42 passed`;隔离 PostgreSQL 16 在迁移到 `0007` 后跨 Benchmark/唯一 lease `2 passed`;终审后的隔离 Compose `2→1→2` 最终 gauges 分别收敛到 `2/2/2/0/0`、`1/1/1/0/0`、`2/2/2/0/0`,cleanup C/V/N/image tags=`0/0/0/0` | 首次空 PG 未迁移导致 fixture setup `UndefinedTable`,按真实部署顺序迁移后通过;fresh/watermark 与 exited-replica 两项终审问题已修复并复审为 0 Blocker/High/Medium;完整 backend `1003 passed, 35 skipped`、frontend `64 passed`,lint/Mock smoke/build/config/diff check 全绿;`b06594c…` run `33299883513` 4/4 成功 | | 2026-08-30 observational overdraw 修复 | backend `946 passed, 33 skipped`;真实 PG+Redis integration `33 passed`;双方言 migration 往返/check、`make lint`、frontend `39 passed`/build、Mock smoke `1 passed`、本地 real-Compose `9/9`、Compose config 与当前库迁移验真通过 | 当前 SQLite head `0007`,scope `4→0`,7 Responses/7 ledger/407 input/599 output、13 表行数、quick/FK 保持;无真实 Provider;`cb00924…` run `33271095910` 4/4 成功 | 所有自动化模型行为只使用 Mock、MockTransport 或 stub;没有真实 Provider 或 API Key。 @@ -153,4 +161,4 @@ ## 状态 -`in_progress`。P2-01、P2-06 与 observational overdraw 维护已完成,P2-05 主切片已交付。P2-07 状态为 `planned`,功能尚未实现,数据库+keyring backup/restore 和完整恢复演练仍缺失。不得把 Phase 2 标为 `completed`,不得宣称生产 HA、灾难恢复 SLA、无限横向扩展或 Provider exactly-once。 +`in_progress`。P2-01、P2-06、observational overdraw 与 Run Detail 指标维护已完成,P2-05 主切片已交付。P2-07 状态为 `planned`,功能尚未实现,数据库+keyring backup/restore 和完整恢复演练仍缺失。不得把 Phase 2 标为 `completed`,不得宣称生产 HA、灾难恢复 SLA、无限横向扩展或 Provider exactly-once。 diff --git a/docs/phases/PHASE-3-BENCHMARKS.md b/docs/phases/PHASE-3-BENCHMARKS.md index 03f0e6f..ff575df 100644 --- a/docs/phases/PHASE-3-BENCHMARKS.md +++ b/docs/phases/PHASE-3-BENCHMARKS.md @@ -48,16 +48,45 @@ MMLU-Pro/GPQA 的可信本地切片。该决定不满足代码沙箱、全局预 当前进度:P3-01 为本地转换器/协议边界的部分实现;P3-02 已接 MMLU-Pro 与 GPQA-Diamond、 尚无 IFEval;P3-03 的固定下载、缓存、源 SHA 与可复现 ZIP 已完成首个切片。P3-06 现可在 Web 选择标准 Benchmark、按数据集建议输出预算/读取超时、从主导航找回全部状态 Run,并以每页 100 条 -浏览逐题证据;CLI 仍提供分组报告,但 Web 尚无分组聚合/子集指标 UI。P3-04/P3-05/P3-07 的代码与 +浏览逐题证据。Run Detail 现把未得分拆为普通答错和执行异常,并在部分 usage 时显示 Run-wide 已知 +Token 小计、输入/输出覆盖率和“完整总量未知”,不再把异常数冒充全部错题或把部分小计冒充精确账单; +CLI 仍提供分组报告,但 Web 尚无分组聚合/子集指标 UI。P3-04/P3-05/P3-07 的代码与 沙箱范围未开始。该 UI 切片的完整本地门禁已通过,功能提交 `467d0243b4fb081c2d637b20ee0958c3bd6ee6d1` 已 push; 分支无 PR且精确 SHA 未触发仅监听 PR/main 的 workflow,不能称为远程绿色。 +P3-06 的 Run Detail 热力图/live metrics 切片已完成:`/runs/{id}/progress` 以固定 +`512` 题 absolute-position block index 返回同一数据库读取快照的 evidence-derived 指标与 block counts, +`/runs/{id}/progress/blocks/{block_index}` 只返回指定 block 的轻量白名单 cells,前端只重取计数变化的 block。前端用虚拟化 +ARIA grid 呈现通过/普通答错/执行异常/未执行四态,并在非空 block hydrate 完成前明确显示“同步中”。 +该切片不改变 `llmbenchlab-protocol-v1`,也不把 known Token/cost 小计回填为 Run 精确字段;无 +migration、ADR 或安全边界变更。初版 cursor 的 4 个失败先行测试因 Response 没有数据库单调提交 +序列而在生产实现前废弃;fixed-block backend/frontend 定向回归为 `37/32 passed`(Run Detail `20` + +heatmap `12`),完整回归为 backend `964 passed, 33 skipped`、frontend `64 passed`,lint、Mock smoke、 +build 与 Compose config 均通过。终态且 progress 已 reconciled 时只做一次最终 Run/当前 evidence 页刷新, +同路由切换 `runId` 会把 evidence offset 重置为 0,两条竞态均有回归。 +目标 198 题 Run 的实页、desktop/768/375、键盘/Tooltip/console 验收通过;12,032/20,000 题由自动化 +虚拟化测试覆盖,不冒充大型真实 Run 的手工性能测量。实现 SHA +[`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) +已普通 push 到 `codex/complete-evaluation-workflow` 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5); +精确 SHA [Actions run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) +的 backend、backend-integration、full-stack-reliability、frontend 四个必需 job 全部成功,因此 P3-06 +切片为 `completed`。 + 2026-08-30 的个人本地实例已把 `artifacts/benchmarks/` 中现有的三份可复现 ZIP 经正式 API 加载: GPQA-Diamond `198` 题,以及 MMLU-Pro Direct/Official-CoT 各 `12,032` 题;默认库连同 Demo 共 `4` 个 Benchmark、`24,277` 题。第三方题目和导入前备份继续位于 Git 忽略目录,本次没有重新下载、 运行评测或调用 Provider;记录 commit `0163b67c00eb59ae59db5f3adb679ad85c799142` 的精确 SHA CI run `33266167547` 已 4/4 成功。这仍不代表 IFEval、Plugin SDK、沙箱、分组 UI 或本阶段验收已经完成。 +同日又从固定来源 revision 确定性准备并通过现有正式 API 导入六套各 100 题的个人本地小型 Benchmark: +GSM8K、中文 MGSM、HellaSwag、WinoGrande、TruthfulQA Binary 五套 mini 子集与完整中文 XCOPA +validation,共 600 题。源缓存、转换器、ZIP、provenance 和备份继续 Git 忽略;Loader/Evaluator、 +API/数据库/Hash 与 SQLite 完整性均已 +验证。该工作只复用现有 numeric/multiple-choice 合同,没有实现通用 Plugin SDK、IFEval、代码沙箱或 +新的产品协议;公开 mini 成绩也不得冒充官方全量榜单。导入任务没有创建、取消或重置 Run;当时既有 +12,032 题 Run 继续推进,导入后的取消请求与 MGSM mini Run 创建属于另一个并发客户端时间线。记录 +commit `8faa2093b2c3308994d50e42a31063cdbf5264a6` 已 push,精确 SHA CI run `33296049611` 4/4 成功。 + ## 验收标准 - [ ] 每个标准 Benchmark 可从固定来源/版本重复导入并得到相同 Hash。 @@ -67,6 +96,7 @@ run `33266167547` 已 4/4 成功。这仍不代表 IFEval、Plugin SDK、沙箱 - [ ] 超时、fork bomb、磁盘/输出耗尽、恶意系统调用等路径有验证。 - [ ] 代码结果保留编译、stdout/stderr、测试状态和沙箱错误的受限快照。 - [ ] 分组指标计算正确,不同协议/数据版本不会无提示混排。 +- [x] 12,032–20,000 题 Run Detail 以固定 block/虚拟化可访问热力图动态展示四态和后端同快照指标,乱序/并发/终态竞态不漏格且不轮询全量正文;大型边界为自动化验证。 - [ ] 核心 Mock 离线回归测试继续通过。 ## 风险 @@ -90,6 +120,7 @@ run `33266167547` 已 4/4 成功。这仍不代表 IFEval、Plugin SDK、沙箱 ## 状态 `in_progress`。仓库不再只有计划:MMLU-Pro 与 GPQA-Diamond 的固定来源转换、可信本地运行和 -报告切片已经实现,但不会提交第三方题目;Web 已有标准数据选择、建议配置、Run 列表和分页证据的 -已 push 的当前切片。IFEval、通用 Plugin SDK、代码题/沙箱、分组/子集完整 UI 与红队验收仍缺失; -本轮精确 SHA 没有 workflow run,且其余范围仍缺失,故本阶段不得标为 `completed`。 +报告切片已经实现,但不会提交第三方题目;Web 已有标准数据选择、建议配置、Run 列表和分页证据。 +热力图/live metrics 的 P3-06 切片已完成实现、回归、目标 Run 浏览器验收、普通 push 与精确 SHA CI, +状态为 `completed`。IFEval、通用 Plugin SDK、代码题/沙箱、分组/子集完整 UI 与红队验收仍缺失, +故 Phase 3 整体不得标为 `completed`。 diff --git a/docs/plans/2026-08-30-fix-run-detail-metrics.md b/docs/plans/2026-08-30-fix-run-detail-metrics.md new file mode 100644 index 0000000..5d441d2 --- /dev/null +++ b/docs/plans/2026-08-30-fix-run-detail-metrics.md @@ -0,0 +1,115 @@ +# 修复 Run Detail 错题与部分 Token 展示执行计划 + +- Owner: Codex +- Status: completed +- Created: 2026-08-30 +- Updated: 2026-08-30 +- Related requirements: FR-RUN-08、FR-RUN-10、FR-API-08、FR-UI-05、NFR-UX-01、TST-04 +- Related phase: [Phase 2 — Reliability](../phases/PHASE-2-RELIABILITY.md)、[Phase 3 — Benchmarks](../phases/PHASE-3-BENCHMARKS.md) +- Worklog: [2026-08-30-fix-run-detail-metrics.md](../worklogs/2026-08-30-fix-run-detail-metrics.md) +- ADRs: 无;本修复保留已接受的 `llmbenchlab-protocol-v1` 聚合语义,只扩充只读展示证据 + +## Context + +Run `a3de7e4d-40b2-4d8c-994b-c713047393ae` 有 198 条 Response:179 条正确、17 条可解析但答错、2 条 Provider 异常。现有 Run Detail 把只统计异常的 `error_questions` 泛称为“错误题”,导致用户误以为未得分题只有 2 条。196 条 Response 有完整 usage,2 条异常没有 usage;协议要求精确 Run Token 在任一 usage 缺失时保持 `null`,但 UI 只显示破折号,未展示仍可审计的已知小计与覆盖率。 + +## Objective + +在不改写历史 Response、不把部分 Token 冒充精确总量、也不改变 protocol-v1 排名口径的前提下,让 Run Detail 明确显示未得分、普通答错、执行异常,以及全量 Response 的已知 Token 小计和 usage 覆盖率。 + +## Scope + +- 扩充 `GET /runs/{run_id}/responses` 的列表元数据,返回全量 Response 的已知 input/output Token 小计与各自报告题数。 +- Run Detail 使用 `completed_questions - correct_questions` 显示“未得分”,并拆分普通答错与执行异常。 +- Run 精确 Token 非空时继续显示精确总量;为 `null` 时显示已知小计、覆盖率和“完整总量未知”。 +- 更新后端/前端回归测试、API/测试文档及强制状态文档。 + +## Non-goals + +- 不修改 `EvaluationRun.input_tokens/output_tokens` 的 all-or-nothing 语义。 +- 不回填或猜测两条历史异常调用的 usage,不更改成绩、Response、ledger 或数据库 schema。 +- 不调用真实 Provider,不修改价格或成本口径。 +- 不改变排行榜或 Dashboard 的完整 Token 聚合语义。 + +## Assumptions + +| 假设 | 依据 | 验证方法 | 不成立时的处理 | +|---|---|---|---| +| built-in protocol-v1 每题分数为 0 或 1 | FR-EVL-07 与协议定义 | 后端既有测试及目标 fixture | 若出现分数扩展,改为新增明确的零分计数聚合,不能用整数相减 | +| Response 列表端点可承载与分页无关的只读聚合元数据 | 端点已返回全量 `total`,详情页与证据同源 | API schema/分页测试 | 若造成不可接受查询开销,改为专用 summary 端点并记录偏差 | +| 部分 Token 小计只能作为已知下界/证据覆盖,不是账单真值 | 两条异常 usage 为 `null`,Provider 调用非 exactly-once | 文案与测试断言“完整总量未知” | 禁止显示为无条件精确总 Token | + +## Requirements + +- [x] FR-UI-05:完成 Run 显示“未得分 19、普通答错 17、执行异常 2、正确 179”,口径互相可解释。 +- [x] FR-RUN-08 / FR-RUN-10:保留精确 Run Token `null`,同时返回已知 input/output 小计与报告覆盖率。 +- [x] FR-API-08:新增字段有明确、非负、分页无关的 Schema 与文档,不泄漏 Provider 正文或秘密。 +- [x] NFR-UX-01:部分 usage 明确标为“已知小计”且提示完整总量未知;零 Response/全 usage/部分 usage 均可读。 +- [x] TST-04:后端 API、前端组件、类型检查和 production build 覆盖新行为。 + +## Implementation steps + +1. [completed] **冻结 API 与 UI 语义并建立回归夹具** + - 修改范围:后端 responses Schema/API 测试、前端 Run Detail 测试设计。 + - 操作:定义 `known_input_tokens`、`known_output_tokens`、`input_token_reported_responses`、`output_token_reported_responses`,确认其为全量而非当前页聚合。 + - 完成判据:测试能复现 179/17/2 与 196/198 部分 usage 的展示要求。 +2. [completed] **实现后端只读 usage 汇总与前端展示** + - 修改范围:`backend/app/api/v1/runs.py`、`backend/app/schemas/evaluation_response.py`、`frontend/src/api/types.ts`、`frontend/src/api/client.ts`、`frontend/src/pages/RunDetailPage.tsx`、必要格式函数。 + - 操作:单次聚合返回 count/sum;UI 区分精确与部分 Token,并拆分未得分/普通答错/异常。 + - 完成判据:目标后端与前端测试通过,既有分页/轮询行为不变。 +3. [completed] **文档、完整验证与交付复核** + - 修改范围:API、TESTING、CHANGELOG、PROJECT_STATUS、Phase 2/3、NEXT_TASK、工作日志和本计划。 + - 操作:记录兼容语义、运行目标测试、lint、完整 test、smoke、build、Compose config,检查 diff/秘密/状态。 + - 完成判据:本地门禁通过并记录真实结果;按仓库规则 commit/push 后等待精确 SHA CI。 + +## Risks + +| 风险 | 可能性 | 影响 | 预防措施 | 触发后的处理 | +|---|---|---|---|---| +| 部分小计被误读为精确账单 | 中 | 高 | 主值和辅助文案同时标“已知/完整总量未知” | 回退部分展示,保留覆盖率诊断 | +| 分页页码影响全局小计 | 低 | 中 | 聚合查询不应用 offset/limit,分页测试跨两页断言相同 summary | 修正为独立聚合查询 | +| 新字段破坏现有通用 ListResponse 类型 | 中 | 中 | 为 responses 定义专用前端类型,保留既有 `items/total/offset/limit` | 调整 client 泛型而非污染所有列表 | +| 聚合查询增加详情读取成本 | 低 | 中 | 与现有 count 合并为一个常数列聚合查询 | 若实测异常,增加针对 run_id 的执行计划检查或专用缓存设计 | + +## Validation + +| 验收项 | 命令/检查 | 预期结果 | 实际结果与证据 | +|---|---|---|---| +| 后端 responses summary | `cd backend && uv run pytest tests/test_response_metadata_api.py` | 新旧分页与 Token 覆盖断言通过 | 与 Smoke 合并运行,后端目标共 `11 passed`;全量亦通过 | +| 后端聚合回归 | `cd backend && uv run pytest tests/test_smoke.py` | protocol-v1 精确 Token nullable 语义保持 | 合并目标 `11 passed`;离线 Smoke `1 passed, 7 deselected` | +| 前端 Run Detail | `cd frontend && npm test -- --run tests/run-detail-page.test.tsx tests/format.test.ts` | 错题拆分、部分/完整 Token 与分页通过 | `20 passed` | +| 静态与构建 | `make lint`、`cd frontend && npm run build` | Ruff/ESLint/TS/build 通过 | 首次 lint 仅 2 个 Ruff format 差异;格式化后全绿,build 2192 modules 成功并保留既有 chunk warning | +| 完整回归 | `make test`、`make smoke` | 全量自动化只用 Mock/Stub 且通过 | backend `951 passed, 33 skipped`;frontend `47 passed`;Smoke `1 passed, 7 deselected` | +| 部署配置 | `docker compose config --quiet` | exit 0 | exit 0 | +| 目标实页核对 | 本地 API + Browser 读取目标 Run | 179/17/2、196/198 和已知小计可见 | API `45,509/4,561,625`、`196/198`;页面 19/17/2、460.7万,第二页 8 未得分/2 执行异常,console error 0 | +| 秘密与无关改动检查 | `git diff --check`、`git status --short` 及敏感词检查 | 无格式错误、无凭据、范围正确 | added diff + untracked 高置信扫描无命中;`diff --check` 通过;18 个候选文件均在计划范围 | + +## Rollback + +本任务没有数据库迁移或数据写入。回滚只需反向应用本任务明确文件的补丁;现有 Run/Response/ledger 与 `EvaluationRun.input_tokens/output_tokens` 均不受影响。不得使用会覆盖其他工作的 reset/checkout。 + +## Documentation updates + +- [x] `docs/API.md`:Response 列表新增全量 usage 汇总字段与精确/部分语义 +- [x] `docs/TESTING.md`:部分 usage 与错题拆分回归 +- [x] `CHANGELOG.md` +- [x] `docs/PROJECT_STATUS.md` 与 Phase 2/3 文档 +- [x] `docs/NEXT_TASK.md` 与本次工作日志 +- [x] README:补充用户可见的错题拆分与部分 Token 语义 + +## Completion evidence + +- 修改文件:后端 Responses Schema/API、前端 API 类型/Run Detail、后端与前端回归,以及 README/API/TESTING/CHANGELOG/PROJECT_STATUS/Phase/NEXT_TASK/计划/工作日志共 18 个文件。 +- 实际命令:目标后端 `11 passed`、目标前端 `20 passed`;`make lint`、`make test`(backend `951 passed, 33 skipped`;frontend `47 passed`)、`make smoke`、frontend build、Compose config、目标本地 API/Browser 实页核对均通过。 +- 验收对应:目标 Run 实页显示未得分/普通答错/执行异常/正确=`19/17/2/179`,已知 input/output=`45,509/4,561,625`,覆盖=`196/198`;第二页显示 8 条未得分与 2 条执行异常。 +- 远程证据:实现 commit [`0003e4291769a851005ba46c7e59b156a6b789eb`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/0003e4291769a851005ba46c7e59b156a6b789eb) 已 push 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5);精确 SHA 的 [CI run `33286730109`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33286730109) 四个 job 全部成功。 +- 未运行:真实 Provider(有意);没有 API Key、历史数据写入或模型费用。实现精确 SHA 已由远程真实 PostgreSQL/Redis integration 与 real-Compose acceptance 覆盖。 +- 已知问题:两条历史异常调用的真实 usage 不可恢复;已知小计不等于 Provider 账单 + +## Decision and discovery log + +| 日期时间 | 类型 | 记录 | 影响/后续 | +|---|---|---|---| +| 2026-08-30 09:31 CST | discovery | 目标 Run 为 179 正确、17 普通答错、2 异常;196/198 有 usage,已知合计 4,607,134 | UI 必须拆分口径,Token 必须标部分覆盖 | +| 2026-08-30 09:31 CST | decision | 保留 protocol-v1 精确 Token all-or-nothing,只在 responses 读取 API 增加已知小计与覆盖率 | 无迁移、无历史数据改写、无需 ADR | +| 2026-08-30 09:44 CST | discovery | Run 与 Responses 并行读取没有共同事务快照;输入/输出报告数相等也不证明来自同一批题 | 精确展示要求题数、两侧全覆盖及小计一致;部分覆盖文案始终明确输入/输出边际 | diff --git a/docs/plans/2026-08-30-multi-worker-evaluation.md b/docs/plans/2026-08-30-multi-worker-evaluation.md new file mode 100644 index 0000000..b84ec81 --- /dev/null +++ b/docs/plans/2026-08-30-multi-worker-evaluation.md @@ -0,0 +1,120 @@ +# 多 Worker 并行评测执行计划 + +- Owner: Codex +- Status: completed +- Created: 2026-08-30 +- Updated: 2026-08-30 +- Related requirements: FR-RUN-04、NFR-PERF-03、Phase 2 P2-03 +- Related phase: [Phase 2 — Reliability](../phases/PHASE-2-RELIABILITY.md) +- Worklog: [2026-08-30 multi-worker evaluation](../worklogs/2026-08-30-multi-worker-evaluation.md) +- ADRs: [ADR-0005](../decisions/ADR-0005-durable-task-execution.md)、[ADR-0009](../decisions/ADR-0009-database-governance-audit-fair-scheduling.md);不新增 ADR,本任务只暴露既有 PostgreSQL 多 Worker 决定,不改变数据库事实、租约或协议语义 + +## Context + +当前 Runner、数据库租约、fencing、Redis consumer group 与 Worker ID 已支持多个独立 Worker 竞争不同 Run,并在真实 PostgreSQL/Redis 双 Worker Mock 验收中通过。日常入口仍把 `make dev` 固定为一个 SQLite Worker,`make docker-up` 也没有传递 Compose `--scale`,所以一个长期 Provider 请求会占住唯一 Worker,使其他 Benchmark Run 只能排队。 + +## Objective + +提供可配置且可观测一致的多 Worker 启动入口,使 PostgreSQL 模式下至少两个不同 Benchmark Run 能由不同 Worker 同时执行;SQLite 继续在启动前拒绝多 Worker,避免产生虚假的并发安全保证。 + +## Scope + +- 扩展本地开发启动器,在 PostgreSQL DSN 下管理 1–32 个独立 Worker 进程、独立私有日志、信号转发与退出清理。 +- 增加 Compose 多 Worker 启动包装器,默认两个 Worker,并以同一副本数设置 `worker_expected_processes`、执行 `--scale` 和启动后 gauges 校验。 +- 更新 Make 入口、环境变量示例、部署/架构/测试/运维说明和阶段状态文档。 +- 增加启动器回归,以及真实 PostgreSQL 下不同 Benchmark Run 的并发领取证据。 + +## Non-goals + +- 不让 SQLite 支持多 Worker;现有 SQLite 数据迁移仍使用 stopped-source、empty-target 的显式 importer。 +- 不修改 Run/Response schema、REST API、`llmbenchlab-protocol-v1`、Provider 请求语义或治理限额。 +- 不把双 Worker Mock 证据外推为三个以上 Worker、真实 Provider、HA 或生产 SLA。 +- 不自动停止当前用户进程、取消 Run、迁移当前 SQLite 数据或调用真实 Provider。 + +## Assumptions + +| 假设 | 依据 | 验证方法 | 不成立时的处理 | +|---|---|---|---| +| PostgreSQL 租约已支持不同 Worker 并发领取不同 Run | ADR-0005、Phase 2 acceptance/capacity | 真实 PostgreSQL integration 回归 | 若失败则停止启动层交付并修复租约竞争 | +| 当前问题来自入口固定单 Worker,而非 API 限制 | `scripts/dev.sh`、Makefile、Compose 默认配置 | 启动脚本测试与 Compose config | 若存在额外串行门禁,补充最小实现和测试 | +| 两个 Worker 是当前唯一经过容量资格的默认值 | P2-local-control-plane-v2 | 保留默认 2 并记录更高规模限制 | 不把 3+ 标记为已资格 | + +## Requirements + +- [x] PostgreSQL 下 `make dev DEV_WORKERS=2` 启动两个独立 Worker,任一子进程退出时其余服务被清理并传播退出码。 +- [x] SQLite 下请求两个以上 Worker 必须在启动任何服务或创建日志前稳定失败,且不输出 DSN。 +- [x] `make docker-up` / `make dev-multi` 默认启动两个 Worker,允许显式 `WORKERS=N`,按 ADR-0016 的扩/缩顺序保持 scale 与 API expected 一致,并验证 expected/registered/live/stalled/shortfall=`N/N/N/0/0`。 +- [x] 多 Worker 不允许同一 Run 出现两个有效 lease;不同 Benchmark Run 可由不同 owner 并发领取。 +- [x] 自动化只使用 Mock/Stub;不读取真实 Key、不调用真实 Provider。 +- [x] 文档明确总 Provider 并发约为 Worker 数 × Run 内并发,并仍受 governance policy 与数据库连接容量约束。 + +## Implementation steps + +1. [completed] **冻结启动合同与失败边界** + - 修改范围:计划、工作日志、启动器测试设计。 + - 操作:复用 ADR-0005 的 PostgreSQL 多 Worker边界;确定本地/Compose 参数、上限、日志与 gauges 合同。 + - 完成判据:计划和工作日志记录明确,测试先覆盖非法输入与进程管理。 +2. [completed] **实现本地与 Compose 多 Worker 启动** + - 修改范围:`scripts/dev.sh`、Compose 包装脚本、`Makefile`、`compose.yaml`、`.env.example`。 + - 操作:管理多个独立 Worker,校验数据库方言/副本数,同步 expected,启动后验证 live/shortfall。 + - 完成判据:目标脚本测试、`bash -n`、Compose config 通过。 +3. [completed] **补充并发正确性回归** + - 修改范围:后端 Worker/租约或 PostgreSQL integration 测试。 + - 操作:验证两个不同 Benchmark Run 被不同 Worker领取且无同 Run重复 owner。 + - 完成判据:目标后端测试通过;真实基础设施测试在可用环境执行并记录。 +4. [completed] **同步文档并完成门禁** + - 修改范围:README、Architecture、Deployment、Operations、Testing、Changelog、Project Status、Phase 2、Next Task、本计划与工作日志。 + - 操作:记录命令、支持边界、回滚与实际验证;运行相关完整门禁。 + - 完成判据:lint/test/smoke/build/config 与精确提交远程 CI 符合仓库 DoD。 + +## Risks + +| 风险 | 可能性 | 影响 | 预防措施 | 触发后的处理 | +|---|---|---|---|---| +| 多开 SQLite Worker造成锁竞争或重复副作用 | 中 | 高 | 方言检查在子进程/日志创建前 fail-fast | 保持单 Worker并提示使用 PostgreSQL | +| scale 与 expected 不一致造成监控误报 | 中 | 中 | 单一副本参数同时驱动环境与 `--scale`,启动后校验 gauges | 启动命令失败并保留栈供诊断 | +| 任一 Worker退出后遗留同组进程 | 中 | 高 | 数组化 PID 管理、统一 TERM/wait、信号回归 | 失败即终止整个本地 dev 会话 | +| Worker 数 × Run 并发放大 Provider成本 | 中 | 高 | 默认只用已资格的 2 Worker,文档强调治理与成本上界 | 要求操作者设置 policy/Run concurrency | +| 当前 SQLite 数据与新 PostgreSQL 栈分叉 | 高 | 中 | 不自动迁移,文档指向显式 stopped-source importer | 用户决定维护窗口后另行迁移 | + +## Validation + +| 验收项 | 命令/检查 | 预期结果 | 实际结果与证据 | +|---|---|---|---| +| 本地多进程启动器 | `cd backend && uv run pytest tests/test_dev_script.py` | 多 Worker/失败/清理回归全部通过 | 与 Compose 包装器合并目标套件 `42 passed`;含 SIGINT→TERM 全清理、显式空 Make 参数、stale generation 与 exited replica | +| Compose 启动包装器 | 目标脚本测试、`docker compose config --quiet` | 默认 2、显式 N、非法输入和 gauges 同步通过 | fake Docker目标套件包含扩/缩顺序、scan、五 gauges、stale/超时/上游失败;隔离真实 Compose 冷启动 `2/2/2/0/0`,随后 `2→1→2` 得到 `1/1/1/0/0` 与 `2/2/2/0/0`,cleanup C/V/N=`0/0/0`;config 通过 | +| 租约并发 | PostgreSQL integration 目标测试 | 不同 Benchmark Run 不同 owner;同 Run 唯一 lease | 首次空库未迁移在 fixture setup 报两项 `UndefinedTable`;迁移到 `20260830_0007` 后相同两项 `2 passed` | +| 质量门禁 | `make lint && make test && make smoke` | 全部通过且无真实 Provider | 终审后重跑:lint/typecheck 通过;backend `1003 passed, 35 skipped`、frontend `64 passed`;offline Mock smoke `1 passed, 7 deselected`;frontend build 与 Compose config 通过 | +| 秘密与无关改动检查 | `git diff --check`、`git status --short` 及敏感词检查 | 无格式错误、无密钥、范围正确 | 19-file staged diff/范围已复核;secret/key、敏感路径、debug marker 三类扫描均为 0;`git diff --cached --check` 通过 | +| 精确 SHA 远程 CI | implementation commit push 后查询该 SHA | 四个必需 job 全部成功 | `b06594c2df67d6e2a8b117651b193cd0fa409bf5` 的 [run `33299883513`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33299883513) 4/4 success | + +## Rollback + +本任务不迁移 schema 或用户数据。回退启动器/Make/Compose/文档即可恢复单 Worker默认;停止多 Worker时使用现有 SIGTERM grace,未完成 Run 保留数据库 lease 并由剩余 Worker或自然过期恢复。不得用删除 PostgreSQL volume 回退。 + +## Documentation updates + +- [x] README / 用户操作说明 +- [x] Architecture / Deployment / Operations / Testing +- [x] CHANGELOG、PROJECT_STATUS、Phase 2、NEXT_TASK、工作日志(远程证据将在 closeout commit 补齐) +- [x] ADR:不新增;复用 ADR-0005/0009,原因见上文 +- [x] API / Benchmark protocol / Security:接口、评分和秘密边界不变 + +## Completion evidence + +- 修改文件:Make/环境示例/Compose、本地与 Compose launcher、launcher/真实 PG tests、README 与 Phase 2 运维/部署/测试/状态文档 +- 实际命令:目标 launcher `42 passed`;迁移后真实 PG `2 passed`;终审修复后的真实 Compose 冷启动与 `2→1→2` 五 gauges 通过且隔离 cleanup 为零;终审后完整 lint/test/smoke/build/config/diff check 通过;implementation exact-SHA CI 4/4 success +- 验收对应:本地/Compose入口、SQLite fail-fast、按方向 scale/expected、active scan、五 gauges 与跨 Benchmark lease 已有本地证据 +- 未运行:真实 Provider(有意);3+ Worker容量资格(不在范围) +- 已知问题:SQLite 继续单 Worker;3+ Worker和真实 Provider尚未资格 + +## Decision and discovery log + +| 日期时间 | 类型 | 记录 | 影响/后续 | +|---|---|---|---| +| 2026-08-30 15:00 CST | discovery | 核心租约与 Compose 已支持多 Worker;日常启动入口固定为一个 Worker | 实现聚焦启动/配置/验证,不重写 Runner | +| 2026-08-30 15:00 CST | decision | SQLite 保持单 Worker;多 Worker只在 PostgreSQL 下启用 | 本地启动器对 SQLite fail-fast,Compose 作为默认多 Worker入口 | +| 2026-08-30 15:00 CST | decision | 默认副本数为已通过资格的 2,而非未测量的更高数量 | 允许显式 N,但文档不外推容量结论 | +| 2026-08-30 15:25 CST | review | 初版包装器只检查三项 gauges、单次同时 scale/API,且超长数字可触发 Bash 回绕 | 按 ADR-0016 改为扩容 Worker scan→API、缩容 API→Worker,并验证 `N/N/N/0/0`;字符串校验先于算术 | +| 2026-08-30 15:35 CST | review | 后台子进程可能忽略 SIGINT,转发 INT 后无界 wait | launcher 对主进程保留 130/143,但一律用 TERM 清理全部子进程,并新增忽略 INT 回归 | +| 2026-08-30 16:20 CST | review | scan gate 可计入 stale generation,且只看 running container 会把含 exited replica 的缩容误判为非缩容 | 使用应用 DB 时钟 watermark 与 fresh scan 双门禁;方向同时统计 all/running replica,增加两个回归并由独立复审确认 0 Blocker/High/Medium | diff --git a/docs/plans/2026-08-30-provider-api-protocols.md b/docs/plans/2026-08-30-provider-api-protocols.md new file mode 100644 index 0000000..ac4cb40 --- /dev/null +++ b/docs/plans/2026-08-30-provider-api-protocols.md @@ -0,0 +1,112 @@ +# Provider API 三协议适配执行计划 + +- Owner: Codex +- Status: completed +- Created: 2026-08-30 +- Updated: 2026-08-30 +- Related phase: [Phase 2 — Reliability](../phases/PHASE-2-RELIABILITY.md) +- Worklog: [工作日志](../worklogs/2026-08-30-provider-api-protocols.md) +- ADRs: [ADR-0019](../decisions/ADR-0019-explicit-provider-api-protocol-adapters.md) + +## Context + +当前 `openai_compatible` Adapter 只实现 Chat Completions。OpenCode Go 按模型分别要求 `/chat/completions`、`/responses` 或 `/messages`,完整的后两类地址会被现有 URL 拼接器继续追加 `/chat/completions`。本任务必须在不改变旧 Chat Run 的前提下,为公共 Model API、Worker/CLI、Run snapshot 与 Web 表单增加显式协议能力。 + +## Objective + +用户可显式注册并离线验证 Chat Completions、OpenAI Responses、Anthropic Messages 三类远程 Model;每类请求使用正确 endpoint/payload/headers/JSON/SSE parser,并保留既有安全、治理、恢复和证据合同。 + +## Scope + +- 扩展 Provider/Adapter 类型和数据库 check constraint,提供向前 migration 与受保护 downgrade。 +- 实现 Responses/Messages JSON 与 typed SSE、usage/metadata 归一化、URL/参数校验。 +- 让 Adapter registry、Runner、可信本地 CLI/preflight、Run snapshot 与 API CRUD 支持新类型。 +- 更新 Models/New Run UI、前端类型、组件测试、API/架构/安全/测试/状态文档。 + +## Non-goals + +- 不调用真实 OpenCode Go 或其他付费 Provider。 +- 不实现自动模型名→协议映射、跨协议 fallback、工具调用、多模态或完整供应商私有扩展。 +- 不改变题目 prompt、Evaluator、评分分母或 `llmbenchlab-protocol-v1`。 + +## Assumptions + +- `provider_type` 是现有 Adapter registry key;扩展它比增加重复的协议列更符合当前模型。 +- OpenCode Go 文档中的 `/responses` 与 `/messages` 分别遵循对应官方协议的文本生成共同子集;通过 MockTransport 固化这个子集。 +- 当前工作区起点干净,运行中的本地服务不重启、不迁移其正在使用的数据库。 + +## Requirements + +- FR-MOD-02~11:显式 Adapter、配置校验、凭据、重试、usage 和参数合同。 +- FR-RUN-02:单题隔离、稳定错误与恢复。 +- FR-REP-01~04:Run 快照与可比性不漂移。 +- NFR-SEC-01~05:无真实 Key、无真实 Provider、HTTPS/loopback、脱敏和有界响应。 + +## Implementation steps + +1. [completed] 建立失败先行协议/迁移/API/UI 测试与 ADR + - Files/modules: `backend/tests/`, `frontend/tests/`, `docs/decisions/ADR-0019*` + - Validation: 新测试在旧实现上因缺少新类型/Adapter/控件而失败。 +2. [completed] 实现后端三协议与持久化/API/Runner/CLI 联动 + - Files/modules: `backend/app/adapters/`, `models/`, `schemas/`, `runners/`, `providers/`, `cli/`, Alembic + - Validation: 定向 Adapter/API/migration/Runner/CLI 测试通过。 +3. [completed] 实现前端协议选择与参数边界 + - Files/modules: `frontend/src/api/`, `frontend/src/pages/`, `frontend/tests/` + - Validation: Models/New Run 组件测试、ESLint、TypeScript 通过。 +4. [completed] 文档与完整本地门禁 + - Files/modules: README、API、ARCHITECTURE、SECURITY、TESTING、状态/阶段/Changelog/Next Task/工作日志 + - Validation: lint、完整 test、migration、Mock smoke、frontend build、Compose config、diff/秘密检查通过。 +5. [completed] 提交、push 与精确 SHA CI + - Files/modules: Git/Actions + - Validation: 普通 push;该精确 commit 的四个必需 job 全部成功。 + +## Risks + +| 风险 | 可能性/影响 | 预防措施 | 触发后的处理 | +|---|---|---|---| +| 协议字段映射错误导致付费 4xx/重复调用 | 中/高 | 无 fallback;MockTransport 精确断言;有限 retry 只重试既有可重试分类 | 稳定失败并保留单题错误,不切换协议 | +| 新解析器接受截断流 | 中/高 | 每协议独立终止事件和 EOF 失败测试 | 丢弃部分内容,记录 `incomplete_provider_stream` | +| 迁移破坏旧 Model | 低/高 | 精确 `VARCHAR(17)→18` 列宽变更并替换 Provider 类型/远程配置两个 check;旧值逐行保持;双方言往返 | downgrade 对新类型 fail closed,否则精确恢复列宽与两个旧 check | +| 新 header/错误泄漏 Key | 低/高 | 复用 SecretStr、禁日志、递归脱敏与假 Key 回显测试 | 测试失败即不提交 | +| 参数在协议间静默丢失 | 中/中 | 新协议默认省略未配置的采样字段;unsupported seed、Messages `temperature>1`/nullable max tokens 显式拒绝并在 UI 解释 | 返回稳定配置错误,不发网络 | + +## Validation + +| 验收项 | 命令或检查 | 预期结果 | 实际结果 | +|---|---|---|---| +| Adapter 三协议 | `cd backend && uv run pytest -q tests/test_provider_protocol_adapters.py tests/test_provider_protocol_plumbing.py ...` | 全部 MockTransport 用例通过 | 通过;合并目标套件零失败 | +| API/迁移/Runner/CLI | 同一后端目标套件及隔离 PostgreSQL 16 往返 | 新类型持久化、快照与执行通过 | 通过;`0008` upgrade/check、populated downgrade 拒绝、清空后 downgrade/upgrade/check 均符合合同 | +| 前端 | `cd frontend && npm test` | 协议选择和参数 UX 通过 | `72 passed` | +| 完整门禁 | `make lint && make test && make smoke` | 零失败、无真实 Provider | 通过;backend `1079 passed, 36 skipped`,frontend `72 passed`,Mock smoke `1 passed, 7 deselected` | +| 构建/部署静态检查 | `cd frontend && npm run build`; `docker compose config --quiet` | exit 0 | 均为 exit 0;保留既有 Vite chunk warning | +| 远程门禁 | GitHub Actions exact SHA | 4/4 required jobs success | 实现 SHA [`6943aa29a154c82bdfbe5efb2578c916c3cbf632`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/6943aa29a154c82bdfbe5efb2578c916c3cbf632) 已普通 push;[run `33304667092`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33304667092) 四个必需 job 全部成功 | + +## Rollback + +停止 API/Worker 后回退应用与 migration。若数据库存在新类型 Model,downgrade 先拒绝;操作者必须先在新应用中确认无 active Run,并显式删除或转换这些 Model。旧 Chat Model/Run/Response 不重写、不删除。 + +## Documentation updates + +- [x] README / 用户操作说明 +- [x] API / Requirements +- [x] Architecture / Security / ADR +- [x] Testing / migration / CLI 说明 +- [x] CHANGELOG、PROJECT_STATUS、阶段文档、NEXT_TASK、工作日志 + +## Completion evidence + +- Changed files: Adapter、Model/API/Runner/CLI/preflight、`0008` migration、Models/New Run UI、自动化测试与合同/运维文档 +- Commands run: 目标 pytest、`make lint`、`make test`、`make smoke`、frontend build、Compose config、隔离 PostgreSQL 16 migration 往返 +- Acceptance evidence: 本地门禁全绿;实现 SHA `6943aa29a154c82bdfbe5efb2578c916c3cbf632` 的 exact-SHA Actions run `33304667092` 4/4 全绿 +- Not run: 真实 Provider(有意不运行) +- Known issues: Responses/Messages 仅覆盖纯文本评测共同子集;未对真实 OpenCode Go 当日模型、额度或账单做自动化验证 + +## Decision and discovery log + +| 日期 | 类型 | 记录 | 影响/后续 | +|---|---|---|---| +| 2026-08-30 | decision | 用三个显式 `provider_type` Adapter 值,不新增重复协议列 | 旧 `openai_compatible` 保持 Chat 默认;migration 扩展列宽并替换 Provider 类型/远程配置两个 check | +| 2026-08-30 | decision | 不做 URL/模型名自动协议推断或失败 fallback | 防止不透明的重复付费请求,Run 快照可复现 | +| 2026-08-30 | discovery | 新协议 canary 的 `max_tokens=null` 不能直接进入有限输出请求 | canary 归一化为 16;正式 Messages 请求仍要求有限非空值 | +| 2026-08-30 | decision | 只对白名单 typed transient error 重试,未知 SSE error fail closed | Responses 与 Messages 保留既有逐 attempt ledger 语义且不扩大重试面 | +| 2026-08-30 | review | Responses/Messages 模型可能拒绝隐式 Chat sampling default;discovery 的 Messages 协议也不使用 Bearer | 请求/Model 默认都省略时冻结 `temperature/top_p/seed=null`;discovery 按显式协议鉴权并有界分页 | diff --git a/docs/plans/2026-08-30-run-progress-heatmap-live-metrics.md b/docs/plans/2026-08-30-run-progress-heatmap-live-metrics.md new file mode 100644 index 0000000..3cc2465 --- /dev/null +++ b/docs/plans/2026-08-30-run-progress-heatmap-live-metrics.md @@ -0,0 +1,126 @@ +# Run Detail 热力图与实时指标执行计划 + +- Owner: Codex +- Status: completed +- Created: 2026-08-30 +- Updated: 2026-08-30 +- Related requirements: FR-API-05、FR-API-08、FR-UI-05、FR-UI-07、US-04、US-05、NFR-PERF-02、NFR-PERF-03、NFR-REL-01、NFR-UX-01、TST-03 +- Related phase: [Phase 3 — Benchmarks](../phases/PHASE-3-BENCHMARKS.md) +- Worklog: [2026-08-30-run-progress-heatmap-live-metrics.md](../worklogs/2026-08-30-run-progress-heatmap-live-metrics.md) +- ADRs: 无;本切片新增只读、向后兼容的进度投影,不改变持久化结构、评分协议或安全边界 + +## Context + +Run Detail 已每秒轮询 Run 与当前 100 条逐题证据,但运行中只有 `completed_questions` 随 Response 提交更新;`correct_questions`、严格总分、回答准确率、完成率、平均延迟和精确 Token/cost 要到终态重聚合后才完整。因此当前进度条会变化,指标卡却可能长时间保持初始值。逐题证据接口还包含 prompt/raw/reference 等大字段,不能为 12,032–20,000 题热力图每秒全量重复拉取。 + +Question 已有 Benchmark 内唯一 `position`,EvaluationResponse 对 `(run_id, question_id)` 唯一且只在题目完成后追加。可以用计划题数生成未执行槽位,以轻量 Response 投影按 absolute position 覆盖已完成格。为避免把应用时间戳或 UUID 误当成无遗漏的提交序列,进度读取采用固定 512 题 block:索引在同一数据库读取快照中返回 live metrics 与每个 block 的 `response_count`,客户端只补齐或重取计数变化的 block。实时指标由后端复用 protocol-v1 证据聚合语义,终态仍与持久化 Run 汇总核对。 + +## Objective + +在 Run Detail 增加可访问、可悬停的逐题进度热力图,并让运行中的严格总分、准确率、完成率、错误数、平均延迟、已知 Token/成本随已持久化 Response 每秒更新,同时保持大型 Benchmark 的轮询负载有界且不下载题目或回答正文。 + +## Scope + +- 新增 `GET /runs/{run_id}/progress` 只读索引接口,固定返回 `block_size=512`、计划/已完成题数、同一读取快照的证据派生 live metrics,以及全部计划 block 的 `block_index/response_count`;空 block 也以 count 0 返回。 +- 新增 `GET /runs/{run_id}/progress/blocks/{block_index}` 只读 payload 接口,只返回该绝对位置范围内已持久化格子的 `position/outcome/score/latency_ms/input_tokens/output_tokens/estimated_cost/error_type`,并按 position 升序。范围由 `block_index * 512` 派生,未返回 position 隐式为 `not_run`。 +- Run Detail 渲染绿/红/黑/白矩阵、中文图例、状态计数及鼠标悬停/键盘聚焦 Tooltip;状态不只依赖颜色。 +- Run Detail 使用索引中的后端派生 score、completion rate、answered accuracy、平均延迟、正确/异常数与 usage/cost 已知覆盖;不得从未完全 hydrate 的前端格子子集重算主指标。 +- 客户端比较索引 block count 与本地已 hydrate count,只读取非空或计数变化的 block;全部非空 block 同步完成前显示“同步中”,不能把尚未加载格误画成 `not_run`。终态先到时继续追齐 block,再停止进度轮询。 +- 保留现有逐题详情分页、取消、治理提示和终态轮询停止行为。 +- 更新 API、架构、README、测试、Roadmap/Phase/状态/Changelog/NEXT_TASK 和工作日志。 + +## Non-goals + +- 不新增数据库列或 migration,不改写历史 Run/Response/ledger。 +- 不改变 `llmbenchlab-protocol-v1` 的评分分母、完成率、answered accuracy 或 Run 精确 Token/cost all-or-nothing 语义。 +- 不新增 WebSocket/SSE、Redis UI 通道、题目执行中间态或 Provider 流式 token 级进度。 +- 不在热力图接口返回 prompt、choices、raw/parsed/reference answer、error message 或 Provider metadata。 +- 不调用真实 Provider,不把本切片并入 P2-07,也不改变 Phase 2/3 整体状态。 + +## Assumptions + +| 假设 | 依据 | 验证方法 | 不成立时的处理 | +|---|---|---|---| +| Question.position 是 Run Benchmark 内稳定、0-based、唯一的计划槽位 | 数据库唯一约束与 Runner 按 position 执行 | API 测试含乱序完成、缺口和边界 position | 若发现非法/越界 position,接口 fail closed,不把格子错位 | +| Response 对 Run/Question 唯一且追加后不更新 | `uq_responses_run_question` 与 Runner 持久化路径 | 单元/Smoke/现有恢复测试 | 若未来允许更新,block 不能只靠 `response_count` 判定变化,需新增持久化 revision/migration | +| 运行中实时指标可由当前持久化 Response 按终态同一公式派生 | NFR-REL-01 与 `aggregate_run_evidence` | 后端 fixture 与前端公式测试对照终态 Run | 任何公式漂移先统一共享合同,不创造第二套评分语义 | +| 最多 20,000 题可由 40 个固定 block 覆盖,且每秒只比较小型 index | Dataset `MAX_QUESTIONS=20_000`、正式集 12,032 题约 24 blocks | API payload 白名单、block 竞态测试、虚拟化浏览器验收 | 若实测不足,调整前端窗口化/刷新节流;不退回正文全量轮询 | + +## Requirements + +- [x] FR-API-05 / FR-API-08:进度 index/block 接口有明确 Schema、404 Run/422 block 边界、固定 512 block、`Cache-Control: no-store` 和 OpenAPI 测试,只暴露固定轻量字段。 +- [x] FR-UI-05 / US-04 / US-05:运行中热力图和指标随 Response 持久化更新,终态停止轮询,逐题详情仍可审计。 +- [x] FR-UI-07 / NFR-UX-01:四种状态有图例、文字/ARIA,Tooltip 支持 hover 与 keyboard focus,桌面/移动均可用。 +- [x] NFR-PERF-02 / NFR-PERF-03:每秒 index 有界;客户端只 hydrate 非空/计数变化的 512 题 block,并以虚拟化 ARIA grid 控制 DOM;不轮询 prompt/raw/reference 正文全集。 +- [x] NFR-REL-01:index live metrics 与 block counts 来自同一读取快照;index→block 之间的新提交不会漏格,旧 Run/旧 block 响应不能污染当前页面;terminal + reconciled 只做一次最终 Run/evidence 刷新,同路由新 `runId` 重置 evidence offset 0。 +- [x] TST-03:后端 API、前端组件、轮询竞态、空/部分/终态和离线 Smoke 均有回归。 + +## Implementation steps + +1. [completed] **冻结公共进度合同并添加失败回归** + - Files/modules: `backend/app/schemas/`、`backend/tests/`、`frontend/tests/run-detail-page.test.tsx`。 + - Validation: 测试复现四种颜色、Tooltip、fixed-block 同步、运行中指标不更新和竞态重取需求,并在实现前按预期失败。初版 cursor 合同 4 个后端 red tests 已失败;因无单调提交序列,在生产实现前替换为本 fixed-block 合同。 +2. [completed] **实现轻量后端 block 投影** + - Files/modules: `backend/app/api/v1/runs.py`、新进度 Schema、日志路由合同。 + - Validation: index/block、空 Run、稀疏/乱序 position、状态优先级、负数/越界 block 422、并发插入、OpenAPI、no-store 与秘密字段断言通过;无 migration。 +3. [completed] **实现热力图与实时指标** + - Files/modules: 前端 API types/client、独立 Heatmap 组件、Run Detail、CSS。 + - Validation: hover/focus Tooltip、绿红黑白、计数、实时准确率/完成率/Token/cost、分页并发与终态停止测试通过。 +4. [completed] **文档、完整验证、真实页面与交付闭环** + - Files/modules: README/API/ARCHITECTURE/TESTING/CHANGELOG/ROADMAP/PROJECT_STATUS/PHASE-3/NEXT_TASK/计划/工作日志。 + - Validation: 目标测试、lint、完整 test、Mock smoke、frontend build、Compose config、真实浏览器验收、秘密/diff 检查、commit/push 与精确 SHA CI 全绿。 + +## Risks + +| 风险 | 可能性/影响 | 预防措施 | 触发后的处理 | +|---|---|---|---| +| index→block 之间并发提交导致 block 比 index 更新 | 中/高 | Response 只追加且 count 单调;block payload 可比旧 index 更新,客户端采纳更大实际 count,下轮 index 收敛 | 保留并发插入/幂等 reducer 测试;未来若允许更新则新增 revision/migration | +| 12k–20k DOM 格子拖慢页面 | 中/中 | 轻量 block、虚拟化 ARIA grid、事件委托、CSS containment、只刷新变化 block | 浏览器实测不足时调窗口/节流;不能减少状态正确性或恢复正文轮询 | +| live metrics 偏离终态聚合 | 低/高 | 后端复用 `aggregate_run_evidence` 语义并在 index 同快照派生;终态 fixture 逐字段对照 | 统一后端聚合实现,禁止前端从部分 cells 创造第二套主指标 | +| 颜色对色觉/键盘用户不可用 | 中/高 | 图例、状态计数、ARIA label、focus Tooltip、可见 focus ring | 无障碍测试失败则不交付该组件 | +| Token/cost 部分证据被误当完整账单 | 中/高 | 运行中始终标“已知小计/覆盖”,只有全题全覆盖且 Run 精确字段一致才显示精确值 | 回退精确标签,保留已知小计与覆盖率 | + +## Validation + +| 验收项 | 命令或检查 | 预期结果 | 实际结果 | +|---|---|---|---| +| 后端进度 API | `cd backend && uv run pytest tests/test_run_progress_api.py tests/test_response_metadata_api.py -q` | index/block、状态、指标、竞态、字段边界、404 Run/422 block/OpenAPI/no-store 通过 | `37 passed`;初版 cursor red tests 的 `4 failed` 保留为已废弃合同的失败先行记录 | +| 前端 Run Detail | `cd frontend && npm test -- --run tests/run-detail-page.test.tsx tests/run-progress-heatmap.test.tsx` | 热力图、Tooltip、实时指标、轮询/竞态通过 | `32 passed`(Run Detail `20` + heatmap `12`);含 terminal reconciled 单次最终刷新与同路由 `runId` 切换 offset 归零回归 | +| 静态与构建 | `make lint`、`cd frontend && npm run build` | Ruff/format、ESLint、TS、production build 通过 | 两项均通过 | +| 完整回归 | `make test`、`make smoke` | 全量与离线 Mock 纵向链路通过 | backend `964 passed, 33 skipped`;frontend `64 passed`;Smoke `1 passed, 7 deselected` | +| 部署配置 | `docker compose config --quiet` | exit 0 | 通过 | +| 实页交互 | Browser 打开历史 198 题 Run | 状态/指标、四种颜色、Tooltip、移动宽度、console 无错 | Run `a3de7e4d-40b2-4d8c-994b-c713047393ae` 显示 179 passed / 17 wrong / 2 error,Token `45,509 / 4,561,625`、覆盖 `196/198`;desktop/768/375 无横向溢出,console 无 warning/error,键盘与 Tooltip 通过 | +| 大型虚拟化 | 前端自动化使用 12,032 / 20,000 题 fixture | DOM 节点有界且键盘定位正确 | 通过;这是自动化组件验证,不是 12,032/20,000 题实页浏览器性能测量 | +| 安全与范围 | `git diff --check`、staged diff、秘密扫描、`git status --short` | 无凭据/正文泄漏、无无关改动 | 本地范围/秘密/diff 与实现提交前 staged tree 复核通过 | +| 远端精确 SHA 门禁 | 普通 push 后检查 PR 与 GitHub Actions | backend、backend-integration、full-stack-reliability、frontend 全绿 | 实现 SHA [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) 已 push 到 `codex/complete-evaluation-workflow` 并进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5);[run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) 4/4 成功 | + +## Rollback + +无数据库迁移或数据写入。可反向应用本任务的 API/UI/文档补丁;旧 `/runs/{id}` 与 `/responses` 合同继续兼容。不得使用 reset/checkout 覆盖其他工作。 + +## Documentation updates + +- [x] README / `docs/REQUIREMENTS.md` / 用户操作与验收边界 +- [x] `docs/API.md` / `docs/ARCHITECTURE.md` +- [x] `docs/TESTING.md` +- [x] `CHANGELOG.md`、`docs/ROADMAP.md`、`docs/PROJECT_STATUS.md`、Phase 3、`docs/NEXT_TASK.md`、工作日志 +- [x] `docs/SECURITY.md`:无需修改;接口为同一可信本地 Run Detail 的只读固定白名单,不扩大既有安全边界 + +## Completion evidence + +- Changed files: 后端 progress Schema/service/routes、共享证据聚合、日志合同与测试;前端 API types/client、polling hook、虚拟化 Heatmap、Run Detail/CSS 与测试;本计划列出的强制文档 +- Commands run: 后端/前端定向测试、`make test`、`make lint`、`make smoke`、frontend build、`docker compose config --quiet`、可信 loopback 浏览器验收 +- Acceptance evidence: backend target `37 passed`;frontend target `32 passed`(Run Detail `20` + heatmap `12`);完整 backend `964 passed, 33 skipped`、frontend `64 passed`;Smoke `1 passed, 7 deselected`;lint/build/Compose config 通过;目标 Run 实页与三档宽度、键盘/Tooltip/console 验收通过 +- Remote evidence: commit `99791964621165c9cc7ec36b4b2d27fe04e6acd5` 已普通 push,PR #5 的 exact-SHA Actions run `33289522923` 对四个必需 job 全部成功 +- Not run: 未调用真实 Provider;12,032/20,000 只做了自动化虚拟化边界测试,未将其描述为大型真实 Run 的 DevTools 性能测量 +- Known issues: 本切片无未完成门禁;Phase 3 其余范围继续按 Roadmap 追踪,P2-07 为下一项且仍为 `planned` + +## Decision and discovery log + +| 日期时间 | 类型 | 记录 | 影响/后续 | +|---|---|---|---| +| 2026-08-30 | discovery | 运行中仅 `completed_questions` 逐题写回;正确数、准确率、平均延迟等终态才聚合 | 实时指标必须基于持久化 Response 只读投影,不能只改 CSS | +| 2026-08-30 | deviation | 初版先写了 `(created_at,id)` cursor 合同,4 个后端 red tests 按预期失败;复核发现 Response 没有数据库单调提交序列,应用 `created_at` 与 UUID 不能证明并发提交无遗漏 | 在任何生产实现前废弃 cursor,测试与文档改为固定 512 absolute-position block;本切片仍无需 migration | +| 2026-08-30 | decision | `/progress` index 在同一读取快照返回 live metrics 与所有 block counts;`/progress/blocks/{block_index}` 返回固定白名单 absolute-position cells | 每秒 index 有界,乱序完成和 index→block 并发可通过单调 response_count 最终收敛 | +| 2026-08-30 | decision | live 主指标由后端按 protocol-v1 证据派生;前端只展示 index 结果和同步 block,不从部分 Map 重算 | 避免同步窗口制造成绩漂移;Run 精确 nullable Token/cost 语义不变 | +| 2026-08-30 | decision | 未执行格只需要隐式 position,不返回未执行题正文/答案;已完成格只返回 Tooltip 必需字段 | 热力图不扩大正文暴露面,轮询负载有界 | diff --git a/docs/worklogs/2026-08-30-fix-run-detail-metrics.md b/docs/worklogs/2026-08-30-fix-run-detail-metrics.md new file mode 100644 index 0000000..cdd6483 --- /dev/null +++ b/docs/worklogs/2026-08-30-fix-run-detail-metrics.md @@ -0,0 +1,150 @@ +# 2026-08-30 — 修复 Run Detail 错题与部分 Token 展示工作日志 + +> 本日志记录实际发生的工作,不是事后美化的总结。所有命令以仓库根目录为基准。 + +## 元信息 + +- 日期:2026-08-30 +- 执行者:Codex +- 关联阶段:[Phase 2 — Reliability](../phases/PHASE-2-RELIABILITY.md)、[Phase 3 — Benchmarks](../phases/PHASE-3-BENCHMARKS.md) +- 关联计划:[2026-08-30-fix-run-detail-metrics.md](../plans/2026-08-30-fix-run-detail-metrics.md) +- 关联 ADR:无;不改变既有协议/架构决定 +- 最终状态:completed + +## 初始仓库状态 + +- 当前分支:`codex/complete-evaluation-workflow`,HEAD `bbf6e87`,跟踪 `origin/codex/complete-evaluation-workflow` +- `git status --short --branch` 摘要:只有分支行,无未提交改动 +- 已有未提交改动:无 +- 相关功能与测试现状:Run Detail 直接把 `error_questions` 标为“错误题”;Response 列表 API 只有逐题 nullable usage,没有全量已知小计/覆盖率;Smoke 固化精确 Run Token 任一缺失即为 `null` +- 环境约束:本地 SQLite/API 可只读核查;自动测试禁止真实 Provider;本任务不修改现有数据库 + +## 本次目标与背景 + +用户指出 Run `a3de7e4d-40b2-4d8c-994b-c713047393ae` 的错误题数量与 Token 展示不正确。只读核查确认页面把 2 条执行异常误导性地呈现为全部错题,而 196 条已知 usage 因另外 2 条缺失而被精确总量的 `null` 完全遮蔽。本任务修复信息表达并保留协议完整性。 + +## 范围 + +- Response 列表 API 增加与分页无关的已知 Token 小计和 usage 报告数。 +- Run Detail 显示未得分、普通答错、执行异常和正确数。 +- Run Detail 在精确 Token 缺失时展示已知小计、覆盖率与完整总量未知提示。 +- 更新自动化测试和相关 API/测试/状态文档。 + +## 非目标 + +- 不回填、估算或修改历史 usage、Response、ledger、成绩或数据库 schema。 +- 不改变 Run 精确 Token、成本、排行榜或 Dashboard 的现有协议语义。 +- 不调用真实 Provider,不检查或记录凭据。 + +## 验收标准 + +- [x] 目标口径能表达正确 179、普通答错 17、执行异常 2、未得分 19。 +- [x] 部分 usage 能表达已知输入 45,509、输出 4,561,625、196/198 已报告且完整总量未知。 +- [x] 全 usage、部分 usage、零 Response 与分页场景有后端/前端回归。 +- [x] API/测试/状态文档与实现一致,既有 protocol-v1 精确聚合不变。 +- [x] 目标测试、lint、完整 test、smoke、frontend build 和 Compose config 有真实结果。 + +## 假设 + +- `completed_questions - correct_questions` 表示已持久化 Response 中未得分题数;依据为 protocol-v1 单题二元评分。 +- 已知 Token 小计只作为 Response 证据下界展示;Provider retry、异常调用和账单真值不由该小计覆盖。 +- responses API 的 aggregate 不应用 offset/limit,因此跨页返回相同全量 summary。 + +## 风险 + +| 风险 | 影响 | 缓解措施 | 结果 | +|---|---|---|---| +| 部分 Token 被误当精确总量 | 费用/规模判断失真 | 明确标“已知小计”并同时显示覆盖率与“完整总量未知” | 目标实页与组件测试确认 | +| 错题与异常继续混淆 | 用户无法核对成绩 | 主指标使用未得分,辅助拆分普通答错/执行异常 | 目标实页显示 19/17/2 | +| API 新字段影响前端通用类型 | 类型或调用回归 | responses 使用专用 response 类型,字段仅追加 | typecheck、全量测试与 build 通过 | + +## 实施步骤 + +1. [completed] 冻结 API/UI 语义并添加失败回归。 +2. [completed] 实现后端只读聚合和前端展示。 +3. [completed] 更新文档并完成本地/远程门禁。 + +## 实际修改 + +| 文件/模块 | 修改内容 | 对应需求/原因 | +|---|---|---| +| `docs/plans/2026-08-30-fix-run-detail-metrics.md` | 建立跨后端/API/前端执行计划 | AGENTS/PLANS 强制流程 | +| 本工作日志 | 冻结目标、范围、风险和验收 | AGENTS 强制流程 | +| `backend/app/schemas/evaluation_response.py` | 为 Responses 列表增加四个非负、带说明的 Run-wide usage summary 字段 | 公共 API 明确部分 usage 证据 | +| `backend/app/api/v1/runs.py` | 用一次聚合查询同时返回总数、输入/输出已知小计与独立上报数 | 不增加查询次数且不受分页影响 | +| `backend/tests/test_response_metadata_api.py`、`backend/tests/test_smoke.py` | 覆盖 OpenAPI、零/全/部分/非对称 usage、合法零 Token 与分页 | 防止字段漂移或部分小计回填 Run 精确值 | +| `frontend/src/api/types.ts`、`frontend/src/api/client.ts` | 为 Responses 定义专用列表类型 | 不污染通用 `ListResponse` | +| `frontend/src/pages/RunDetailPage.tsx` | 显示未得分/普通答错/执行异常;显示精确或明确不完整的 Token 小计 | 修复用户可见误导并处理并行快照竞态 | +| `frontend/tests/run-detail-page.test.tsx` | 覆盖目标 179/17/2、页内拆分、精确/部分/零/非对称 Token、快照不一致和分页 | 固化可见文案与保守语义 | +| `README.md`、`docs/API.md`、`docs/TESTING.md`、`CHANGELOG.md` | 同步用户、API、测试与变更说明 | 文档与实现一致 | + +## 决定、偏差与发现 + +| 时间 | 类型 | 事实与理由 | 后续影响 | +|---|---|---|---| +| 09:31 CST | discovery | 198 条 Response 中 179 正确、17 普通答错、2 异常;196 条有 usage | 修复展示而不改成绩事实 | +| 09:31 CST | decision | 追加分页无关的 usage evidence summary,保留 Run 精确 Token `null` | 公共 API 只做向后兼容字段扩充,需更新 API 文档和测试 | +| 09:40 CST | test | 失败先行回归按预期失败:后端缺少 summary 字段,前端仍显示旧“错误题”与破折号 | 证明测试能捕获原缺陷后才实施 | +| 09:43 CST | discovery | Run 与 Responses 并行请求可能来自相邻快照;两个边际 reported count 相等不等于同题完整 usage | 精确 Token 需题数、全覆盖、小计一致;部分文案分别说明输入/输出覆盖 | + +## 实际运行命令 + +| 命令 | 目的 | 退出码 | 结果摘要 | +|---|---|---:|---| +| `git status --short --branch` | 确认初始工作区 | 0 | 分支干净,无未提交改动 | +| `git rev-parse --show-toplevel` / `git log -1 --oneline --decorate` | 确认仓库与 HEAD | 0 | 仓库根为当前目录,HEAD `bbf6e87` | +| `cd backend && uv run pytest tests/test_response_metadata_api.py -q`(实现前) | 验证失败回归 | 1 | `1 passed, 1 failed`;新增 summary KeyError,符合预期 | +| `cd frontend && npm test -- --run tests/run-detail-page.test.tsx`(实现前) | 验证失败回归 | 1 | `5 passed, 4 failed`;旧标题/Token/页头不满足新要求 | +| `cd backend && uv run pytest tests/test_response_metadata_api.py tests/test_smoke.py -q` | 目标 API/纵向回归 | 0 | `11 passed`;仅有既有上游弃用 warning | +| `cd frontend && npm test -- --run tests/run-detail-page.test.tsx tests/format.test.ts` | 目标 UI/格式回归 | 0 | `20 passed` | +| `git diff --check && make lint`(首次) | 格式/静态门禁 | 2 | 逻辑 lint 通过,仅两个 Python 测试文件需 Ruff format;如实保留 | +| `cd backend && uv run ruff format tests/test_response_metadata_api.py tests/test_smoke.py && cd .. && make lint` | 修正并重跑静态门禁 | 0 | Ruff/format、ESLint、TypeScript 全绿 | +| `make test` | 完整自动化回归 | 0 | backend `951 passed, 33 skipped`;frontend `47 passed`;仅既有上游弃用 warning | +| `make smoke` | 离线纵向验收 | 0 | `1 passed, 7 deselected`,只用 Mock | +| `cd frontend && npm run build` | production build | 0 | 2192 modules 成功;保留既有 663.81 kB chunk warning | +| `docker compose config --quiet` | 部署配置解析 | 0 | 无输出,配置有效 | +| 本地 API 读取目标 Run Responses summary | 真实目标记录只读验真 | 0 | total 198;known input/output `45,509/4,561,625`;两侧 `196/198` | +| Browser 打开目标 Run Detail、翻到第二页并检查 console | 实页视觉/交互验收 | 0 | 首屏 19/17/2、已知小计 460.7万;第二页 8 未得分/2 执行异常;console error 0 | +| `git diff --check` + added diff/untracked 高置信 secret scan | 最终范围、空白与秘密复核 | 0 | 无 whitespace error、无 Key/Bearer/private-key 命中;18 个候选文件均在计划范围 | +| `git commit -m "fix: clarify run detail result metrics"` / `git push origin codex/complete-evaluation-workflow` | 形成并发布实现阶段 | 0 | commit `0003e4291769a851005ba46c7e59b156a6b789eb`;远端分支与本地一致 | +| `gh pr create ...` | 触发分支远程门禁 | 0 | [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5) 已创建;未执行合并 | +| `gh run watch 33286730109 --exit-status` | 等待实现精确 SHA CI | 0 | backend、真实 PostgreSQL/Redis integration、real-Compose acceptance、frontend 4/4 success | + +## 测试结果 + +- 通过:后端目标 `11 passed`;前端目标 `20 passed`;完整 backend `951 passed, 33 skipped`、frontend `47 passed`;lint/build/smoke/config/实页验收全绿 +- 失败:实现前失败回归分别为后端 `1 failed`、前端 `4 failed`,均在实现后转绿 +- Lint/typecheck/build:全绿;build 只保留既有大 chunk warning +- Smoke/Docker:离线 Smoke 与 Compose config 通过 + +## 未运行验证 + +- 真实 Provider 未运行(有意):没有 API Key,也不需要产生模型费用。本地没有另跑真实 PostgreSQL/Redis integration 或 real-Compose acceptance;实现精确 SHA 的远程 CI 已实际运行这两项并通过。 + +## 未完成项 + +- 无功能、测试或远程门禁未完成项。PR #5 保持打开;合并不在本次授权范围,也不是本阶段 commit/push/exact-SHA CI 的完成条件。 + +## 已知问题与限制 + +- 两条历史 Provider 异常没有 usage,无法恢复精确完整 Token;只能展示已知小计与覆盖率。 + +## 安全检查 + +- 真实密钥扫描:added diff/untracked 高置信模式无命中;实现/API/浏览器核对未读取或记录凭据正文 +- 真实 Provider API 调用:否;只读取本地 API 的目标 Run 汇总 +- 日志/API 脱敏:本任务不新增正文或 Provider-controlled 聚合字段 +- 危险 Git 操作(force push/reset 等):无 +- 阶段 push:实现 commit `0003e429…` 已普通 push;无 force push +- 远程 CI:[run `33286730109`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33286730109) 对实现精确 SHA 4/4 success;只有 Node.js 20 action deprecation annotation,无失败 +- 遗留安全风险:已知 Token 小计不是 Provider 账单真值,必须保持明确文案 + +## 结果与下一步 + +本维护为 `completed`:实现、文档、本地门禁、普通 push 和实现精确 SHA CI 均已闭环。下一独立任务仍是 `docs/NEXT_TASK.md` 中的 P2-07 最小只读 recovery verifier;本次不合并 PR、不开始 P2-07。 + +## 最终 Git 状态 + +```text +实现 commit/push 后 clean;证据收尾文档将形成独立 commit,最终状态在交付回复中复核。 +``` diff --git a/docs/worklogs/2026-08-30-multi-worker-evaluation.md b/docs/worklogs/2026-08-30-multi-worker-evaluation.md new file mode 100644 index 0000000..43c4a27 --- /dev/null +++ b/docs/worklogs/2026-08-30-multi-worker-evaluation.md @@ -0,0 +1,157 @@ +# 2026-08-30 — 多 Worker 并行评测工作日志 + +> 本日志记录实际发生的工作,不是事后美化的总结。所有命令以仓库根目录为基准。 + +## 元信息 + +- 日期:2026-08-30 +- 执行者:Codex +- 关联阶段:[Phase 2 — Reliability](../phases/PHASE-2-RELIABILITY.md) +- 关联计划:[多 Worker 并行评测](../plans/2026-08-30-multi-worker-evaluation.md) +- 关联 ADR:[ADR-0005](../decisions/ADR-0005-durable-task-execution.md)、[ADR-0009](../decisions/ADR-0009-database-governance-audit-fair-scheduling.md) +- 最终状态:completed + +## 初始仓库状态 + +- 当前分支:`codex/complete-evaluation-workflow` +- `git status --short --branch` 摘要:`## codex/complete-evaluation-workflow...origin/codex/complete-evaluation-workflow`,工作树干净 +- 已有未提交改动:无 +- 相关功能与测试现状:PostgreSQL/Redis/租约/fencing 已有双 Worker Mock acceptance/capacity 证据;`scripts/dev.sh` 与普通 `make docker-up` 均只启动一个 Worker +- 环境约束:当前个人 `.env` 使用 SQLite 且未配置 Redis;不得直接多开 SQLite Worker,不得改动当前 Run 或调用真实 Provider + +## 本次目标与背景 + +当前一个长期 Provider SSE 请求会占住唯一 Worker,使其他 Benchmark Run 保持 pending。目标是在既有数据库租约架构上补齐可配置多 Worker启动与验证,让 PostgreSQL 模式下多个数据集 Run 可并行执行,并保持 SQLite 兼容路径安全失败。 + +## 范围 + +- 本地 PostgreSQL dev 多 Worker进程管理、独立日志与退出清理 +- Compose 默认双 Worker与显式副本配置、expected/live gauges 同步 +- 不同 Benchmark Run 的多 Worker领取回归 +- README、架构、部署、测试、运维和状态文档同步 + +## 非目标 + +- 不支持 SQLite 多 Worker,不自动迁移当前 SQLite 数据 +- 不改变数据库 schema、REST API、评分协议、Provider transport 或 governance policy +- 不重启当前用户服务、不取消/重试现有 Run、不访问真实 Provider + +## 验收标准 + +- [x] PostgreSQL 本地 dev 可配置启动至少两个独立 Worker,且正确清理所有进程 +- [x] SQLite 多 Worker与非法副本数在任何服务启动前失败 +- [x] Compose 默认两个 Worker,按方向同步 API expected/scale,并自动验证 expected/registered/live/stalled/shortfall=`2/2/2/0/0` +- [x] 两个不同 Benchmark Run 可由不同 Worker并发领取,同一 Run 不会重复拥有有效 lease +- [x] 自动化只用 Mock/Stub,相关 lint/test/smoke/config 通过 +- [x] 强制文档、commit/push 与精确 SHA CI 完成 + +## 假设 + +- 多 Worker核心正确性已由 ADR-0005 和现有 PostgreSQL tests/acceptance 给出;本任务增加不同 Benchmark 和日常入口覆盖。 +- 当前容量资格只覆盖两个 Worker,因此默认 2;更高数量允许显式配置但不宣称已资格。 + +## 风险 + +| 风险 | 影响 | 缓解措施 | 结果 | +|---|---|---|---| +| SQLite 被误用为多 Worker共享数据库 | 锁争用、不可支持的恢复语义 | 启动前方言检查并固定错误 | launcher 回归确认在 keyring/log/子进程前失败且不输出 DSN | +| Worker规模与 expected metrics 漂移 | shortfall 误报或漏报 | 同一副本参数驱动两者并做启动后检查 | fake/真实 Compose 均确认五 gauges 必须精确收敛 | +| Provider总并发/费用被放大 | 限流、费用和长尾增加 | 默认 2,记录 Worker×Run并发与治理边界 | 文档明确总并发近似 Worker×Run concurrency;真实 Provider 未运行 | + +## 实施步骤 + +1. [completed] 完成只读勘察、计划、日志与测试合同 +2. [completed] 实现本地/Compose 多 Worker启动入口 +3. [completed] 增加 launcher、真实 PostgreSQL 与真实 Compose 回归 +4. [completed] 同步文档、运行完整门禁、commit/push 并等待精确 SHA CI + +## 实际修改 + +| 文件/模块 | 修改内容 | 对应需求/原因 | +|---|---|---| +| `docs/plans/2026-08-30-multi-worker-evaluation.md` | 新建复杂任务执行计划 | AGENTS/PLANS 强制流程 | +| `docs/worklogs/2026-08-30-multi-worker-evaluation.md` | 新建事实工作日志 | AGENTS 强制流程 | +| `Makefile`、`.env.example` | 暴露 `DEV_WORKERS` / `WORKERS` 入口与安全默认 | 日常可操作入口 | +| `scripts/dev.sh` | PostgreSQL 1–32 Worker进程、独立日志、expected、SQLite fail-fast、TERM cleanup | 本地开发并行执行与无遗留进程 | +| `scripts/compose_up.sh`、`compose.yaml` | 默认 2、方向化扩缩、active scan、API-only expected、五 gauges 有界门禁 | 落实 ADR-0016 且 fail closed | +| `backend/tests/test_dev_script.py`、`test_compose_up_script.py` | 假子进程/Docker、输入/环境/Make、SIGINT/TERM、扩缩顺序和失败路径 | 启动器离线回归 | +| `backend/tests/integration/test_postgres_leases.py` | 两个 Benchmark Run 的不同 owner 并发领取和同 Run lease 唯一 | 真实 PostgreSQL 正确性 | +| `README.md`、`docs/ARCHITECTURE.md`、`docs/DEPLOYMENT.md`、`docs/OPERATIONS.md`、`docs/TESTING.md` | 使用方法、执行槽、扩缩顺序、容量/迁移/Provider并发边界 | 用户与运维合同同步 | +| `CHANGELOG.md`、`docs/PROJECT_STATUS.md`、Phase 2、`docs/NEXT_TASK.md` | 进行中事实、证据和下一任务保持 | 仓库状态闭环 | + +## 决定、偏差与发现 + +| 时间 | 类型 | 事实与理由 | 后续影响 | +|---|---|---|---| +| 15:00 CST | discovery | 当前个人模式为 SQLite + 单 Worker;底层 PostgreSQL/Compose 已有双 Worker资格 | 不重写调度核心,补齐入口与跨 Benchmark证据 | +| 15:00 CST | decision | 不直接 fork 多个 SQLite Worker | 多 Worker模式必须验证 PostgreSQL DSN | +| 15:00 CST | decision | 不新增 ADR | 既有 ADR-0005/0009 已明确批准受限 PostgreSQL 多 Worker,本任务不改变其不变量 | +| 15:05 CST | diagnosis | API gauges 为 pending/due/running=`3/3/1`;唯一 Worker仍 live/续租但最近题进度停滞,当前 Run 为 concurrency 4、read timeout 300 秒、最多 3 次 HTTP attempt | 单个上游长请求批次最坏约占用一个 Worker 15 分钟;多 Worker隔离执行槽,但不把总超时冒充已解决 | +| 15:25 CST | review | 初版 Compose只校验三 gauges、未按 ADR-0016 排序,超长十进制还可在 Bash 3.2 回绕 | 改为字符串 `1..32`、扩容 scan→API、缩容 API→Worker、最终 `N/N/N/0/0`;stale generation 不能通过 | +| 15:35 CST | review | 子进程可能继承忽略 SIGINT | 主 launcher 保留 130/143,cleanup 固定 TERM 并回归所有子进程终止 | +| 16:20 CST | review | 终审发现旧 stale generation 可误满足 scan count,且仅统计 running container 会误判含 exited replica 的缩容方向 | 改为应用 DB 时钟 watermark + fresh live scan 双门禁,并分别统计 all/running replica;新增两项回归,复审确认 0 Blocker/High/Medium | + +## 实际运行命令 + +| 命令 | 目的 | 退出码 | 结果摘要 | +|---|---|---:|---| +| `git status --short --branch` | 初始状态 | 0 | 当前分支跟踪 origin,工作树干净 | +| `docker compose config --quiet`(只读勘察代理) | 现有 Compose 配置校验 | 0 | 配置可解析;默认仍为一个 Worker | +| 现有 acceptance/capacity 目标测试(只读勘察代理) | 核对双 Worker现有合同 | 0 | 127 项通过;未修改文件 | +| `cd backend && uv run pytest tests/test_dev_script.py tests/test_compose_up_script.py -q`(最终目标版本) | launcher/Make/fake Docker完整边界 | 0 | `42 passed`;仅既有 deprecation warnings | +| `bash -n`、ShellCheck、Ruff check/format | shell/Python静态检查 | 0 | 两 launcher 与两个目标测试文件通过 | +| 隔离 PostgreSQL 16 目标 lease tests(空库首次) | 真实方言并发 | 1 | fixture setup 两项 `UndefinedTable`;原因是临时库未先迁移,未进入测试断言 | +| 隔离 PostgreSQL 16 `alembic upgrade head` 后重跑相同 tests | 真实方言并发 | 0 | `2 passed`;临时容器按精确名称清理 | +| 隔离 Compose 冷启动(第一次) | 默认双 Worker真实入口 | 2 | 一个 Worker container exit 1,wrapper正确失败;trap cleanup C/V/N 全空,但容器日志随首次清理丢失,未把它写成通过 | +| 隔离 Compose 冷启动(第二次) | 复现并在失败时保留日志 | 0 | `expected/registered/live/stalled/shortfall=2/2/2/0/0`,随后精确清理 | +| 隔离 Compose `cold 2 → scale 1 → scale 2` | ADR-0016 方向化扩缩 | 0 | gauges 依次 `2/2/2/0/0`、`1/1/1/0/0`、`2/2/2/0/0`;cleanup C/V/N=`0/0/0` | +| 终审修复后隔离 Compose `cold 2 → scale 1 → scale 2` | fresh/watermark 与 all/running replica 真实验证 | 0 | gauges 再次为 `2/2/2/0/0`、`1/1/1/0/0`、`2/2/2/0/0`;唯一 project `llmbenchlab-mwfix-7f3a21` cleanup C/V/N/image tags=`0/0/0/0`;image tag 可由构建恢复 | +| `make lint` | Ruff/format、ESLint、TypeScript | 0 | 160 个 Python 文件、前端 lint/typecheck 全部通过 | +| `make test`(终审修复后重跑) | 完整后端/前端回归 | 0 | backend `1003 passed, 35 skipped`;frontend 10 files / `64 passed` | +| `make smoke` | 离线端到端链路 | 0 | 明确使用 Mock + 临时 SQLite;`1 passed, 7 deselected` | +| `cd frontend && npm run build` | production build | 0 | 2194 modules;构建通过,仅既有 >500 kB chunk warning | +| `docker compose config --quiet`、`git diff --check` | 部署/格式门禁 | 0 | 无输出、无 warning/whitespace error | +| `git diff --cached --check`、staged name/stat/diff 与三类秘密/debug 扫描 | 提交前范围与安全复核 | 0 | 19 files;secret/key patterns、敏感路径、debug marker 均为 0;无 unstaged diff | +| `git commit`、`git push origin codex/complete-evaluation-workflow` | 实现交付 | 0 | implementation `b06594c2df67d6e2a8b117651b193cd0fa409bf5` 已普通 push;无 force push | +| `gh run watch 33299883513 --exit-status`、`gh run view 33299883513` | 精确实现 SHA 远程门禁 | 0 | backend、PostgreSQL/Redis integration、frontend、real Compose reliability 四个必需 job 全 success | + +## 测试结果 + +- 通过:launcher `42`、迁移后真实 PostgreSQL `2`、终审修复前后真实 Compose 冷启动与 `2→1→2` 扩缩;隔离资源均清理 +- 失败:首次空 PostgreSQL 因未迁移 setup error,按部署顺序修正后通过;第一次真实 Compose冷启动有一个 Worker exit 1,wrapper安全失败,随后两次冷启动和完整扩缩均通过 +- Lint/typecheck/build:完整门禁全部通过;build 仅有既有 chunk-size warning +- Smoke/Docker:真实 Compose只使用空 PostgreSQL/Redis且没有 Model/Run;未调用真实 Provider +- 远程:implementation exact-SHA run `33299883513` 4/4 success;Node 20 action deprecation annotation 为 GitHub Action维护提示,不是功能失败 + +## 未运行验证 + +- 真实 Provider与 3+ Worker容量资格未运行(有意,均不在本任务支持声明内)。 + +## 未完成项 + +- 无功能未完成项;P2-07 保持下一独立任务。 + +## 已知问题与限制 + +- 当前个人 SQLite 中的 Model/Benchmark/Run 不会自动出现在新的 PostgreSQL Compose volume;迁移必须另选停写维护窗口执行现有 importer。 +- 三个以上 Worker和真实 Provider容量未资格。 + +## 安全检查 + +- 真实密钥扫描:初始与 staged diff 均未发现;未读取 `.env` 内容或 keyring +- 真实 API 调用:否 +- 日志/API 脱敏:未改变 +- 危险 Git 操作(force push/reset 等):无 +- 阶段 push:implementation SHA `b06594c…` 已普通 push +- 远程 CI:run `33299883513` 四个必需 job 全 success +- 遗留安全风险:现有可信 loopback与 Provider SSRF边界不变 + +## 结果与下一步 + +任务完成:PostgreSQL 日常多 Worker入口、本地/Compose安全边界、并发租约证据和本地/远程门禁全部闭环;P2-07 仍是下一独立任务。 + +## 最终 Git 状态 + +```text +implementation commit/push 后为空;本 closeout 文档提交完成后再次核对 +``` diff --git a/docs/worklogs/2026-08-30-provider-api-protocols.md b/docs/worklogs/2026-08-30-provider-api-protocols.md new file mode 100644 index 0000000..afccfc4 --- /dev/null +++ b/docs/worklogs/2026-08-30-provider-api-protocols.md @@ -0,0 +1,149 @@ +# 2026-08-30 — Provider API 三协议适配工作日志 + +> 本日志记录实际发生的工作,不是事后美化的总结。所有命令以仓库根目录为基准。 + +## 元信息 + +- 日期:2026-08-30 +- 执行者:Codex +- 关联阶段:[Phase 2 — Reliability](../phases/PHASE-2-RELIABILITY.md) +- 关联计划:[执行计划](../plans/2026-08-30-provider-api-protocols.md) +- 关联 ADR:[ADR-0019](../decisions/ADR-0019-explicit-provider-api-protocol-adapters.md) +- 最终状态:completed(实现提交已普通 push,exact-SHA CI 四个必需 job 全绿) + +## 初始仓库状态 + +- 当前分支:`codex/complete-evaluation-workflow`,跟踪 `origin/codex/complete-evaluation-workflow` +- `git status --short --branch` 摘要:分支同步,工作区无已有改动 +- 已有未提交改动:无 +- 相关功能与测试现状:`openai_compatible` 仅实现 Chat Completions;离线定向测试 `2 passed`,明确断言 `/chat/completions` 与 Chat payload +- 环境约束:可联网读取官方文档;自动化不得调用真实 Provider;不重启当前本地 API/Worker/frontend,不迁移其活动数据库 + +## 本次目标与背景 + +OpenCode Go 按模型要求 `/chat/completions`、`/responses` 或 `/messages`。用户确认直接修改项目,使三种协议都能在现有可审计评测链路中显式配置和执行。 + +## 范围 + +- Adapter 类型、URL/payload/headers/JSON/SSE/usage 归一化 +- Model/API/Run snapshot/Runner/CLI/preflight 与 Alembic migration +- Models/New Run UI 与前端类型/测试 +- ADR、API、安全、测试、架构、状态及发布文档 + +## 非目标 + +- 不真实调用 OpenCode Go,不验证套餐余额、当日模型可用性或真实费用 +- 不实现 tools、多模态、自动协议推断或协议 fallback +- 不改变 Benchmark/评分协议 + +## 验收标准 + +- [x] 旧 `openai_compatible` Chat 行为与历史数据无回归 +- [x] Responses/Messages 普通 JSON 与 typed SSE 都能归一化文本、usage 和 metadata +- [x] 已知错误 endpoint、unsupported seed、Messages null max tokens 在外发前失败 +- [x] Model API、Run snapshot、Worker/CLI 与 Web 可显式选择三类 Adapter +- [x] 双方言 migration、完整 lint/test/Mock smoke/build/Compose config 通过 +- [x] 普通 push 后精确 SHA GitHub Actions 四个必需 job 全绿 + +## 假设 + +- `provider_type` 是 Adapter key,不另增 `api_protocol` 列;由 ADR-0019 固化。 +- Responses/Messages 只实现文本评测所需的官方共同子集;上游私有扩展仍不承诺。 + +## 风险 + +| 风险 | 影响 | 缓解措施 | 结果 | +|---|---|---|---| +| 协议映射错误 | 请求失败或额外费用 | 精确 MockTransport、无 fallback、有限 retry | 三协议 endpoint/payload/header 与已知错误 suffix 回归通过 | +| 截断流被当成功 | 错误评分证据 | 每协议终止事件与 EOF fail-closed | `[DONE]` / `response.completed` / `message_stop` 与截断流回归通过 | +| 迁移回退丢新配置 | 配置丢失 | populated downgrade guard | 隔离 PostgreSQL 中有新类型时 DDL 前拒绝,清空后可安全往返 | +| Key 经新增 header/错误泄漏 | 凭据泄漏 | 不记录 headers、递归脱敏、假 Key 测试 | 假 Key 反射/错误脱敏与 credential 回归通过 | + +## 实施步骤 + +1. [completed] 建立 ADR/计划/日志和失败先行测试 +2. [completed] 实现后端三协议、迁移与执行链联动 +3. [completed] 实现前端协议与参数 UX +4. [completed] 文档、完整本地门禁与技术/安全终审 +5. [completed] commit、普通 push 与 exact-SHA CI + +## 实际修改 + +| 文件/模块 | 修改内容 | 对应需求/原因 | +|---|---|---| +| `docs/decisions/ADR-0019-*` | 固化显式 Adapter 类型、参数、终止与回滚边界 | 架构/公共 API/迁移变更前置决定 | +| `docs/plans/2026-08-30-provider-api-protocols.md` | 建立可持续执行计划 | 跨数据库/后端/前端复杂任务 | +| `backend/app/adapters/` | 新增 Responses/Messages Adapter,保留 Chat 默认;实现 endpoint、JSON/SSE、usage、typed retry 与参数 fail-fast | 三协议执行核心 | +| `backend/app/models/`、`schemas/`、`api/v1/`、`runners/`、`services/` | 扩展显式类型并冻结到 Run snapshot | 公共 API 与可靠执行链一致 | +| `backend/app/providers/`、`cli/evaluate.py` | `/models` 推导、协议鉴权、Messages bounded pagination 和按显式协议执行 canary | 可信本地正式入口不再绑定 Chat | +| `backend/alembic/versions/20260830_0008_provider_api_protocols.py` | `provider_type` `VARCHAR(17)→18` 并替换 Provider 类型/远程配置两个 check;有新类型时 downgrade fail closed | 历史 Chat 数据兼容与可审计回滚 | +| `frontend/src/`、`frontend/tests/` | Models 显式协议选择、New Run sampling/seed/max token 边界及回归 | Web 可配置且不静默丢参数 | +| README 与 `docs/` | 更新 API、架构、安全、测试、部署、状态和运维合同 | 用户/运维说明与实现一致 | + +## 决定、偏差与发现 + +| 时间 | 类型 | 事实与理由 | 后续影响 | +|---|---|---|---| +| 2026-08-30 Asia/Shanghai | discovery | OpenCode Go 当前模型分属三种 endpoint;现有 Adapter 只识别 Chat | 需要独立 payload/parser,不能只改 URL | +| 2026-08-30 Asia/Shanghai | decision | 扩展 Adapter `provider_type`,旧值继续表示 Chat | 保持旧 API/DB/Run snapshot 兼容 | +| 2026-08-30 Asia/Shanghai | discovery | 前端失败先行用例暴露缺少 Chat Completions 选择项及新协议 seed 边界 | 增加三项显式选择并在非 Chat 时禁用/清空 seed | +| 2026-08-30 Asia/Shanghai | discovery | canary 直接转发 `max_tokens=null` 会形成无界探测或 Messages 配置错误 | 新协议 canary 固定为最小有限 16;正式 Messages null 在外发前拒绝 | +| 2026-08-30 Asia/Shanghai | review | 只按 HTTP 状态不足以表达 Responses/Messages SSE transient error | 增加协议 typed transient 白名单;未知流错误保持 fail closed,每次重试独立 ledger 结算 | +| 2026-08-30 Asia/Shanghai | review | Responses/Messages 的隐式 Chat sampling default 会让部分模型在请求解析阶段拒绝;Messages discovery 也不接受 Bearer-only 假设 | 请求/Model 默认都省略时冻结 `temperature/top_p/seed=null`;Messages 另限制 `temperature<=1`,discovery 按协议鉴权并跟随有界 `after_id` 分页 | +| 2026-08-30 Asia/Shanghai | review | 多页 Messages discovery 和解析异常仍需独立资源/秘密边界 | `after_id` 聚合限制为 100 页/60 秒/2 MiB/10k entries 并拒绝重复 cursor;malformed JSON/SSE、oversized 与 transport 错误均不链回原始 Provider 内容 | + +## 实际运行命令 + +| 命令 | 目的 | 退出码 | 结果摘要 | +|---|---|---:|---| +| `cd backend && uv run pytest -q tests/test_adapters.py::test_openai_compatible_sends_chat_completion_fields tests/test_provider_preflight.py::test_chat_canary_uses_run_fields_and_requires_parseable_a` | 确认现有 Chat 合同 | 0 | `2 passed`;仅 MockTransport | +| `cd frontend && npm test -- --run tests/models-page.test.tsx tests/new-run-page.test.tsx`(失败先行) | 固化 Web 协议选择/参数边界 | 1 | `2 failed, 12 passed`;缺少 Chat 选项和 seed 行为,符合预期红灯 | +| `cd backend && uv run pytest -q tests/test_provider_protocol_adapters.py tests/test_provider_protocol_plumbing.py tests/test_provider_preflight.py tests/test_evaluation_cli.py tests/test_api.py tests/test_migrations.py tests/test_web_credentials.py tests/test_evaluation_runner_reliability.py` | 三协议执行、API、迁移、Runner、CLI 合并目标回归 | 0 | 全部通过;仅 MockTransport/本地数据库 | +| `make lint`(首次) | 静态门禁 | 1 | 仅 7 个本次 Python 文件需要 Ruff format;随后机械格式化并重跑通过 | +| `make lint`(最终) | Ruff/format、ESLint、TypeScript | 0 | 全部通过;164 个 Python 文件 format clean | +| `make test` | 完整后端/前端回归 | 0 | backend `1079 passed, 36 skipped`;frontend `72 passed` | +| `make smoke` | 离线垂直链路 | 0 | `1 passed, 7 deselected`;Mock-only | +| `cd frontend && npm run build` | 生产前端构建 | 0 | 通过;保留既有大 chunk warning | +| `docker compose config --quiet` | Compose 静态配置 | 0 | 通过 | +| 高置信 staged-candidate 秘密扫描 | 检查本次 modified/untracked 文件中的真实 Key/私钥格式 | 1(无匹配) | 仅更宽的初筛命中明确命名的测试 canary/secret marker;高置信格式无匹配 | +| 隔离 PostgreSQL 16:preflight、`upgrade head`、`check`、populated downgrade、清空后 downgrade/upgrade/check | 验证真实 PostgreSQL `0008` 约束与回滚门禁 | 0/预期拒绝/0 | 新类型存在时 downgrade 以 RuntimeError 在 DDL 前拒绝;清空两条测试 Model 后往返与 check 通过;测试容器已停止并删除 | +| `git commit`、普通 `git push`;`gh run watch 33304667092 --exit-status` | 发布实现并验证精确 SHA | 0 | 实现 SHA `6943aa29a154c82bdfbe5efb2578c916c3cbf632` 已 push;backend、backend integration、real-Compose reliability、frontend 四个必需 job 全部成功 | + +## 测试结果 + +- 通过:三协议目标回归、完整 backend/frontend、Mock smoke、build、Compose config 与隔离 PostgreSQL 16 migration 门禁均通过。 +- 失败并已修复:前端失败先行 `2 failed, 12 passed`;首次 `make lint` 仅要求 7 文件格式化。 +- 已知 warning:Python 3.14 上游弃用/async warning 与既有 Vite large-chunk warning;无新增失败。 + +## 未运行验证 + +- 真实 Provider:按安全规则有意不运行。 + +## 未完成项 + +- 无;证据文档作为独立收尾提交,提交后同样接受 exact-SHA CI 门禁。 + +## 已知问题与限制 + +- Responses/Messages 只实现纯文本评测共同子集,不含 tools、多模态或供应商私有扩展。 +- 本地自动化不能证明 OpenCode Go 当日模型、额度或真实账单兼容性。 + +## 安全检查 + +- 真实密钥扫描:已对本次 modified/untracked 文件执行高置信格式扫描,无匹配;较宽初筛只命中明确命名的假 canary/secret marker 测试值 +- 真实 API 调用:否 +- 日志/API 脱敏:三协议 malformed JSON/SSE、oversized、transport 与反射假 Key 回归通过,安全异常不保留原始 Provider cause/context +- 危险 Git 操作(force push/reset 等):无 +- 阶段 push:实现 SHA `6943aa29a154c82bdfbe5efb2578c916c3cbf632` 已普通 push +- 远程 CI:[run `33304667092`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33304667092) 对该实现 SHA 4/4 成功 +- 遗留安全风险:自定义 HTTPS `base_url` 的 SSRF/数据外发风险不变 + +## 结果与下一步 + +三协议实现、本地门禁、隔离 PostgreSQL 迁移、最终审查、普通 push 与实现 exact-SHA CI 全部完成。下一独立任务恢复为 P2-07 最小只读 recovery verifier。 + +## 最终 Git 状态 + +```text +实现提交后工作树干净;本次仅追加独立 evidence closeout 文档提交 +``` diff --git a/docs/worklogs/2026-08-30-run-progress-heatmap-live-metrics.md b/docs/worklogs/2026-08-30-run-progress-heatmap-live-metrics.md new file mode 100644 index 0000000..ed64318 --- /dev/null +++ b/docs/worklogs/2026-08-30-run-progress-heatmap-live-metrics.md @@ -0,0 +1,157 @@ +# 2026-08-30 — Run Detail 热力图与实时指标工作日志 + +> 本日志记录实际发生的工作,不是事后美化的总结。所有命令以仓库根目录为基准。 + +## 元信息 + +- 日期:2026-08-30 +- 执行者:Codex +- 关联阶段:[Phase 3 — Benchmarks](../phases/PHASE-3-BENCHMARKS.md) +- 关联计划:[2026-08-30-run-progress-heatmap-live-metrics.md](../plans/2026-08-30-run-progress-heatmap-live-metrics.md) +- 关联 ADR:无;新增只读进度投影,不改变持久化/协议/安全决定 +- 最终状态:completed + +## 初始仓库状态 + +- 当前分支:`codex/complete-evaluation-workflow`,HEAD `a59a706921937924b752466b20f8523349c9de29`,跟踪同名 origin 分支 +- `git status --short --branch` 摘要:只有分支行,无未提交改动 +- 已有未提交改动:无;本任务从已通过精确 SHA CI 的 Run Detail 指标修复之后继续 +- 相关功能与测试现状:Run Detail 每秒并行轮询 Run 与当前 100 条 Responses;只有完成题数运行中变化,其余汇总多在终态聚合;无全题热力图或轻量增量接口 +- 环境约束:自动化只用 Mock/Stub;本任务不调用真实 Provider、不修改本地历史数据、不新增依赖或 migration + +## 本次目标与背景 + +用户要求评测展示界面增加逐题热力图,绿色表示通过、红色表示答案错误、黑色表示执行异常、白色表示未执行;鼠标悬停格子可查看 Token、运行时间等信息,并希望准确率等数字在评测过程中动态变化。 + +## 范围 + +- 新增不含 prompt/raw/reference/provider metadata 的轻量 Run progress index/block API;固定每 block 512 个 absolute positions。 +- Run Detail 增加可 hover/focus 的四态题目矩阵及图例/计数。 +- 后端在 progress index 同一读取快照中从已持久化 Response 派生 protocol-v1 live metrics 和 Token/cost 已知覆盖。 +- 保留现有详情分页、治理、取消和终态停止轮询。 +- 增加后端/前端测试并同步强制文档。 + +## 非目标 + +- 不增加题目“执行中”第五种状态,不提供 Provider token 流式进度。 +- 不改写 Run/Response/ledger,不增加 migration,不改变评分或精确 Token/cost 语义。 +- 不引入 WebSocket/SSE、图表依赖或虚拟列表依赖。 +- 不开始 P2-07、不改变 Phase 2/3 整体状态、不合并 PR。 + +## 验收标准 + +- [x] 绿/红/黑/白格子与正确计数来自全 Run 计划位置,而非当前详情页。 +- [x] hover 与 keyboard focus 均显示题号、状态、Token、延迟等,且状态不只靠颜色。 +- [x] 运行中新增 Response 后 score/accuracy/completion/延迟/错题/Token/cost 在下一次 index 轮询更新。 +- [x] 每秒只读轻量 index;只 hydrate 非空或 `response_count` 变化的 512 题 block,index→block 并发提交最终收敛且不漏格。 +- [x] 非空 block 初始同步完成前显示“同步中”,终态先到时仍追齐全部目标 count;hidden 暂停、visible 恢复同步。 +- [x] 现有分页、取消、治理、终态停止轮询和目标历史 Run 展示不回归。 +- [x] API/架构/用户/测试/状态文档一致,本地门禁全绿,实施 commit 已普通 push 且远端精确 SHA CI 4/4 成功。 + +## 假设 + +- Question.position 为 0-based 稳定槽位;数据库对 Benchmark 内 position 有唯一约束。 +- EvaluationResponse 为 Run/Question 唯一追加事实;后续不原地更新已完成 Response。 +- 实时指标必须复制 `aggregate_run_evidence` 的现有规则:strict score 以计划题为分母,completion 统计非空 raw,answered accuracy 只统计无 error 的非空 raw。 +- 20,000 题对应最多 40 个 index 项;每秒只比较 block counts,不能全量拉取逐题正文。 + +## 风险 + +| 风险 | 影响 | 缓解措施 | 结果 | +|---|---|---|---| +| index→block 之间并发提交 | 热力图短暂比 index 更新或少格 | Response 追加计数单调;客户端采纳 block 实际内容并由下轮 index 收敛,reducer 幂等 | 定向自动化通过 | +| 大 Run DOM 性能 | 页面卡顿 | 虚拟化 ARIA grid、事件委托、CSS containment、无正文 payload、只刷新变化 block | 12,032/20,000 题虚拟化自动化通过;实页手工仅验证目标 198 题 Run,不冒充大型 DevTools 性能测量 | +| live/终态公式漂移 | 结果误导 | 后端复用同一证据公式;index 指标与 counts 使用同一读取快照;终态 fixture 对照 | 后端定向与完整回归通过 | +| 颜色不可访问 | 用户无法识别状态 | 中文图例/计数、ARIA label、focus ring 与 Tooltip | 自动化 ARIA/键盘覆盖及实页键盘/Tooltip 通过 | + +## 实施步骤 + +1. [completed] 冻结 API/UI 合同并添加失败测试;初版 cursor red tests 已失败,合同已在实现前切换为 fixed blocks。 +2. [completed] 实现后端轻量 progress index/block API 与同快照 live metrics。 +3. [completed] 实现虚拟化热力图、Tooltip、独立轮询和实时指标。 +4. [completed] 文档、本地目标/完整门禁、浏览器验收、普通 push、最终 SHA 记录和远端精确 SHA CI 均已完成。 + +## 实际修改 + +| 文件/模块 | 修改内容 | 对应需求/原因 | +|---|---|---| +| 本计划与工作日志 | 冻结范围、fixed-block 性能/竞态/协议边界和验收 | AGENTS/PLANS 强制流程 | +| 后端 progress Schema/service/routes、共享证据聚合、日志合同与测试 | 固定 512 block index/payload、同快照 live metrics、typed 边界和固定白名单 | 运行中动态指标与大型 Run 轻量同步 | +| 前端 API types/client、`useRunProgress`、`RunProgressHeatmap`、Run Detail/CSS 与测试 | block reducer/poller、终态追齐、虚拟化 ARIA grid、Tooltip 与响应式布局 | 四态进度和动态指标 UI | +| README/API/TESTING/REQUIREMENTS/ARCHITECTURE/Phase/Roadmap/Status/Changelog/NEXT_TASK | 按实现、本地证据与精确 SHA 远端门禁同步产品/API/测试/状态边界 | 强制文档与仓库级 closeout 已完成 | + +## 决定、偏差与发现 + +| 时间 | 类型 | 事实与理由 | 后续影响 | +|---|---|---|---| +| 10:13 CST | discovery | 当前 Runner 每题只写回 completed count;其余 Run 汇总在终态 `aggregate_run_evidence` | 新 read model 必须从 Response 证据实时派生并由前端展示,不能只轮询旧 Run 汇总字段 | +| 10:16 CST | decision | 新接口只暴露 Tooltip/公式需要的固定字段,不包含题目/回答正文 | 保持轮询轻量且不扩大敏感正文面 | +| 10:18 CST | deviation | 初版 `(created_at,id)` opaque cursor 的 4 个后端 red tests 按预期失败;进一步复核发现应用 `created_at` 与 UUID 都不是数据库单调提交序列,并发提交无法证明绝对不漏 | 在生产实现前废弃 cursor 测试/合同,改用固定 512 absolute-position blocks;无需 migration | +| 10:25 CST | decision | `GET /runs/{id}/progress` 返回同快照 live metrics 与全部 block counts;`GET /runs/{id}/progress/blocks/{block_index}` 返回 absolute-position 固定白名单 cells | 前端只补齐非空/变化 block;12,032/20,000 题分别约 24/40 blocks | +| 10:27 CST | decision | outcome 优先级为 `error_type != null -> error`、否则 `score == 1 -> passed`、否则 `wrong`;没有 Response 的计划 position 为 `not_run` | 执行异常不会被重复算成普通答错,四态计数互斥 | +| 10:29 CST | decision | live 主指标由后端证据聚合并与 block index 使用同一读取快照;前端不从部分 hydrate Map 重算 | 防止“同步中”窗口产生虚假分数;Run 精确 nullable Token/cost 保持不变 | +| 10:32 CST | discovery | 用户 Run `a3de7e4d-40b2-4d8c-994b-c713047393ae` 的 Run/Response 对账为 total/completed/correct/error=`198/198/179/2`,198 条 Response 中 `score < 1` 为 19、`error_type` 非空为 2 | 四态应为通过 179、普通答错 17、执行异常 2、未执行 0;旧页面把执行异常 2 当成全部“错误题”,已复现用户问题 | +| 10:32 CST | discovery | 同一 Run 的已知 input/output Token 小计为 `45,509 / 4,561,625`,两列覆盖均为 `196/198`,平均延迟 `181,454.235 ms`;Run 精确 input/output/cost 均为 `null` | UI 应显示已知小计和覆盖率,不能把两条 usage 缺失解释为 0,也不能回填精确 Run 字段;未记录任何敏感 Response 正文 | +| 本地验收 | validation | 目标 Run 实页显示通过 179、普通答错 17、执行异常 2、未执行 0;Token `45,509 / 4,561,625`、输入/输出覆盖均为 `196/198` | 用户报告的两处展示问题已在真实本地页面闭环 | +| 本地验收 | validation | desktop、768px、375px 无横向溢出,console 无 warning/error,键盘定位与 Tooltip 通过 | 常见本地宽度和关键非鼠标路径已验证;不把它扩大为 VoiceOver/NVDA 认证 | +| 前端终审 | fix/validation | terminal + progress reconciled 后只执行一次最终 Run/当前 evidence 页刷新 | 避免终态 `Promise.all` 交错让较旧证据覆盖最终证据;新增回归通过 | +| 前端终审 | fix/validation | 同一路由切换 `runId` 时把 evidence offset 重置为 0 | 防止从旧 Run 的后续页请求新 Run;新增回归通过 | + +## 实际运行命令 + +| 命令 | 目的 | 退出码 | 结果摘要 | +|---|---|---:|---| +| `git status --short --branch` | 确认初始工作区 | 0 | 分支干净,跟踪 origin 同名分支 | +| `rg`/`sed` 读取 AGENTS、README、状态、Roadmap、Phase、API、Testing、Security、Architecture 与相关代码/测试 | 冻结约束和现状 | 0 | 确认跨 API/UI 需计划;无 migration/ADR;现有 Run 运行中汇总不完整 | +| 后端定向 red tests(初版 cursor 合同) | 失败先行验证接口尚不存在 | 失败(预期) | `4 failed`;因无单调提交序列,已在实现前废弃该合同,随后 fixed-block 定向测试 `37 passed` | +| `git diff --check -- <本切片文档>` | 检查文档补丁空白/冲突 | 0 | 文档先行补丁 clean;最终代码/测试 diff 仍由主任务收尾复核 | +| `cd backend && uv run pytest tests/test_run_progress_api.py tests/test_response_metadata_api.py -q` | progress API/聚合/边界定向回归 | 0 | `37 passed` | +| `cd frontend && npm test -- --run tests/run-detail-page.test.tsx tests/run-progress-heatmap.test.tsx` | 热力图、动态指标与轮询定向回归 | 0 | `32 passed`(Run Detail `20` + heatmap `12`);含两条终审竞态与 12,032/20,000 题虚拟化自动化边界 | +| `make test` | 完整本地回归 | 0 | backend `964 passed, 33 skipped`;frontend `64 passed` | +| `make lint` | Ruff/format、ESLint、TypeScript | 0 | 通过 | +| `make smoke` | 完全离线 Mock 纵向链路 | 0 | `1 passed, 7 deselected` | +| `cd frontend && npm run build` | production build | 0 | 通过 | +| `docker compose config --quiet` | Compose 配置 | 0 | 通过 | +| 可信 loopback 浏览器验收 | 目标历史 Run、响应式、键盘、Tooltip 与 console | 0 | 179/17/2、Token/覆盖正确;desktop/768/375 无横向溢出;console 无 warning/error | +| `git commit`、普通 `git push` | 固化并推送实现树 | 0 | 实现 commit [`99791964621165c9cc7ec36b4b2d27fe04e6acd5`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/commit/99791964621165c9cc7ec36b4b2d27fe04e6acd5) 已 push 到 `codex/complete-evaluation-workflow`,进入 [PR #5](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/pull/5) | +| GitHub Actions exact-SHA gate | 验证远端精确实现 SHA | 0 | [run `33289522923`](https://github.com/CWNU-Open-Source-Community/LLMBenchLab/actions/runs/33289522923) completed/success;backend、backend-integration、full-stack-reliability、frontend 4/4 成功 | + +## 测试结果 + +- 通过:backend target `37 passed`;frontend target `32 passed`(Run Detail `20` + heatmap `12`);完整 backend `964 passed, 33 skipped`、frontend `64 passed` +- 失败:初版 cursor 后端 red tests `4 failed`(预期且已废弃);fixed-block 实现后的目标/完整回归零失败 +- Lint/typecheck/build:`make lint` 与 frontend production build 通过 +- Smoke/Docker:Mock smoke `1 passed, 7 deselected`;`docker compose config --quiet` 通过 +- 远端:实现 SHA `99791964621165c9cc7ec36b4b2d27fe04e6acd5` 的 Actions run `33289522923` 四个必需 job 全部成功 + +## 未运行验证 + +- 未调用真实 Provider。12,032/20,000 题是自动化虚拟化边界测试;没有把它写成大型真实 Run 的手工 DevTools 性能/内存测量。 + +## 未完成项 + +- 本 P3-06 切片无未完成项;Phase 3 其余 IFEval、Plugin SDK、代码题/沙箱、分组 UI 与红队范围继续单独追踪。 + +## 已知问题与限制 + +- 已持久化 Response 前无法从当前数据模型区分“正在 Provider 执行”与“尚未开始”,二者按用户指定统一显示白色未执行;本任务不新增中间态事实。 + +## 安全检查 + +- 真实密钥扫描:本地范围/秘密与 staged diff 复核通过 +- 真实 API 调用:否;计划内仅 Mock/Stub 与本地只读页面 +- 日志/API 脱敏:progress 合同禁止 question/external ID、prompt/choices/raw/parsed/reference/error message 与 Provider metadata;只返回 absolute position 和指标白名单 +- 危险 Git 操作(force push/reset 等):无 +- 阶段 push:完成;实现 commit `99791964621165c9cc7ec36b4b2d27fe04e6acd5` 已普通 push +- 远程 CI:完成;精确 SHA run `33289522923` 四个必需 job 全部成功 +- 遗留安全风险:与现有 Run 证据相同,仅适合可信 loopback;见 `docs/SECURITY.md` + +## 结果与下一步 + +`completed`。本地实现、验证、浏览器闭环、普通 push 与精确 SHA CI 均已完成;P3-06 切片关闭。Phase 3 整体仍为 `in_progress`,P2-07 恢复为既定下一可靠性切片且保持 `planned`。 + +## 最终 Git 状态 + +```text +实现 commit 99791964621165c9cc7ec36b4b2d27fe04e6acd5 已普通 push 到 codex/complete-evaluation-workflow;PR #5 的精确 SHA Actions run 33289522923 四个必需 job 全部成功。 +``` diff --git a/docs/worklogs/2026-08-30-small-benchmark-datasets.md b/docs/worklogs/2026-08-30-small-benchmark-datasets.md new file mode 100644 index 0000000..0dd8b4d --- /dev/null +++ b/docs/worklogs/2026-08-30-small-benchmark-datasets.md @@ -0,0 +1,123 @@ +# 2026-08-30:准备小型模型评测数据集 + +## 元数据 + +- 日期:2026-08-30(Asia/Shanghai) +- 分支:`codex/complete-evaluation-workflow` +- 当前阶段:Phase 3 标准 Benchmark 的个人本地小样本补充 +- 关联计划:不适用;本次只生成并导入本地第三方数据 ZIP,不改变架构、协议、Schema 或公共 API +- 初始工作区:与 `origin/codex/complete-evaluation-workflow` 同步,开始时无未提交改动 + +## 目标与背景 + +当前本地正式 Benchmark 为 GPQA-Diamond 198 题及两份 MMLU-Pro 12,032 题版本,运行成本和等待时间不适合频繁横向比较模型。用户希望补充若干每套不超过 100 题的小型数据集,用当前内置客观评分器快速测试不同模型。 + +## 范围 + +- 从官方或维护者发布位置固定来源 revision,选择可转换为 multiple-choice 或 numeric 的公开评测集。 +- 对超过 100 题的来源使用固定 seed 和稳定散列排序,从有公开标签的评测 split/文件确定性抽取 100 题;恰好 100 题的 XCOPA 中文 validation 使用全量并只平衡选项呈现位置。 +- 生成并完整校验 `llmbenchlab-dataset-v1` ZIP,保存在 Git 忽略的 `artifacts/benchmarks/`。 +- 通过现有 Benchmark 导入 API 加载到当前默认本地数据库,并核对 API/数据库题数与 Hash。 +- 记录来源、许可、抽样、Hash、导入和验证证据,不记录题目正文。 + +## 非目标 + +- 本任务不新建模型评测、不主动调用 Provider 或产生额外模型费用;用户既有活动 Run 继续执行,不在本任务范围内停止。 +- 不修改 Dataset Loader、Evaluator、Benchmark 协议、数据库 Schema、API、前端或生产依赖。 +- 不加入需要 LLM Judge、代码执行沙箱或 IFEval 专用验证器的题目。 +- 不把第三方原始数据、转换后的题目或数据库备份提交到 Git。 +- 小样本结果不冒充原 Benchmark 全量榜单结果,也不宣称模型差异具有统计显著性。 + +## 验收标准 + +- 至少 5 个能力维度,包含英文和中文;每个新 Benchmark 恰好 100 题且可由当前内置评分器自动判分。 +- 每个来源 revision、源文件 SHA-256、split、抽样 seed/算法、prompt profile 和许可均可追溯。 +- 每个 ZIP 由当前 Loader 完整校验,题目 ID 唯一、答案合法,且不输出题目正文。 +- 新 Benchmark 经正式 API 导入后可查询,API/manifest/数据库逐集题数和 Dataset Hash 一致。 +- 导入前保留一致性数据库备份;导入后 SQLite `quick_check=ok`、外键错误为 0;本任务不创建、停止、重置或修改任何 Run。导入完成检查点的 Model/Run 数量应相对备份保持,Response 只允许由既有活动 Run 增长;更晚的并发客户端变更必须由独立 Run/audit 时间线解释。 + +## 假设 + +- “一种数据集不超过 100 道题”解释为每个可独立选择的 Benchmark 版本最多 100 道计分题;本次统一为 100 题,便于分数直接理解为答对题数。 +- 用户希望数据准备后可直接在当前 Web/API 中选择,而不只是获得候选链接。 +- 选择有公开标签的 test/validation/dev split 或官方 benchmark 文件;不使用隐藏标签测试集,不从训练 split 抽题。TruthfulQA 固定 CSV 没有官方 split。 +- 相同 100 题、相同 prompt 和 evaluator 用于所有模型;生成参数仍需由用户在 Run 时保持一致。 + +## 风险与控制 + +- 许可证或来源边界:只选有明确许可的官方/维护者来源,并在 manifest 与记录中披露;受非商业限制的数据默认不选入。 +- 公共 Benchmark 污染:明确披露高分可能受训练污染影响,小样本只用于快速筛查。 +- 抽样偏差和高方差:固定 seed 42、源记录标识,以及 `sha256-sort-v1`、`sha256-stratified-v1` 或 `full-split+sha256-balanced-position-v1` 的逐集算法,保证复现;不与全量结果混排。 +- 答案格式不兼容:只使用 multiple-choice/numeric,prompt 明确要求最终字母或数字,导入前逐题做结构与参考答案检查。 +- 本地数据库写入中断或与既有 Run 并发:导入前使用 SQLite online backup,逐 ZIP 单事务导入,导入后核对完整性、Benchmark/Question 增量和原有业务计数;不把既有 Run 正常新增的 Response 误归因于导入。 +- 第三方题目泄漏:源缓存、ZIP、备份和转换清单留在 Git 忽略目录并使用 `0600` 权限;日志只记录元数据与 Hash。 + +## 实施步骤 + +1. 核对仓库状态、当前数据、格式/协议限制和候选数据的官方来源与许可。 +2. 固定来源 revision,下载并校验原始文件;保存只含元数据的来源与转换清单。 +3. 以 seed 42 对超过 100 题的公开标签评测 split/文件做稳定散列抽样,对 XCOPA 中文 validation 使用完整 100 题,生成 dataset-v1 ZIP。 +4. 用当前 Loader 校验全部 ZIP,并检查题型、ID、答案、题数、语言和元数据分布。 +5. 创建并校验导入前数据库备份,通过正式 API 逐个导入。 +6. 对账 API、数据库、manifest、Hash、完整性和原有 Run/Response,确认本任务未新建 Run 或触发 Provider;单独记录既有活动 Run 的并发推进。 +7. 更新本日志与状态文档,复核 tracked diff、秘密边界和 Git 忽略范围。 + +## 实际结果 + +### 数据集与来源 + +共准备并导入 6 套、600 道计分题,覆盖英文/中文、数学推理、常识续写、代词消歧、真实性常识与因果推理。所有来源均固定到不可变 revision 或官方发布归档;下表 Hash 为下载到本地的原始源文件 SHA-256。 + +| Benchmark slug | 能力 / 语言 / 题型 | 来源文件 / split(原规模) | 固定 revision / 发布物 | 源 SHA-256 | 许可 | +| --- | --- | --- | --- | --- | --- | +| `gsm8k-mini-100` | 数学推理 / 英文 / numeric | GSM8K `test`(1,319) | `openai/grade-school-math@3101c7d5072418e28b9008a6636bde82a006892c` | `3730d312f6e3440559ace48831e51066acaca737f6eabec99bccb9e4b3c39d14` | MIT | +| `mgsm-zh-mini-100` | 数学推理 / 中文 / numeric | `mgsm/mgsm_zh.tsv`(benchmark/test set,250) | `google-research/url-nlp@452a21ad3dae5668c06ceeac21ff073e1e40f9be` | `b2fa63151022370a0de1f4211c8c284eae74b0f5a3b003b1d5982c0d4a73f661` | CC BY 4.0(数据子目录) | +| `hellaswag-mini-100` | 常识续写 / 英文 / 四选一 | HellaSwag `validation`(10,042) | `rowanz/hellaswag@a29ff8e9a04bba4bd6588223785ce105328adc57` | `0aa3b88843990f3f10a97b9575c94d7b71fb2205240ba04ae4884d9e9c992588` | MIT | +| `winogrande-mini-100` | 代词消歧 / 英文 / 二选一 | WinoGrande v1.1 `dev`(1,267) | 官方 `winogrande_1.1.zip`;仓库指针 `727e837f77521ef38bcc56df3b275c8da43f45af`,归档字节由右侧 SHA 固定 | `3619ab104d8be2977b25c90ff420cb42d491707dcc75362a1e5d22bc082b7318` | CC BY 2.0(数据) | +| `truthfulqa-binary-mini-100` | 真实性常识 / 英文 / 二选一 | `TruthfulQA.csv`(无官方 split,790) | `sylinrl/TruthfulQA@d71c110897f5d31c5d7f309e7bc316c152f6f031` | `b8d8ef1e12f98b4f2a9f47abc9765da0640b182b6c5d9b92f0c1a1f2f1e02e5c` | Apache-2.0(仓库) | +| `xcopa-zh-validation-100` | 因果常识 / 中文 / 二选一 | `data/zh/val.zh.jsonl`(100,全量) | `cambridgeltl/xcopa@e2e9d7f105a758ee869cd13e0fe251ef93a29840` | `8f638466c196342104bdbe9276e7d91fe0abcc044e9a9ad715f30c8ad618bcdd` | CC BY 4.0(仓库) | + +### 可复现转换与产物 + +- 本地转换器为 Git 忽略的 `artifacts/tools/prepare_small_benchmarks.py`;统一 seed 为 `42`,使用带数据集命名空间的确定性 SHA-256 排序。GSM8K test 与 MGSM benchmark/test 文件各取 100 题;TruthfulQA 从固定、无官方 split 的 CSV 取 100 题;HellaSwag 按正确选项分层为 `25/25/25/25`;WinoGrande 为 `50/50`;XCOPA 使用恰好 100 题的完整 `data/zh/val.zh.jsonl`,并确定性平衡呈现后的正确选项为 `50/50`。 +- TruthfulQA 转换成当前 Evaluator 可评分的 binary MC:每题使用 `Best Answer` 与 `Best Incorrect Answer`,并确定性平衡正确答案位置。它不是官方 MC1/MC2 全选项成绩,名称显式保留 `binary-mini`,不得与官方全量分数混用。 +- numeric prompt 要求最后给出不带千位分隔符的数字;multiple-choice prompt 要求最终选项字母。没有加入答案解析或链式思维参考正文。 +- 元数据清单为 Git 忽略的 `artifacts/benchmarks/small-benchmarks-provenance-v1.json`,记录源 URL/revision/Hash、所选源行、许可、题型/答案分布、archive Hash 与 Dataset Hash;题目正文和答案均未写入 tracked 文档。 +- 转换器重复运行后六个 ZIP 与 Dataset Hash 均未变化。产物、源缓存、转换器和清单权限均为 `0600`,且由 Git ignore 覆盖。 + +| Benchmark slug | 版本 | Dataset Hash | ZIP SHA-256 | +| --- | --- | --- | --- | +| `gsm8k-mini-100` | `1.0.0-b4d4154c57b3` | `16c0cd13616024a3a4f3f4f4533552ce6d1552c7569e9e721c8e255932115a95` | `93b7221ca25cc37044c697cf98424a3847eae0981be6e395edd25a2e2187f8d8` | +| `mgsm-zh-mini-100` | `1.0.0-4bd34714ce14` | `a79b41e07a4e15f454d13eceacdca76780f8fdc6fdc679374184be25207bcafc` | `272a1ce194a39959112d1429f3b51fc20a3f70ef4680bf9939ba53c650b63a87` | +| `hellaswag-mini-100` | `1.0.0-89e902992f2d` | `37311f31662e3f9f11882cc7ec5f3bd9ee34647b241ab37b991ac76c150ff0c8` | `51c77744d33fb8550de8f40836bb5f9c6210f3ad23ff91be1c3f6da9bf99f642` | +| `winogrande-mini-100` | `1.0.0-cff50d08ab54` | `2e0b597263b16d39d6e4bd4d5bc9f8eb39632e6e29123e57e3a1aefbde4fd77d` | `440e2fc2fe5ff28842927b4078aa15022535576113b9f7864df4ddd3f9593a3c` | +| `truthfulqa-binary-mini-100` | `1.0.0-ead5ea285da8` | `909b44c76c6b0bbe39baa1a82aabb0a14a967866d6b4bce6f364b2d86c4fdef6` | `b1933adcf5c74907e5751c6bee424474ec5dc7f7e89250d4d2340b5ad125cf31` | +| `xcopa-zh-validation-100` | `1.0.0-34d05db11a88` | `ffacf866600eacf9023e944db31f2cc17412927c3ffb9f4e32aac8eae2a6da9d` | `624dd7174d1eaddf0f5129ac093b387ebc2ae71b39bccd7452a30781b318b4d6` | + +### 数据库备份与正式导入 + +- 导入前通过 SQLite online backup 创建 `backend/data/llmbenchlab.db.pre-small-benchmark-import-20260830T054837Z.bak`,权限 `0600`、大小 `109,219,840` bytes、SHA-256 `5456688d871c3040c6d092df744399a315d48feddf37db7062fc37844cf04d2c`。冻结快照为 Models/Benchmarks/Questions/Runs/Responses=`2/4/24,277/5/1,000`,唯一活动 Run `ab72c8ea-1d64-42d9-9946-e9cb4f1f23cb` 为 `765/12,032`、`error_questions=0`;`quick_check=ok`、外键错误 0、head=`20260830_0007`。 +- 六个 ZIP 均通过正式 `POST /api/v1/benchmarks/import` 导入,HTTP 结果为 `201` × 6。API、manifest 与数据库逐集均为 100 题且 Dataset Hash 一致;默认库因此从 4 增至 10 个 Benchmarks、从 24,277 增至 24,877 道 Questions,Models 与 Runs 数量保持 `2/5`。 +- 既有 12,032 题 Run 在准备和导入期间继续执行;本任务没有调用 Run 创建/取消/重置接口。取消请求之前的只读复核返回 `running`、`789/12,032`、`error_questions=0`,全库 Runs/Responses=`5/1,024`;相对备份新增的 24 条 Response 正好来自该活动 Run,不能归因于 Benchmark 导入。该检查与 Writer 并发,因此不把客户端命令开始时间冒充数据库行的精确 `created_at` 截止点。 +- 导入完成后的另一个并发客户端时间线在 `2026-08-30T05:53:55Z` 创建了一条 MGSM mini Run;数据库 audit 又在 `05:54:49Z` 记录原大 Run 的 `run_cancel_requested`,并于 `05:54:52Z` 记录 terminal `cancelled`,最终 `798/12,032`、正确 `723`、`error_questions=1`,reconcile payload 为 `released_reservations=0`、`conservative_settlements=1`。随后还创建了另一条 MGSM mini Run。`05:55:48Z` 的动态库快照为 Runs/Responses=`7/1,041`。这些后续状态变化没有由本任务的工具调用发起,也不能归因于 Benchmark 导入;它们反而说明新 Benchmark 已可被其他客户端直接选择。本任务没有直接发起 Provider 请求,但不能把整台服务描述为“期间无 Provider 流量”。 +- 导入后 `quick_check=ok`、外键错误 0、Alembic head=`20260830_0007`。没有修改 Schema、API、协议、Evaluator 或产品代码。 + +### 验证结果 + +- 六个 ZIP 均由当前 Dataset Loader 完整加载;600 个题目 ID 在各自数据集内唯一,题型/答案合法,manifest、archive 与持久化 Hash 对账一致。 +- provenance 的六个 `source.rows` 已完整记录为 `1,319/250/10,042/1,267/790/100`;首次独立复核发现三个非 JSONL 源为 `null` 后,已修正本地转换器并重复生成 metadata-only 清单,六个 ZIP/archive/Dataset Hash 均保持不变。 +- 以每题金标 reference answer 走当前内置 Evaluator 做格式自检,`600/600` 均获接受;这不是任何模型答对 600/600 的评测成绩。HellaSwag 正确选项为 `A/B/C/D=25/25/25/25`,WinoGrande、TruthfulQA binary 与 XCOPA 均为 `A/B=50/50`。 +- `uv run pytest -q tests/test_dataset_loader.py tests/test_standard_datasets.py`:`40 passed`,只有既有上游弃用 warning。 +- 转换脚本通过 Python 编译检查;重复生成 Hash 不变。首次从 `backend/` 工作目录误用 `backend/.venv/bin/python`,因相对路径重复而失败;随后改用 `.venv/bin/python` 重跑同一 reference-answer 检查并得到 `600/600`。该纠正过程如实保留,不把首次失败记成产品回归。 +- 最终 archive Hash shell 复核的首次循环把 `path` 用作变量名;zsh 会把它与 `PATH` 绑定,导致循环内 `shasum`/`awk` 无法查找并报出伪 mismatch。改名为 `archive_file` 并使用绝对工具路径后,六个 archive Hash 全部通过;同时把转换器权限从默认创建的 `0644` 收紧为 `0600`。这是本地校验命令/权限修正,不是数据内容变化。 +- staged 文档措辞检索首次把含 Markdown 反引号的模式放入双引号,zsh 因而尝试执行无害的 `zh` 并报告 command not found;改为单引号后同一只读检索正常完成,未产生文件或外部状态变化。 +- 本次不改实现代码、Schema、公共合同或依赖,因此没有重复运行不相关的完整 Compose/capacity 门禁;目标 Loader/标准数据集测试和真实本地数据库/API 对账与风险相称。 + +### 结论与边界 + +验收标准全部满足。六套数据可直接在当前 Benchmarks/New Run 页面选择,且每套恰好 100 题。它们适合低成本、同配置的模型初筛和逐题配对比较;公开 Benchmark 可能被训练数据污染,100 题准确率的标准误差在最不利的 50% 正确率附近约为 5 个百分点,因此不应把小分差解释为稳定排名,也不应将 mini/binary 子集成绩冒充官方全量成绩。 + +### 仓库收尾 + +- 文档记录 commit `8faa2093b2c3308994d50e42a31063cdbf5264a6` 已普通 push 到 `codex/complete-evaluation-workflow`。其精确 SHA 的 GitHub Actions run `33296049611` 对 backend、真实 PostgreSQL/Redis integration、real-Compose reliability 和 frontend 四个必需 job 全部成功。 +- 首次按 exact SHA 查询 Actions 时手工展开了错误的完整 SHA `8faa20978a8d87665453130c2c226c17689b205f`,因此没有匹配到 run;随后以 `git rev-parse HEAD` 取得真实 SHA 并正确定位 `33296049611`。错误查询只读且没有触发或修改远程状态。 diff --git a/frontend/src/api/client.ts b/frontend/src/api/client.ts index 5673d5d..1ee17bc 100644 --- a/frontend/src/api/client.ts +++ b/frontend/src/api/client.ts @@ -1,12 +1,14 @@ import type { Benchmark, DashboardSummary, - EvaluationResponse, + EvaluationResponseList, EvaluationRun, LeaderboardEntry, ListResponse, ModelConfig, ModelPayload, + RunProgressBlock, + RunProgressIndex, RunPayload, RunStatus, } from "./types"; @@ -53,6 +55,7 @@ async function request(path: string, init: RequestInit = {}): Promise { try { response = await fetch(`${API_BASE}${path}`, { ...init, headers }); } catch (error) { + if (init.signal?.aborted) throw error; throw new ApiError( error instanceof Error ? `无法连接后端:${error.message}` : "无法连接后端服务。", 0, @@ -106,13 +109,17 @@ export const api = { benchmark_id?: string; protocol_version?: string; } = {}) => request>(`/runs${query({ ...params, limit: params.limit ?? 20 })}`), - run: (id: string) => request(`/runs/${id}`), + run: (id: string, signal?: AbortSignal) => request(`/runs/${id}`, { signal }), createRun: (payload: RunPayload) => request("/runs", { method: "POST", body: JSON.stringify(payload) }), cancelRun: (id: string) => request(`/runs/${id}/cancel`, { method: "POST" }), responses: (runId: string, params: { offset?: number; limit?: number } = {}) => - request>( + request( `/runs/${runId}/responses${query({ offset: params.offset, limit: params.limit ?? 100 })}`, ), + runProgressIndex: (runId: string, signal?: AbortSignal) => + request(`/runs/${runId}/progress`, { signal }), + runProgressBlock: (runId: string, blockIndex: number, signal?: AbortSignal) => + request(`/runs/${runId}/progress/blocks/${blockIndex}`, { signal }), leaderboard: (params: { model_id?: string; benchmark_id?: string; order?: string } = {}) => request>(`/leaderboard${query({ ...params, limit: 100 })}`), }; diff --git a/frontend/src/api/types.ts b/frontend/src/api/types.ts index de71933..98f380d 100644 --- a/frontend/src/api/types.ts +++ b/frontend/src/api/types.ts @@ -1,8 +1,13 @@ -export type ProviderType = "mock" | "openai_compatible"; +export type ProviderType = + | "mock" + | "openai_compatible" + | "openai_responses" + | "anthropic_messages"; export type CredentialSource = "none" | "environment" | "stored"; export type RunStatus = "pending" | "running" | "completed" | "failed" | "cancelled"; export type GovernanceRunStatus = "legacy_unmanaged" | "managed" | "delayed" | "exhausted"; export type QuestionType = "exact_match" | "multiple_choice" | "numeric"; +export type RunProgressOutcome = "passed" | "wrong" | "error"; export interface ListResponse { items: T[]; @@ -114,8 +119,8 @@ export interface EvaluationRun { export interface RunPayload { model_id: string; benchmark_id: string; - temperature: number; - top_p: number; + temperature: number | null; + top_p: number | null; max_tokens: number | null; seed: number | null; system_prompt?: string | null; @@ -149,6 +154,53 @@ export interface EvaluationResponse { created_at: string; } +export interface EvaluationResponseList extends ListResponse { + known_input_tokens: number; + known_output_tokens: number; + input_token_reported_responses: number; + output_token_reported_responses: number; +} + +export interface RunProgressBlockSummary { + block_index: number; + response_count: number; +} + +export interface RunProgressIndex { + block_size: number; + total_questions: number; + completed_questions: number; + correct_questions: number; + error_questions: number; + score: number; + completion_rate: number; + answered_accuracy: number | null; + average_latency_ms: number | null; + known_input_tokens: number; + known_output_tokens: number; + input_token_reported_responses: number; + output_token_reported_responses: number; + known_estimated_cost: number; + estimated_cost_reported_responses: number; + blocks: RunProgressBlockSummary[]; +} + +export interface RunProgressCell { + position: number; + outcome: RunProgressOutcome; + score: number; + latency_ms: number | null; + input_tokens: number | null; + output_tokens: number | null; + estimated_cost: number | null; + error_type: string | null; +} + +export interface RunProgressBlock { + block_index: number; + items: RunProgressCell[]; +} + export interface LeaderboardEntry { run_id: string; model_id: string; diff --git a/frontend/src/components/RunProgressHeatmap.tsx b/frontend/src/components/RunProgressHeatmap.tsx new file mode 100644 index 0000000..4921083 --- /dev/null +++ b/frontend/src/components/RunProgressHeatmap.tsx @@ -0,0 +1,270 @@ +import { memo, useEffect, useId, useMemo, useRef, useState } from "react"; + +import type { RunProgressCell, RunProgressIndex, RunProgressOutcome } from "../api/types"; +import { formatCost, formatLatency, formatTokens } from "../lib/format"; + +const CELL_GAP = 3; +const ROW_HEIGHT = 19; +const VIEWPORT_HEIGHT = 266; +const OVERSCAN_ROWS = 4; +const FALLBACK_WIDTH = 640; + +type DisplayOutcome = RunProgressOutcome | "pending"; + +const outcomeLabels: Record = { + passed: "通过", + wrong: "答案错误", + error: "执行异常", + pending: "未执行", +}; + +type RunProgressHeatmapProps = { + index: RunProgressIndex; + items: RunProgressCell[]; + syncing?: boolean; +}; + +function reported(value: number | null, formatter: (known: number) => string): string { + return value == null ? "未上报" : formatter(value); +} + +function cellLabel(position: number, cell: RunProgressCell | undefined): string { + const outcome: DisplayOutcome = cell?.outcome ?? "pending"; + if (!cell) return `第 ${position + 1} 题,${outcomeLabels[outcome]},Token 未上报,运行时间未上报`; + return [ + `第 ${position + 1} 题`, + outcomeLabels[outcome], + `输入 Token ${reported(cell.input_tokens, formatTokens)}`, + `输出 Token ${reported(cell.output_tokens, formatTokens)}`, + `运行时间 ${reported(cell.latency_ms, formatLatency)}`, + ].join(","); +} + +function RunProgressHeatmapView({ index, items, syncing = false }: RunProgressHeatmapProps) { + const headingId = useId(); + const tooltipId = useId(); + const cellIdPrefix = useId(); + const viewportRef = useRef(null); + const [viewportWidth, setViewportWidth] = useState(FALLBACK_WIDTH); + const [scrollTop, setScrollTop] = useState(0); + const [activePosition, setActivePosition] = useState(0); + const [hoverPosition, setHoverPosition] = useState(null); + const [hasFocus, setHasFocus] = useState(false); + const [pinned, setPinned] = useState(false); + const cellByPosition = useMemo( + () => new Map(items.map((item) => [item.position, item])), + [items], + ); + const columns = Math.max(8, Math.floor(viewportWidth / 19)); + const rowCount = Math.ceil(index.total_questions / columns); + const viewportHeight = Math.min( + VIEWPORT_HEIGHT, + Math.max(ROW_HEIGHT, rowCount * ROW_HEIGHT + 4), + ); + const visibleRowCount = Math.ceil(viewportHeight / ROW_HEIGHT); + const firstVisibleRow = Math.max(0, Math.floor(scrollTop / ROW_HEIGHT) - OVERSCAN_ROWS); + const lastVisibleRow = Math.min( + rowCount - 1, + Math.ceil((scrollTop + viewportHeight) / ROW_HEIGHT) + OVERSCAN_ROWS, + ); + const activeRow = Math.floor(activePosition / columns); + const renderedRows = useMemo(() => { + const rows = new Set(); + for (let row = firstVisibleRow; row <= lastVisibleRow; row += 1) rows.add(row); + if (index.total_questions > 0) rows.add(activeRow); + return [...rows].sort((left, right) => left - right); + }, [activeRow, firstVisibleRow, index.total_questions, lastVisibleRow]); + const selectedPosition = hoverPosition ?? ((hasFocus || pinned) ? activePosition : null); + const selectedCell = selectedPosition == null ? undefined : cellByPosition.get(selectedPosition); + const selectedOutcome: DisplayOutcome = selectedCell?.outcome ?? "pending"; + const passed = index.correct_questions; + const errors = index.error_questions; + const wrong = Math.max(0, index.completed_questions - passed - errors); + const pending = Math.max(0, index.total_questions - index.completed_questions); + + useEffect(() => { + const viewport = viewportRef.current; + if (!viewport) return; + const measure = () => setViewportWidth(viewport.clientWidth || FALLBACK_WIDTH); + measure(); + if (typeof ResizeObserver === "undefined") return; + const observer = new ResizeObserver(measure); + observer.observe(viewport); + return () => observer.disconnect(); + }, []); + + useEffect(() => { + setActivePosition((current) => Math.min(current, Math.max(0, index.total_questions - 1))); + }, [index.total_questions]); + + const revealPosition = (position: number) => { + const viewport = viewportRef.current; + if (!viewport) return; + const row = Math.floor(position / columns); + const top = row * ROW_HEIGHT; + const bottom = top + ROW_HEIGHT; + if (top < viewport.scrollTop) viewport.scrollTop = top; + else if (bottom > viewport.scrollTop + viewportHeight) { + viewport.scrollTop = Math.max(0, bottom - viewportHeight); + } + setScrollTop(viewport.scrollTop); + }; + + const selectPosition = (position: number) => { + if (index.total_questions === 0) return; + const next = Math.max(0, Math.min(index.total_questions - 1, position)); + setActivePosition(next); + revealPosition(next); + }; + + const handleKeyDown = (event: React.KeyboardEvent) => { + if (index.total_questions === 0) return; + const currentRow = Math.floor(activePosition / columns); + let next: number | null = null; + switch (event.key) { + case "ArrowRight": next = activePosition + 1; break; + case "ArrowLeft": next = activePosition - 1; break; + case "ArrowDown": next = activePosition + columns; break; + case "ArrowUp": next = activePosition - columns; break; + case "Home": next = event.ctrlKey || event.metaKey ? 0 : currentRow * columns; break; + case "End": next = event.ctrlKey || event.metaKey + ? index.total_questions - 1 + : Math.min(index.total_questions - 1, (currentRow + 1) * columns - 1); break; + case "PageDown": next = activePosition + visibleRowCount * columns; break; + case "PageUp": next = activePosition - visibleRowCount * columns; break; + case "Enter": + case " ": + event.preventDefault(); + setHoverPosition(null); + setPinned(true); + return; + case "Escape": + event.preventDefault(); + setPinned(false); + setHoverPosition(null); + return; + default: + return; + } + event.preventDefault(); + setHoverPosition(null); + setPinned(false); + selectPosition(next); + }; + + return ( +
+
+
+ PROGRESS MAP +

逐题进度热力图

+

每格对应 Benchmark 中的绝对题号;白格也可能是正在执行但尚未保存结果。

+
+ +
+
+ 通过 {passed} + 答案错误 {wrong} + ×执行异常 {errors} + 未执行 {pending} +
+ {index.total_questions > 0 ? ( +
setHasFocus(true)} + onBlur={(event) => { + if (!event.currentTarget.contains(event.relatedTarget)) setHasFocus(false); + }} + onKeyDown={handleKeyDown} + onMouseLeave={() => setHoverPosition(null)} + onScroll={(event) => setScrollTop(event.currentTarget.scrollTop)} + > +
+ {renderedRows.map((row) => { + const rowStart = row * columns; + const rowEnd = Math.min(index.total_questions, rowStart + columns); + return ( +
+ {Array.from({ length: rowEnd - rowStart }, (_, offset) => { + const position = rowStart + offset; + const cell = cellByPosition.get(position); + const outcome: DisplayOutcome = cell?.outcome ?? "pending"; + return ( + + ); + })} +
+ ); + })} +
+
+ ) :
该 Run 没有计划题目。
} + + 已完成 {index.completed_questions} / {index.total_questions} 题 + + {selectedPosition != null && index.total_questions > 0 && ( + + )} +
+ ); +} + +export const RunProgressHeatmap = memo(RunProgressHeatmapView); diff --git a/frontend/src/hooks/useRunProgress.ts b/frontend/src/hooks/useRunProgress.ts new file mode 100644 index 0000000..85c91d3 --- /dev/null +++ b/frontend/src/hooks/useRunProgress.ts @@ -0,0 +1,273 @@ +import { useCallback, useEffect, useRef, useState } from "react"; + +import { api } from "../api/client"; +import type { + RunProgressBlock, + RunProgressBlockSummary, + RunProgressCell, + RunProgressIndex, + RunStatus, +} from "../api/types"; + +const POLL_INTERVAL_MS = 1000; +const BLOCK_REQUEST_CONCURRENCY = 4; +const TERMINAL_STATUSES: ReadonlySet = new Set(["completed", "failed", "cancelled"]); + +type ProgressState = { + index: RunProgressIndex | null; + cells: RunProgressCell[]; + ready: boolean; + syncing: boolean; + error: string | null; + terminalVerified: boolean; +}; + +const initialState: ProgressState = { + index: null, + cells: [], + ready: false, + syncing: false, + error: null, + terminalVerified: false, +}; + +function isAbortError(reason: unknown): boolean { + return reason instanceof Error && reason.name === "AbortError"; +} + +function normalizedBlocks(index: RunProgressIndex): RunProgressBlockSummary[] { + if (index.block_size <= 0 || index.total_questions < 0) return []; + const blockCount = Math.ceil(index.total_questions / index.block_size); + const reported = new Map(index.blocks.map((block) => [block.block_index, block.response_count])); + return Array.from({ length: blockCount }, (_, blockIndex) => ({ + block_index: blockIndex, + response_count: reported.get(blockIndex) ?? 0, + })); +} + +function validateBlock( + block: RunProgressBlock, + blockIndex: number, + index: RunProgressIndex, +): RunProgressCell[] { + if (block.block_index !== blockIndex) throw new Error("progress_block_mismatch"); + const start = blockIndex * index.block_size; + const end = Math.min(start + index.block_size, index.total_questions); + const seen = new Set(); + for (const cell of block.items) { + if (cell.position < start || cell.position >= end || seen.has(cell.position)) { + throw new Error("progress_cell_out_of_range"); + } + seen.add(cell.position); + } + return [...block.items].sort((left, right) => left.position - right.position); +} + +async function fetchWithConcurrency( + blocks: RunProgressBlockSummary[], + worker: (block: RunProgressBlockSummary) => Promise, +): Promise { + let cursor = 0; + const workers = Array.from( + { length: Math.min(BLOCK_REQUEST_CONCURRENCY, blocks.length) }, + async () => { + while (cursor < blocks.length) { + const block = blocks[cursor]; + cursor += 1; + await worker(block); + } + }, + ); + await Promise.all(workers); +} + +function flattenBlocks(blocks: Map): RunProgressCell[] { + return [...blocks.entries()] + .sort(([left], [right]) => left - right) + .flatMap(([, cells]) => cells); +} + +export type UseRunProgressResult = { + index: RunProgressIndex | null; + cells: RunProgressCell[]; + ready: boolean; + syncing: boolean; + error: string | null; + reconciled: boolean; + refresh: () => void; +}; + +export function useRunProgress( + runId: string, + runStatus: RunStatus | null, + expectedCompletedQuestions: number | null, +): UseRunProgressResult { + const [state, setState] = useState(initialState); + const [visible, setVisible] = useState(() => document.visibilityState !== "hidden"); + const blockCells = useRef(new Map()); + const loadedBlockCounts = useRef(new Map()); + const requestSequence = useRef(0); + const requestInFlight = useRef(false); + const abortController = useRef(null); + const publishedRunId = useRef(null); + const terminalRef = useRef(false); + const expectedCompletedRef = useRef(expectedCompletedQuestions); + + const terminal = runStatus != null && TERMINAL_STATUSES.has(runStatus); + terminalRef.current = terminal; + expectedCompletedRef.current = expectedCompletedQuestions; + + const sync = useCallback(async () => { + if (!runId || document.visibilityState === "hidden" || requestInFlight.current) return; + const requestId = ++requestSequence.current; + const controller = new AbortController(); + abortController.current = controller; + requestInFlight.current = true; + setState((current) => ({ ...current, syncing: true })); + + try { + const index = await api.runProgressIndex(runId, controller.signal); + if (requestId !== requestSequence.current || controller.signal.aborted) return; + + const blocks = normalizedBlocks(index); + const expectedResponseCount = blocks.reduce( + (total, block) => total + block.response_count, + 0, + ); + const changed = blocks.filter( + (block) => loadedBlockCounts.current.get(block.block_index) !== block.response_count, + ); + let blockRequestFailed = false; + + for (const block of changed.filter((item) => item.response_count === 0)) { + blockCells.current.delete(block.block_index); + loadedBlockCounts.current.set(block.block_index, 0); + } + + await fetchWithConcurrency( + changed.filter((item) => item.response_count > 0), + async (summary) => { + try { + const block = await api.runProgressBlock(runId, summary.block_index, controller.signal); + if (requestId !== requestSequence.current || controller.signal.aborted) return; + const cells = validateBlock(block, summary.block_index, index); + blockCells.current.set(summary.block_index, cells); + // A Response may commit between the index and block reads. Keeping + // the observed count forces the next index poll to reconcile it. + loadedBlockCounts.current.set(summary.block_index, cells.length); + } catch (reason) { + if (!isAbortError(reason) && !controller.signal.aborted) blockRequestFailed = true; + } + }, + ); + if (requestId !== requestSequence.current || controller.signal.aborted) return; + + const blocksMatch = blocks.every( + (block) => loadedBlockCounts.current.get(block.block_index) === block.response_count, + ); + const indexConsistent = expectedResponseCount === index.completed_questions; + const snapshotCurrent = blocksMatch && indexConsistent && !blockRequestFailed; + if (!snapshotCurrent) { + setState((current) => ({ + ...current, + syncing: false, + error: blockRequestFailed + ? "部分题目进度暂时无法同步,将自动重试。" + : "题目进度索引暂未收敛,将自动重试。", + terminalVerified: false, + })); + return; + } + publishedRunId.current = runId; + const terminalVerified = terminalRef.current + && expectedCompletedRef.current === index.completed_questions; + + setState({ + index, + cells: flattenBlocks(blockCells.current), + ready: true, + syncing: false, + error: null, + terminalVerified, + }); + } catch (reason) { + if (requestId === requestSequence.current && !isAbortError(reason) && !controller.signal.aborted) { + setState((current) => ({ + ...current, + syncing: false, + error: "题目进度暂时无法同步,将自动重试。", + terminalVerified: false, + })); + } + } finally { + if (requestId === requestSequence.current) { + requestInFlight.current = false; + if (abortController.current === controller) abortController.current = null; + } + } + }, [runId]); + + useEffect(() => { + requestSequence.current += 1; + abortController.current?.abort(); + abortController.current = null; + requestInFlight.current = false; + blockCells.current = new Map(); + loadedBlockCounts.current = new Map(); + publishedRunId.current = null; + setState(initialState); + }, [runId]); + + useEffect(() => { + if (terminal) { + setState((current) => ({ ...current, terminalVerified: false })); + void sync(); + } + }, [expectedCompletedQuestions, sync, terminal]); + + const stateMatchesRun = publishedRunId.current === runId; + const reconciled = stateMatchesRun && state.ready + && (!terminal || ( + state.terminalVerified + && state.index?.completed_questions === expectedCompletedQuestions + )); + const shouldPoll = visible && (!terminal || !reconciled); + + useEffect(() => { + if (!shouldPoll) return; + void sync(); + const timer = window.setInterval(() => void sync(), POLL_INTERVAL_MS); + return () => window.clearInterval(timer); + }, [shouldPoll, sync]); + + useEffect(() => { + const handleVisibility = () => { + const nextVisible = document.visibilityState !== "hidden"; + if (!nextVisible) { + requestSequence.current += 1; + abortController.current?.abort(); + abortController.current = null; + requestInFlight.current = false; + setState((current) => ({ ...current, syncing: false, terminalVerified: false })); + } + setVisible(nextVisible); + }; + document.addEventListener("visibilitychange", handleVisibility); + return () => document.removeEventListener("visibilitychange", handleVisibility); + }, []); + + useEffect(() => () => { + requestSequence.current += 1; + abortController.current?.abort(); + }, []); + + return { + index: stateMatchesRun ? state.index : null, + cells: stateMatchesRun ? state.cells : [], + ready: stateMatchesRun && state.ready, + syncing: state.syncing, + error: state.error, + reconciled, + refresh: () => void sync(), + }; +} diff --git a/frontend/src/pages/ModelsPage.tsx b/frontend/src/pages/ModelsPage.tsx index 216e232..a76a967 100644 --- a/frontend/src/pages/ModelsPage.tsx +++ b/frontend/src/pages/ModelsPage.tsx @@ -38,6 +38,73 @@ function apiKeyLabel(model: ModelConfig): string { return "未配置"; } +function providerLabel(providerType: ProviderType): string { + switch (providerType) { + case "mock": + return "Mock"; + case "openai_compatible": + return "Chat Completions"; + case "openai_responses": + return "OpenAI Responses"; + case "anthropic_messages": + return "Anthropic Messages"; + } +} + +const providerEndpointSuffixes = { + openai_compatible: "/chat/completions", + openai_responses: "/responses", + anthropic_messages: "/messages", +} satisfies Record, string>; + +function endpointSuffix(providerType: ProviderType): string | null { + return providerType === "mock" ? null : providerEndpointSuffixes[providerType]; +} + +function replaceKnownEndpointSuffix( + baseUrl: string | null, + providerType: ProviderType, +): string | null { + if (!baseUrl || providerType === "mock") return baseUrl; + try { + const url = new URL(baseUrl); + const normalizedPath = url.pathname.replace(/\/+$/, ""); + const currentSuffix = Object.values(providerEndpointSuffixes).find((suffix) => + normalizedPath.endsWith(suffix), + ); + if (!currentSuffix) return baseUrl; + url.pathname = `${normalizedPath.slice(0, -currentSuffix.length)}${providerEndpointSuffixes[providerType]}`; + return url.toString(); + } catch { + return baseUrl; + } +} + +function normalizeDefaultParametersForProvider( + defaultParameters: Record, + providerType: ProviderType, +): Record { + if (providerType !== "openai_responses" && providerType !== "anthropic_messages") { + return defaultParameters; + } + const normalized = { ...defaultParameters }; + if (normalized.seed !== null && normalized.seed !== undefined) { + normalized.seed = null; + } + if (providerType === "anthropic_messages") { + if ( + typeof normalized.temperature === "number" && + normalized.temperature > 1 + ) { + delete normalized.temperature; + } + if (normalized.max_tokens === null) { + delete normalized.max_tokens; + } + } + return normalized; +} + function providerOrigin(baseUrl: string | null): string | null { if (!baseUrl) return null; try { @@ -117,28 +184,42 @@ export function ModelsPage() { const setProvider = (provider_type: ProviderType) => { if (provider_type === "mock") setApiKey(""); - setPayload((current) => ({ - ...current, - provider_type, - ...(provider_type === "mock" - ? { - base_url: null, - remote_model_name: null, - input_price_per_million: 0, - output_price_per_million: 0, - } - : { input_price_per_million: null, output_price_per_million: null }), - })); + setPayload((current) => { + if (current.provider_type === provider_type) return current; + if (provider_type === "mock") { + return { + ...current, + provider_type, + base_url: null, + remote_model_name: null, + input_price_per_million: 0, + output_price_per_million: 0, + }; + } + return { + ...current, + provider_type, + base_url: replaceKnownEndpointSuffix(current.base_url, provider_type), + default_parameters: normalizeDefaultParametersForProvider( + current.default_parameters, + provider_type, + ), + ...(current.provider_type === "mock" + ? { input_price_per_million: null, output_price_per_million: null } + : {}), + }; + }); }; - const isCompatible = payload.provider_type === "openai_compatible"; + const isRemote = payload.provider_type !== "mock"; const hasExistingCredential = editing?.credential_source === "environment" || (editing?.credential_source === "stored" && editing.has_api_key); const originalProviderOrigin = providerOrigin(editing?.base_url ?? null); const currentProviderOrigin = providerOrigin(payload.base_url); const canReuseExistingCredential = - editing?.provider_type === "openai_compatible" && + editing?.provider_type !== "mock" && + isRemote && hasExistingCredential && originalProviderOrigin !== null && originalProviderOrigin === currentProviderOrigin; @@ -146,7 +227,7 @@ export function ModelsPage() { const canSave = useMemo( () => Boolean(payload.name.trim()) && - (!isCompatible || + (!isRemote || (Boolean(payload.base_url) && Boolean(payload.remote_model_name) && (canReuseExistingCredential || hasSubmittedApiKey))), @@ -154,7 +235,7 @@ export function ModelsPage() { payload.name, payload.base_url, payload.remote_model_name, - isCompatible, + isRemote, canReuseExistingCredential, hasSubmittedApiKey, ], @@ -172,7 +253,7 @@ export function ModelsPage() { requestControllerRef.current = requestController; try { const submittedPayload: ModelPayload = - isCompatible && apiKeyForRequest.trim() + isRemote && apiKeyForRequest.trim() ? { ...payload, api_key: apiKeyForRequest } : payload; if (editing) { @@ -256,7 +337,7 @@ export function ModelsPage() {

{model.provider_type === "mock" ? "Mock · 完全离线" - : `OpenAI-compatible · ${model.remote_model_name}`} + : `${providerLabel(model.provider_type)} · ${model.remote_model_name}`}

@@ -321,26 +402,48 @@ export function ModelsPage() { />
- Provider 类型 + Provider 类型与 API 协议
+ +
- {isCompatible && ( + {isRemote && (
+

+ 填写 API 根地址或匹配的完整 endpoint;当前协议使用{" "} + {endpointSuffix(payload.provider_type)} 端点。 +