From 6b2f6b6cb3e8733d73680a12eb7b31893a709f43 Mon Sep 17 00:00:00 2001 From: olddonkey Date: Sun, 26 Jul 2026 01:40:21 -0700 Subject: [PATCH 1/4] fix(logs): show effective reasoning effort --- .../src/content/docs/guides/web-dashboard.md | 2 +- .../content/docs/ja/guides/web-dashboard.md | 2 +- .../content/docs/ko/guides/web-dashboard.md | 2 +- .../content/docs/ru/guides/web-dashboard.md | 2 +- .../docs/zh-cn/guides/web-dashboard.md | 2 +- gui/src/pages/Logs.tsx | 22 ++++++- src/adapters/base.ts | 6 ++ src/adapters/mimo-free.ts | 1 + src/adapters/openai-chat.ts | 30 +++++++++- src/server/request-log.ts | 32 ++++++++++ src/server/responses/core.ts | 7 +++ src/usage/log.ts | 15 +++++ src/web-search/loop.ts | 5 +- tests/mimo-free-provider.test.ts | 9 ++- tests/reasoning-effort.test.ts | 60 ++++++++++++++++--- tests/request-log.test.ts | 55 +++++++++++++++++ tests/usage-log.test.ts | 10 +++- tests/web-search.test.ts | 19 +++++- 18 files changed, 260 insertions(+), 21 deletions(-) diff --git a/docs-site/src/content/docs/guides/web-dashboard.md b/docs-site/src/content/docs/guides/web-dashboard.md index a932fc3e74f..c69b2c169b5 100644 --- a/docs-site/src/content/docs/guides/web-dashboard.md +++ b/docs-site/src/content/docs/guides/web-dashboard.md @@ -37,7 +37,7 @@ bun run dev:gui | **Codex Auth** | Add ChatGPT/Codex pool accounts, select the next-session account, refresh 5h / weekly / 30d quotas, enable or disable quota auto-switch, set its 1–100% threshold, and configure transient-failure failover. | | **Subagents** | Feature up to five bare native or namespaced routed models in the `spawn_agent` override list. | | **Models** | Toggle native GPT and routed models, set provider allowlists and context caps, choose v1/base/v2, and configure the v2 thread limit. Configured providers stay visible as zero-model groups when discovery is off or returns no rows. | -| **Logs** | Auto-refresh recent requests with tokens, requested effort, resolved model, provider, status, request id, duration, and error details. | +| **Logs** | Auto-refresh recent requests with tokens, requested effort and (when available) effective outbound effort, resolved model, provider, status, request id, duration, and error details. The detail view includes the exact reasoning wire field when the adapter emits one. | | **Usage / Debug** | Inspect token-usage coverage and trends, or enable opt-in provider transport and usage-extraction diagnostics. | | **Stop** | Gracefully stop the proxy and installed background service, restore native Codex, and exit (`POST /api/stop`). | diff --git a/docs-site/src/content/docs/ja/guides/web-dashboard.md b/docs-site/src/content/docs/ja/guides/web-dashboard.md index 397d24ffe92..23057579ee8 100644 --- a/docs-site/src/content/docs/ja/guides/web-dashboard.md +++ b/docs-site/src/content/docs/ja/guides/web-dashboard.md @@ -37,7 +37,7 @@ bun run dev:gui | **Codex 認証** | ChatGPT/Codex プールアカウントを追加し、次回セッションアカウントを選び、5 時間 / 週間 / 30 日クォータを更新し、クォータ自動切り替えのオン/オフと 1~100% のしきい値、一時的失敗フェイルオーバーを設定します。 | | **サブエージェント** | `spawn_agent` オーバーライド一覧にネイティブまたはルーティングモデルを最大 5 つまで優先公開します。 | | **モデル** | ネイティブ GPT とルーティングモデルをオン/オフし、プロバイダー許可リストとコンテキスト上限、v1/base/v2、v2 スレッド数を設定します。 | -| **ログ** | トークン、リクエスト強度、実際のモデル、プロバイダー、状態、リクエスト ID、所要時間、エラー詳細を含む最近のリクエストを自動更新します。 | +| **ログ** | トークン、要求された強度と(利用可能な場合は)実際に送信された強度、実際のモデル、プロバイダー、状態、リクエスト ID、所要時間、エラー詳細を含む最近のリクエストを自動更新します。アダプターが reasoning パラメーターを送信した場合、詳細表示に正確な wire field も表示されます。 | | **使用量 / デバッグ** | トークン使用量の測定範囲と推移を見るか、オプションのプロバイダートランスポート/使用量抽出診断をオンにします。 | | **停止** | プロキシとインストールされたバックグラウンドサービスを正常終了しネイティブ Codex を復元した後終了します(`POST /api/stop`)。 | diff --git a/docs-site/src/content/docs/ko/guides/web-dashboard.md b/docs-site/src/content/docs/ko/guides/web-dashboard.md index b334ec1294f..55667adbe08 100644 --- a/docs-site/src/content/docs/ko/guides/web-dashboard.md +++ b/docs-site/src/content/docs/ko/guides/web-dashboard.md @@ -37,7 +37,7 @@ bun run dev:gui | **Codex Auth** | ChatGPT/Codex 풀 계정을 추가하고, 다음 세션 계정을 선택하고, 5시간 / 주간 / 30일 할당량을 갱신하며, 할당량 자동 전환을 켜거나 끄고 1~100% 임계값과 일시적 실패 failover를 설정합니다. | | **Subagents** | `spawn_agent` override 목록에 네이티브 또는 라우팅 모델을 최대 5개까지 우선 노출합니다. | | **Models** | 네이티브 GPT와 라우팅 모델을 켜고 끄고, 프로바이더 allowlist와 컨텍스트 상한, v1/base/v2, v2 thread 수를 설정합니다. | -| **Logs** | 토큰, 요청 강도, 실제 모델, 프로바이더, 상태, 요청 id, 소요 시간, 오류 상세가 포함된 최근 요청을 자동 갱신합니다. | +| **Logs** | 토큰, 요청한 강도와 (사용 가능한 경우) 실제 전송 강도, 실제 모델, 프로바이더, 상태, 요청 id, 소요 시간, 오류 상세가 포함된 최근 요청을 자동 갱신합니다. 어댑터가 reasoning 매개변수를 전송한 경우 상세 보기에 정확한 wire field도 표시됩니다. | | **Usage / Debug** | 토큰 사용량의 측정 범위와 추이를 보거나, 선택적 프로바이더 전송/사용량 추출 진단을 켭니다. | | **Stop** | 프록시와 설치된 백그라운드 서비스를 정상 종료하고 네이티브 Codex를 복원한 뒤 끝냅니다(`POST /api/stop`). | diff --git a/docs-site/src/content/docs/ru/guides/web-dashboard.md b/docs-site/src/content/docs/ru/guides/web-dashboard.md index 8d54e5b2cb0..ea0c17ec92d 100644 --- a/docs-site/src/content/docs/ru/guides/web-dashboard.md +++ b/docs-site/src/content/docs/ru/guides/web-dashboard.md @@ -37,7 +37,7 @@ bun run dev:gui | **Codex Auth** | Добавление аккаунтов пула ChatGPT/Codex, выбор аккаунта для следующей сессии, обновление квот 5 ч / недельных / 30-дневных, включение или отключение автопереключения, настройка его порога 1–100% и failover при временных сбоях. | | **Subagents** | Выделение до пяти «голых» нативных или маршрутизируемых моделей с пространством имён в списке переопределений `spawn_agent`. | | **Models** | Включение и отключение нативных GPT и маршрутизируемых моделей, настройка списков разрешённых провайдеров и лимитов контекста, выбор v1/base/v2 и настройка лимита потоков v2. | -| **Logs** | Автообновляемый список недавних запросов: токены, запрошенный уровень рассуждений, фактическая модель, провайдер, статус, id запроса, длительность и подробности ошибок. | +| **Logs** | Автообновляемый список недавних запросов: токены, запрошенный и, когда доступен, фактически отправленный уровень рассуждений, фактическая модель, провайдер, статус, id запроса, длительность и подробности ошибок. Если адаптер отправляет параметр рассуждений, в подробностях также отображается точное wire-поле. | | **Usage / Debug** | Просмотр покрытия и трендов расхода токенов либо включение опциональной диагностики транспорта провайдеров и извлечения данных об использовании. | | **Stop** | Корректная остановка прокси и установленного фонового сервиса, восстановление нативного Codex и выход (`POST /api/stop`). | diff --git a/docs-site/src/content/docs/zh-cn/guides/web-dashboard.md b/docs-site/src/content/docs/zh-cn/guides/web-dashboard.md index c6ff397e878..224754896dc 100644 --- a/docs-site/src/content/docs/zh-cn/guides/web-dashboard.md +++ b/docs-site/src/content/docs/zh-cn/guides/web-dashboard.md @@ -36,7 +36,7 @@ bun run dev:gui | **Codex Auth** | 添加 ChatGPT/Codex 池账号,选择下一 session 的账号,刷新 5h / 每周 / 30d 配额,启用或停用配额自动切换,设置其 1–100% 阈值和临时故障 failover。 | | **Subagents** | 在 `spawn_agent` override 列表中置顶最多五个原生或路由模型。 | | **Models** | 开关原生 GPT 与路由模型,配置 provider allowlist、上下文上限、v1/base/v2 以及 v2 thread 数量。 | -| **Logs** | 自动刷新近期请求,显示 token、请求强度、实际模型、provider、状态、request id、耗时和错误详情。 | +| **Logs** | 自动刷新近期请求,显示 token、请求强度以及(可用时)实际发送强度、实际模型、provider、状态、request id、耗时和错误详情。适配器发送 reasoning 参数时,详情中还会显示准确的 wire field。 | | **Usage / Debug** | 查看 token usage 覆盖率与趋势,或启用可选的 provider transport 和 usage 提取诊断。 | | **Stop** | 优雅地停止代理和已安装的后台服务,恢复原生 Codex 并退出(`POST /api/stop`)。 | diff --git a/gui/src/pages/Logs.tsx b/gui/src/pages/Logs.tsx index 8190f762e75..497aceae65d 100644 --- a/gui/src/pages/Logs.tsx +++ b/gui/src/pages/Logs.tsx @@ -95,6 +95,9 @@ interface LogEntry { provider: string; surface?: "claude"; requestedEffort?: string; + effectiveEffort?: string; + reasoningWireField?: string; + reasoningWireValue?: string | number; requestedServiceTier?: string; requestedSpeedLabel?: string; configuredServiceTier?: string; @@ -163,6 +166,19 @@ function speedLabel(log: LogEntry): string | undefined { return undefined; } +function effortLabel(log: LogEntry): string { + const requested = log.requestedEffort?.replace(/\s*->\s*/g, " → "); + const effective = log.effectiveEffort; + if (!requested) return effective ?? "-"; + if (!effective || requested === effective || requested.split(" → ").at(-1) === effective) return requested; + return `${requested} → ${effective}`; +} + +function reasoningWireLabel(log: LogEntry): string | undefined { + if (!log.reasoningWireField || log.reasoningWireValue === undefined) return undefined; + return `${log.reasoningWireField}=${log.reasoningWireValue}`; +} + function formatTokPerSecond(result: TokPerSecondResult | undefined, localeTag?: string): string { if (!result || result.kind === "unavailable" || !Number.isFinite(result.value) || result.value <= 0) return "\u2014"; const digits = result.value >= 100 ? 0 : 1; @@ -468,7 +484,7 @@ export default function Logs({ apiBase }: { apiBase: string }) { {speedLabel(log) && {speedLabel(log)}} - {log.requestedEffort ?? "-"} + {effortLabel(log)} {log.provider} @@ -533,6 +549,7 @@ function LogDetailDialog({ const [copied, setCopied] = useState(false); const tokenSplit = cacheSplit(detail); const cost = detail.displayMetrics?.cost; + const reasoningWire = reasoningWireLabel(detail); const copyRequestId = async () => { if (!detail.requestId) return; @@ -577,6 +594,9 @@ function LogDetailDialog({ {t("logs.col.model")}{modelLabel(detail.resolvedModel ?? detail.model)} {t("logs.col.provider")}{detail.provider} + {(detail.requestedEffort || detail.effectiveEffort) && ( + <>{t("logs.col.effort")}{effortLabel(detail)}{reasoningWire ? ` (${reasoningWire})` : ""} + )} {detail.errorCode && (<>{t("logs.col.error")}{detail.errorCode})} {detail.upstreamError && (<>{t("logs.col.upstreamReason")}{detail.upstreamError})} diff --git a/src/adapters/base.ts b/src/adapters/base.ts index 56c2c9df943..20ca5e01733 100644 --- a/src/adapters/base.ts +++ b/src/adapters/base.ts @@ -44,6 +44,12 @@ export interface AdapterRequest { method: string; headers: Record; body: string; + /** Exact reasoning parameter emitted by the adapter, for request-log diagnostics only. */ + reasoningLog?: { + effectiveEffort: string; + wireField: "reasoning_effort" | "thinking_budget" | "thinking.type"; + wireValue: string | number; + }; usageLog?: { inputTokens?: number; estimated?: boolean; diff --git a/src/adapters/mimo-free.ts b/src/adapters/mimo-free.ts index e0d6d899aef..56dac58a38b 100644 --- a/src/adapters/mimo-free.ts +++ b/src/adapters/mimo-free.ts @@ -191,6 +191,7 @@ export function createMimoFreeAdapter(provider: OcxProviderConfig): ProviderAdap method: "POST", headers, body: JSON.stringify(markedBody), + ...(baseReq.reasoningLog ? { reasoningLog: baseReq.reasoningLog } : {}), }; }, diff --git a/src/adapters/openai-chat.ts b/src/adapters/openai-chat.ts index 9944e10b9ef..d98d8c7e5fc 100644 --- a/src/adapters/openai-chat.ts +++ b/src/adapters/openai-chat.ts @@ -1,4 +1,4 @@ -import type { ProviderAdapter } from "./base"; +import type { AdapterRequest, ProviderAdapter } from "./base"; import type { AdapterEvent, OcxAssistantMessage, OcxContentPart, OcxMessage, OcxParsedRequest, OcxProviderConfig, OcxTextContent, OcxThinkingContent, OcxToolCall, OcxUsage } from "../types"; import { isAllowedToolChoice, modelInList, namespacedToolName, resolveToolChoiceWireName, toolAllowedByChoice } from "../types"; import { mapReasoningEffort, modelRecordValue } from "../reasoning-effort"; @@ -544,19 +544,37 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd } if (parsed.options.stopSequences !== undefined) body.stop = parsed.options.stopSequences; const reasoningEffort = mapReasoningEffort(provider, parsed.modelId, parsed.options.reasoning); + let reasoningLog: AdapterRequest["reasoningLog"]; if (reasoningEffort !== undefined) { if (modelInList(provider.thinkingBudgetModels, parsed.modelId)) { const budget = thinkingBudgetForEffort(parsed, reasoningEffort, maxTokens); - if (budget !== undefined) body.thinking_budget = budget; + if (budget !== undefined) { + body.thinking_budget = budget; + reasoningLog = { + effectiveEffort: parsed.options.reasoning === "minimal" ? "minimal" : reasoningEffort, + wireField: "thinking_budget", + wireValue: budget, + }; + } } else if (modelInList(provider.thinkingToggleModels, parsed.modelId)) { // Vendor thinking-toggle wire: the mapped value is sent as `thinking: {type}` because // these models ignore/reject reasoning_effort. Most use enabled/disabled; MiniMax-M3 // uses adaptive/disabled. if (reasoningEffort === "enabled" || reasoningEffort === "disabled" || reasoningEffort === "adaptive") { body.thinking = { type: reasoningEffort }; + reasoningLog = { + effectiveEffort: reasoningEffort, + wireField: "thinking.type", + wireValue: reasoningEffort, + }; } } else { body.reasoning_effort = reasoningEffort; + reasoningLog = { + effectiveEffort: reasoningEffort, + wireField: "reasoning_effort", + wireValue: reasoningEffort, + }; } } if (parsed.options.presencePenalty !== undefined && !modelInList(provider.noPenaltyModels, parsed.modelId)) { @@ -610,7 +628,13 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd }); } - return { url, method: "POST", headers, body: bodyJson }; + return { + url, + method: "POST", + headers, + body: bodyJson, + ...(reasoningLog ? { reasoningLog } : {}), + }; }, async *parseStream(response: Response): AsyncGenerator { diff --git a/src/server/request-log.ts b/src/server/request-log.ts index fcbc1f58349..53d66f0f542 100644 --- a/src/server/request-log.ts +++ b/src/server/request-log.ts @@ -8,6 +8,7 @@ import { import { CODEX_CONFIG_PATH, readRootTomlString } from "../codex/paths"; import { readCodexCatalogPath } from "../codex/catalog"; import type { OcxUsage } from "../types"; +import type { AdapterRequest } from "../adapters/base"; import { redactSecretString } from "../lib/redact"; import { appendUsageEntry, @@ -38,6 +39,9 @@ export interface RequestLogContext { /** Internal structural combo identity; omitted from RequestLogEntry/JSONL. */ comboId?: string; requestedEffort?: string; + effectiveEffort?: string; + reasoningWireField?: string; + reasoningWireValue?: string | number; requestedServiceTier?: string; requestedSpeedLabel?: string; configuredServiceTier?: string; @@ -84,6 +88,9 @@ export interface RequestLogEntry { surface?: "claude" | "claude-desktop"; requestedModel?: string; requestedEffort?: string; + effectiveEffort?: string; + reasoningWireField?: string; + reasoningWireValue?: string | number; requestedServiceTier?: string; requestedSpeedLabel?: string; configuredServiceTier?: string; @@ -147,6 +154,9 @@ export function requestLogEntryFromPersistedUsage(entry: PersistedUsageEntry): R ...(entry.surface === "claude" || entry.surface === "claude-desktop" ? { surface: entry.surface } : {}), ...(entry.requestedModel ? { requestedModel: entry.requestedModel } : {}), ...(entry.requestedEffort ? { requestedEffort: entry.requestedEffort } : {}), + ...(entry.effectiveEffort ? { effectiveEffort: entry.effectiveEffort } : {}), + ...(entry.reasoningWireField ? { reasoningWireField: entry.reasoningWireField } : {}), + ...(entry.reasoningWireValue !== undefined ? { reasoningWireValue: entry.reasoningWireValue } : {}), ...(entry.requestedServiceTier ? { requestedServiceTier: entry.requestedServiceTier } : {}), ...(entry.requestedSpeedLabel ? { requestedSpeedLabel: entry.requestedSpeedLabel } : {}), ...(entry.configuredServiceTier ? { configuredServiceTier: entry.configuredServiceTier } : {}), @@ -224,6 +234,9 @@ export function addRequestLog(entry: RequestLogEntry) { ...(entry.resolvedModel ? { resolvedModel: entry.resolvedModel } : {}), ...(entry.requestedModel ? { requestedModel: entry.requestedModel } : {}), ...(entry.requestedEffort ? { requestedEffort: entry.requestedEffort } : {}), + ...(entry.effectiveEffort ? { effectiveEffort: entry.effectiveEffort } : {}), + ...(entry.reasoningWireField ? { reasoningWireField: entry.reasoningWireField } : {}), + ...(entry.reasoningWireValue !== undefined ? { reasoningWireValue: entry.reasoningWireValue } : {}), ...(entry.requestedServiceTier ? { requestedServiceTier: entry.requestedServiceTier } : {}), ...(entry.requestedSpeedLabel ? { requestedSpeedLabel: entry.requestedSpeedLabel } : {}), ...(entry.configuredServiceTier ? { configuredServiceTier: entry.configuredServiceTier } : {}), @@ -271,6 +284,22 @@ export function recordFirstOutput( } } +/** Copy the adapter's exact outbound reasoning parameter into the durable request log. */ +export function recordAdapterReasoning( + logCtx: RequestLogContext, + request: AdapterRequest, +): void { + delete logCtx.effectiveEffort; + delete logCtx.reasoningWireField; + delete logCtx.reasoningWireValue; + if (!request.reasoningLog) return; + logCtx.effectiveEffort = redactSecretString(request.reasoningLog.effectiveEffort).slice(0, 64); + logCtx.reasoningWireField = request.reasoningLog.wireField; + logCtx.reasoningWireValue = typeof request.reasoningLog.wireValue === "string" + ? redactSecretString(request.reasoningLog.wireValue).slice(0, 64) + : request.reasoningLog.wireValue; +} + export function requestLogErrorCode(status: number, upstreamError?: string): string | undefined { if (status >= 200 && status < 400) return undefined; // Defense in depth: mid-stream web-search aborts used to land as 502 with this message. @@ -584,6 +613,9 @@ export function addFinalRequestLog( ...(logCtx.surface ? { surface: logCtx.surface } : {}), ...(logCtx.requestedModel ? { requestedModel: logCtx.requestedModel } : {}), ...(logCtx.requestedEffort ? { requestedEffort: logCtx.requestedEffort } : {}), + ...(logCtx.effectiveEffort ? { effectiveEffort: logCtx.effectiveEffort } : {}), + ...(logCtx.reasoningWireField ? { reasoningWireField: logCtx.reasoningWireField } : {}), + ...(logCtx.reasoningWireValue !== undefined ? { reasoningWireValue: logCtx.reasoningWireValue } : {}), ...(logCtx.requestedServiceTier ? { requestedServiceTier: logCtx.requestedServiceTier } : {}), ...(logCtx.requestedSpeedLabel ? { requestedSpeedLabel: logCtx.requestedSpeedLabel } : {}), ...(logCtx.configuredServiceTier ? { configuredServiceTier: logCtx.configuredServiceTier } : {}), diff --git a/src/server/responses/core.ts b/src/server/responses/core.ts index d80807d7da8..cb709f491ac 100644 --- a/src/server/responses/core.ts +++ b/src/server/responses/core.ts @@ -91,6 +91,7 @@ import { inspectResponseLogJson, noteAttemptSend, readConfiguredCodexServiceTier, + recordAdapterReasoning, requestLogSpeedLabel, sealRequestAttemptIdentity, usageFromResponsesPayload, @@ -1123,6 +1124,7 @@ export async function handleResponses( ); } let request = await adapter.buildRequest(parsed, { headers: selectedForwardHeaders }); + recordAdapterReasoning(logCtx, request); const passthroughEstimate = typeof request.usageLog?.inputTokens === "number" ? request.usageLog.inputTokens : undefined; @@ -1212,6 +1214,7 @@ export async function handleResponses( config.cacheRetention, ); request = await retryAdapter.buildRequest(parsed, { headers: retryHeaders }); + recordAdapterReasoning(logCtx, request); await upstreamResponse.body?.cancel().catch(() => undefined); authCtx = retryAuthCtx; @@ -1613,6 +1616,7 @@ export async function handleResponses( forceEmptyResponseId: true, abortSignal: options.abortSignal, ...(options.onFirstOutput ? { onFirstOutput: options.onFirstOutput } : {}), + onRequestBuilt: request => recordAdapterReasoning(logCtx, request), onUsage: usage => { logCtx.usageFromBridge = true; if (usage) { @@ -1657,6 +1661,7 @@ export async function handleResponses( let activeAdapter = adapter; const request = await activeAdapter.buildRequest(parsed, { headers: selectedForwardHeaders }); + recordAdapterReasoning(logCtx, request); const inputTokenEstimate = typeof request.usageLog?.inputTokens === "number" ? request.usageLog.inputTokens : undefined; @@ -1709,6 +1714,7 @@ export async function handleResponses( headers: selectedForwardHeaders, ...(imageTierBias > 0 ? { imageTierBias } : {}), }); + recordAdapterReasoning(logCtx, retryRequest); const retryEstimate = typeof retryRequest.usageLog?.inputTokens === "number" ? retryRequest.usageLog.inputTokens : undefined; @@ -1849,6 +1855,7 @@ export async function handleResponses( headers: selectedForwardHeaders, ...(imageTierBias > 0 ? { imageTierBias } : {}), }); + recordAdapterReasoning(logCtx, continuationRequest); const continuationEstimate = typeof continuationRequest.usageLog?.inputTokens === "number" ? continuationRequest.usageLog.inputTokens : undefined; diff --git a/src/usage/log.ts b/src/usage/log.ts index 478b8cc94e7..a91eb5a6a7e 100644 --- a/src/usage/log.ts +++ b/src/usage/log.ts @@ -41,6 +41,10 @@ export interface PersistedUsageEntry { requestedModel?: string; /** Reasoning effort / service-tier metadata for GUI Logs after restart. */ requestedEffort?: string; + /** Adapter-normalized tier and exact upstream parameter emitted for this request. */ + effectiveEffort?: string; + reasoningWireField?: string; + reasoningWireValue?: string | number; requestedServiceTier?: string; requestedSpeedLabel?: string; configuredServiceTier?: string; @@ -223,6 +227,17 @@ function normalizeUsageEntry(entry: PersistedUsageEntry): PersistedUsageEntry { ...(typeof entry.requestedEffort === "string" && entry.requestedEffort ? { requestedEffort: capMetadataString(entry.requestedEffort) } : {}), + ...(typeof entry.effectiveEffort === "string" && entry.effectiveEffort + ? { effectiveEffort: capMetadataString(entry.effectiveEffort) } + : {}), + ...(typeof entry.reasoningWireField === "string" && entry.reasoningWireField + ? { reasoningWireField: capMetadataString(entry.reasoningWireField) } + : {}), + ...(typeof entry.reasoningWireValue === "string" && entry.reasoningWireValue + ? { reasoningWireValue: capMetadataString(entry.reasoningWireValue) } + : isNonNegativeFiniteNumber(entry.reasoningWireValue) + ? { reasoningWireValue: entry.reasoningWireValue } + : {}), ...(typeof entry.requestedServiceTier === "string" && entry.requestedServiceTier ? { requestedServiceTier: capMetadataString(entry.requestedServiceTier) } : {}), diff --git a/src/web-search/loop.ts b/src/web-search/loop.ts index ae8120e19e0..12eb414049f 100644 --- a/src/web-search/loop.ts +++ b/src/web-search/loop.ts @@ -1,4 +1,4 @@ -import type { ProviderAdapter } from "../adapters/base"; +import type { AdapterRequest, ProviderAdapter } from "../adapters/base"; import type { AdapterEvent, OcxMessage, OcxParsedRequest, OcxProviderConfig, OcxThinkingContent, OcxUsage } from "../types"; import { namespacedToolName } from "../types"; import { bridgeToResponsesSSE } from "../bridge"; @@ -191,6 +191,8 @@ export interface WebSearchLoopDeps { onFirstOutput?: () => void; /** Raw adapter usage at the terminal event, pre wire-normalization (see bridgeToResponsesSSE onUsage). */ onUsage?: (usage: OcxUsage | undefined) => void; + /** Observe the exact adapter request selected for each routed-model iteration. */ + onRequestBuilt?: (request: AdapterRequest) => void; /** * 429 key-failover hook: rotate the provider's active pool key and return a rebuilt adapter, * or null when the pool is exhausted (same semantics as the normal routed path). @@ -271,6 +273,7 @@ export async function runWithWebSearch(deps: WebSearchLoopDeps): Promise { try { const provider: OcxProviderConfig = providerConfigSeed(PROVIDER_REGISTRY.find(e => e.id === "mimo-free")!); const adapter = createMimoFreeAdapter(provider); - const req = await adapter.buildRequest(minimalRequest()); + const parsed = minimalRequest(); + parsed.options.reasoning = "high"; + const req = await adapter.buildRequest(parsed); const headers = req.headers as Record; expect(req.url).toBe(MIMO_CHAT_URL); @@ -308,6 +310,11 @@ describe("mimo-free adapter request building", () => { const body = JSON.parse(req.body as string) as { messages: { role: string; content: string }[] }; expect(body.messages[0]?.role).toBe("system"); expect(body.messages[0]?.content).toBe(MIMO_SYSTEM_MARKER); + expect(req.reasoningLog).toEqual({ + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }); } finally { globalThis.fetch = originalFetch; resetMimoJwtCache(); diff --git a/tests/reasoning-effort.test.ts b/tests/reasoning-effort.test.ts index 567f4edb1d7..ba3385a7f56 100644 --- a/tests/reasoning-effort.test.ts +++ b/tests/reasoning-effort.test.ts @@ -2,6 +2,7 @@ import { describe, expect, test } from "bun:test"; import { buildCatalogEntries } from "../src/codex/catalog"; import { createAnthropicAdapter } from "../src/adapters/anthropic"; import { createOpenAIChatAdapter } from "../src/adapters/openai-chat"; +import type { AdapterRequest } from "../src/adapters/base"; import { configuredReasoningEfforts, mapReasoningEffort, sanitizeCodexReasoningEfforts } from "../src/reasoning-effort"; import { routeModel } from "../src/router"; import { resolveWireProtocolOverride } from "../src/server/adapter-resolve"; @@ -34,10 +35,18 @@ function parsed(modelId: string, providerOptions: OcxParsedRequest["options"]): } function buildBody(provider: OcxProviderConfig, modelId: string, options: OcxParsedRequest["options"]): Record { - const req = createOpenAIChatAdapter(provider).buildRequest(parsed(modelId, options)); + const req = buildChatRequest(provider, modelId, options); return JSON.parse(req.body as string) as Record; } +function buildChatRequest( + provider: OcxProviderConfig, + modelId: string, + options: OcxParsedRequest["options"], +): AdapterRequest { + return createOpenAIChatAdapter(provider).buildRequest(parsed(modelId, options)) as AdapterRequest; +} + describe("provider-specific reasoning effort mapping", () => { test("Codex catalog advertises only the efforts actually supported by a routed model", () => { const entries = buildCatalogEntries(nativeTemplate(), [], [ @@ -73,8 +82,21 @@ describe("provider-specific reasoning effort mapping", () => { reasoningEfforts: ["low", "medium", "high"], }; - expect(buildBody(provider, "glm-5.2", { reasoning: "xhigh" }).reasoning_effort).toBe("high"); - expect(buildBody(provider, "glm-5.2", { reasoning: "max" }).reasoning_effort).toBe("high"); + const xhigh = buildChatRequest(provider, "glm-5.2", { reasoning: "xhigh" }); + const max = buildChatRequest(provider, "glm-5.2", { reasoning: "max" }); + + expect(JSON.parse(xhigh.body).reasoning_effort).toBe("high"); + expect(JSON.parse(max.body).reasoning_effort).toBe("high"); + expect(xhigh.reasoningLog).toEqual({ + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }); + expect(max.reasoningLog).toEqual({ + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }); }); test("Neuralwatt GLM-5.2 sends direct max and preserves reasoning history", () => { @@ -248,16 +270,22 @@ describe("provider-specific reasoning effort mapping", () => { max: "max", ultra: "max", })) { - const body = buildBody(route.provider, route.modelId, { + const req = buildChatRequest(route.provider, route.modelId, { reasoning: requested, temperature: 0.2, topP: 0.7, presencePenalty: 1, frequencyPenalty: 1, }); + const body = JSON.parse(req.body) as Record; expect(body.model).toBe("k3"); expect(body.reasoning_effort).toBe(wire); + expect(req.reasoningLog).toEqual({ + effectiveEffort: wire, + wireField: "reasoning_effort", + wireValue: wire, + }); expect(body).not.toHaveProperty("temperature"); expect(body).not.toHaveProperty("top_p"); expect(body).not.toHaveProperty("presence_penalty"); @@ -458,9 +486,15 @@ describe("thinking-toggle models (260707)", () => { }; test("high effort emits thinking enabled, never reasoning_effort", () => { - const body = buildBody(toggleProvider, "mimo-v2.5", { reasoning: "high" }); + const req = buildChatRequest(toggleProvider, "mimo-v2.5", { reasoning: "high" }); + const body = JSON.parse(req.body) as Record; expect(body.thinking).toEqual({ type: "enabled" }); expect(body).not.toHaveProperty("reasoning_effort"); + expect(req.reasoningLog).toEqual({ + effectiveEffort: "enabled", + wireField: "thinking.type", + wireValue: "enabled", + }); }); test("low effort emits thinking disabled", () => { @@ -546,10 +580,16 @@ describe("thinking-budget models (260709)", () => { ] as const; for (const [reasoning, budget] of cases) { - const body = buildBody(budgetProvider, "qwen3.5-397b", { reasoning, maxOutputTokens: 10000 }); + const req = buildChatRequest(budgetProvider, "qwen3.5-397b", { reasoning, maxOutputTokens: 10000 }); + const body = JSON.parse(req.body) as Record; expect(body.thinking_budget).toBe(budget); expect(body).not.toHaveProperty("reasoning_effort"); expect(body).not.toHaveProperty("thinking"); + expect(req.reasoningLog).toEqual({ + effectiveEffort: reasoning, + wireField: "thinking_budget", + wireValue: budget, + }); } }); @@ -560,9 +600,15 @@ describe("thinking-budget models (260709)", () => { }); test("minimal Qwen reasoning maps to a zero budget", () => { - const body = buildBody(budgetProvider, "qwen3.5-397b", { reasoning: "minimal", maxOutputTokens: 10000 }); + const req = buildChatRequest(budgetProvider, "qwen3.5-397b", { reasoning: "minimal", maxOutputTokens: 10000 }); + const body = JSON.parse(req.body) as Record; expect(body.thinking_budget).toBe(0); expect(body).not.toHaveProperty("reasoning_effort"); + expect(req.reasoningLog).toEqual({ + effectiveEffort: "minimal", + wireField: "thinking_budget", + wireValue: 0, + }); }); test("routed Qwen models advertise five levels and send thinking_budget over openai-chat", () => { diff --git a/tests/request-log.test.ts b/tests/request-log.test.ts index cf7689fc135..4e162160282 100644 --- a/tests/request-log.test.ts +++ b/tests/request-log.test.ts @@ -17,6 +17,7 @@ import { getRequestLogEntries, hydrateRequestLogsFromDisk, noteAttemptSend, + recordAdapterReasoning, recordFirstOutput, requestLogEntryFromPersistedUsage, sealRequestAttemptIdentity, @@ -44,6 +45,48 @@ function log(overrides: Partial): RequestLogEntry { } describe("request log metadata", () => { + test("records the adapter's exact outbound reasoning parameter", () => { + const logCtx: RequestLogContext = { + model: "grok-4.5", + provider: "xai", + requestedEffort: "max", + }; + + recordAdapterReasoning(logCtx, { + url: "https://api.x.ai/v1/chat/completions", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: { + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }, + }); + + expect(logCtx).toMatchObject({ + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", + }); + + const sensitiveAlias = ["sk", "proj", "redaction-fixture"].join("-"); + recordAdapterReasoning(logCtx, { + url: "https://provider.test/v1/chat/completions", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: { + effectiveEffort: sensitiveAlias, + wireField: "reasoning_effort", + wireValue: sensitiveAlias, + }, + }); + expect(logCtx.effectiveEffort).not.toContain("redaction-fixture"); + expect(logCtx.reasoningWireValue).not.toContain("redaction-fixture"); + }); + test("recordFirstOutput is one-shot for request and active attempt (WP4 TTFT)", () => { const attempt = beginRequestAttempt(1, "a", "m1", "openai-chat"); const logCtx: RequestLogContext = { @@ -398,6 +441,9 @@ describe("request log metadata", () => { provider: "chatgpt-p000001", requestedModel: "gpt-5.5", requestedEffort: "xhigh", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", requestedServiceTier: "priority", requestedSpeedLabel: requestLogSpeedLabel("priority"), configuredServiceTier: "fast", @@ -421,6 +467,9 @@ describe("request log metadata", () => { expect(entries[0]).toMatchObject({ requestedModel: "gpt-5.5", requestedEffort: "xhigh", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", requestedServiceTier: "priority", requestedSpeedLabel: "fast", configuredServiceTier: "fast", @@ -879,6 +928,9 @@ describe("request log restart hydrate", () => { model: "gpt-5.6-sol", requestedModel: "gpt-5.6-sol", requestedEffort: "high", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", requestedServiceTier: "priority", requestedSpeedLabel: "fast", configuredServiceTier: "auto", @@ -898,6 +950,9 @@ describe("request log restart hydrate", () => { model: "gpt-5.6-sol", requestedModel: "gpt-5.6-sol", requestedEffort: "high", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", requestedServiceTier: "priority", requestedSpeedLabel: "fast", configuredServiceTier: "auto", diff --git a/tests/usage-log.test.ts b/tests/usage-log.test.ts index 5dd574e354f..10bc0d3d94b 100644 --- a/tests/usage-log.test.ts +++ b/tests/usage-log.test.ts @@ -421,7 +421,10 @@ describe("usage log", () => { provider: "openai", model: "gpt-5.6-sol", requestedModel: "gpt-5.6-sol", - requestedEffort: "high", + requestedEffort: "xhigh", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", requestedServiceTier: "priority", requestedSpeedLabel: "fast", configuredServiceTier: "auto", @@ -433,7 +436,10 @@ describe("usage log", () => { }); expect(readUsageEntries()[0]).toMatchObject({ requestId: "ocx-effort", - requestedEffort: "high", + requestedEffort: "xhigh", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", requestedServiceTier: "priority", requestedSpeedLabel: "fast", configuredServiceTier: "auto", diff --git a/tests/web-search.test.ts b/tests/web-search.test.ts index 57bea3ecb61..93acfea819b 100644 --- a/tests/web-search.test.ts +++ b/tests/web-search.test.ts @@ -327,13 +327,24 @@ describe("BUG-R86 routed web-search timeout semantics", () => { test("routed iterations use upstream streaming and never call parseResponse", async () => { const seenStream: boolean[] = []; + const reasoningLogs: unknown[] = []; let parseStreamCalls = 0; let parseResponseCalls = 0; const adapter: ProviderAdapter = { name: "stream-only", buildRequest(parsed) { seenStream.push(parsed.stream); - return { url: "https://routed.test/v1", method: "POST", headers: {}, body: "{}" }; + return { + url: "https://routed.test/v1", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: { + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }, + }; }, fetchResponse: async () => new Response("wire", { status: 200 }), async *parseStream() { @@ -355,11 +366,17 @@ describe("BUG-R86 routed web-search timeout semantics", () => { selectedForwardHeaders: new Headers({ authorization: "Bearer token" }), settings: { model: "gpt-5.6-luna", reasoning: "low", timeoutMs: 30_000 }, maxSearches: 1, + onRequestBuilt: request => reasoningLogs.push(request.reasoningLog), }); expect(response.status).toBe(200); const frames = await collectSse(response.body!); expect(seenStream).toEqual([true]); + expect(reasoningLogs).toEqual([{ + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }]); expect(parseStreamCalls).toBe(1); expect(parseResponseCalls).toBe(0); expect(frames.some(frame => frame.event === "response.completed")).toBe(true); From 3ab721a65d226fba8cf8d1631ab1bc3507033e71 Mon Sep 17 00:00:00 2001 From: olddonkey Date: Sun, 26 Jul 2026 02:07:53 -0700 Subject: [PATCH 2/4] fix(logs): harden reasoning diagnostics --- gui/src/pages/Logs.tsx | 24 ++++- gui/tests/logs-auto-refresh.test.tsx | 73 +++++++++++++++ src/server/request-log.ts | 61 +++++++++++-- src/server/responses/core.ts | 2 + src/usage/log.ts | 19 ++++ tests/request-log.test.ts | 113 ++++++++++++++++++++++-- tests/server-combo-failover-e2e.test.ts | 36 ++++++++ tests/usage-log.test.ts | 42 +++++++++ 8 files changed, 356 insertions(+), 14 deletions(-) diff --git a/gui/src/pages/Logs.tsx b/gui/src/pages/Logs.tsx index 497aceae65d..8bbb2068ef5 100644 --- a/gui/src/pages/Logs.tsx +++ b/gui/src/pages/Logs.tsx @@ -85,6 +85,10 @@ interface LogAttempt { totalTokens?: number; errorCode?: string; firstOutputMs?: number; + requestedEffort?: string; + effectiveEffort?: string; + reasoningWireField?: string; + reasoningWireValue?: string | number; displayMetrics?: LogDisplayMetrics; } @@ -166,7 +170,14 @@ function speedLabel(log: LogEntry): string | undefined { return undefined; } -function effortLabel(log: LogEntry): string { +interface ReasoningLogFields { + requestedEffort?: string; + effectiveEffort?: string; + reasoningWireField?: string; + reasoningWireValue?: string | number; +} + +function effortLabel(log: ReasoningLogFields): string { const requested = log.requestedEffort?.replace(/\s*->\s*/g, " → "); const effective = log.effectiveEffort; if (!requested) return effective ?? "-"; @@ -174,7 +185,7 @@ function effortLabel(log: LogEntry): string { return `${requested} → ${effective}`; } -function reasoningWireLabel(log: LogEntry): string | undefined { +function reasoningWireLabel(log: ReasoningLogFields): string | undefined { if (!log.reasoningWireField || log.reasoningWireValue === undefined) return undefined; return `${log.reasoningWireField}=${log.reasoningWireValue}`; } @@ -667,6 +678,7 @@ function LogDetailDialog({ {detail.attempts.toSorted((a, b) => a.ordinal - b.ordinal).map(attempt => { const attemptCost = attempt.displayMetrics?.cost; + const attemptReasoningWire = reasoningWireLabel(attempt); const matched = attemptCost?.kind === "value" ? attemptCost.estimate.price : undefined; const reason = attempt.errorCode ?? (attempt.recoveryKinds.length ? attempt.recoveryKinds.join(", ") : undefined) @@ -677,6 +689,14 @@ function LogDetailDialog({ {attempt.provider}
{attempt.model} + {(attempt.requestedEffort || attempt.effectiveEffort) && ( + <> +
+ + {effortLabel(attempt)}{attemptReasoningWire ? ` (${attemptReasoningWire})` : ""} + + + )} {matched && ( <>
diff --git a/gui/tests/logs-auto-refresh.test.tsx b/gui/tests/logs-auto-refresh.test.tsx index 16905a2ac04..5af6a5730a7 100644 --- a/gui/tests/logs-auto-refresh.test.tsx +++ b/gui/tests/logs-auto-refresh.test.tsx @@ -332,3 +332,76 @@ test("Logs: switching to the Debug tab stops scheduled log requests", async () = await act(async () => { root.unmount(); }); }); + +test("Logs: attempt details render exact reasoning wire values without legacy placeholders", async () => { + const attemptsLog = { + ...sampleLog, + requestedEffort: "high", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", + attempts: [ + { + ordinal: 1, + provider: "budget-provider", + model: "budget-model", + adapter: "openai-chat", + status: 503, + durationMs: 10, + sendCount: 1, + recoveryKinds: [], + usageStatus: "unreported", + requestedEffort: "minimal", + effectiveEffort: "low", + reasoningWireField: "thinking_budget", + reasoningWireValue: 0, + }, + { + ordinal: 2, + provider: "toggle-provider", + model: "toggle-model", + adapter: "openai-chat", + status: 503, + durationMs: 11, + sendCount: 1, + recoveryKinds: [], + usageStatus: "unreported", + requestedEffort: "high", + effectiveEffort: "enabled", + reasoningWireField: "thinking.type", + reasoningWireValue: "enabled", + }, + { + ordinal: 3, + provider: "legacy-provider", + model: "legacy-model", + adapter: "openai-chat", + status: 200, + durationMs: 12, + sendCount: 1, + recoveryKinds: [], + usageStatus: "unreported", + }, + ], + }; + globalThis.fetch = (async (input) => { + if (!String(input).includes("/api/logs")) return new Response(null, { status: 404 }); + return jsonResponse([attemptsLog]); + }) as typeof fetch; + + const { root, container } = await mountLogs(); + await flushMicrotasks(); + await act(async () => { + container.querySelector(".log-detail-btn")!.click(); + }); + + const rows = [...container.querySelectorAll(".log-detail-attempts tbody tr")]; + expect(rows).toHaveLength(3); + expect(rows[0]?.textContent).toContain("minimal → low (thinking_budget=0)"); + expect(rows[1]?.textContent).toContain("high → enabled (thinking.type=enabled)"); + expect(rows[2]?.textContent).toContain("legacy-model"); + expect(rows[2]?.querySelectorAll("br")).toHaveLength(1); + expect(rows[2]?.textContent).not.toContain("undefined"); + + await act(async () => { root.unmount(); }); +}); diff --git a/src/server/request-log.ts b/src/server/request-log.ts index 53d66f0f542..0e0acd30b0f 100644 --- a/src/server/request-log.ts +++ b/src/server/request-log.ts @@ -284,6 +284,20 @@ export function recordFirstOutput( } } +/** Snapshot target-specific requested effort even for runTurn adapters with no AdapterRequest. */ +export function recordAttemptRequestedEffort(logCtx: RequestLogContext): void { + const attempt = logCtx.activeAttempt; + if (!attempt) return; + delete attempt.requestedEffort; + try { + if (typeof logCtx.requestedEffort === "string" && logCtx.requestedEffort) { + attempt.requestedEffort = redactSecretString(logCtx.requestedEffort).slice(0, 64); + } + } catch { + // Request logging is best-effort and must not affect request delivery. + } +} + /** Copy the adapter's exact outbound reasoning parameter into the durable request log. */ export function recordAdapterReasoning( logCtx: RequestLogContext, @@ -292,12 +306,47 @@ export function recordAdapterReasoning( delete logCtx.effectiveEffort; delete logCtx.reasoningWireField; delete logCtx.reasoningWireValue; - if (!request.reasoningLog) return; - logCtx.effectiveEffort = redactSecretString(request.reasoningLog.effectiveEffort).slice(0, 64); - logCtx.reasoningWireField = request.reasoningLog.wireField; - logCtx.reasoningWireValue = typeof request.reasoningLog.wireValue === "string" - ? redactSecretString(request.reasoningLog.wireValue).slice(0, 64) - : request.reasoningLog.wireValue; + const attempt = logCtx.activeAttempt; + if (attempt) { + delete attempt.effectiveEffort; + delete attempt.reasoningWireField; + delete attempt.reasoningWireValue; + } + recordAttemptRequestedEffort(logCtx); + + // Diagnostics must never make an otherwise valid upstream request fail. Config files + // written by older versions (or edited by hand) can contain values that violate the + // current TypeScript shape, so validate the runtime object before redacting strings. + try { + const raw: unknown = request.reasoningLog; + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return; + const reasoning = raw as Record; + if (typeof reasoning.effectiveEffort !== "string" || !reasoning.effectiveEffort + || (reasoning.wireField !== "reasoning_effort" + && reasoning.wireField !== "thinking_budget" + && reasoning.wireField !== "thinking.type") + || (typeof reasoning.wireValue !== "string" + && !(typeof reasoning.wireValue === "number" + && Number.isFinite(reasoning.wireValue) + && reasoning.wireValue >= 0))) { + return; + } + + const effectiveEffort = redactSecretString(reasoning.effectiveEffort).slice(0, 64); + const wireValue = typeof reasoning.wireValue === "string" + ? redactSecretString(reasoning.wireValue).slice(0, 64) + : reasoning.wireValue; + logCtx.effectiveEffort = effectiveEffort; + logCtx.reasoningWireField = reasoning.wireField; + logCtx.reasoningWireValue = wireValue; + if (attempt) { + attempt.effectiveEffort = effectiveEffort; + attempt.reasoningWireField = reasoning.wireField; + attempt.reasoningWireValue = wireValue; + } + } catch { + // Request logging is best-effort and must not affect request delivery. + } } export function requestLogErrorCode(status: number, upstreamError?: string): string | undefined { diff --git a/src/server/responses/core.ts b/src/server/responses/core.ts index cb709f491ac..92c84fc8515 100644 --- a/src/server/responses/core.ts +++ b/src/server/responses/core.ts @@ -92,6 +92,7 @@ import { noteAttemptSend, readConfiguredCodexServiceTier, recordAdapterReasoning, + recordAttemptRequestedEffort, requestLogSpeedLabel, sealRequestAttemptIdentity, usageFromResponsesPayload, @@ -593,6 +594,7 @@ async function applyFinalRouteRequestNormalization(args: { logCtx.requestedEffort = `${logCtx.requestedEffort ?? "max"}->${clamped}`; } } + recordAttemptRequestedEffort(logCtx); logCtx.modelSupportsServiceTier = catalogModelSupportsServiceTier( route.modelId, logCtx.requestedServiceTier ?? logCtx.configuredServiceTier, diff --git a/src/usage/log.ts b/src/usage/log.ts index a91eb5a6a7e..0c4dca43727 100644 --- a/src/usage/log.ts +++ b/src/usage/log.ts @@ -29,6 +29,11 @@ export interface PersistedUsageAttempt { usage?: OcxUsage; totalTokens?: number; errorCode?: string; + /** Target-specific reasoning intent and exact adapter-normalized wire parameter. */ + requestedEffort?: string; + effectiveEffort?: string; + reasoningWireField?: string; + reasoningWireValue?: string | number; } export interface PersistedUsageEntry { @@ -200,6 +205,20 @@ function normalizeUsageAttempt(raw: unknown): PersistedUsageAttempt | null { ? { totalTokens: attempt.totalTokens } : {}), ...(typeof attempt.errorCode === "string" ? { errorCode: attempt.errorCode } : {}), + ...(typeof attempt.requestedEffort === "string" && attempt.requestedEffort + ? { requestedEffort: capMetadataString(attempt.requestedEffort) } + : {}), + ...(typeof attempt.effectiveEffort === "string" && attempt.effectiveEffort + ? { effectiveEffort: capMetadataString(attempt.effectiveEffort) } + : {}), + ...(typeof attempt.reasoningWireField === "string" && attempt.reasoningWireField + ? { reasoningWireField: capMetadataString(attempt.reasoningWireField) } + : {}), + ...(typeof attempt.reasoningWireValue === "string" && attempt.reasoningWireValue + ? { reasoningWireValue: capMetadataString(attempt.reasoningWireValue) } + : isNonNegativeFiniteNumber(attempt.reasoningWireValue) + ? { reasoningWireValue: attempt.reasoningWireValue } + : {}), }; } diff --git a/tests/request-log.test.ts b/tests/request-log.test.ts index 4e162160282..6190d291db8 100644 --- a/tests/request-log.test.ts +++ b/tests/request-log.test.ts @@ -46,10 +46,12 @@ function log(overrides: Partial): RequestLogEntry { describe("request log metadata", () => { test("records the adapter's exact outbound reasoning parameter", () => { + const attempt = beginRequestAttempt(1, "xai", "grok-4.5", "openai-chat"); const logCtx: RequestLogContext = { model: "grok-4.5", provider: "xai", requestedEffort: "max", + activeAttempt: attempt, }; recordAdapterReasoning(logCtx, { @@ -70,6 +72,12 @@ describe("request log metadata", () => { reasoningWireField: "reasoning_effort", reasoningWireValue: "high", }); + expect(attempt).toMatchObject({ + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", + }); const sensitiveAlias = ["sk", "proj", "redaction-fixture"].join("-"); recordAdapterReasoning(logCtx, { @@ -85,6 +93,54 @@ describe("request log metadata", () => { }); expect(logCtx.effectiveEffort).not.toContain("redaction-fixture"); expect(logCtx.reasoningWireValue).not.toContain("redaction-fixture"); + expect(attempt.effectiveEffort).not.toContain("redaction-fixture"); + expect(attempt.reasoningWireValue).not.toContain("redaction-fixture"); + }); + + test("malformed adapter reasoning metadata never interrupts request logging", () => { + const malformed = [ + { effectiveEffort: 123, wireField: "reasoning_effort", wireValue: 123 }, + { effectiveEffort: null, wireField: "reasoning_effort", wireValue: "high" }, + { effectiveEffort: {}, wireField: "reasoning_effort", wireValue: "high" }, + { effectiveEffort: "high", wireField: "unknown", wireValue: "high" }, + { effectiveEffort: "high", wireField: "thinking_budget", wireValue: null }, + { effectiveEffort: "high", wireField: "thinking_budget", wireValue: {} }, + { effectiveEffort: "high", wireField: "thinking_budget", wireValue: Number.NaN }, + { effectiveEffort: "high", wireField: "thinking_budget", wireValue: -1 }, + ]; + + for (const reasoningLog of malformed) { + const attempt = beginRequestAttempt(1, "xai", "grok-4.5", "openai-chat"); + const logCtx: RequestLogContext = { + model: "grok-4.5", + provider: "xai", + requestedEffort: "max", + effectiveEffort: "stale", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "stale", + activeAttempt: attempt, + }; + Object.assign(attempt, { + effectiveEffort: "stale", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "stale", + }); + + expect(() => recordAdapterReasoning(logCtx, { + url: "https://provider.test/v1/chat/completions", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: reasoningLog as never, + })).not.toThrow(); + expect(logCtx.effectiveEffort).toBeUndefined(); + expect(logCtx.reasoningWireField).toBeUndefined(); + expect(logCtx.reasoningWireValue).toBeUndefined(); + expect(attempt.requestedEffort).toBe("max"); + expect(attempt.effectiveEffort).toBeUndefined(); + expect(attempt.reasoningWireField).toBeUndefined(); + expect(attempt.reasoningWireValue).toBeUndefined(); + } }); test("recordFirstOutput is one-shot for request and active attempt (WP4 TTFT)", () => { @@ -187,8 +243,25 @@ describe("request log metadata", () => { test("final combo logging keeps one logical row and finalizes its active attempt", () => { const entries: RequestLogEntry[] = []; - const a = finishRequestAttempt( - beginRequestAttempt(1, "a", "model-a", "openai-chat"), + const a = beginRequestAttempt(1, "a", "model-a", "openai-chat"); + recordAdapterReasoning({ + model: "model-a", + provider: "a", + requestedEffort: "minimal", + activeAttempt: a, + }, { + url: "https://provider-a.test/v1/chat/completions", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: { + effectiveEffort: "low", + wireField: "thinking_budget", + wireValue: 0, + }, + }); + finishRequestAttempt( + a, 503, 3, { inputTokens: 4, outputTokens: 1 }, @@ -196,10 +269,11 @@ describe("request log metadata", () => { const b = beginRequestAttempt(2, "b", "model-b", "openai-chat"); noteAttemptSend(b, undefined); const start = Date.now(); - addFinalRequestLog("combo-parent", start, { + const logCtx: RequestLogContext = { model: "combo/free", provider: "combo", requestedModel: "combo/free", + requestedEffort: "max", comboId: "free", resolvedModel: "model-b", providerAdapter: "openai-chat", @@ -207,7 +281,19 @@ describe("request log metadata", () => { attempts: [a, b], activeAttempt: b, activeAttemptStartedAt: start, - }, 200, undefined, entry => entries.push(entry)); + }; + recordAdapterReasoning(logCtx, { + url: "https://provider.test/v1/chat/completions", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: { + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }, + }); + addFinalRequestLog("combo-parent", start, logCtx, 200, undefined, entry => entries.push(entry)); expect(entries).toHaveLength(1); expect(entries[0]).toMatchObject({ @@ -219,8 +305,23 @@ describe("request log metadata", () => { usage: { inputTokens: 14, outputTokens: 3, totalTokens: 17 }, totalTokens: 17, attempts: [ - { provider: "a", status: 503 }, - { provider: "b", status: 200, usage: { inputTokens: 10, outputTokens: 2 } }, + { + provider: "a", + status: 503, + requestedEffort: "minimal", + effectiveEffort: "low", + reasoningWireField: "thinking_budget", + reasoningWireValue: 0, + }, + { + provider: "b", + status: 200, + usage: { inputTokens: 10, outputTokens: 2 }, + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", + }, ], }); }); diff --git a/tests/server-combo-failover-e2e.test.ts b/tests/server-combo-failover-e2e.test.ts index 5032e526726..433759c38cd 100644 --- a/tests/server-combo-failover-e2e.test.ts +++ b/tests/server-combo-failover-e2e.test.ts @@ -816,6 +816,42 @@ describe("server combo failover 030 activation matrix", () => { expect(bHits).toBe(1); }); + test("runTurn combo attempts retain requested effort without adapter wire metadata", async () => { + customRunTurn = async (parsed, _incoming, emit) => { + if (parsed.modelId === "m1") { + emit({ type: "error", message: "first target unavailable" }); + return; + } + emit({ type: "text_delta", text: "runTurn backup" }); + emit({ type: "done" }); + }; + const config = comboConfig({ + a: provider("test-run-turn", "https://a.test/v1", "key-a"), + b: provider("test-run-turn", "https://b.test/v1", "key-b"), + }); + + const response = await postLogged(config, { + reasoning: { effort: "high" }, + }); + expect(response.status).toBe(200); + expect(JSON.stringify(await response.json())).toContain("runTurn backup"); + const { log, usage } = await latestAttemptReceipts(config); + + for (const receipt of [log, usage]) { + expect(receipt).toMatchObject({ + attempts: [ + { ordinal: 1, provider: "a", requestedEffort: "high" }, + { ordinal: 2, provider: "b", requestedEffort: "high" }, + ], + }); + for (const attempt of receipt.attempts as Array>) { + expect(attempt).not.toHaveProperty("effectiveEffort"); + expect(attempt).not.toHaveProperty("reasoningWireField"); + expect(attempt).not.toHaveProperty("reasoningWireValue"); + } + } + }); + test("hosted web-search eager model failure hops through the loop path", async () => { const modelHits: Array<{ model?: string; hasWebTool: boolean }> = []; const routed = serve(async request => { diff --git a/tests/usage-log.test.ts b/tests/usage-log.test.ts index 10bc0d3d94b..4cd0e25b84f 100644 --- a/tests/usage-log.test.ts +++ b/tests/usage-log.test.ts @@ -54,6 +54,10 @@ describe("usage log", () => { inputTokenEstimate: 5, usage: { inputTokens: 5, outputTokens: 0, estimated: true }, totalTokens: 5, + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", headers: { authorization: "Bearer attempt-token" }, body: "attempt body secret", messages: ["attempt message secret"], @@ -86,9 +90,47 @@ describe("usage log", () => { inputTokenEstimate: 5, usage: { inputTokens: 5, outputTokens: 0, estimated: true }, totalTokens: 5, + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", }]); }); + test("omits malformed optional attempt reasoning metadata without dropping the attempt", () => { + appendUsageEntry({ + requestId: "ocx-attempt-reasoning", + timestamp: 1, + provider: "combo", + model: "combo/free", + status: 200, + durationMs: 4, + usageStatus: "unreported", + attempts: [{ + ordinal: 1, + provider: "a", + model: "m1", + adapter: "openai-chat", + status: 200, + durationMs: 3, + sendCount: 1, + recoveryKinds: [], + usageStatus: "unreported", + requestedEffort: 123, + effectiveEffort: null, + reasoningWireField: {}, + reasoningWireValue: -1, + } as never], + }); + + const attempt = readUsageEntries()[0]?.attempts?.[0]; + expect(attempt?.ordinal).toBe(1); + expect(attempt).not.toHaveProperty("requestedEffort"); + expect(attempt).not.toHaveProperty("effectiveEffort"); + expect(attempt).not.toHaveProperty("reasoningWireField"); + expect(attempt).not.toHaveProperty("reasoningWireValue"); + }); + test("drops only malformed persisted attempts while preserving valid siblings", () => { const valid = (ordinal: number) => ({ ordinal, From 9be7196ab160c75e05967139c83e7f4b308f8608 Mon Sep 17 00:00:00 2001 From: olddonkey Date: Sun, 26 Jul 2026 03:09:51 -0700 Subject: [PATCH 3/4] fix(logs): harden reasoning diagnostics --- src/server/request-log.ts | 2 +- src/web-search/loop.ts | 6 +++++- tests/request-log.test.ts | 1 + tests/web-search.test.ts | 45 +++++++++++++++++++++++++++++++++++---- 4 files changed, 48 insertions(+), 6 deletions(-) diff --git a/src/server/request-log.ts b/src/server/request-log.ts index 0e0acd30b0f..a14c532c226 100644 --- a/src/server/request-log.ts +++ b/src/server/request-log.ts @@ -325,7 +325,7 @@ export function recordAdapterReasoning( || (reasoning.wireField !== "reasoning_effort" && reasoning.wireField !== "thinking_budget" && reasoning.wireField !== "thinking.type") - || (typeof reasoning.wireValue !== "string" + || (!(typeof reasoning.wireValue === "string" && reasoning.wireValue) && !(typeof reasoning.wireValue === "number" && Number.isFinite(reasoning.wireValue) && reasoning.wireValue >= 0))) { diff --git a/src/web-search/loop.ts b/src/web-search/loop.ts index 12eb414049f..5284660775f 100644 --- a/src/web-search/loop.ts +++ b/src/web-search/loop.ts @@ -273,7 +273,11 @@ export async function runWithWebSearch(deps: WebSearchLoopDeps): Promise { { effectiveEffort: null, wireField: "reasoning_effort", wireValue: "high" }, { effectiveEffort: {}, wireField: "reasoning_effort", wireValue: "high" }, { effectiveEffort: "high", wireField: "unknown", wireValue: "high" }, + { effectiveEffort: "high", wireField: "reasoning_effort", wireValue: "" }, { effectiveEffort: "high", wireField: "thinking_budget", wireValue: null }, { effectiveEffort: "high", wireField: "thinking_budget", wireValue: {} }, { effectiveEffort: "high", wireField: "thinking_budget", wireValue: Number.NaN }, diff --git a/tests/web-search.test.ts b/tests/web-search.test.ts index 93acfea819b..02664c7e1d7 100644 --- a/tests/web-search.test.ts +++ b/tests/web-search.test.ts @@ -325,7 +325,7 @@ describe("BUG-R86 routed web-search timeout semantics", () => { expect(text).toContain("event: response.completed"); }); - test("routed iterations use upstream streaming and never call parseResponse", async () => { + test("routed iterations isolate diagnostic failures and never call parseResponse", async () => { const seenStream: boolean[] = []; const reasoningLogs: unknown[] = []; let parseStreamCalls = 0; @@ -366,7 +366,10 @@ describe("BUG-R86 routed web-search timeout semantics", () => { selectedForwardHeaders: new Headers({ authorization: "Bearer token" }), settings: { model: "gpt-5.6-luna", reasoning: "low", timeoutMs: 30_000 }, maxSearches: 1, - onRequestBuilt: request => reasoningLogs.push(request.reasoningLog), + onRequestBuilt: request => { + reasoningLogs.push(request.reasoningLog); + throw new Error("diagnostic hook failure must not abort delivery"); + }, }); expect(response.status).toBe(200); @@ -498,16 +501,37 @@ describe("web-search sidecar native web_search_call emission", () => { ))) as typeof fetch; // First adapter always 429s via fetchResponse; the rotated adapter answers. + const reasoningLogs: unknown[] = []; const firstAdapter: ProviderAdapter = { name: "mock-429", - buildRequest: () => ({ url: "https://routed.test/v1", method: "POST", headers: {}, body: "{}" }), + buildRequest: () => ({ + url: "https://routed.test/v1", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: { + effectiveEffort: "low", + wireField: "reasoning_effort", + wireValue: "low", + }, + }), fetchResponse: async () => new Response("rate limited", { status: 429, headers: { "retry-after": "30" } }), async *parseStream() { /* unused */ }, async parseResponse() { return [{ type: "text_delta", text: "should not reach" }, { type: "done" }] as AdapterEvent[]; }, }; const rotatedAdapter: ProviderAdapter = { name: "mock-rotated", - buildRequest: () => ({ url: "https://routed.test/v1", method: "POST", headers: {}, body: "{}" }), + buildRequest: () => ({ + url: "https://routed.test/v1", + method: "POST", + headers: {}, + body: "{}", + reasoningLog: { + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }, + }), fetchResponse: async () => new Response("{}", { status: 200 }), async *parseStream() { yield { type: "text_delta", text: "answer from rotated key" }; @@ -525,6 +549,7 @@ describe("web-search sidecar native web_search_call emission", () => { selectedForwardHeaders: new Headers({ authorization: "Bearer token" }), settings: { model: "gpt-5.4-mini", reasoning: "low", timeoutMs: 30_000 }, maxSearches: 1, + onRequestBuilt: request => reasoningLogs.push(request.reasoningLog), on429: retryAfter => { rotations++; expect(retryAfter).toBe("30"); @@ -537,6 +562,18 @@ describe("web-search sidecar native web_search_call emission", () => { const output = completed.output as { type: string; content?: { text?: string }[] }[]; expect(output.find(o => o.type === "message")?.content?.[0]?.text).toBe("answer from rotated key"); expect(rotations).toBe(1); + expect(reasoningLogs).toEqual([ + { + effectiveEffort: "low", + wireField: "reasoning_effort", + wireValue: "low", + }, + { + effectiveEffort: "high", + wireField: "reasoning_effort", + wireValue: "high", + }, + ]); }); test("loop 429 with exhausted pool (on429 null) surfaces the provider error", async () => { From cf06abbd56bfbdd45b61f49c86e691ea6e738860 Mon Sep 17 00:00:00 2001 From: olddonkey Date: Sun, 26 Jul 2026 14:15:01 -0700 Subject: [PATCH 4/4] fix(logs): address maintainer review --- gui/src/pages/Logs.tsx | 10 ++- gui/tests/logs-auto-refresh.test.tsx | 6 +- src/server/responses/core.ts | 27 +++--- tests/server-combo-failover-e2e.test.ts | 112 +++++++++++++++++++++++- 4 files changed, 142 insertions(+), 13 deletions(-) diff --git a/gui/src/pages/Logs.tsx b/gui/src/pages/Logs.tsx index 8bbb2068ef5..9a6a670ca9c 100644 --- a/gui/src/pages/Logs.tsx +++ b/gui/src/pages/Logs.tsx @@ -181,6 +181,8 @@ function effortLabel(log: ReasoningLogFields): string { const requested = log.requestedEffort?.replace(/\s*->\s*/g, " → "); const effective = log.effectiveEffort; if (!requested) return effective ?? "-"; + // requestedEffort may already contain a cap/clamp chain (for example max->high). + // Only append the adapter result when it differs from that chain's terminal value. if (!effective || requested === effective || requested.split(" → ").at(-1) === effective) return requested; return `${requested} → ${effective}`; } @@ -447,6 +449,7 @@ export default function Logs({ apiBase }: { apiBase: string }) { )} {virtualRows.map(virtualRow => { const log = filteredLogs[filteredLogs.length - 1 - virtualRow.index]; + const reasoningWire = reasoningWireLabel(log); return ( {speedLabel(log)}} - {effortLabel(log)} + + + {effortLabel(log)} + {reasoningWire && {reasoningWire}} + + {log.provider} diff --git a/gui/tests/logs-auto-refresh.test.tsx b/gui/tests/logs-auto-refresh.test.tsx index 5af6a5730a7..c210ba140de 100644 --- a/gui/tests/logs-auto-refresh.test.tsx +++ b/gui/tests/logs-auto-refresh.test.tsx @@ -336,7 +336,7 @@ test("Logs: switching to the Debug tab stops scheduled log requests", async () = test("Logs: attempt details render exact reasoning wire values without legacy placeholders", async () => { const attemptsLog = { ...sampleLog, - requestedEffort: "high", + requestedEffort: "max->high", effectiveEffort: "high", reasoningWireField: "reasoning_effort", reasoningWireValue: "high", @@ -391,6 +391,10 @@ test("Logs: attempt details render exact reasoning wire values without legacy pl const { root, container } = await mountLogs(); await flushMicrotasks(); + const overviewReasoning = container.querySelector(".log-reasoning-cell"); + expect(overviewReasoning?.textContent).toContain("max → high"); + expect(overviewReasoning?.textContent).toContain("reasoning_effort=high"); + expect(overviewReasoning?.textContent).not.toContain("max → high → high"); await act(async () => { container.querySelector(".log-detail-btn")!.click(); }); diff --git a/src/server/responses/core.ts b/src/server/responses/core.ts index 92c84fc8515..06041527130 100644 --- a/src/server/responses/core.ts +++ b/src/server/responses/core.ts @@ -624,6 +624,19 @@ export async function handleComboResponses( if (!combo) { return formatErrorResponse(404, "invalid_request_error", `Unknown combo: ${comboId}`); } + const adoptFailedChildLog = (childLog: RequestLogContext): void => { + // Attempts remain the complete physical history; the logical row mirrors the most recent + // failed target so an exhausted combo still has useful top-level reasoning diagnostics. + Object.assign(logCtx, childLog, { + requestedModel, + model: requestedModel, + provider: "combo", + comboId, + attempts: logCtx.attempts, + activeAttempt: undefined, + activeAttemptStartedAt: undefined, + }); + }; const unreadableEncryptedAgentTask = hasUnreadableEncryptedAgentTask( (rawBody as { input?: unknown } | undefined)?.input, @@ -794,25 +807,19 @@ export async function handleComboResponses( if (comboFailureDecision(failure.response.status, failure.classificationText, { code: failure.upstreamCode, }) === "stop") { - Object.assign(logCtx, childLog, { - requestedModel, - model: requestedModel, - provider: "combo", - comboId, - attempts: logCtx.attempts, - activeAttempt: undefined, - activeAttemptStartedAt: undefined, - }); + adoptFailedChildLog(childLog); return lastFailure; } console.warn( `[combo] ${comboId}: ${targetKey(pick.target)} failed with ${response.status} after ${Date.now() - started}ms`, ); - pick = advanceComboAfterFailure(config, pick, { + const nextPick = advanceComboAfterFailure(config, pick, { retryAfter: failure.retryAfter, now: Date.now(), eligible: payloadEligible, }); + if (!nextPick) adoptFailedChildLog(childLog); + pick = nextPick; } return lastFailure!; } diff --git a/tests/server-combo-failover-e2e.test.ts b/tests/server-combo-failover-e2e.test.ts index 433759c38cd..c80eb82a5c2 100644 --- a/tests/server-combo-failover-e2e.test.ts +++ b/tests/server-combo-failover-e2e.test.ts @@ -15,7 +15,7 @@ import { XAI_OAUTH_DISCOVERY_URL } from "../src/oauth/xai"; import { XAI_GROK_CLI_BASE_URL } from "../src/providers/xai-transport"; import type { AdapterEvent, OcxConfig, OcxProviderConfig } from "../src/types"; import { installIsolatedCodexHome, type IsolatedCodexHome } from "./helpers/isolated-codex-home"; -import { clearRequestLogsForTests, type RequestLogContext } from "../src/server/request-log"; +import { clearRequestLogsForTests, hydrateRequestLogsFromDisk, type RequestLogContext } from "../src/server/request-log"; import { responseWithDeferredRequestLog } from "../src/server/relay"; import { readUsageEntries } from "../src/usage/log"; import { saveCodexAccountCredential } from "../src/codex/account-store"; @@ -416,6 +416,116 @@ describe("server combo failover 030 activation matrix", () => { } }); + test("preserves distinct failed and winning reasoning wires through restart hydration", async () => { + const bodies: Array<{ provider: string; effort?: unknown }> = []; + const a = serve(async request => { + const body = await request.json() as Record; + bodies.push({ provider: "a", effort: body.reasoning_effort }); + return Response.json({ error: { message: "overloaded" } }, { status: 503 }); + }); + const b = serve(async request => { + const body = await request.json() as Record; + bodies.push({ provider: "b", effort: body.reasoning_effort }); + return chatSuccess("mapped backup", "m2"); + }); + const config = comboConfig({ + a: provider("openai-chat", baseUrl(a), "key-a", { + reasoningEfforts: ["low", "high"], + reasoningEffortMap: { max: "low" }, + }), + b: provider("openai-chat", baseUrl(b), "key-b", { + reasoningEfforts: ["low", "high"], + reasoningEffortMap: { max: "high" }, + }), + }); + + const response = await postLogged(config, { reasoning: { effort: "max" } }); + expect(response.status).toBe(200); + expect(await response.text()).toContain("mapped backup"); + expect(bodies).toEqual([ + { provider: "a", effort: "low" }, + { provider: "b", effort: "high" }, + ]); + + const expectMappedReceipt = (receipt: Record) => { + expect(receipt).toMatchObject({ + provider: "combo", + model: "combo/free", + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", + attempts: [ + { + ordinal: 1, + provider: "a", + status: 503, + requestedEffort: "max", + effectiveEffort: "low", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "low", + }, + { + ordinal: 2, + provider: "b", + status: 200, + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", + }, + ], + }); + }; + + const { log, usage } = await latestAttemptReceipts(config); + expectMappedReceipt(log); + expectMappedReceipt(usage); + expect(log).not.toHaveProperty("upstreamError"); + + clearRequestLogsForTests(); + expect(hydrateRequestLogsFromDisk()).toBe(1); + const hydratedResponse = await management(config, "GET", "/api/logs?tail=1"); + const hydrated = await hydratedResponse!.json() as Array>; + expect(hydrated).toHaveLength(1); + expectMappedReceipt(hydrated[0]!); + }); + + test("all-target exhaustion promotes the final attempt reasoning wire to the logical row", async () => { + const a = serve(() => Response.json({ error: { message: "first overloaded" } }, { status: 503 })); + const b = serve(() => Response.json({ error: { message: "last overloaded" } }, { status: 503 })); + const config = comboConfig({ + a: provider("openai-chat", baseUrl(a), "key-a", { + reasoningEfforts: ["low", "high"], + reasoningEffortMap: { max: "low" }, + }), + b: provider("openai-chat", baseUrl(b), "key-b", { + reasoningEfforts: ["low", "high"], + reasoningEffortMap: { max: "high" }, + }), + }); + + const response = await postLogged(config, { reasoning: { effort: "max" } }); + expect(response.status).toBe(503); + await response.text(); + const { log, usage } = await latestAttemptReceipts(config); + + for (const receipt of [log, usage]) { + expect(receipt).toMatchObject({ + provider: "combo", + model: "combo/free", + requestedEffort: "max", + effectiveEffort: "high", + reasoningWireField: "reasoning_effort", + reasoningWireValue: "high", + attempts: [ + { provider: "a", status: 503, effectiveEffort: "low", reasoningWireValue: "low" }, + { provider: "b", status: 503, effectiveEffort: "high", reasoningWireValue: "high" }, + ], + }); + } + }); + test("bare alias runs full failover and preserves structural combo log identity", async () => { const targetBodies: Array<{ provider: string; model?: unknown }> = []; const a = serve(async request => {