From 0b73370f24edeb14ef5b168132af5d0ed0b85cff Mon Sep 17 00:00:00 2001 From: Julius Marminge Date: Mon, 5 Oct 2026 14:43:15 -0700 Subject: [PATCH 1/4] fix(server): metrics count interrupted work on the monotonic clock withMetrics timed work with Clock.currentTimeNanos (the wall clock) and recorded after Effect.exit. When the fiber running the work was interrupted, the interrupt was re-raised at the next step, so nothing was recorded: interrupted RPCs and git commands never showed up with outcome "interrupt". The provider session/turn metrics had no writers at all since the orchestrator-v2 merge deleted the v1 ProviderService that updated them, while the observability docs still told people to watch them. withMetrics now reads Clock.monotonicTimeNanos and records in an Effect.onExit finalizer, so success, failure and interruption are all counted with a duration. RPC stream metrics use the monotonic clock and Effect.onError for the same reason. The provider metrics are wired back in at ProviderSessionManagerV2, the one place every provider's sessions and turns pass through: session open and release, and turn send (timed until the provider accepts it), steer, interrupt and runtime-request responses. The two metrics nothing in v2 can feed (runtime events, reactor events processed) are deleted. The shell and editor-discovery TTL caches, whose comments already called currentTimeNanos monotonic, now use the monotonic clock. Co-Authored-By: Claude Opus 5.5 (1M context) --- apps/server/src/observability/Metrics.test.ts | 59 ++++++++++ apps/server/src/observability/Metrics.ts | 74 ++++-------- .../src/observability/RpcInstrumentation.ts | 22 ++-- .../ProviderSessionManager.test.ts | 105 ++++++++++++++++++ .../ProviderSessionManager.ts | 48 +++++++- apps/server/src/process/externalLauncher.ts | 7 +- packages/shared/src/shell.ts | 4 +- 7 files changed, 242 insertions(+), 77 deletions(-) diff --git a/apps/server/src/observability/Metrics.test.ts b/apps/server/src/observability/Metrics.test.ts index 3fe6b9a8cd88..fa2c82a1a6cd 100644 --- a/apps/server/src/observability/Metrics.test.ts +++ b/apps/server/src/observability/Metrics.test.ts @@ -1,5 +1,6 @@ import { assert, describe, it } from "@effect/vitest"; import { ProviderDriverKind } from "@t3tools/contracts"; +import * as Deferred from "effect/Deferred"; import * as Duration from "effect/Duration"; import * as Effect from "effect/Effect"; import * as Fiber from "effect/Fiber"; @@ -127,6 +128,64 @@ describe("withMetrics", () => { }), ); + it.effect("counts interrupted work with an interrupt outcome and its duration", () => + Effect.gen(function* () { + const counter = Metric.counter("with_metrics_interrupt_total"); + const timer = Metric.timer("with_metrics_interrupt_duration"); + const started = yield* Deferred.make(); + + const fiber = yield* Deferred.succeed(started, undefined).pipe( + Effect.andThen(Effect.never), + withMetrics({ counter, timer, attributes: { operation: "interrupt" } }), + Effect.forkChild, + ); + yield* Deferred.await(started); + yield* TestClock.adjust(Duration.millis(5)); + yield* Fiber.interrupt(fiber); + + const snapshots = yield* Metric.snapshot; + assert.equal( + hasMetricSnapshot(snapshots, "with_metrics_interrupt_total", { + operation: "interrupt", + outcome: "interrupt", + }), + true, + ); + const duration = findHistogramSnapshot(snapshots, "with_metrics_interrupt_duration", { + operation: "interrupt", + }); + assert.equal(duration?.state.count, 1); + assert.equal(duration?.state.sum, 5); + }), + ); + + it.effect("measures durations on the monotonic clock, not the wall clock", () => + Effect.gen(function* () { + const timer = Metric.timer("with_metrics_monotonic_duration"); + const started = yield* Deferred.make(); + const finish = yield* Deferred.make(); + + const fiber = yield* Deferred.succeed(started, undefined).pipe( + Effect.andThen(Deferred.await(finish)), + withMetrics({ timer, attributes: { operation: "monotonic" } }), + Effect.forkChild, + ); + yield* Deferred.await(started); + yield* TestClock.adjust(Duration.millis(10)); + // A backward wall-clock correction must not shorten the measured duration. + yield* TestClock.setTime(0); + yield* Deferred.succeed(finish, undefined); + yield* Fiber.join(fiber); + + const snapshots = yield* Metric.snapshot; + const duration = findHistogramSnapshot(snapshots, "with_metrics_monotonic_duration", { + operation: "monotonic", + }); + assert.equal(duration?.state.count, 1); + assert.equal(duration?.state.sum, 10); + }), + ); + it.effect("records timer durations from nanosecond clock readings", () => Effect.gen(function* () { const duration = Duration.nanos(1_500_000n); diff --git a/apps/server/src/observability/Metrics.ts b/apps/server/src/observability/Metrics.ts index b75ace399ccc..9e3166fc355e 100644 --- a/apps/server/src/observability/Metrics.ts +++ b/apps/server/src/observability/Metrics.ts @@ -5,11 +5,7 @@ import * as Exit from "effect/Exit"; import * as Metric from "effect/Metric"; import { dual } from "effect/Function"; -import { - compactMetricAttributes, - normalizeModelMetricLabel, - outcomeFromExit, -} from "./Attributes.ts"; +import { compactMetricAttributes, outcomeFromExit } from "./Attributes.ts"; export const rpcRequestsTotal = Metric.counter("t3_rpc_requests_total", { description: "Total RPC requests handled by the websocket RPC server.", @@ -19,13 +15,6 @@ export const rpcRequestDuration = Metric.timer("t3_rpc_request_duration", { description: "RPC request handling duration.", }); -const orchestrationEventsProcessedTotal = Metric.counter( - "t3_orchestration_events_processed_total", - { - description: "Total orchestration intent events processed by runtime reactors.", - }, -); - export const orchestrationEffectClaimsTotal = Metric.counter( "t3_orchestration_effect_claims_total", { @@ -38,22 +27,18 @@ export const orchestrationEffectQueueWait = Metric.timer("t3_orchestration_effec "Time from an orchestration effect's temporal availability until claim, including same-thread blocking.", }); -const providerSessionsTotal = Metric.counter("t3_provider_sessions_total", { +export const providerSessionsTotal = Metric.counter("t3_provider_sessions_total", { description: "Total provider session lifecycle operations.", }); -const providerTurnsTotal = Metric.counter("t3_provider_turns_total", { +export const providerTurnsTotal = Metric.counter("t3_provider_turns_total", { description: "Total provider turn lifecycle operations.", }); -const providerTurnDuration = Metric.timer("t3_provider_turn_duration", { +export const providerTurnDuration = Metric.timer("t3_provider_turn_duration", { description: "Provider turn request duration.", }); -const providerRuntimeEventsTotal = Metric.counter("t3_provider_runtime_events_total", { - description: "Total canonical provider runtime events processed.", -}); - export const gitCommandsTotal = Metric.counter("t3_git_commands_total", { description: "Total git commands executed by the server runtime.", }); @@ -91,16 +76,13 @@ export interface WithMetricsOptions { ) => Readonly>; } -const withMetricsImpl = ( - effect: Effect.Effect, +const recordMetrics = ( options: WithMetricsOptions, -): Effect.Effect => + startedAt: bigint, + exit: Exit.Exit, +) => Effect.gen(function* () { - const startedAt = yield* Clock.currentTimeNanos; - const exit = yield* Effect.exit(effect); - const endedAt = yield* Clock.currentTimeNanos; - const elapsedNanos = endedAt > startedAt ? endedAt - startedAt : 0n; - const duration = Duration.nanos(elapsedNanos); + const duration = Duration.nanos((yield* Clock.monotonicTimeNanos) - startedAt); const baseAttributes = typeof options.attributes === "function" ? options.attributes() : (options.attributes ?? {}); @@ -125,35 +107,21 @@ const withMetricsImpl = ( 1, ); } - - if (Exit.isSuccess(exit)) { - return exit.value; - } - return yield* Effect.failCause(exit.cause); }); +// Durations come from the monotonic clock, so wall-clock corrections cannot skew them, and +// metrics are recorded in an exit finalizer, so interrupted work is counted as "interrupt". +const withMetricsImpl = ( + effect: Effect.Effect, + options: WithMetricsOptions, +): Effect.Effect => + Effect.flatMap(Clock.monotonicTimeNanos, (startedAt) => + Effect.onExit(effect, (exit) => recordMetrics(options, startedAt, exit)), + ); + export const withMetrics: { - ( + ( options: WithMetricsOptions, - ): (effect: Effect.Effect) => Effect.Effect; + ): (effect: Effect.Effect) => Effect.Effect; (effect: Effect.Effect, options: WithMetricsOptions): Effect.Effect; } = dual(2, withMetricsImpl); - -const providerMetricAttributes = (provider: string, extra?: Readonly>) => - compactMetricAttributes({ - provider, - ...extra, - }); - -const providerTurnMetricAttributes = (input: { - readonly provider: string; - readonly model: string | null | undefined; - readonly extra?: Readonly>; -}) => { - const modelFamily = normalizeModelMetricLabel(input.model); - return compactMetricAttributes({ - provider: input.provider, - ...(modelFamily ? { modelFamily } : {}), - ...input.extra, - }); -}; diff --git a/apps/server/src/observability/RpcInstrumentation.ts b/apps/server/src/observability/RpcInstrumentation.ts index edbd705b3ee4..1f3d172b5763 100644 --- a/apps/server/src/observability/RpcInstrumentation.ts +++ b/apps/server/src/observability/RpcInstrumentation.ts @@ -67,12 +67,9 @@ const recordRpcStreamMetrics = ( exit: Exit.Exit, ): Effect.Effect => Effect.gen(function* () { - const endedAt = yield* Clock.currentTimeNanos; - const elapsedNanos = endedAt > startedAt ? endedAt - startedAt : 0n; - yield* Metric.update( Metric.withAttributes(rpcRequestDuration, metricAttributes({ method })), - Duration.nanos(elapsedNanos), + Duration.nanos((yield* Clock.monotonicTimeNanos) - startedAt), ); yield* Metric.update( Metric.withAttributes( @@ -111,7 +108,7 @@ export const observeRpcStream = ( ): Stream.Stream => { const instrumented = Stream.unwrap( Effect.gen(function* () { - const startedAt = yield* Clock.currentTimeNanos; + const startedAt = yield* Clock.monotonicTimeNanos; return stream.pipe(Stream.onExit((exit) => recordRpcStreamMetrics(method, startedAt, exit))); }), ); @@ -126,15 +123,12 @@ export const observeRpcStreamEffect = => { const instrumented = Stream.unwrap( Effect.gen(function* () { - const startedAt = yield* Clock.currentTimeNanos; - const exit = yield* Effect.exit(effect); - - if (Exit.isFailure(exit)) { - yield* recordRpcStreamMetrics(method, startedAt, exit); - return yield* Effect.failCause(exit.cause); - } - - return exit.value.pipe( + const startedAt = yield* Clock.monotonicTimeNanos; + // onError also runs when the stream is interrupted before it is produced. + const stream = yield* effect.pipe( + Effect.onError((cause) => recordRpcStreamMetrics(method, startedAt, Exit.failCause(cause))), + ); + return stream.pipe( Stream.onExit((streamExit) => recordRpcStreamMetrics(method, startedAt, streamExit)), ); }), diff --git a/apps/server/src/orchestration-v2/ProviderSessionManager.test.ts b/apps/server/src/orchestration-v2/ProviderSessionManager.test.ts index 527b95f4a3fb..808bd5c30d87 100644 --- a/apps/server/src/orchestration-v2/ProviderSessionManager.test.ts +++ b/apps/server/src/orchestration-v2/ProviderSessionManager.test.ts @@ -24,6 +24,7 @@ import * as Exit from "effect/Exit"; import * as Fiber from "effect/Fiber"; import * as FileSystem from "effect/FileSystem"; import * as Layer from "effect/Layer"; +import * as Metric from "effect/Metric"; import * as Option from "effect/Option"; import * as Queue from "effect/Queue"; import * as Ref from "effect/Ref"; @@ -832,6 +833,110 @@ it.effect("ProviderSessionManagerV2 closes every live session for a provider ins }), ); +it.effect("ProviderSessionManagerV2 records provider session and turn metrics", () => + Effect.gen(function* () { + const state = yield* Ref.make(emptyState); + const effect = Effect.gen(function* () { + const eventSink = yield* EventSink.EventSinkV2; + const idAllocator = yield* IdAllocator.IdAllocatorV2; + const manager = yield* ProviderSessionManager.ProviderSessionManagerV2; + const projectionStore = yield* ProjectionStore.ProjectionStoreV2; + const now = yield* DateTime.now; + const threadId = ThreadId.make("thread-provider-session-manager-metrics"); + const providerSessionId = yield* idAllocator.allocate.providerSession({ + providerInstanceId: modelSelection.instanceId, + threadId, + }); + const providerThread = makeProviderThread({ idAllocator, threadId, providerSessionId, now }); + const runId = idAllocator.derive.run({ threadId, ordinal: 1 }); + yield* eventSink.write({ + events: [yield* makeThreadCreatedEvent({ idAllocator, threadId, now })], + }); + + const runtime = yield* manager.open({ + threadId, + providerSessionId, + modelSelection, + runtimePolicy, + }); + yield* runtime.startTurn({ + appThread: (yield* projectionStore.getThreadProjection(threadId)).thread, + threadId, + runId, + runOrdinal: 1, + providerTurnOrdinal: 1, + attemptId: idAllocator.derive.runAttempt({ runId, attemptOrdinal: 1 }), + rootNodeId: idAllocator.derive.rootNode({ runId }), + providerThread, + message: { + createdBy: "user", + creationSource: "web", + messageId: yield* idAllocator.allocate.message({ threadId, ordinal: 1 }), + text: "hello", + attachments: [], + }, + modelSelection, + runtimePolicy, + }); + yield* runtime.interruptTurn({ + providerThread, + providerTurnId: idAllocator.derive.providerTurn({ + driver: CODEX_DRIVER, + nativeTurnId: "native-turn", + }), + }); + yield* manager.close(providerSessionId); + + const snapshots = yield* Metric.snapshot; + const has = (id: string, attributes: Readonly>) => + snapshots.some( + (snapshot) => + snapshot.id === id && + Object.entries(attributes).every( + ([key, value]) => snapshot.attributes?.[key] === value, + ), + ); + assert.isTrue( + has("t3_provider_sessions_total", { + provider: "codex", + operation: "open", + outcome: "success", + }), + ); + assert.isTrue( + has("t3_provider_sessions_total", { + provider: "codex", + operation: "release", + reason: "manual_shutdown", + outcome: "success", + }), + ); + assert.isTrue( + has("t3_provider_turns_total", { + provider: "codex", + operation: "send", + modelFamily: "gpt", + outcome: "success", + }), + ); + assert.isTrue(has("t3_provider_turn_duration", { provider: "codex", operation: "send" })); + assert.isTrue( + has("t3_provider_turns_total", { + provider: "codex", + operation: "interrupt", + outcome: "success", + }), + ); + }); + + yield* effect.pipe( + Effect.provide(makeTestLayer({ state, idleTimeoutMs: 60_000 })), + // A private registry keeps other tests' provider metrics out of the assertions. + Effect.provideService(Metric.MetricRegistry, new Map()), + ); + }), +); + it.effect("ProviderSessionManagerV2 opens a duplicate session only once", () => Effect.gen(function* () { const state = yield* Ref.make(emptyState); diff --git a/apps/server/src/orchestration-v2/ProviderSessionManager.ts b/apps/server/src/orchestration-v2/ProviderSessionManager.ts index 307bbb708b18..5730432aad87 100644 --- a/apps/server/src/orchestration-v2/ProviderSessionManager.ts +++ b/apps/server/src/orchestration-v2/ProviderSessionManager.ts @@ -30,6 +30,13 @@ import * as Scope from "effect/Scope"; import * as Semaphore from "effect/Semaphore"; import * as Stream from "effect/Stream"; +import { normalizeModelMetricLabel } from "../observability/Attributes.ts"; +import { + providerSessionsTotal, + providerTurnDuration, + providerTurnsTotal, + withMetrics, +} from "../observability/Metrics.ts"; import { ProviderWorkspaceMissingError } from "../provider/Errors.ts"; import * as ProjectService from "../project/ProjectService.ts"; import * as McpProviderSession from "../mcp/McpProviderSession.ts"; @@ -941,7 +948,16 @@ export const layerWithOptions = ( if (Option.isSome(closeExit) && Exit.isFailure(closeExit.value)) { return yield* Effect.failCause(closeExit.value.cause); } - }), + }).pipe( + withMetrics({ + counter: providerSessionsTotal, + attributes: { + provider: entry.runtime.driver, + operation: "release", + reason: input.reason, + }, + }), + ), }), ([entry]) => Option.match(entry, { @@ -1402,6 +1418,18 @@ export const layerWithOptions = ( ): ProviderAdapterV2SessionRuntime => { const providerSessionId = runtime.providerSessionId; const subscribeEvents = makeEventSubscription(eventSubscribers); + // Every provider's turn operations pass through here, so this is where they are + // counted. Only turn starts are timed: until the provider accepts the turn. + const turnMetrics = (operation: string, model?: string) => + withMetrics({ + counter: providerTurnsTotal, + ...(operation === "send" ? { timer: providerTurnDuration } : {}), + attributes: { + provider: runtime.driver, + operation, + modelFamily: normalizeModelMetricLabel(model), + }, + }); return { ...runtime, subscribeEvents, @@ -1509,7 +1537,9 @@ export const layerWithOptions = ( }), ).pipe( Effect.andThen(observeActivity(providerSessionId, markBusy(providerSessionId))), - Effect.andThen(runtime.startTurn(input)), + Effect.andThen( + runtime.startTurn(input).pipe(turnMetrics("send", input.modelSelection.model)), + ), Effect.catch((error) => observeActivity(providerSessionId, markIdle(providerSessionId)).pipe( Effect.andThen(Effect.fail(error)), @@ -1518,15 +1548,19 @@ export const layerWithOptions = ( ), steerTurn: (input) => observeActivity(providerSessionId, touchActivity(providerSessionId)).pipe( - Effect.andThen(runtime.steerTurn(input)), + Effect.andThen(runtime.steerTurn(input).pipe(turnMetrics("steer"))), ), interruptTurn: (input) => observeActivity(providerSessionId, touchActivity(providerSessionId)).pipe( - Effect.andThen(runtime.interruptTurn(input)), + Effect.andThen(runtime.interruptTurn(input).pipe(turnMetrics("interrupt"))), ), respondToRuntimeRequest: (input) => observeActivity(providerSessionId, touchActivity(providerSessionId)).pipe( - Effect.andThen(runtime.respondToRuntimeRequest(input)), + Effect.andThen( + runtime + .respondToRuntimeRequest(input) + .pipe(turnMetrics("runtime-request-response")), + ), ), }; }; @@ -1794,6 +1828,10 @@ export const layerWithOptions = ( cause, }), ), + withMetrics({ + counter: providerSessionsTotal, + attributes: { provider: adapter.driver, operation: "open" }, + }), ); const eventSubscribers = yield* Ref.make< ReadonlyMap> diff --git a/apps/server/src/process/externalLauncher.ts b/apps/server/src/process/externalLauncher.ts index ffd7a6213e2b..6215f8653ffc 100644 --- a/apps/server/src/process/externalLauncher.ts +++ b/apps/server/src/process/externalLauncher.ts @@ -471,7 +471,7 @@ const resolveFileManagerRevealKind = Effect.fn("externalLauncher.resolveFileMana // waiting on, or throw away work a slow host (a busy server at startup, a // long PATH) needs more than one connect to finish. A failed scan clears the // entry so the next caller starts over rather than replaying the failure. -// Expiry uses the monotonic clock (Clock.currentTimeNanos), matching the +// Expiry uses the monotonic clock (Clock.monotonicTimeNanos), matching the // command-resolution cache in @t3tools/shared/shell, so a backward wall-clock // adjustment cannot keep an expired entry alive. const EDITOR_DISCOVERY_CACHE_TTL_NANOS = 60_000_000_000n; @@ -772,7 +772,8 @@ export const make = Effect.gen(function* () { Effect.provideService(ChildProcessSpawner.ChildProcessSpawner, spawner), Effect.onExit((exit) => Effect.gen(function* () { - const expiresAtNanos = (yield* Clock.currentTimeNanos) + EDITOR_DISCOVERY_CACHE_TTL_NANOS; + const expiresAtNanos = + (yield* Clock.monotonicTimeNanos) + EDITOR_DISCOVERY_CACHE_TTL_NANOS; yield* Ref.update(editorDiscoveryCache, (current) => Option.isNone(current) || current.value.scan !== scan ? current @@ -789,7 +790,7 @@ export const make = Effect.gen(function* () { // Claiming the cache entry and starting its scan must not be split by an // interrupt, or the entry would wait on a scan that never runs. const acquireEditorDiscovery = Effect.gen(function* () { - const nowNanos = yield* Clock.currentTimeNanos; + const nowNanos = yield* Clock.monotonicTimeNanos; const [scan, isNewScan] = yield* Ref.modify( editorDiscoveryCache, ( diff --git a/packages/shared/src/shell.ts b/packages/shared/src/shell.ts index 6cb08d2890be..34ac96f4dee8 100644 --- a/packages/shared/src/shell.ts +++ b/packages/shared/src/shell.ts @@ -492,7 +492,7 @@ function resolveCommandCandidates( // just written (e.g. managed binary installs). A "not-found" outcome is also // cached for the TTL, so a just-installed binary can stay invisible for up to // 30s unless resolved by explicit path. -// TTL expiry uses the monotonic clock (Clock.currentTimeNanos) so backward +// TTL expiry uses the monotonic clock (Clock.monotonicTimeNanos) so backward // wall-clock adjustments cannot keep expired entries alive. const COMMAND_RESOLUTION_CACHE_TTL_NANOS = 30_000_000_000n; const COMMAND_RESOLUTION_CACHE_MAX_ENTRIES = 512; @@ -630,7 +630,7 @@ const resolveCommandPathForPlatform = Effect.fn("shell.resolveCommandPathForPlat COMMAND_RESOLUTION_CACHE_KEY_SEPARATOR, ); const cache = yield* CommandResolutionCache; - const nowNanos = yield* Clock.currentTimeNanos; + const nowNanos = yield* Clock.monotonicTimeNanos; const cached = cache.get(cacheKey); if (cached !== undefined && cached.expiresAtNanos > nowNanos) { if (cached.resolvedPath === null) { From c6d338409420997d8a503a0b7504ba230f0eae33 Mon Sep 17 00:00:00 2001 From: Julius Marminge Date: Mon, 5 Oct 2026 14:43:21 -0700 Subject: [PATCH 2/4] fix(server): retry settings and keybindings reads after a failure The settings and keybindings caches were built with Cache.make, whose default time-to-live is infinite for every exit, failures included. One failed read (a transient file or database error) was replayed to every later caller until a file-watcher event or a write invalidated the entry, and a failure during startup stuck until the file was touched. Both caches now use Cache.makeWith with a time-to-live of zero for failed lookups, so a failed read is dropped and the next read goes back to disk. Successful reads keep the existing behavior: cached until a write or a watcher event replaces them. Co-Authored-By: Claude Opus 5.5 (1M context) --- apps/server/src/keybindings.test.ts | 22 ++++++++++++++++++++++ apps/server/src/keybindings.ts | 7 ++++--- apps/server/src/serverSettings.test.ts | 22 ++++++++++++++++++++++ apps/server/src/serverSettings.ts | 12 ++++++++---- 4 files changed, 56 insertions(+), 7 deletions(-) diff --git a/apps/server/src/keybindings.test.ts b/apps/server/src/keybindings.test.ts index 202b5de45c5b..19eef4502dbd 100644 --- a/apps/server/src/keybindings.test.ts +++ b/apps/server/src/keybindings.test.ts @@ -604,6 +604,28 @@ it.layer(NodeServices.layer)("keybindings", (it) => { }).pipe(Effect.provide(makeKeybindingsLayer())), ); + it.effect("retries a failed config read instead of keeping the failure", () => + Effect.gen(function* () { + const fileSystem = yield* FileSystem.FileSystem; + const { keybindingsConfigPath } = yield* ServerConfig.ServerConfig; + // A directory where the file should be makes the read itself fail. + yield* fileSystem.makeDirectory(keybindingsConfigPath, { recursive: true }); + + const keybindings = yield* Keybindings.Keybindings; + const failed = yield* toDetailResult(keybindings.loadConfigState); + assertFailure(failed, "failed to read keybindings config"); + + yield* fileSystem.remove(keybindingsConfigPath, { recursive: true }); + yield* writeKeybindingsConfig(keybindingsConfigPath, [ + { key: "mod+j", command: "terminal.toggle" }, + ]); + + const configState = yield* keybindings.loadConfigState; + assert.deepEqual(configState.issues, []); + assert.isTrue(configState.keybindings.some((entry) => entry.command === "terminal.toggle")); + }).pipe(Effect.provide(makeKeybindingsLayer())), + ); + it.effect("updates cached resolved config after upsert", () => Effect.gen(function* () { const { keybindingsConfigPath } = yield* ServerConfig.ServerConfig; diff --git a/apps/server/src/keybindings.ts b/apps/server/src/keybindings.ts index 0484296edd7e..3697b62cb403 100644 --- a/apps/server/src/keybindings.ts +++ b/apps/server/src/keybindings.ts @@ -480,13 +480,14 @@ const make = Effect.gen(function* () { })), ); - const resolvedConfigCache = yield* Cache.make< + // A failed read is not kept: the next read retries instead of replaying the failure. + const resolvedConfigCache = yield* Cache.makeWith< typeof resolvedConfigCacheKey, KeybindingsConfigState, KeybindingsConfigError - >({ + >(() => loadConfigStateFromDisk, { capacity: 1, - lookup: () => loadConfigStateFromDisk, + timeToLive: (exit) => (Exit.isSuccess(exit) ? Duration.infinity : Duration.zero), }); const loadConfigStateFromCacheOrDisk = Cache.get(resolvedConfigCache, resolvedConfigCacheKey); diff --git a/apps/server/src/serverSettings.test.ts b/apps/server/src/serverSettings.test.ts index 35df600e76a6..03622c9bd9db 100644 --- a/apps/server/src/serverSettings.test.ts +++ b/apps/server/src/serverSettings.test.ts @@ -303,6 +303,28 @@ it.layer(NodeServices.layer)("server settings", (it) => { }).pipe(Effect.provide(makeServerSettingsLayer())), ); + it.effect("retries a failed settings read instead of keeping the failure", () => + Effect.gen(function* () { + const serverConfig = yield* ServerConfig.ServerConfig; + const fileSystem = yield* FileSystem.FileSystem; + const serverSettings = yield* ServerSettingsModule.ServerSettingsService; + // A directory where the file should be makes the read itself fail. + yield* fileSystem.makeDirectory(serverConfig.settingsPath); + + const error = yield* Effect.flip(serverSettings.getSettings); + assert.deepInclude(error, { _tag: "ServerSettingsError", operation: "read-file" }); + + yield* fileSystem.remove(serverConfig.settingsPath, { recursive: true }); + yield* fileSystem.writeFileString( + serverConfig.settingsPath, + `{ "responseStreamingMode": "turn" }`, + ); + + const settings = yield* serverSettings.getSettings; + assert.equal(settings.responseStreamingMode, "turn"); + }).pipe(Effect.provide(makeServerSettingsLayer())), + ); + it.effect("decodes nested settings patches", () => Effect.gen(function* () { assert.deepEqual( diff --git a/apps/server/src/serverSettings.ts b/apps/server/src/serverSettings.ts index 04530f9f790a..6452c8d775a2 100644 --- a/apps/server/src/serverSettings.ts +++ b/apps/server/src/serverSettings.ts @@ -802,10 +802,14 @@ const make = Effect.gen(function* () { return migrated; }); - const settingsCache = yield* Cache.make({ - capacity: 1, - lookup: () => loadSettingsFromDisk, - }); + // A failed read is not kept: the next read retries instead of replaying the failure. + const settingsCache = yield* Cache.makeWith( + () => loadSettingsFromDisk, + { + capacity: 1, + timeToLive: (exit) => (Exit.isSuccess(exit) ? Duration.infinity : Duration.zero), + }, + ); const getSettingsFromCache = Cache.get(settingsCache, cacheKey); From 3cf6001b932e817a314fd330aefaee7fdae59de0 Mon Sep 17 00:00:00 2001 From: Julius Marminge Date: Mon, 5 Oct 2026 17:05:17 -0700 Subject: [PATCH 3/4] docs(server): say the provider turn timer measures turn start Co-Authored-By: Claude Opus 5.5 (1M context) --- apps/server/src/observability/Metrics.ts | 2 +- docs/operations/observability.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/apps/server/src/observability/Metrics.ts b/apps/server/src/observability/Metrics.ts index 9e3166fc355e..04a8cf98e14a 100644 --- a/apps/server/src/observability/Metrics.ts +++ b/apps/server/src/observability/Metrics.ts @@ -36,7 +36,7 @@ export const providerTurnsTotal = Metric.counter("t3_provider_turns_total", { }); export const providerTurnDuration = Metric.timer("t3_provider_turn_duration", { - description: "Provider turn request duration.", + description: "Time for a provider to accept a new turn, not how long the turn runs.", }); export const gitCommandsTotal = Metric.counter("t3_git_commands_total", { diff --git a/docs/operations/observability.md b/docs/operations/observability.md index b83a9015abb2..9987a36ad80d 100644 --- a/docs/operations/observability.md +++ b/docs/operations/observability.md @@ -377,7 +377,7 @@ Traces are best for one request. Metrics are best for trends. Good metric families to watch: - `t3_rpc_request_duration` -- `t3_provider_turn_duration` +- `t3_provider_turn_duration` (how long a provider takes to accept a new turn, not the turn's run time) - `t3_git_command_duration` Counters tell you volume and failure rate: From d0379d1e1d3aeba99db90e32118df3eacdb8ab74 Mon Sep 17 00:00:00 2001 From: Julius Marminge Date: Mon, 5 Oct 2026 17:23:34 -0700 Subject: [PATCH 4/4] docs(server): the turn timer covers the adapter's start, not provider acceptance Co-Authored-By: Claude Opus 5.5 (1M context) --- apps/server/src/observability/Metrics.ts | 2 +- docs/operations/observability.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/apps/server/src/observability/Metrics.ts b/apps/server/src/observability/Metrics.ts index 04a8cf98e14a..a01c7bb75758 100644 --- a/apps/server/src/observability/Metrics.ts +++ b/apps/server/src/observability/Metrics.ts @@ -36,7 +36,7 @@ export const providerTurnsTotal = Metric.counter("t3_provider_turns_total", { }); export const providerTurnDuration = Metric.timer("t3_provider_turn_duration", { - description: "Time for a provider to accept a new turn, not how long the turn runs.", + description: "Time for the provider adapter to start a turn, not how long the turn runs.", }); export const gitCommandsTotal = Metric.counter("t3_git_commands_total", { diff --git a/docs/operations/observability.md b/docs/operations/observability.md index 9987a36ad80d..5cb393e07f39 100644 --- a/docs/operations/observability.md +++ b/docs/operations/observability.md @@ -377,7 +377,7 @@ Traces are best for one request. Metrics are best for trends. Good metric families to watch: - `t3_rpc_request_duration` -- `t3_provider_turn_duration` (how long a provider takes to accept a new turn, not the turn's run time) +- `t3_provider_turn_duration` (how long the provider adapter takes to start a turn, not the turn's run time) - `t3_git_command_duration` Counters tell you volume and failure rate: