From 790aef5eebb8579e0e40bf45aee4fe7e633d7136 Mon Sep 17 00:00:00 2001 From: "Jakub A. W" Date: Tue, 22 Sep 2026 12:10:53 +0200 Subject: [PATCH 1/3] feat(jev): add the Jev / Kev System One provider --- .env.template | 12 +- README.md | 1 + config/config.example.yaml | 18 ++ config/config.go | 1 + config/config_test.go | 2 +- config/server.go | 2 +- docs/docs.json | 1 + docs/features/passthrough-api.mdx | 4 +- docs/providers/jev.mdx | 173 +++++++++++++++ docs/providers/overview.mdx | 7 + internal/providers/jev/jev.go | 153 ++++++++++++++ internal/providers/jev/jev_test.go | 197 ++++++++++++++++++ internal/providers/jev/models.go | 104 +++++++++ internal/providers/jev/models_test.go | 103 +++++++++ .../providers/jev/newtestprovider_test.go | 16 ++ .../providers/jev/passthrough_semantics.go | 23 ++ internal/server/handlers_test.go | 2 +- internal/server/passthrough_support.go | 2 +- internal/server/passthrough_support_test.go | 17 +- run/providers.go | 2 + run/providers_test.go | 2 +- 21 files changed, 826 insertions(+), 16 deletions(-) create mode 100644 docs/providers/jev.mdx create mode 100644 internal/providers/jev/jev.go create mode 100644 internal/providers/jev/jev_test.go create mode 100644 internal/providers/jev/models.go create mode 100644 internal/providers/jev/models_test.go create mode 100644 internal/providers/jev/newtestprovider_test.go create mode 100644 internal/providers/jev/passthrough_semantics.go diff --git a/.env.template b/.env.template index f333266e9..47e5f87bb 100644 --- a/.env.template +++ b/.env.template @@ -106,11 +106,11 @@ # Allow optional /p/{provider}/v1/... passthrough aliases while keeping /p/{provider}/... canonical (default: true) # ALLOW_PASSTHROUGH_V1_ALIAS=true -# Comma-separated list of provider types enabled for /p/{provider}/... passthrough (default: openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek) +# Comma-separated list of provider types enabled for /p/{provider}/... passthrough (default: openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek,jev) # Cohere and audio.cpp native passthrough are opt-in; add cohere or audiocpp when # those routes are needed. audio.cpp's native surface includes model management # and server-local file paths, so enable it only for trusted callers. -# ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,cohere,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,audiocpp,deepseek,hetzner +# ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,cohere,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,audiocpp,deepseek,hetzner,jev # Enable the realtime (speech-to-speech) endpoints (default: true): the /v1/realtime # websocket (and /p/{provider}/v1/realtime passthrough upgrade), the WebRTC SDP @@ -700,6 +700,14 @@ # ELEVENLABS_API_KEY=... # ELEVENLABS_BASE_URL=https://api.elevenlabs.io +# Jev (TypeSafe System One decision API; default base URL: https://api.typesafe.ai) +# The API is not OpenAI-compatible: send System One requests to +# POST /p/jev/v1/systemone (or point the TypeSafe SDK at http://localhost:8080/p/jev). +# JEV_API_KEY=... +# A self-hosted Kev server (github.com/jaredpalmer/kev) speaks the same API and +# has no authentication of its own: set the base URL and leave the key unset. +# JEV_BASE_URL=http://localhost:8009 + # Xiaomi MiMo (default base URL: https://api.xiaomimimo.com/v1) # XIAOMI_API_KEY=... # XIAOMI_BASE_URL=https://api.xiaomimimo.com/v1 diff --git a/README.md b/README.md index 73c27a459..1e9cafc0a 100644 --- a/README.md +++ b/README.md @@ -132,6 +132,7 @@ The official SDKs therefore work unchanged. Configure their base URLs as follows - Amazon Bedrock Runtime and Bedrock Mantle - ChatGPT (the Codex backend) and Claude - ElevenLabs (text-to-speech and speech-to-text) +- Jev (TypeSafe System One decision API) and self-hosted Kev - All OpenAI-compatible providers See the [Providers Overview](https://gomodel.enterpilot.io/docs/providers/overview?utm_source=readme) for the full diff --git a/config/config.example.yaml b/config/config.example.yaml index 3235e4824..77a81a62a 100644 --- a/config/config.example.yaml +++ b/config/config.example.yaml @@ -570,6 +570,24 @@ providers: # gateway's own) and a trailing /v1 is accepted. api_key is optional and # usually unset, since audio.cpp has no authentication of its own. + jev: + type: jev + api_key: "${JEV_API_KEY}" + # base_url defaults to "https://api.typesafe.ai". TypeSafe's System One + # API is a decision API with no OpenAI-compatible surface: requests go to + # POST /p/jev/v1/systemone, or point the TypeSafe SDK at /p/jev. A + # self-hosted Kev server speaks the same API without authentication: + # set base_url (e.g. "http://localhost:8009") and omit api_key. + # Jev is priced per input token and is not in the upstream model catalog; + # declare its pricing here to have the gateway cost System One requests. + # models: + # - id: "jev-1.13.0" + # metadata: + # pricing: + # currency: USD + # input_per_mtok: 0.042 + # output_per_mtok: 0 + meta: type: meta api_key: "..." diff --git a/config/config.go b/config/config.go index d1b0955b2..41fe80def 100644 --- a/config/config.go +++ b/config/config.go @@ -127,6 +127,7 @@ func buildDefaultConfig() *Config { "llamacpp", "llmd", "deepseek", + "jev", }, }, Models: ModelsConfig{ diff --git a/config/config_test.go b/config/config_test.go index 7e76e9762..5063048ec 100644 --- a/config/config_test.go +++ b/config/config_test.go @@ -132,7 +132,7 @@ func TestBuildDefaultConfig(t *testing.T) { assert.Equal(t, DefaultStreamStallTimeoutSeconds, cfg.Server.StreamStallTimeout) assert.True(t, cfg.Server.EnablePassthroughRoutes) assert.True(t, cfg.Server.AllowPassthroughV1Alias) - assert.Equal(t, []string{"openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek"}, cfg.Server.EnabledPassthroughProviders) + assert.Equal(t, []string{"openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek", "jev"}, cfg.Server.EnabledPassthroughProviders) assert.Equal(t, ConfiguredProviderModelsModeFallback, cfg.Models.ConfiguredProviderModelsMode) assert.Nil(t, cfg.Cache.Model.Local) assert.Equal(t, 3600, cfg.Cache.Model.RefreshInterval) diff --git a/config/server.go b/config/server.go index 917412a25..45a0e8a06 100644 --- a/config/server.go +++ b/config/server.go @@ -45,7 +45,7 @@ type ServerConfig struct { UserPathHeader string `yaml:"user_path_header" env:"USER_PATH_HEADER"` // EnabledPassthroughProviders lists the provider types enabled on // /p/{provider}/... passthrough routes. Default: - // ["openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llmd", "deepseek"]. + // ["openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek", "jev"]. EnabledPassthroughProviders []string `yaml:"enabled_passthrough_providers" env:"ENABLED_PASSTHROUGH_PROVIDERS"` // RealtimeEnabled exposes the realtime (speech-to-speech) websocket endpoints // at /v1/realtime and /v1/realtime/translations, their WebRTC signaling diff --git a/docs/docs.json b/docs/docs.json index fb97165e5..252cfd651 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -214,6 +214,7 @@ "providers/minimax", "providers/elevenlabs", "providers/audiocpp", + "providers/jev", "providers/opencode-go", "providers/sglang", "providers/vllm", diff --git a/docs/features/passthrough-api.mdx b/docs/features/passthrough-api.mdx index 5a432887f..0c8853307 100644 --- a/docs/features/passthrough-api.mdx +++ b/docs/features/passthrough-api.mdx @@ -136,7 +136,7 @@ from passthrough requests before forwarding them upstream. Passthrough is intentionally narrow while the API is in beta. -- `openai`, `anthropic`, `openrouter`, `kilo`, `zai`, `sglang`, `vllm`, `llamacpp`, `llmd`, and `deepseek` +- `openai`, `anthropic`, `openrouter`, `kilo`, `zai`, `sglang`, `vllm`, `llamacpp`, `llmd`, `deepseek`, and `jev` are enabled by default. - Chutes supports passthrough but requires explicit operator opt-in because passthrough can forward provider-native routes that do not identify a model. @@ -155,7 +155,7 @@ Passthrough routes are enabled by default: ```env ENABLE_PASSTHROUGH_ROUTES=true ALLOW_PASSTHROUGH_V1_ALIAS=true -ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek +ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek,jev ``` Set `ENABLED_PASSTHROUGH_PROVIDERS` to the provider types you want to expose. diff --git a/docs/providers/jev.mdx b/docs/providers/jev.mdx new file mode 100644 index 000000000..c2c2b16c0 --- /dev/null +++ b/docs/providers/jev.mdx @@ -0,0 +1,173 @@ +--- +title: "Jev / Kev (TypeSafe System One)" +sidebarTitle: "Jev / Kev" +description: "Route TypeSafe System One decision requests through GoModel, to the hosted Jev API or a self-hosted Kev server." +icon: "scale-balanced" +keywords: ["Jev", "Kev", "TypeSafe", "System One", "decision model", "classification", "noul", "choice", "score", "self-hosted"] +--- + +[Jev](https://docs.typesafe.ai/introduction) is TypeSafe's System One model: a +decision model rather than a text generator. A request carries a `state` (the +text or record to evaluate) and a map of typed questions, and the answer is a +calibrated probability per question. [Kev](https://github.com/jaredpalmer/kev) +is a family of small open-weight models that implement the same API, so one +`jev` provider type covers both. + +There are three question types: + +| Type | Asks | Answer | +| --- | --- | --- | +| `noul` | A yes/no question | `noul`: the probability of yes | +| `choice` | Pick one option from a set you define | `choice`, plus `probabilities` and `confidence` | +| `score` | Rate against ordered levels | `score`, plus `legend`, `probabilities` and `confidence` | + +The API is not OpenAI-compatible, and its answers have no chat equivalent, so +GoModel does not translate it: System One requests go through +[passthrough](/features/passthrough-api) at `/p/jev/...`, which is enabled by +default for this provider. Chat, `/responses`, and `/v1/embeddings` return +`invalid_request_error` for `jev` models. + +## Configure + +For the hosted API, the key is the whole setup: + +```bash +JEV_API_KEY=ts-... +GOMODEL_MASTER_KEY=change-me +``` + +For a self-hosted Kev server, set the base URL instead. Kev has no +authentication of its own, so leave the key unset: + +```bash +JEV_BASE_URL=http://host.docker.internal:8009 +GOMODEL_MASTER_KEY=change-me +``` + + + The default base URL is `https://api.typesafe.ai`, the origin TypeSafe's SDKs + use; a trailing `/v1` is accepted and trimmed, so both spellings address the + same server. To run the hosted API and a local Kev side by side, register the + second under a suffixed name: `JEV_KEV_BASE_URL=...` creates provider + `jev-kev`, reached at `/p/jev-kev/...`. + + +## Verify + +```bash +curl -s http://localhost:8080/p/jev/v1/systemone \ + -H "Authorization: Bearer change-me" \ + -H "Content-Type: application/json" \ + -d '{ + "state": "Shoes arrived two weeks late and in the wrong size. Also I see two charges on my card.", + "model": "jev-latest", + "questions": { + "department": {"type": "choice", "instructions": "Which team should handle this?", + "criteria": {"returns": "Exchanges, refunds, wrong or damaged items", + "shipping": "Delivery status, delays, lost packages", + "billing": "Charges, invoices, payment problems"}}, + "escalate": {"type": "noul", "instructions": "Does this need urgent human attention?"}, + "frustration": {"type": "score", "instructions": "How frustrated is the customer?", + "criteria": ["Calm", "Frustrated", "Very angry"]} + } + }' +``` + +```json +{ + "model": "jev-1.13.0", + "answers": { + "department": {"type": "choice", "choice": "returns", "confidence": 0.21, + "probabilities": {"returns": 0.47, "shipping": 0.28, "billing": 0.25}}, + "escalate": {"type": "noul", "noul": 0.93}, + "frustration": {"type": "score", "score": 1.44, "confidence": 0.78, + "legend": {"0": "Calm", "1": "Frustrated", "2": "Very angry"}, + "probabilities": {"0": 0.00, "1": 0.56, "2": 0.44}} + }, + "usage": {"input_tokens": 101, "output_tokens": 161} +} +``` + +The `/v1` segment is optional: `/p/jev/systemone` is the same route. Use +`kev-latest` as the model on a Kev server; it also answers to `jev-latest`. + +## Using the TypeSafe SDKs + +The SDKs send `POST {base_url}/v1/systemone`, so point them at the provider's +passthrough root and authenticate with your GoModel key: + + +```python Python +from typesafe_sdk import Noul, TypeSafeClient + +client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080/p/jev") +response = client.system_one( + state="I was charged twice. Please fix this ASAP.", + questions={"billing": Noul(instructions="Is this ticket about billing?")}, +) +print(response.nouls["billing"].noul) +``` + +```typescript JavaScript +import { TypeSafeClient, noul } from "@typesafe-ai/sdk"; + +const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080/p/jev" }); +const result = await client.systemOne({ + state: "I was charged twice. Please fix this ASAP.", + questions: { billing: noul({ instructions: "Is this ticket about billing?" }) }, +}); +console.log(result.answers.billing.noul); +``` + + +The same works with `TYPESAFE_BASE_URL=http://localhost:8080/p/jev` and +`TYPESAFE_API_KEY=change-me` in the environment. + +## Native routes + +| Route | What it does | +| --- | --- | +| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions | +| `GET /p/jev/v1/models` | The names the `model` field accepts, in the upstream's own shape | +| `POST /p/jev/v1/systemone/permute` | Kev only: run one Choice question with several option orders | +| `POST /p/jev/v1/systemone/separate` | Kev only: run each question in its own forward pass | + +Upstream errors keep their status code, with the provider's body carried in +the gateway error message: a malformed question comes back as TypeSafe's `422` +naming the offending field, and `429` or `529` mean back off and retry. + +## Models, access control, and cost + +`GET /v1/models` lists what the upstream reports, as `jev/jev-latest` and so +on. TypeSafe lists its aliases (`jev-latest`, `jev-preview`); a Kev server +lists its checkpoint (`kev-latest`) and the aliases it answers to. Versioned +IDs such as `jev-1.13.0` are accepted by the `model` field whether or not they +are listed. The models are categorized as utility models with no generation +mode, since there is no OpenAI endpoint to route them to. + +Every System One request names its model, so the passthrough surface applies +the caller's [model allowlist](/features/users) to it like any other +request. + +The response's `usage.input_tokens` and `usage.output_tokens` are recorded, so +System One calls appear in the usage API and dashboard under the model that +answered (`jev-1.13.0`, or the Kev checkpoint). Jev is priced per input token +and is not in the upstream model catalog; declare its pricing on the provider +to have those rows costed, or set it in the +[pricing override editor](/features/cost-tracking): + +```yaml +providers: + jev: + type: jev + api_key: "${JEV_API_KEY}" + models: + - id: "jev-1.13.0" + metadata: + pricing: + currency: USD + input_per_mtok: 0.042 + output_per_mtok: 0 +``` + +A local Kev server costs nothing per token, so it needs no pricing. diff --git a/docs/providers/overview.mdx b/docs/providers/overview.mdx index e559c230a..7b7bf43f0 100644 --- a/docs/providers/overview.mdx +++ b/docs/providers/overview.mdx @@ -61,6 +61,7 @@ support, not every individual model capability exposed by an upstream provider. | Xiaomi MiMo | `XIAOMI_API_KEY` (`XIAOMI_BASE_URL` optional) | `mimo-v2.5-pro` | ✅ | ✅ | ❌ | ❌ | ❌ | ✅ | [Xiaomi MiMo](/providers/xiaomi) | | ElevenLabs (voice only) | `ELEVENLABS_API_KEY` (`ELEVENLABS_BASE_URL` optional) | `eleven_multilingual_v2` | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | [ElevenLabs](/providers/elevenlabs) | | audio.cpp (audio only) | `AUDIOCPP_BASE_URL` (`AUDIOCPP_API_KEY` optional) | `pocket-tts` | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | [audio.cpp](/providers/audiocpp) | +| Jev / Kev (System One only) | `JEV_API_KEY` (`JEV_BASE_URL` optional) | `jev-latest` | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | [Jev](/providers/jev) | | OpenCode Go | `OPENCODE_GO_API_KEY` (`OPENCODE_GO_BASE_URL` optional) | `glm-5.1` | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | [OpenCode Go](/providers/opencode-go) | | Kimi Code | `KIMICODE_API_KEY` | `kimi-for-coding` | ✅ | ✅ | ✅ | ❌ | ❌ | ✅ | [Kimi Code](/providers/kimicode) | | Hetzner (experimental) | `HETZNER_API_KEY` (`HETZNER_BASE_URL` optional) | `Qwen/Qwen3.6-35B-A3B-FP8` | ✅ | ✅ | ❌ | ❌ | ❌ | ✅ | [Hetzner](/providers/hetzner) | @@ -206,6 +207,12 @@ support, not every individual model capability exposed by an upstream provider. its own. Audio only: `/v1/audio/speech` and `/v1/audio/transcriptions` route to it, and its detail, alignment, and live-streaming routes are reachable through passthrough. See [audio.cpp](/providers/audiocpp). +- **Jev / Kev** — TypeSafe's System One API is a decision API (state plus + typed questions in, calibrated probabilities out) with no OpenAI-compatible + surface, so it is reached only through passthrough at + `POST /p/jev/v1/systemone`. `JEV_API_KEY` configures the hosted API; for a + self-hosted Kev server, which speaks the same API without authentication, + set `JEV_BASE_URL` and leave the key unset. See [Jev](/providers/jev). - **llama.cpp / LM Studio** — `LLAMACPP_BASE_URL` is required (llama-server's default port collides with GoModel's own 8080, so there is no default); `LLAMACPP_API_KEY` is optional. Do not register these servers as `ollama`, diff --git a/internal/providers/jev/jev.go b/internal/providers/jev/jev.go new file mode 100644 index 000000000..32b3ea576 --- /dev/null +++ b/internal/providers/jev/jev.go @@ -0,0 +1,153 @@ +// Package jev provides TypeSafe's Jev (System One) API integration for the +// gateway, and covers the self-hosted Kev servers that implement the same +// API. System One is a decision API rather than a text-generation one: a +// request carries a state and a map of typed questions (noul, choice, score) +// and the answer is a calibrated probability per question. It has no +// OpenAI-compatible surface, so the gateway reaches it through native +// passthrough at /p/jev/systemone. +package jev + +import ( + "context" + "io" + "net/http" + "strings" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/llmclient" + "github.com/enterpilot/gomodel/internal/providers" +) + +const defaultBaseURL = "https://api.typesafe.ai" + +// Registration provides factory registration for the Jev provider. The hosted +// TypeSafe API needs a key; a local Kev server has no authentication of its +// own, so a base URL alone configures the provider. +var Registration = providers.Registration{ + Type: "jev", + New: New, + PassthroughSemanticEnricher: passthroughSemanticEnricher, + Discovery: providers.DiscoveryConfig{ + DefaultBaseURL: defaultBaseURL, + AllowAPIKeyless: true, + }, +} + +// Provider implements the model listing and native passthrough against one +// System One server. Chat, Responses, and embeddings return +// invalid_request_error because the API has no such endpoints. +type Provider struct { + client *llmclient.Client + keys *providers.Keyring +} + +var ( + _ core.Provider = (*Provider)(nil) + _ core.PassthroughProvider = (*Provider)(nil) +) + +// New creates a Jev provider. The client is rooted at the API origin, which +// is how TypeSafe's SDKs are configured (https://api.typesafe.ai, with /v1 +// added per request); a base URL written with the /v1 suffix every other +// provider here uses is accepted and trimmed, so both spellings address the +// same server. +func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Provider { + p := &Provider{keys: opts.Keyring(cfg.APIKey)} + clientCfg := llmclient.Config{ + ProviderName: opts.ClientName("jev"), + BaseURL: baseURL(cfg.BaseURL), + Retry: opts.Resilience.Retry, + Hooks: opts.Hooks, + CircuitBreaker: opts.Resilience.CircuitBreaker, + } + p.client = llmclient.NewWithOptionalHTTPClient(opts.HTTPClient, clientCfg, p.setHeaders) + return p +} + +// SetBaseURL allows configuring a custom base URL for the provider. +func (p *Provider) SetBaseURL(url string) { + p.client.SetBaseURL(baseURL(url)) +} + +func baseURL(configured string) string { + return providers.PassthroughBaseURL(providers.ResolveBaseURL(configured, defaultBaseURL)) +} + +// setHeaders resolves the credential per request rather than capturing it, so +// several configured keys rotate across calls. A keyless local Kev server +// gets no Authorization header at all. +func (p *Provider) setHeaders(req *http.Request) { + providers.SetAuthHeaders(req, p.keys.NextForContext(req.Context()), providers.AuthHeaderConfig{ + AuthScheme: "Bearer ", + RequestIDHeader: "X-Request-Id", + OptionalAPIKey: true, + }) +} + +// ChatCompletion reports that System One has no chat API. +func (p *Provider) ChatCompletion(_ context.Context, _ *core.ChatRequest) (*core.ChatResponse, error) { + return nil, unsupported("chat completions") +} + +// StreamChatCompletion reports that System One has no chat API. +func (p *Provider) StreamChatCompletion(_ context.Context, _ *core.ChatRequest) (io.ReadCloser, error) { + return nil, unsupported("chat completions") +} + +// Responses reports that System One has no Responses API. +func (p *Provider) Responses(_ context.Context, _ *core.ResponsesRequest) (*core.ResponsesResponse, error) { + return nil, unsupported("the responses API") +} + +// StreamResponses reports that System One has no Responses API. +func (p *Provider) StreamResponses(_ context.Context, _ *core.ResponsesRequest) (io.ReadCloser, error) { + return nil, unsupported("the responses API") +} + +// Embeddings reports that System One has no embeddings API. +func (p *Provider) Embeddings(_ context.Context, _ *core.EmbeddingRequest) (*core.EmbeddingResponse, error) { + return nil, unsupported("embeddings") +} + +func unsupported(surface string) error { + return core.NewInvalidRequestError("jev does not support "+surface+"; send System One requests to /p/jev/systemone", nil) +} + +// Passthrough forwards a System One request as the client wrote it. It is the +// only way to reach the evaluation endpoint, since the request and answer +// shapes have no OpenAI equivalent. +func (p *Provider) Passthrough(ctx context.Context, req *core.PassthroughRequest) (*core.PassthroughResponse, error) { + if req == nil { + return nil, core.NewInvalidRequestError("passthrough request is required", nil) + } + resp, err := p.client.DoPassthrough(ctx, llmclient.Request{ + Method: req.Method, + Endpoint: passthroughPath(req.Endpoint), + Operation: req.Operation, + Model: req.Model, + Stream: req.Stream, + StreamUncertain: req.StreamUncertain, + RawBodyReader: req.Body, + Headers: req.Headers, + }) + if err != nil { + return nil, err + } + return &core.PassthroughResponse{ + StatusCode: resp.StatusCode, + Headers: providers.CloneHTTPHeaders(resp.Header), + Body: resp.Body, + }, nil +} + +// passthroughPath resolves the server path a passthrough endpoint addresses. +// Every System One route lives under /v1, and the gateway strips the optional +// v1 alias before a provider sees the endpoint, so the prefix is restored +// unless the endpoint already carries it. +func passthroughPath(endpoint string) string { + path := providers.PassthroughEndpoint(endpoint) + if path == "/v1" || strings.HasPrefix(path, "/v1/") || strings.HasPrefix(path, "/v1?") { + return path + } + return "/v1" + path +} diff --git a/internal/providers/jev/jev_test.go b/internal/providers/jev/jev_test.go new file mode 100644 index 000000000..43c603028 --- /dev/null +++ b/internal/providers/jev/jev_test.go @@ -0,0 +1,197 @@ +package jev + +import ( + "context" + "io" + "net/http" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/llmclient" + "github.com/enterpilot/gomodel/internal/providers" + "github.com/enterpilot/gomodel/internal/providers/providertest" +) + +var ( + _ core.Provider = (*Provider)(nil) + _ core.PassthroughProvider = (*Provider)(nil) +) + +// The hosted API has a default origin and needs a key; a local Kev server has +// no authentication, so a base URL alone must configure the provider. +func TestRegistration_DescribesHostedAndKeylessServers(t *testing.T) { + assert.Equal(t, "jev", Registration.Type) + require.NotNil(t, Registration.New) + assert.Equal(t, "https://api.typesafe.ai", Registration.Discovery.DefaultBaseURL) + assert.True(t, Registration.Discovery.AllowAPIKeyless) + assert.False(t, Registration.Discovery.RequireBaseURL) + + // Zero options are what a keyless provider is built with outside the + // factory: no keyring, no resilience settings, and no test transport. + provider := Registration.New(providers.ProviderConfig{BaseURL: "http://localhost:8009"}, providers.ProviderOptions{}) + assert.NotNil(t, provider) +} + +// The inference surfaces System One does not implement must fail as typed +// invalid-request errors that point at the native route, rather than 404s +// from an upstream that never had those endpoints. +func TestUnsupportedCapabilities_ReturnInvalidRequestErrors(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, `{}`) + provider := newTestProvider("", server.URL, server.Client(), llmclient.Hooks{}) + + _, err := provider.ChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) + providertest.AssertUnsupported(t, err) + assert.Contains(t, err.Error(), "/p/jev/systemone") + _, err = provider.StreamChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) + providertest.AssertUnsupported(t, err) + _, err = provider.Responses(context.Background(), &core.ResponsesRequest{Model: "jev-latest"}) + providertest.AssertUnsupported(t, err) + _, err = provider.StreamResponses(context.Background(), &core.ResponsesRequest{Model: "jev-latest"}) + providertest.AssertUnsupported(t, err) + _, err = provider.Embeddings(context.Background(), &core.EmbeddingRequest{Model: "jev-latest"}) + providertest.AssertUnsupported(t, err) + + assert.Zero(t, capture.Count(), "unsupported surfaces must not reach the upstream") +} + +// TypeSafe's SDKs are configured with the origin and add /v1 per request; +// every other provider here is configured with the /v1 suffix. Both address +// the same server, so a configured suffix is trimmed rather than doubled. +func TestBaseURL_AcceptsOriginAndV1Suffix(t *testing.T) { + for _, suffix := range []string{"", "/", "/v1", "/v1/"} { + t.Run("suffix "+suffix, func(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, `{"models":[]}`) + provider := newTestProvider("", server.URL+suffix, server.Client(), llmclient.Hooks{}) + + _, err := provider.ListModels(context.Background()) + require.NoError(t, err) + assert.Equal(t, "/v1/models", capture.Last(t).Path) + }) + } +} + +func TestSetBaseURL_ChangesRequestTarget(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, `{"models":[]}`) + provider := newTestProvider("", "http://unused.invalid/v1", server.Client(), llmclient.Hooks{}) + provider.SetBaseURL(server.URL + "/v1") + + _, err := provider.ListModels(context.Background()) + require.NoError(t, err) + assert.Equal(t, "/v1/models", capture.Last(t).Path) +} + +// Passthrough is how the evaluation endpoint is reached. The gateway strips +// the optional v1 alias before the provider sees the endpoint, so both +// spellings a client may use land on the same upstream route. +func TestPassthrough_ForwardsNativeEndpoints(t *testing.T) { + tests := []struct { + name string + endpoint string + wantPath string + wantQuery string + }{ + {name: "evaluation", endpoint: "systemone", wantPath: "/v1/systemone"}, + {name: "evaluation with explicit v1 prefix", endpoint: "v1/systemone", wantPath: "/v1/systemone"}, + {name: "leading slash", endpoint: "/systemone", wantPath: "/v1/systemone"}, + {name: "kev permute", endpoint: "systemone/permute", wantPath: "/v1/systemone/permute"}, + {name: "kev separate", endpoint: "systemone/separate", wantPath: "/v1/systemone/separate"}, + {name: "models with query", endpoint: "models?limit=5", wantPath: "/v1/models", wantQuery: "limit=5"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, `{"model":"jev-1.13.0","answers":{}}`) + provider := newTestProvider("ts-key", server.URL, server.Client(), llmclient.Hooks{}) + + resp, err := provider.Passthrough(context.Background(), &core.PassthroughRequest{ + Method: http.MethodPost, + Endpoint: tt.endpoint, + Body: io.NopCloser(strings.NewReader(`{"state":"hi","model":"jev-latest","questions":{}}`)), + Headers: http.Header{"Content-Type": []string{"application/json"}}, + }) + require.NoError(t, err) + defer resp.Body.Close() + + req := capture.Last(t) + assert.Equal(t, http.MethodPost, req.Method) + assert.Equal(t, tt.wantPath, req.Path) + assert.Equal(t, tt.wantQuery, req.Query.Encode()) + assert.Equal(t, "Bearer ts-key", req.Header.Get("Authorization")) + assert.Equal(t, "application/json", req.Header.Get("Content-Type")) + assert.JSONEq(t, `{"state":"hi","model":"jev-latest","questions":{}}`, string(req.Body)) + assert.Equal(t, http.StatusOK, resp.StatusCode) + }) + } +} + +// A local Kev server is the keyless deployment: nothing must be sent unless +// the operator configured a token for a proxy in front of it. +func TestPassthrough_SendsNoCredentialWhenNoneIsConfigured(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, `{"models":[]}`) + provider := New(providers.ProviderConfig{BaseURL: server.URL}, providertest.Options(llmclient.Hooks{})).(*Provider) + + resp, err := provider.Passthrough(context.Background(), &core.PassthroughRequest{ + Method: http.MethodGet, + Endpoint: "models", + Headers: http.Header{}, + }) + require.NoError(t, err) + defer resp.Body.Close() + + assert.Equal(t, http.StatusOK, resp.StatusCode) + assert.Empty(t, capture.Last(t).Header.Get("Authorization")) +} + +// Provider-native errors relay status and body verbatim: System One reports +// a malformed question as 422 with the offending field in the body, which the +// client needs as written. +func TestPassthrough_RelaysNativeErrors(t *testing.T) { + server, _ := providertest.Server(t, func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusUnprocessableEntity) + _, _ = w.Write([]byte(`{"detail":[{"loc":["body","questions","tone","criteria"],"msg":"field required"}]}`)) + }) + provider := newTestProvider("ts-key", server.URL, server.Client(), llmclient.Hooks{}) + + resp, err := provider.Passthrough(context.Background(), &core.PassthroughRequest{ + Method: http.MethodPost, + Endpoint: "systemone", + Headers: http.Header{}, + }) + require.NoError(t, err) + defer resp.Body.Close() + + body, _ := io.ReadAll(resp.Body) + assert.Equal(t, http.StatusUnprocessableEntity, resp.StatusCode) + assert.Contains(t, string(body), "field required") +} + +func TestPassthrough_RequiresRequest(t *testing.T) { + provider := New(providers.ProviderConfig{BaseURL: "http://localhost:8009"}, providers.ProviderOptions{}).(*Provider) + _, err := provider.Passthrough(context.Background(), nil) + require.Error(t, err) + assert.Contains(t, err.Error(), "passthrough request is required") +} + +// The enricher names the evaluation route so the audit log records what a +// passthrough call did rather than an opaque /p/jev/... path. +func TestPassthroughSemantics_NameNativeRoutes(t *testing.T) { + enriched := passthroughSemanticEnricher.Enrich(nil, nil, &core.PassthroughRouteInfo{RawEndpoint: "/systemone"}) + require.NotNil(t, enriched) + assert.Equal(t, "jev.systemone", enriched.SemanticOperation) + assert.Empty(t, enriched.GenAIOperation) + assert.Equal(t, "/v1/systemone", enriched.AuditPath) + + permute := passthroughSemanticEnricher.Enrich(nil, nil, &core.PassthroughRouteInfo{RawEndpoint: "systemone/permute"}) + require.NotNil(t, permute) + assert.Equal(t, "jev.systemone_permute", permute.SemanticOperation) + assert.Equal(t, "/v1/systemone/permute", permute.AuditPath) + + unknown := passthroughSemanticEnricher.Enrich(nil, nil, &core.PassthroughRouteInfo{RawEndpoint: "/models"}) + require.NotNil(t, unknown) + assert.Equal(t, "/p/jev/models", unknown.AuditPath) +} diff --git a/internal/providers/jev/models.go b/internal/providers/jev/models.go new file mode 100644 index 000000000..c78560321 --- /dev/null +++ b/internal/providers/jev/models.go @@ -0,0 +1,104 @@ +package jev + +import ( + "context" + "net/http" + "strings" + "time" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/llmclient" +) + +// modelsResponse mirrors GET /v1/models. Neither server returns the OpenAI +// list shape: TypeSafe names each model or alias with "name" and describes +// it, while a Kev server names its checkpoint with "id" and lists the aliases +// it answers to, so both spellings are read. +type modelsResponse struct { + Models []modelEntry `json:"models"` +} + +type modelEntry struct { + Name string `json:"name"` + ID string `json:"id"` + Description string `json:"description"` + ReleaseDate string `json:"release_date"` + Aliases []string `json:"aliases"` +} + +// ListModels returns the names a request's model field may carry. Aliases +// are listed as models of their own, so a request that names one (jev-latest +// on a Kev server) resolves against the catalog like any other model. +func (p *Provider) ListModels(ctx context.Context) (*core.ModelsResponse, error) { + var raw modelsResponse + if err := p.client.Do(ctx, llmclient.Request{ + Method: http.MethodGet, + Endpoint: "/v1/models", + }, &raw); err != nil { + return nil, err + } + return raw.toCore(), nil +} + +func (r *modelsResponse) toCore() *core.ModelsResponse { + resp := &core.ModelsResponse{Object: "list", Data: make([]core.Model, 0, len(r.Models))} + seen := make(map[string]struct{}, len(r.Models)) + for _, entry := range r.Models { + for _, id := range entry.ids() { + if _, dup := seen[id]; dup { + continue + } + seen[id] = struct{}{} + resp.Data = append(resp.Data, entry.toCore(id)) + } + } + return resp +} + +// ids returns the model's own name followed by its aliases, blanks dropped. +func (e modelEntry) ids() []string { + ids := make([]string, 0, 1+len(e.Aliases)) + name := strings.TrimSpace(e.Name) + if name == "" { + name = strings.TrimSpace(e.ID) + } + if name != "" { + ids = append(ids, name) + } + for _, alias := range e.Aliases { + if alias = strings.TrimSpace(alias); alias != "" { + ids = append(ids, alias) + } + } + return ids +} + +// toCore describes one name from the entry. System One models are decision +// models with no generation mode the gateway could route an OpenAI request +// to, so they are categorized as utility models and claim no mode. +func (e modelEntry) toCore(id string) core.Model { + return core.Model{ + ID: id, + Object: "model", + Created: releaseTimestamp(e.ReleaseDate), + Metadata: &core.ModelMetadata{ + Description: strings.TrimSpace(e.Description), + Categories: []core.ModelCategory{core.CategoryUtility}, + }, + } +} + +// releaseTimestamp converts a release date to the Unix timestamp the OpenAI +// model shape carries, or 0 when the entry has none or it is not a date. +func releaseTimestamp(releaseDate string) int64 { + releaseDate = strings.TrimSpace(releaseDate) + if releaseDate == "" { + return 0 + } + for _, layout := range []string{time.DateOnly, time.RFC3339} { + if parsed, err := time.Parse(layout, releaseDate); err == nil { + return parsed.Unix() + } + } + return 0 +} diff --git a/internal/providers/jev/models_test.go b/internal/providers/jev/models_test.go new file mode 100644 index 000000000..4094a5e89 --- /dev/null +++ b/internal/providers/jev/models_test.go @@ -0,0 +1,103 @@ +package jev + +import ( + "context" + "net/http" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/llmclient" + "github.com/enterpilot/gomodel/internal/providers/providertest" +) + +// The hosted catalog lists aliases by name with a description and release +// date; versioned IDs are accepted by the model field without being listed. +const typesafeModelsJSON = `{ + "models": [ + {"name": "jev-latest", "description": "The most recent stable release.", "release_date": "2026-08-12"}, + {"name": "jev-preview", "description": "The most recent release, official or not.", "release_date": "2026-08-12T00:00:00Z"} + ] +}` + +// A Kev server names its checkpoint and lists the aliases it answers to, +// alongside serving details the gateway has no use for. +const kevModelsJSON = `{ + "models": [ + {"id": "kev-latest", "aliases": ["jev-latest"], "run": "jaredpalmer/kev-4b", "base": "Qwen/Qwen3.5-4B-Base", + "device": "mps", "temperature": 2.3, "prefix_cache": {"size": 4, "hits": 0}} + ] +}` + +func TestListModels_ReadsTheHostedCatalog(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, typesafeModelsJSON) + provider := newTestProvider("ts-key", server.URL, server.Client(), llmclient.Hooks{}) + + resp, err := provider.ListModels(context.Background()) + require.NoError(t, err) + assert.Equal(t, http.MethodGet, capture.Last(t).Method) + assert.Equal(t, "/v1/models", capture.Last(t).Path) + assert.Equal(t, "Bearer ts-key", capture.Last(t).Header.Get("Authorization")) + + assert.Equal(t, "list", resp.Object) + require.Len(t, resp.Data, 2) + + latest := resp.Data[0] + assert.Equal(t, "jev-latest", latest.ID) + assert.Equal(t, "model", latest.Object) + assert.Equal(t, time.Date(2026, 8, 12, 0, 0, 0, 0, time.UTC).Unix(), latest.Created) + require.NotNil(t, latest.Metadata) + assert.Equal(t, "The most recent stable release.", latest.Metadata.Description) + assert.Equal(t, []core.ModelCategory{core.CategoryUtility}, latest.Metadata.Categories) + assert.Empty(t, latest.Metadata.Modes, "a decision model claims no generation mode") + + preview := resp.Data[1] + assert.Equal(t, "jev-preview", preview.ID) + assert.Equal(t, time.Date(2026, 8, 12, 0, 0, 0, 0, time.UTC).Unix(), preview.Created, "RFC 3339 release dates are read too") +} + +// A Kev checkpoint and each alias it answers to are listed as models of +// their own, so a request naming jev-latest resolves against the catalog. +func TestListModels_ListsKevAliasesAsModels(t *testing.T) { + server, _ := providertest.JSONServer(t, http.StatusOK, kevModelsJSON) + provider := newTestProvider("", server.URL, server.Client(), llmclient.Hooks{}) + + resp, err := provider.ListModels(context.Background()) + require.NoError(t, err) + require.Len(t, resp.Data, 2) + assert.Equal(t, "kev-latest", resp.Data[0].ID) + assert.Equal(t, "jev-latest", resp.Data[1].ID) + for _, model := range resp.Data { + assert.Equal(t, "model", model.Object) + assert.Zero(t, model.Created, "no release date is reported") + require.NotNil(t, model.Metadata) + assert.Equal(t, []core.ModelCategory{core.CategoryUtility}, model.Metadata.Categories) + } +} + +func TestListModels_SkipsNamelessAndDuplicateEntries(t *testing.T) { + server, _ := providertest.JSONServer(t, http.StatusOK, `{"models":[ + {"description": "no name"}, + {"name": " jev-latest ", "aliases": [" ", "jev-latest", "jev-preview"]}, + {"id": "jev-preview"} + ]}`) + provider := newTestProvider("", server.URL, server.Client(), llmclient.Hooks{}) + + resp, err := provider.ListModels(context.Background()) + require.NoError(t, err) + require.Len(t, resp.Data, 2) + assert.Equal(t, "jev-latest", resp.Data[0].ID) + assert.Equal(t, "jev-preview", resp.Data[1].ID) +} + +func TestListModels_PropagatesUpstreamFailure(t *testing.T) { + server, _ := providertest.JSONServer(t, http.StatusUnauthorized, `{"error":"Missing or invalid API key"}`) + provider := newTestProvider("", server.URL, server.Client(), llmclient.Hooks{}) + + _, err := provider.ListModels(context.Background()) + require.Error(t, err) + assert.Contains(t, err.Error(), "Missing or invalid API key") +} diff --git a/internal/providers/jev/newtestprovider_test.go b/internal/providers/jev/newtestprovider_test.go new file mode 100644 index 000000000..5a0a9ddae --- /dev/null +++ b/internal/providers/jev/newtestprovider_test.go @@ -0,0 +1,16 @@ +package jev + +import ( + "net/http" + + "github.com/enterpilot/gomodel/internal/llmclient" + "github.com/enterpilot/gomodel/internal/providers" + "github.com/enterpilot/gomodel/internal/providers/providertest" +) + +// newTestProvider builds the provider through its own constructor on a test transport. +func newTestProvider(apiKey, baseURL string, httpClient *http.Client, hooks llmclient.Hooks) *Provider { + opts := providertest.Options(hooks) + opts.HTTPClient = httpClient + return New(providers.ProviderConfig{APIKey: apiKey, BaseURL: baseURL}, opts).(*Provider) +} diff --git a/internal/providers/jev/passthrough_semantics.go b/internal/providers/jev/passthrough_semantics.go new file mode 100644 index 000000000..091e58353 --- /dev/null +++ b/internal/providers/jev/passthrough_semantics.go @@ -0,0 +1,23 @@ +package jev + +import "github.com/enterpilot/gomodel/internal/providers" + +// The keys are endpoints as a passthrough request spells them, without the +// optional /v1 alias the gateway strips before a provider sees them. The +// evaluation route is the hosted API's only inference endpoint; permute and +// separate are the diagnostic variants a Kev server adds. System One is not +// one of the standard GenAI operations, so none is claimed. +var passthroughSemanticEnricher = providers.NewSemanticEnricher("jev", map[string]providers.PassthroughEndpointSemantics{ + "/systemone": { + Operation: "jev.systemone", + AuditPath: "/v1/systemone", + }, + "/systemone/permute": { + Operation: "jev.systemone_permute", + AuditPath: "/v1/systemone/permute", + }, + "/systemone/separate": { + Operation: "jev.systemone_separate", + AuditPath: "/v1/systemone/separate", + }, +}) diff --git a/internal/server/handlers_test.go b/internal/server/handlers_test.go index 6af3f6d34..b8cd9e565 100644 --- a/internal/server/handlers_test.go +++ b/internal/server/handlers_test.go @@ -5939,7 +5939,7 @@ func TestProviderPassthrough_RejectsUnsupportedProvider(t *testing.T) { require.Equal(t, http.StatusBadRequest, rec.Code) require.Contains(t, rec.Body.String(), `provider passthrough for \"groq\" is not enabled`) - require.Contains(t, rec.Body.String(), "anthropic, deepseek, hetzner, kilo, llamacpp, llmd, openai, openrouter, sglang, vllm, zai") + require.Contains(t, rec.Body.String(), "anthropic, deepseek, hetzner, jev, kilo, llamacpp, llmd, openai, openrouter, sglang, vllm, zai") } func TestProviderPassthrough_ChutesRequiresExplicitOptIn(t *testing.T) { diff --git a/internal/server/passthrough_support.go b/internal/server/passthrough_support.go index 45aa8da86..51ba65e79 100644 --- a/internal/server/passthrough_support.go +++ b/internal/server/passthrough_support.go @@ -20,7 +20,7 @@ import ( "github.com/enterpilot/gomodel/internal/usage" ) -var defaultEnabledPassthroughProviders = []string{"openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek", "hetzner"} +var defaultEnabledPassthroughProviders = []string{"openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek", "hetzner", "jev"} const llmdDroppedReasonHeader = "X-Llm-D-Request-Dropped-Reason" diff --git a/internal/server/passthrough_support_test.go b/internal/server/passthrough_support_test.go index 3def33f89..09fde6a7a 100644 --- a/internal/server/passthrough_support_test.go +++ b/internal/server/passthrough_support_test.go @@ -32,13 +32,16 @@ func TestBuildPassthroughHeadersSkipsConfiguredUserPathHeader(t *testing.T) { require.Equal(t, "responses=v1", value) } -// TestDefaultEnabledPassthroughProvidersIncludesHetzner asserts that the default -// allowlist contains hetzner — the provider matrix marks hetzner passthrough ✅, -// and the default handler must not reject those requests before contacting the -// upstream. Caught by greptile P1 on PR #701. -func TestDefaultEnabledPassthroughProvidersIncludesHetzner(t *testing.T) { - found := slices.Contains(defaultEnabledPassthroughProviders, "hetzner") - require.True(t, found, "defaultEnabledPassthroughProviders = %v, want hetzner included", defaultEnabledPassthroughProviders) +// TestDefaultEnabledPassthroughProvidersIncludesMatrixProviders asserts that +// the default allowlist contains the providers the matrix marks passthrough ✅, +// so the default handler does not reject those requests before contacting the +// upstream. hetzner was caught by greptile P1 on PR #701; jev is reachable +// only through passthrough, so leaving it out would make the provider inert. +func TestDefaultEnabledPassthroughProvidersIncludesMatrixProviders(t *testing.T) { + for _, providerType := range []string{"hetzner", "jev"} { + found := slices.Contains(defaultEnabledPassthroughProviders, providerType) + require.True(t, found, "defaultEnabledPassthroughProviders = %v, want %s included", defaultEnabledPassthroughProviders, providerType) + } } // A successful non-streaming JSON passthrough response must produce a usage diff --git a/run/providers.go b/run/providers.go index 101dffad6..a1bffd594 100644 --- a/run/providers.go +++ b/run/providers.go @@ -19,6 +19,7 @@ import ( "github.com/enterpilot/gomodel/internal/providers/gemini" "github.com/enterpilot/gomodel/internal/providers/groq" "github.com/enterpilot/gomodel/internal/providers/hetzner" + "github.com/enterpilot/gomodel/internal/providers/jev" "github.com/enterpilot/gomodel/internal/providers/kilo" "github.com/enterpilot/gomodel/internal/providers/kimicode" "github.com/enterpilot/gomodel/internal/providers/llamacpp" @@ -66,6 +67,7 @@ func defaultProviderFactory(cfg *config.Config) *providers.ProviderFactory { factory.Add(vertex.Registration) factory.Add(groq.Registration) factory.Add(hetzner.Registration) + factory.Add(jev.Registration) factory.Add(kilo.Registration) factory.Add(kimicode.Registration) factory.Add(llamacpp.Registration) diff --git a/run/providers_test.go b/run/providers_test.go index 2dac774e1..975eb7b26 100644 --- a/run/providers_test.go +++ b/run/providers_test.go @@ -169,7 +169,7 @@ var credentialPayloadFields = []string{ func TestDefaultProviderFactoryRegistersAllProviderTypes(t *testing.T) { expected := []string{ "anthropic", "audiocpp", "azure", "bailian", "bedrock", "bedrock-mantle", "chatgpt", "chutes", "cohere", "deepseek", "elevenlabs", - "fireworks", "gemini", "groq", "hetzner", "kilo", "kimicode", "llamacpp", "llmd", "meta", "minimax", "ollama", "openai", "opencode_go", + "fireworks", "gemini", "groq", "hetzner", "jev", "kilo", "kimicode", "llamacpp", "llmd", "meta", "minimax", "ollama", "openai", "opencode_go", "openrouter", "oracle", "sglang", "vertex", "vllm", "xai", "xiaomi", "zai", } From f76c6ef8a17c370d07585fb31e8e34605b308406 Mon Sep 17 00:00:00 2001 From: "Jakub A. W" Date: Tue, 22 Sep 2026 14:52:36 +0200 Subject: [PATCH 2/3] fix(passthrough): authorize the model of oversized and repeated-model opaque bodies An opaque JSON body past the 64 KiB peek limit left the model unset, so the passthrough allowlist was skipped; the whole body is now read (bounded by the body limit) before the model is taken from it. A body that repeats the top-level model field is rejected instead of forwarded, since the upstream parser would pick a value the gateway never checked. --- docs/features/passthrough-api.mdx | 3 + internal/core/semantic.go | 27 +++++- internal/core/semantic_test.go | 25 +++++ internal/server/http_test.go | 83 +++++++++++++++- internal/server/passthrough_service.go | 5 + internal/server/request_selector_peek.go | 54 ++++++++--- internal/server/request_selector_peek_test.go | 97 ++++++++++++++++--- internal/server/request_snapshot.go | 32 +++++- 8 files changed, 288 insertions(+), 38 deletions(-) diff --git a/docs/features/passthrough-api.mdx b/docs/features/passthrough-api.mdx index 0c8853307..9f6d4729c 100644 --- a/docs/features/passthrough-api.mdx +++ b/docs/features/passthrough-api.mdx @@ -143,6 +143,9 @@ Passthrough is intentionally narrow while the API is in beta. Add `chutes` to `ENABLED_PASSTHROUGH_PROVIDERS` only when you intend to expose that surface. - GoModel does not translate passthrough request bodies or response bodies. +- The `model` a JSON passthrough body names is checked against the caller's + model allowlist whatever the body size. A body that repeats the top-level + `model` field is rejected, since the upstream would decide which one wins. - Provider-native error bodies and status codes are proxied instead of converted into OpenAI-compatible responses. - Features that depend on OpenAI-compatible request or response shapes may not diff --git a/internal/core/semantic.go b/internal/core/semantic.go index 89e03d0ef..eb8940cda 100644 --- a/internal/core/semantic.go +++ b/internal/core/semantic.go @@ -45,8 +45,13 @@ type PassthroughRouteInfo struct { GenAIOperation string // standard GenAI operation, if this is an inference call Stream bool // explicit streaming intent derived from the request body StreamUncertain bool // bounded opaque-body inspection could not determine stream intent - AuditPath string - Model string + // ModelAmbiguous reports that the opaque body names a model more than once, + // so no single value can be authorized: the upstream's parser decides which + // one wins, and the gateway cannot know. Such a request is rejected rather + // than forwarded unchecked. + ModelAmbiguous bool + AuditPath string + Model string } type semanticCacheKey string @@ -347,6 +352,24 @@ func applyBodyStreamHint(env *WhiteBoxPrompt, stream, uncertain bool) { } } +// MarkPassthroughModelAmbiguous records that the opaque body carries more than +// one top-level model field, so the request's model cannot be authorized. Any +// model hint already taken from the body is dropped with it: a first-match +// peek would have kept whichever value came first, which is not necessarily +// the one the upstream's parser uses. +func MarkPassthroughModelAmbiguous(env *WhiteBoxPrompt) { + if env == nil { + return + } + env.RouteHints.Model = "" + if passthrough := env.CachedPassthroughRouteInfo(); passthrough != nil { + cloned := *passthrough + cloned.ModelAmbiguous = true + cloned.Model = "" + CachePassthroughRouteInfo(env, &cloned) + } +} + // MarkPassthroughStreamUncertain records that bounded opaque-body inspection // stopped before it could determine explicit streaming intent. func MarkPassthroughStreamUncertain(env *WhiteBoxPrompt) { diff --git a/internal/core/semantic_test.go b/internal/core/semantic_test.go index b1b641262..3386bc19c 100644 --- a/internal/core/semantic_test.go +++ b/internal/core/semantic_test.go @@ -311,3 +311,28 @@ func TestDeriveBatchRouteInfoFromTransport_MessagesBatches(t *testing.T) { }) } } + +// A repeated model field leaves no value the gateway can authorize, so the +// marker drops the first-match hint along with recording the ambiguity, and +// a later body refresh keeps the ambiguity on the merged route info. +func TestMarkPassthroughModelAmbiguous_DropsModelAndSurvivesRefresh(t *testing.T) { + env := &WhiteBoxPrompt{} + CachePassthroughRouteInfo(env, &PassthroughRouteInfo{Provider: "jev"}) + ApplyBodySelectorHints(env, "jev-latest", "", false) + require.Equal(t, "jev-latest", env.RouteHints.Model) + + MarkPassthroughModelAmbiguous(env) + + require.Empty(t, env.RouteHints.Model) + info := env.CachedPassthroughRouteInfo() + require.NotNil(t, info) + require.True(t, info.ModelAmbiguous) + require.Empty(t, info.Model) + + snapshot := NewRequestSnapshot(http.MethodPost, "/p/jev/systemone", map[string]string{"provider": "jev", "endpoint": "systemone"}, nil, nil, "application/json", []byte(`{"model":"jev-latest","model":"jev-preview"}`), false, "", nil) + refreshed := RefreshWhiteBoxPrompt(snapshot, env) + require.NotNil(t, refreshed) + require.True(t, refreshed.CachedPassthroughRouteInfo().ModelAmbiguous) + + MarkPassthroughModelAmbiguous(nil) // a missing envelope is ignored +} diff --git a/internal/server/http_test.go b/internal/server/http_test.go index bf5ab4ed5..ef126f991 100644 --- a/internal/server/http_test.go +++ b/internal/server/http_test.go @@ -1052,7 +1052,9 @@ func TestProviderPassthroughRoute_EnabledByDefault(t *testing.T) { require.Equal(t, "openai", got) } -func TestProviderPassthroughRoute_MarksOversizedStreamIntentUncertain(t *testing.T) { +// An opaque body past the peek limit is read whole, so its model and stream +// intent are known and the bytes reach the provider unchanged. +func TestProviderPassthroughRoute_ReadsOversizedBodyCompletely(t *testing.T) { mock := &mockProvider{ passthroughResponse: &core.PassthroughResponse{ StatusCode: http.StatusOK, @@ -1070,7 +1072,84 @@ func TestProviderPassthroughRoute_MarksOversizedStreamIntentUncertain(t *testing require.Equal(t, http.StatusOK, rec.Code) require.NotNil(t, mock.lastPassthroughReq) - require.True(t, mock.lastPassthroughReq.StreamUncertain) + require.Equal(t, "gpt-5-mini", mock.lastPassthroughReq.Model) + require.True(t, mock.lastPassthroughReq.Stream) + require.False(t, mock.lastPassthroughReq.StreamUncertain) + forwarded, err := io.ReadAll(mock.lastPassthroughReq.Body) + require.NoError(t, err) + require.Equal(t, body, string(forwarded)) +} + +// A System One state is routinely longer than the peek limit. The model named +// after it must still be checked against the caller's allowlist, or a key +// restricted to other models could evaluate through /p/jev unchecked. +func TestProviderPassthroughRoute_AuthorizesModelFromOversizedBody(t *testing.T) { + mock := &mockProvider{} + authorizer := &recordingModelAuthorizer{err: core.NewInvalidRequestError("requested model is not available for this API key", nil)} + srv := New(mock, &Config{ModelAuthorizer: authorizer}) + body := `{"state":"` + strings.Repeat("x", 65*1024) + `","model":"jev-latest","questions":{"q":{"type":"noul","instructions":"?"}}}` + req := httptest.NewRequest(http.MethodPost, "/p/jev/v1/systemone", strings.NewReader(body)) + req.Header.Set("Content-Type", "application/json") + rec := httptest.NewRecorder() + + srv.ServeHTTP(rec, req) + + require.Equal(t, http.StatusBadRequest, rec.Code) + require.Contains(t, rec.Body.String(), "not available for this API key") + require.Equal(t, "jev-latest", authorizer.lastSelector.Model) + require.Nil(t, mock.lastPassthroughReq, "a denied request must not reach the provider") +} + +// The body is forwarded as written, so a repeated top-level model would let +// the upstream's parser pick a value the gateway never checked. Such a body +// is rejected whether or not an allowlist is configured. +func TestProviderPassthroughRoute_RejectsRepeatedModelField(t *testing.T) { + tests := []struct { + name string + body string + knownLength bool + }{ + {name: "small body captured inline", body: `{"model":"jev-latest","state":"x","questions":{},"model":"jev-preview"}`, knownLength: true}, + {name: "small body with unknown length", body: `{"model":"jev-latest","state":"x","questions":{},"model":"jev-preview"}`}, + {name: "repeat past the peek limit", body: `{"model":"jev-latest","state":"` + strings.Repeat("x", 65*1024) + `","model":"jev-preview"}`, knownLength: true}, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + mock := &mockProvider{} + srv := New(mock, &Config{}) + req := httptest.NewRequest(http.MethodPost, "/p/jev/systemone", strings.NewReader(test.body)) + if !test.knownLength { + req.ContentLength = -1 + } + req.Header.Set("Content-Type", "application/json") + rec := httptest.NewRecorder() + + srv.ServeHTTP(rec, req) + + require.Equal(t, http.StatusBadRequest, rec.Code) + require.Contains(t, rec.Body.String(), "model field is repeated") + require.Nil(t, mock.lastPassthroughReq) + }) + } +} + +// Reading an oversized body in full stays bounded by the body limit: a body +// past it is refused with the limit's own 413 instead of being forwarded +// with its model unchecked. +func TestProviderPassthroughRoute_OversizedBodyHonorsBodyLimit(t *testing.T) { + mock := &mockProvider{} + srv := New(mock, &Config{BodySizeLimit: "100K"}) + body := `{"state":"` + strings.Repeat("x", 120*1024) + `","model":"jev-latest"}` + req := httptest.NewRequest(http.MethodPost, "/p/jev/systemone", strings.NewReader(body)) + req.ContentLength = -1 // a chunked upload, so the limit is hit while reading rather than declared up front + req.Header.Set("Content-Type", "application/json") + rec := httptest.NewRecorder() + + srv.ServeHTTP(rec, req) + + require.Equal(t, http.StatusRequestEntityTooLarge, rec.Code) + require.Nil(t, mock.lastPassthroughReq) } func TestProviderPassthroughRoute_DisabledRequiresAuthBefore404(t *testing.T) { diff --git a/internal/server/passthrough_service.go b/internal/server/passthrough_service.go index 945c663a4..72d2424a3 100644 --- a/internal/server/passthrough_service.go +++ b/internal/server/passthrough_service.go @@ -33,6 +33,11 @@ func (s *passthroughService) ProviderPassthrough(c *echo.Context) error { if !isEnabledPassthroughProvider(providerType, s.enabledPassthroughProviders) { return handleError(c, s.unsupportedPassthroughProviderError(providerType)) } + // The body is forwarded unchanged, so a repeated model field would let the + // upstream's parser pick a value the gateway never checked. + if info.ModelAmbiguous { + return handleError(c, core.NewInvalidRequestError("passthrough request body must name the model once; the model field is repeated", nil)) + } if s.modelAuthorizer != nil { if selector, ok := passthroughAccessSelector(s.provider, info); ok { if err := s.modelAuthorizer.ValidateModelAccess(c.Request().Context(), selector); err != nil { diff --git a/internal/server/request_selector_peek.go b/internal/server/request_selector_peek.go index 39f66eb5b..50506a420 100644 --- a/internal/server/request_selector_peek.go +++ b/internal/server/request_selector_peek.go @@ -21,15 +21,28 @@ type requestBodySelectorHints struct { streamVerified bool parsed bool complete bool + // modelAmbiguous reports a complete body that names a model more than + // once, so no single value can be trusted. + modelAmbiguous bool } -func seedRequestBodySelectorHints(req *http.Request, bodyMode core.BodyMode, env *core.WhiteBoxPrompt) { +// seedRequestBodySelectorHints records the model, provider, and stream hints a +// JSON request body carries. Managed routes are peeked within the limit, since +// canonical decode later reads the whole body anyway. Opaque passthrough +// bodies are forwarded as they are, so their model is the only thing the +// allowlist can check: it is taken only from a complete body, however large, +// which is buffered in full when it exceeds the peek limit. The error is a +// body read failure, including the body-limit 413. +func seedRequestBodySelectorHints(req *http.Request, bodyMode core.BodyMode, env *core.WhiteBoxPrompt) error { if !shouldPeekRequestBodySelectors(req, bodyMode, env) { - return + return nil } if bodyMode == core.BodyModeOpaque { - hints := peekCompleteRequestBodySelectorHints(req, requestSelectorPeekLimit) + hints, err := peekCompleteRequestBodySelectorHints(req, requestSelectorPeekLimit) + if err != nil { + return err + } if hints.complete { core.ApplyBodySelectorHints(env, hints.model, hints.provider, hints.stream) } else if hints.streamParsed { @@ -42,7 +55,10 @@ func seedRequestBodySelectorHints(req *http.Request, bodyMode core.BodyMode, env if !hints.streamParsed { core.MarkPassthroughStreamUncertain(env) } - return + if hints.modelAmbiguous { + core.MarkPassthroughModelAmbiguous(env) + } + return nil } hints := peekRequestBodySelectorHints(req, requestSelectorPeekLimit) @@ -52,6 +68,7 @@ func seedRequestBodySelectorHints(req *http.Request, bodyMode core.BodyMode, env if !hints.streamParsed { core.MarkPassthroughStreamUncertain(env) } + return nil } func shouldPeekRequestBodySelectors(req *http.Request, bodyMode core.BodyMode, env *core.WhiteBoxPrompt) bool { @@ -91,29 +108,33 @@ func peekRequestBodySelectorHints(req *http.Request, limit int64) requestBodySel } // peekCompleteRequestBodySelectorHints returns authoritative selector hints -// only when the entire body fits within limit and has no duplicate selector -// fields. A unique stream hint may be returned independently from a bounded -// oversized body. The body is restored before returning so passthrough -// forwarding remains byte-for-byte unchanged. -func peekCompleteRequestBodySelectorHints(req *http.Request, limit int64) requestBodySelectorHints { +// from the entire body, which is what makes them safe to authorize on: an +// opaque body is forwarded byte for byte, so a model field the upstream's +// parser would see must be seen here too, including a repeat of it after a +// long value. A body within limit is read once; a larger one is buffered in +// full, bounded by the body-limit middleware ahead of this peek. The body is +// restored before returning so forwarding remains unchanged. A read error is +// returned as is, so the body limit's 413 reaches the client. +func peekCompleteRequestBodySelectorHints(req *http.Request, limit int64) (requestBodySelectorHints, error) { if req == nil || req.Body == nil || limit <= 0 { - return requestBodySelectorHints{} + return requestBodySelectorHints{}, nil } originalBody := req.Body body, err := io.ReadAll(io.LimitReader(originalBody, limit+1)) + if err == nil && int64(len(body)) > limit { + var rest []byte + rest, err = io.ReadAll(originalBody) + body = append(body, rest...) + } req.Body = &combinedReadCloser{ Reader: io.MultiReader(bytes.NewReader(body), originalBody), rc: originalBody, } if err != nil { - return requestBodySelectorHints{} - } - if int64(len(body)) > limit { - hints := decodeCompleteRequestBodySelectorHints(bytes.NewReader(body[:limit])) - return hints.independentStreamHint() + return requestBodySelectorHints{}, err } - return decodeCompleteRequestBodySelectorHints(bytes.NewReader(body)) + return decodeCompleteRequestBodySelectorHints(bytes.NewReader(body)), nil } func (hints requestBodySelectorHints) independentStreamHint() requestBodySelectorHints { @@ -228,6 +249,7 @@ func decodeRequestBodySelectorHintsWithMode(r io.Reader, requireComplete bool) r if modelAmbiguous || providerAmbiguous { hints.model = "" hints.provider = "" + hints.modelAmbiguous = modelAmbiguous return hints } } diff --git a/internal/server/request_selector_peek_test.go b/internal/server/request_selector_peek_test.go index c2cff61e2..ab07a46c4 100644 --- a/internal/server/request_selector_peek_test.go +++ b/internal/server/request_selector_peek_test.go @@ -8,6 +8,7 @@ import ( "strings" "testing" + "github.com/labstack/echo/v5" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -42,7 +43,7 @@ func TestSeedRequestBodySelectorHintsDoesNotMarkModelOnlyPeekAsParsed(t *testing req := httptest.NewRequest(http.MethodPost, "/v1/chat/completions", strings.NewReader(`{"model":"gpt-4o-mini","stream":true}`)) env := &core.WhiteBoxPrompt{} - seedRequestBodySelectorHints(req, core.BodyModeJSON, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeJSON, env)) require.False(t, env.JSONBodyParsed) require.False(t, env.StreamRequested) @@ -56,7 +57,7 @@ func TestSeedRequestBodySelectorHintsAppliesCompleteModelForOpaqueBody(t *testin env := &core.WhiteBoxPrompt{} core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "openai"}) - seedRequestBodySelectorHints(req, core.BodyModeOpaque, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeOpaque, env)) require.True(t, env.JSONBodyParsed) require.True(t, env.StreamRequested) @@ -73,7 +74,10 @@ func TestSeedRequestBodySelectorHintsAppliesCompleteModelForOpaqueBody(t *testin require.Equal(t, `{"model":"gpt-4o-mini","stream":true}`, string(restored)) } -func TestSeedRequestBodySelectorHintsRejectsIncompleteOpaqueModel(t *testing.T) { +// An opaque body larger than the peek limit is read in full, so a model field +// repeated after a long value is still caught: the upstream's parser would +// take the second one, which the gateway never authorized. +func TestSeedRequestBodySelectorHintsDetectsDuplicateModelBeyondPeekLimit(t *testing.T) { tests := []struct { name string prefix string @@ -82,10 +86,9 @@ func TestSeedRequestBodySelectorHintsRejectsIncompleteOpaqueModel(t *testing.T) knownLength bool }{ {name: "model first", prefix: `{"model":"allowed-model","padding":"`, wantUncertain: true}, - {name: "stream first", prefix: `{"stream":true,"model":"allowed-model","padding":"`, wantStream: true, wantUncertain: true}, - {name: "model before stream", prefix: `{"model":"allowed-model","stream":true,"padding":"`, wantStream: true, wantUncertain: true}, - {name: "model before stream with known length", prefix: `{"model":"allowed-model","stream":true,"padding":"`, wantStream: true, wantUncertain: true, knownLength: true}, - {name: "stream before oversized value", prefix: `{"stream":true,"padding":"`, wantStream: true, wantUncertain: true}, + {name: "stream first", prefix: `{"stream":true,"model":"allowed-model","padding":"`, wantStream: true}, + {name: "model before stream", prefix: `{"model":"allowed-model","stream":true,"padding":"`, wantStream: true}, + {name: "model before stream with known length", prefix: `{"model":"allowed-model","stream":true,"padding":"`, wantStream: true, knownLength: true}, } for _, test := range tests { @@ -99,7 +102,7 @@ func TestSeedRequestBodySelectorHintsRejectsIncompleteOpaqueModel(t *testing.T) env := &core.WhiteBoxPrompt{} core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "openai"}) - seedRequestBodySelectorHints(req, core.BodyModeOpaque, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeOpaque, env)) require.False(t, env.JSONBodyParsed) require.Empty(t, env.RouteHints.Model) @@ -107,12 +110,69 @@ func TestSeedRequestBodySelectorHintsRejectsIncompleteOpaqueModel(t *testing.T) info := env.CachedPassthroughRouteInfo() require.NotNil(t, info) require.Empty(t, info.Model) + require.True(t, info.ModelAmbiguous) require.Equal(t, test.wantStream, info.Stream) require.Equal(t, test.wantUncertain, info.StreamUncertain) + + restored, err := io.ReadAll(req.Body) + require.NoError(t, err) + require.Equal(t, body, string(restored)) }) } } +// A single model named after a value longer than the peek limit (a System One +// state, a long prompt) is authoritative: the whole body was read, so the +// allowlist can check it, and the bytes forwarded upstream are unchanged. +func TestSeedRequestBodySelectorHintsAppliesModelFromOversizedOpaqueBody(t *testing.T) { + body := `{"stream":true,"state":"` + strings.Repeat("x", int(requestSelectorPeekLimit)) + `","model":"jev-latest"}` + req := httptest.NewRequest(http.MethodPost, "/p/jev/systemone", strings.NewReader(body)) + req.ContentLength = -1 + req.Header.Set("Content-Type", "application/json") + env := &core.WhiteBoxPrompt{} + core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "jev"}) + + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeOpaque, env)) + + require.True(t, env.JSONBodyParsed) + require.True(t, env.StreamRequested) + require.Equal(t, "jev-latest", env.RouteHints.Model) + + info := env.CachedPassthroughRouteInfo() + require.NotNil(t, info) + require.Equal(t, "jev-latest", info.Model) + require.False(t, info.ModelAmbiguous) + require.True(t, info.Stream) + require.False(t, info.StreamUncertain) + + restored, err := io.ReadAll(req.Body) + require.NoError(t, err) + require.Equal(t, body, string(restored)) +} + +// erroringReader yields err on every read, standing in for the body-limit +// reader once a body exceeds the configured limit. +type erroringReader struct{ err error } + +func (r erroringReader) Read([]byte) (int, error) { return 0, r.err } + +// Reading past the peek limit can fail (the body limit's 413); the error must +// reach the caller instead of leaving the model unset and the request +// forwarded unchecked. +func TestSeedRequestBodySelectorHintsReturnsBodyReadError(t *testing.T) { + prefix := `{"model":"jev-latest","state":"` + strings.Repeat("x", int(requestSelectorPeekLimit)) + req := httptest.NewRequest(http.MethodPost, "/p/jev/systemone", io.MultiReader(strings.NewReader(prefix), erroringReader{err: echo.ErrStatusRequestEntityTooLarge})) + req.ContentLength = -1 + req.Header.Set("Content-Type", "application/json") + env := &core.WhiteBoxPrompt{} + core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "jev"}) + + err := seedRequestBodySelectorHints(req, core.BodyModeOpaque, env) + require.ErrorIs(t, err, echo.ErrStatusRequestEntityTooLarge) + require.False(t, env.JSONBodyParsed) + require.Empty(t, env.CachedPassthroughRouteInfo().Model) +} + func TestSeedRequestBodySelectorHintsRejectsAmbiguousStreamBeforePeekLimit(t *testing.T) { body := `{"stream":true,"stream":false,"padding":"` + strings.Repeat("x", int(requestSelectorPeekLimit)) + `"}` req := httptest.NewRequest(http.MethodPost, "/p/openai/chat/completions", strings.NewReader(body)) @@ -121,7 +181,7 @@ func TestSeedRequestBodySelectorHintsRejectsAmbiguousStreamBeforePeekLimit(t *te env := &core.WhiteBoxPrompt{} core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "openai"}) - seedRequestBodySelectorHints(req, core.BodyModeOpaque, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeOpaque, env)) require.False(t, env.StreamRequested) @@ -131,7 +191,11 @@ func TestSeedRequestBodySelectorHintsRejectsAmbiguousStreamBeforePeekLimit(t *te require.True(t, info.StreamUncertain) } -func TestSeedRequestBodySelectorHintsMarksStreamBeyondPeekBoundaryUncertain(t *testing.T) { +// A stream field repeated after a value longer than the peek limit is read +// whole and treated like any duplicate: no stream intent is claimed, and the +// uncertainty is recorded, rather than trusting the first value the upstream +// would not use. +func TestSeedRequestBodySelectorHintsRejectsDuplicateStreamBeyondPeekLimit(t *testing.T) { body := `{"stream":true,"padding":"` + strings.Repeat("x", int(requestSelectorPeekLimit)) + `","stream":false}` req := httptest.NewRequest(http.MethodPost, "/p/openai/chat/completions", strings.NewReader(body)) req.ContentLength = -1 @@ -139,13 +203,13 @@ func TestSeedRequestBodySelectorHintsMarksStreamBeyondPeekBoundaryUncertain(t *t env := &core.WhiteBoxPrompt{} core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "openai"}) - seedRequestBodySelectorHints(req, core.BodyModeOpaque, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeOpaque, env)) - require.True(t, env.StreamRequested) + require.False(t, env.StreamRequested) info := env.CachedPassthroughRouteInfo() require.NotNil(t, info) - require.True(t, info.Stream) + require.False(t, info.Stream) require.True(t, info.StreamUncertain) restored, err := io.ReadAll(req.Body) @@ -186,7 +250,7 @@ func TestSeedRequestBodySelectorHintsRejectsCompleteDuplicateOpaqueFields(t *tes env := &core.WhiteBoxPrompt{} core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "openai"}) - seedRequestBodySelectorHints(req, core.BodyModeOpaque, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeOpaque, env)) require.False(t, env.JSONBodyParsed) require.Equal(t, "openai", env.RouteHints.Provider) @@ -206,7 +270,7 @@ func TestSeedRequestBodySelectorHintsRejectsDuplicateOpaqueModel(t *testing.T) { env := &core.WhiteBoxPrompt{} core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "openai"}) - seedRequestBodySelectorHints(req, core.BodyModeOpaque, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeOpaque, env)) require.False(t, env.JSONBodyParsed) require.Empty(t, env.RouteHints.Model) @@ -214,6 +278,7 @@ func TestSeedRequestBodySelectorHintsRejectsDuplicateOpaqueModel(t *testing.T) { info := env.CachedPassthroughRouteInfo() require.NotNil(t, info) + require.True(t, info.ModelAmbiguous) require.True(t, info.Stream) require.False(t, info.StreamUncertain) } @@ -260,7 +325,7 @@ func TestSeedRequestBodySelectorHintsTracksStreamConfidenceIndependently(t *test env := &core.WhiteBoxPrompt{} core.CachePassthroughRouteInfo(env, &core.PassthroughRouteInfo{Provider: "openai"}) - seedRequestBodySelectorHints(req, core.BodyModeJSON, env) + require.NoError(t, seedRequestBodySelectorHints(req, core.BodyModeJSON, env)) info := env.CachedPassthroughRouteInfo() require.NotNil(t, info) diff --git a/internal/server/request_snapshot.go b/internal/server/request_snapshot.go index 34c7584fc..a5c2ca5b7 100644 --- a/internal/server/request_snapshot.go +++ b/internal/server/request_snapshot.go @@ -83,8 +83,10 @@ func RequestSnapshotCapture(userPathHeader ...string) echo.MiddlewareFunc { ctx := core.WithUserPathHeaderName(req.Context(), userPathHeaderName) ctx = core.WithRequestSnapshot(ctx, snapshot) if semantics := core.DeriveWhiteBoxPrompt(snapshot); semantics != nil { - if !bodyCaptured { - seedRequestBodySelectorHints(req, desc.BodyMode, semantics) + if bodyCaptured { + verifyCapturedOpaqueSelectorHints(desc.BodyMode, bodyBytes, semantics) + } else if err := seedRequestBodySelectorHints(req, desc.BodyMode, semantics); err != nil { + return handleError(c, requestBodyReadError(err)) } ctx = core.WithWhiteBoxPrompt(ctx, semantics) } @@ -95,6 +97,32 @@ func RequestSnapshotCapture(userPathHeader ...string) echo.MiddlewareFunc { } } +// verifyCapturedOpaqueSelectorHints re-reads the selector fields of a +// captured opaque body with the complete decoder. The snapshot's own peek +// takes the first model it meets, which is fine for a body the gateway will +// decode canonically, but an opaque body is forwarded as written and the +// upstream's parser may take the last one instead; a repeated model is +// therefore recorded as ambiguous so the passthrough route refuses it. +func verifyCapturedOpaqueSelectorHints(bodyMode core.BodyMode, body []byte, env *core.WhiteBoxPrompt) { + if bodyMode != core.BodyModeOpaque || env == nil || !env.JSONBodyParsed { + return + } + if hints := decodeCompleteRequestBodySelectorHints(bytes.NewReader(body)); hints.modelAmbiguous { + core.MarkPassthroughModelAmbiguous(env) + } +} + +// requestBodyReadError shapes a failure to read the request body as a gateway +// error. One that already carries an HTTP status (the body limit's 413) keeps +// it so the client learns why; any other read failure is the client's +// malformed or interrupted request. +func requestBodyReadError(err error) error { + if status := echo.StatusCode(err); status > 0 { + return core.NewInvalidRequestErrorWithStatus(status, echoErrorMessage(err, status), err) + } + return core.NewInvalidRequestError("failed to read request body", err) +} + func configuredUserPathHeaderName(headerNames ...string) string { if len(headerNames) == 0 { return core.UserPathHeader From 915b249ba5b2ac739bca80c2937c09d12165ecfb Mon Sep 17 00:00:00 2001 From: "Jakub A. W" Date: Tue, 22 Sep 2026 15:13:20 +0200 Subject: [PATCH 3/3] docs(passthrough): state the body-limit bound on model allowlist checks --- docs/features/passthrough-api.mdx | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/features/passthrough-api.mdx b/docs/features/passthrough-api.mdx index 9f6d4729c..bc409fdaa 100644 --- a/docs/features/passthrough-api.mdx +++ b/docs/features/passthrough-api.mdx @@ -144,8 +144,10 @@ Passthrough is intentionally narrow while the API is in beta. that surface. - GoModel does not translate passthrough request bodies or response bodies. - The `model` a JSON passthrough body names is checked against the caller's - model allowlist whatever the body size. A body that repeats the top-level - `model` field is rejected, since the upstream would decide which one wins. + model allowlist regardless of body size, up to the configured body limit; + a larger body is refused with the limit's own error. A body that repeats + the top-level `model` field is rejected, since the upstream would decide + which one wins. - Provider-native error bodies and status codes are proxied instead of converted into OpenAI-compatible responses. - Features that depend on OpenAI-compatible request or response shapes may not