Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 10 additions & 2 deletions .env.template
Original file line number Diff line number Diff line change
Expand Up @@ -106,11 +106,11 @@
# Allow optional /p/{provider}/v1/... passthrough aliases while keeping /p/{provider}/... canonical (default: true)
# ALLOW_PASSTHROUGH_V1_ALIAS=true

# Comma-separated list of provider types enabled for /p/{provider}/... passthrough (default: openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek)
# Comma-separated list of provider types enabled for /p/{provider}/... passthrough (default: openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek,jev)
# Cohere and audio.cpp native passthrough are opt-in; add cohere or audiocpp when
# those routes are needed. audio.cpp's native surface includes model management
# and server-local file paths, so enable it only for trusted callers.
# ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,cohere,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,audiocpp,deepseek,hetzner
# ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,cohere,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,audiocpp,deepseek,hetzner,jev

# Enable the realtime (speech-to-speech) endpoints (default: true): the /v1/realtime
# websocket (and /p/{provider}/v1/realtime passthrough upgrade), the WebRTC SDP
Expand Down Expand Up @@ -700,6 +700,14 @@
# ELEVENLABS_API_KEY=...
# ELEVENLABS_BASE_URL=https://api.elevenlabs.io

# Jev (TypeSafe System One decision API; default base URL: https://api.typesafe.ai)
# The API is not OpenAI-compatible: send System One requests to
# POST /p/jev/v1/systemone (or point the TypeSafe SDK at http://localhost:8080/p/jev).
# JEV_API_KEY=...
# A self-hosted Kev server (github.com/jaredpalmer/kev) speaks the same API and
# has no authentication of its own: set the base URL and leave the key unset.
# JEV_BASE_URL=http://localhost:8009

# Xiaomi MiMo (default base URL: https://api.xiaomimimo.com/v1)
# XIAOMI_API_KEY=...
# XIAOMI_BASE_URL=https://api.xiaomimimo.com/v1
Expand Down
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -132,6 +132,7 @@ The official SDKs therefore work unchanged. Configure their base URLs as follows
- Amazon Bedrock Runtime and Bedrock Mantle
- ChatGPT (the Codex backend) and Claude
- ElevenLabs (text-to-speech and speech-to-text)
- Jev (TypeSafe System One decision API) and self-hosted Kev
- All OpenAI-compatible providers

See the [Providers Overview](https://gomodel.enterpilot.io/docs/providers/overview?utm_source=readme) for the full
Expand Down
18 changes: 18 additions & 0 deletions config/config.example.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -570,6 +570,24 @@ providers:
# gateway's own) and a trailing /v1 is accepted. api_key is optional and
# usually unset, since audio.cpp has no authentication of its own.

jev:
type: jev
api_key: "${JEV_API_KEY}"
# base_url defaults to "https://api.typesafe.ai". TypeSafe's System One
# API is a decision API with no OpenAI-compatible surface: requests go to
# POST /p/jev/v1/systemone, or point the TypeSafe SDK at /p/jev. A
# self-hosted Kev server speaks the same API without authentication:
# set base_url (e.g. "http://localhost:8009") and omit api_key.
# Jev is priced per input token and is not in the upstream model catalog;
# declare its pricing here to have the gateway cost System One requests.
# models:
# - id: "jev-1.13.0"
# metadata:
# pricing:
# currency: USD
# input_per_mtok: 0.042
# output_per_mtok: 0

meta:
type: meta
api_key: "..."
Expand Down
1 change: 1 addition & 0 deletions config/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -127,6 +127,7 @@ func buildDefaultConfig() *Config {
"llamacpp",
"llmd",
"deepseek",
"jev",
},
},
Models: ModelsConfig{
Expand Down
2 changes: 1 addition & 1 deletion config/config_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -132,7 +132,7 @@ func TestBuildDefaultConfig(t *testing.T) {
assert.Equal(t, DefaultStreamStallTimeoutSeconds, cfg.Server.StreamStallTimeout)
assert.True(t, cfg.Server.EnablePassthroughRoutes)
assert.True(t, cfg.Server.AllowPassthroughV1Alias)
assert.Equal(t, []string{"openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek"}, cfg.Server.EnabledPassthroughProviders)
assert.Equal(t, []string{"openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek", "jev"}, cfg.Server.EnabledPassthroughProviders)
assert.Equal(t, ConfiguredProviderModelsModeFallback, cfg.Models.ConfiguredProviderModelsMode)
assert.Nil(t, cfg.Cache.Model.Local)
assert.Equal(t, 3600, cfg.Cache.Model.RefreshInterval)
Expand Down
2 changes: 1 addition & 1 deletion config/server.go
Original file line number Diff line number Diff line change
Expand Up @@ -45,7 +45,7 @@ type ServerConfig struct {
UserPathHeader string `yaml:"user_path_header" env:"USER_PATH_HEADER"`
// EnabledPassthroughProviders lists the provider types enabled on
// /p/{provider}/... passthrough routes. Default:
// ["openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llmd", "deepseek"].
// ["openai", "anthropic", "openrouter", "kilo", "zai", "sglang", "vllm", "llamacpp", "llmd", "deepseek", "jev"].
EnabledPassthroughProviders []string `yaml:"enabled_passthrough_providers" env:"ENABLED_PASSTHROUGH_PROVIDERS"`
// RealtimeEnabled exposes the realtime (speech-to-speech) websocket endpoints
// at /v1/realtime and /v1/realtime/translations, their WebRTC signaling
Expand Down
1 change: 1 addition & 0 deletions docs/docs.json
Original file line number Diff line number Diff line change
Expand Up @@ -214,6 +214,7 @@
"providers/minimax",
"providers/elevenlabs",
"providers/audiocpp",
"providers/jev",
"providers/opencode-go",
"providers/sglang",
"providers/vllm",
Expand Down
9 changes: 7 additions & 2 deletions docs/features/passthrough-api.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -136,13 +136,18 @@ from passthrough requests before forwarding them upstream.

Passthrough is intentionally narrow while the API is in beta.

- `openai`, `anthropic`, `openrouter`, `kilo`, `zai`, `sglang`, `vllm`, `llamacpp`, `llmd`, and `deepseek`
- `openai`, `anthropic`, `openrouter`, `kilo`, `zai`, `sglang`, `vllm`, `llamacpp`, `llmd`, `deepseek`, and `jev`
are enabled by default.
- Chutes supports passthrough but requires explicit operator opt-in because
passthrough can forward provider-native routes that do not identify a model.
Add `chutes` to `ENABLED_PASSTHROUGH_PROVIDERS` only when you intend to expose
that surface.
- GoModel does not translate passthrough request bodies or response bodies.
- The `model` a JSON passthrough body names is checked against the caller's
model allowlist regardless of body size, up to the configured body limit;
a larger body is refused with the limit's own error. A body that repeats
the top-level `model` field is rejected, since the upstream would decide
which one wins.
- Provider-native error bodies and status codes are proxied instead of converted
into OpenAI-compatible responses.
- Features that depend on OpenAI-compatible request or response shapes may not
Expand All @@ -155,7 +160,7 @@ Passthrough routes are enabled by default:
```env
ENABLE_PASSTHROUGH_ROUTES=true
ALLOW_PASSTHROUGH_V1_ALIAS=true
ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek
ENABLED_PASSTHROUGH_PROVIDERS=openai,anthropic,openrouter,kilo,zai,sglang,vllm,llamacpp,llmd,deepseek,jev
```

Set `ENABLED_PASSTHROUGH_PROVIDERS` to the provider types you want to expose.
173 changes: 173 additions & 0 deletions docs/providers/jev.mdx
Original file line number Diff line number Diff line change
@@ -0,0 +1,173 @@
---
title: "Jev / Kev (TypeSafe System One)"
sidebarTitle: "Jev / Kev"
description: "Route TypeSafe System One decision requests through GoModel, to the hosted Jev API or a self-hosted Kev server."
icon: "scale-balanced"
keywords: ["Jev", "Kev", "TypeSafe", "System One", "decision model", "classification", "noul", "choice", "score", "self-hosted"]
---

[Jev](https://docs.typesafe.ai/introduction) is TypeSafe's System One model: a
decision model rather than a text generator. A request carries a `state` (the
text or record to evaluate) and a map of typed questions, and the answer is a
calibrated probability per question. [Kev](https://github.com/jaredpalmer/kev)
is a family of small open-weight models that implement the same API, so one
`jev` provider type covers both.

There are three question types:

| Type | Asks | Answer |
| --- | --- | --- |
| `noul` | A yes/no question | `noul`: the probability of yes |
| `choice` | Pick one option from a set you define | `choice`, plus `probabilities` and `confidence` |
| `score` | Rate against ordered levels | `score`, plus `legend`, `probabilities` and `confidence` |

The API is not OpenAI-compatible, and its answers have no chat equivalent, so
GoModel does not translate it: System One requests go through
[passthrough](/features/passthrough-api) at `/p/jev/...`, which is enabled by
default for this provider. Chat, `/responses`, and `/v1/embeddings` return
`invalid_request_error` for `jev` models.

## Configure

For the hosted API, the key is the whole setup:

```bash
JEV_API_KEY=ts-...
GOMODEL_MASTER_KEY=change-me
```

For a self-hosted Kev server, set the base URL instead. Kev has no
authentication of its own, so leave the key unset:

```bash
JEV_BASE_URL=http://host.docker.internal:8009
GOMODEL_MASTER_KEY=change-me
```

<Note>
The default base URL is `https://api.typesafe.ai`, the origin TypeSafe's SDKs
use; a trailing `/v1` is accepted and trimmed, so both spellings address the
same server. To run the hosted API and a local Kev side by side, register the
second under a suffixed name: `JEV_KEV_BASE_URL=...` creates provider
`jev-kev`, reached at `/p/jev-kev/...`.
</Note>

## Verify

```bash
curl -s http://localhost:8080/p/jev/v1/systemone \
-H "Authorization: Bearer change-me" \
-H "Content-Type: application/json" \
-d '{
"state": "Shoes arrived two weeks late and in the wrong size. Also I see two charges on my card.",
"model": "jev-latest",
"questions": {
"department": {"type": "choice", "instructions": "Which team should handle this?",
"criteria": {"returns": "Exchanges, refunds, wrong or damaged items",
"shipping": "Delivery status, delays, lost packages",
"billing": "Charges, invoices, payment problems"}},
"escalate": {"type": "noul", "instructions": "Does this need urgent human attention?"},
"frustration": {"type": "score", "instructions": "How frustrated is the customer?",
"criteria": ["Calm", "Frustrated", "Very angry"]}
}
}'
```

```json
{
"model": "jev-1.13.0",
"answers": {
"department": {"type": "choice", "choice": "returns", "confidence": 0.21,
"probabilities": {"returns": 0.47, "shipping": 0.28, "billing": 0.25}},
"escalate": {"type": "noul", "noul": 0.93},
"frustration": {"type": "score", "score": 1.44, "confidence": 0.78,
"legend": {"0": "Calm", "1": "Frustrated", "2": "Very angry"},
"probabilities": {"0": 0.00, "1": 0.56, "2": 0.44}}
},
"usage": {"input_tokens": 101, "output_tokens": 161}
}
```

The `/v1` segment is optional: `/p/jev/systemone` is the same route. Use
`kev-latest` as the model on a Kev server; it also answers to `jev-latest`.

## Using the TypeSafe SDKs

The SDKs send `POST {base_url}/v1/systemone`, so point them at the provider's
passthrough root and authenticate with your GoModel key:

<CodeGroup>
```python Python
from typesafe_sdk import Noul, TypeSafeClient

client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080/p/jev")
response = client.system_one(
state="I was charged twice. Please fix this ASAP.",
questions={"billing": Noul(instructions="Is this ticket about billing?")},
)
print(response.nouls["billing"].noul)
```

```typescript JavaScript
import { TypeSafeClient, noul } from "@typesafe-ai/sdk";

const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080/p/jev" });
const result = await client.systemOne({
state: "I was charged twice. Please fix this ASAP.",
questions: { billing: noul({ instructions: "Is this ticket about billing?" }) },
});
console.log(result.answers.billing.noul);
```
</CodeGroup>

The same works with `TYPESAFE_BASE_URL=http://localhost:8080/p/jev` and
`TYPESAFE_API_KEY=change-me` in the environment.

## Native routes

| Route | What it does |
| --- | --- |
| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions |
| `GET /p/jev/v1/models` | The names the `model` field accepts, in the upstream's own shape |
| `POST /p/jev/v1/systemone/permute` | Kev only: run one Choice question with several option orders |
| `POST /p/jev/v1/systemone/separate` | Kev only: run each question in its own forward pass |

Upstream errors keep their status code, with the provider's body carried in
the gateway error message: a malformed question comes back as TypeSafe's `422`
naming the offending field, and `429` or `529` mean back off and retry.
Comment thread
coderabbitai[bot] marked this conversation as resolved.

## Models, access control, and cost

`GET /v1/models` lists what the upstream reports, as `jev/jev-latest` and so
on. TypeSafe lists its aliases (`jev-latest`, `jev-preview`); a Kev server
lists its checkpoint (`kev-latest`) and the aliases it answers to. Versioned
IDs such as `jev-1.13.0` are accepted by the `model` field whether or not they
are listed. The models are categorized as utility models with no generation
mode, since there is no OpenAI endpoint to route them to.

Every System One request names its model, so the passthrough surface applies
the caller's [model allowlist](/features/users) to it like any other
request.

The response's `usage.input_tokens` and `usage.output_tokens` are recorded, so
System One calls appear in the usage API and dashboard under the model that
answered (`jev-1.13.0`, or the Kev checkpoint). Jev is priced per input token
and is not in the upstream model catalog; declare its pricing on the provider
to have those rows costed, or set it in the
[pricing override editor](/features/cost-tracking):

```yaml
providers:
jev:
type: jev
api_key: "${JEV_API_KEY}"
models:
- id: "jev-1.13.0"
metadata:
pricing:
currency: USD
input_per_mtok: 0.042
output_per_mtok: 0
```

A local Kev server costs nothing per token, so it needs no pricing.
7 changes: 7 additions & 0 deletions docs/providers/overview.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,7 @@ support, not every individual model capability exposed by an upstream provider.
| Xiaomi MiMo | `XIAOMI_API_KEY` (`XIAOMI_BASE_URL` optional) | `mimo-v2.5-pro` | ✅ | ✅ | ❌ | ❌ | ❌ | ✅ | [Xiaomi MiMo](/providers/xiaomi) |
| ElevenLabs (voice only) | `ELEVENLABS_API_KEY` (`ELEVENLABS_BASE_URL` optional) | `eleven_multilingual_v2` | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | [ElevenLabs](/providers/elevenlabs) |
| audio.cpp (audio only) | `AUDIOCPP_BASE_URL` (`AUDIOCPP_API_KEY` optional) | `pocket-tts` | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | [audio.cpp](/providers/audiocpp) |
| Jev / Kev (System One only) | `JEV_API_KEY` (`JEV_BASE_URL` optional) | `jev-latest` | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | [Jev](/providers/jev) |
| OpenCode Go | `OPENCODE_GO_API_KEY` (`OPENCODE_GO_BASE_URL` optional) | `glm-5.1` | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | [OpenCode Go](/providers/opencode-go) |
| Kimi Code | `KIMICODE_API_KEY` | `kimi-for-coding` | ✅ | ✅ | ✅ | ❌ | ❌ | ✅ | [Kimi Code](/providers/kimicode) |
| Hetzner (experimental) | `HETZNER_API_KEY` (`HETZNER_BASE_URL` optional) | `Qwen/Qwen3.6-35B-A3B-FP8` | ✅ | ✅ | ❌ | ❌ | ❌ | ✅ | [Hetzner](/providers/hetzner) |
Expand Down Expand Up @@ -206,6 +207,12 @@ support, not every individual model capability exposed by an upstream provider.
its own. Audio only: `/v1/audio/speech` and `/v1/audio/transcriptions` route
to it, and its detail, alignment, and live-streaming routes are reachable
through passthrough. See [audio.cpp](/providers/audiocpp).
- **Jev / Kev** — TypeSafe's System One API is a decision API (state plus
typed questions in, calibrated probabilities out) with no OpenAI-compatible
surface, so it is reached only through passthrough at
`POST /p/jev/v1/systemone`. `JEV_API_KEY` configures the hosted API; for a
self-hosted Kev server, which speaks the same API without authentication,
set `JEV_BASE_URL` and leave the key unset. See [Jev](/providers/jev).
- **llama.cpp / LM Studio** — `LLAMACPP_BASE_URL` is required (llama-server's
default port collides with GoModel's own 8080, so there is no default);
`LLAMACPP_API_KEY` is optional. Do not register these servers as `ollama`,
Expand Down
27 changes: 25 additions & 2 deletions internal/core/semantic.go
Original file line number Diff line number Diff line change
Expand Up @@ -45,8 +45,13 @@ type PassthroughRouteInfo struct {
GenAIOperation string // standard GenAI operation, if this is an inference call
Stream bool // explicit streaming intent derived from the request body
StreamUncertain bool // bounded opaque-body inspection could not determine stream intent
AuditPath string
Model string
// ModelAmbiguous reports that the opaque body names a model more than once,
// so no single value can be authorized: the upstream's parser decides which
// one wins, and the gateway cannot know. Such a request is rejected rather
// than forwarded unchecked.
ModelAmbiguous bool
AuditPath string
Model string
}

type semanticCacheKey string
Expand Down Expand Up @@ -347,6 +352,24 @@ func applyBodyStreamHint(env *WhiteBoxPrompt, stream, uncertain bool) {
}
}

// MarkPassthroughModelAmbiguous records that the opaque body carries more than
// one top-level model field, so the request's model cannot be authorized. Any
// model hint already taken from the body is dropped with it: a first-match
// peek would have kept whichever value came first, which is not necessarily
// the one the upstream's parser uses.
func MarkPassthroughModelAmbiguous(env *WhiteBoxPrompt) {
if env == nil {
return
}
env.RouteHints.Model = ""
if passthrough := env.CachedPassthroughRouteInfo(); passthrough != nil {
cloned := *passthrough
cloned.ModelAmbiguous = true
cloned.Model = ""
CachePassthroughRouteInfo(env, &cloned)
}
}

// MarkPassthroughStreamUncertain records that bounded opaque-body inspection
// stopped before it could determine explicit streaming intent.
func MarkPassthroughStreamUncertain(env *WhiteBoxPrompt) {
Expand Down
25 changes: 25 additions & 0 deletions internal/core/semantic_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -311,3 +311,28 @@ func TestDeriveBatchRouteInfoFromTransport_MessagesBatches(t *testing.T) {
})
}
}

// A repeated model field leaves no value the gateway can authorize, so the
// marker drops the first-match hint along with recording the ambiguity, and
// a later body refresh keeps the ambiguity on the merged route info.
func TestMarkPassthroughModelAmbiguous_DropsModelAndSurvivesRefresh(t *testing.T) {
env := &WhiteBoxPrompt{}
CachePassthroughRouteInfo(env, &PassthroughRouteInfo{Provider: "jev"})
ApplyBodySelectorHints(env, "jev-latest", "", false)
require.Equal(t, "jev-latest", env.RouteHints.Model)

MarkPassthroughModelAmbiguous(env)

require.Empty(t, env.RouteHints.Model)
info := env.CachedPassthroughRouteInfo()
require.NotNil(t, info)
require.True(t, info.ModelAmbiguous)
require.Empty(t, info.Model)

snapshot := NewRequestSnapshot(http.MethodPost, "/p/jev/systemone", map[string]string{"provider": "jev", "endpoint": "systemone"}, nil, nil, "application/json", []byte(`{"model":"jev-latest","model":"jev-preview"}`), false, "", nil)
refreshed := RefreshWhiteBoxPrompt(snapshot, env)
require.NotNil(t, refreshed)
require.True(t, refreshed.CachedPassthroughRouteInfo().ModelAmbiguous)

MarkPassthroughModelAmbiguous(nil) // a missing envelope is ignored
}
Loading
Loading