Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -101,6 +101,13 @@ plugins:
protocol: "messages" # "chat-completions" | "messages" | "responses"
endpoint: "/v1/messages" # must start with /

# Optional per-model capability fixes (takes priority over catalog and models.dev)
model-metadata-overrides:
"gpt-5.6-luna":
thinking:
zero-allowed: true
levels: [none, low, medium, high, xhigh, max]

# Execution settings
request-timeout: "5m" # upstream request timeout (default: "5m")
max-response-bytes: 67108864 # max non-streaming response body size in bytes (default: 64 MiB)
Expand All @@ -109,6 +116,8 @@ plugins:

### Configuration Options

The configured `catalog-url` alone determines which models are routable. On each successful discovery, the plugin fills missing limits, modalities, and unambiguous reasoning metadata from the `opencode-go` provider in [models.dev](https://models.dev/); if that request fails, the last good metadata is retained. Field priority is: `model-metadata-overrides` > catalog > models.dev > plugin defaults. `toggle` reasoning options are not inferred as effort levels.

| Option | Type | Default | Description |
|---|---|---|---|
| `api-keys` | `[]object` | *(Required)* | List of API keys (`- value: "..."`). Supports `${ENV_VAR}` expansion. Duplicates and empty values are rejected. |
Expand All @@ -122,6 +131,7 @@ plugins:
| `protocols.messages` | `bool` | `true` | Protocol switch for Messages endpoints. |
| `protocols.responses` | `bool` | `true` | Protocol switch for Responses endpoints. |
| `route-overrides` | `map` | `{}` | Map of model ID to `{ protocol: "...", endpoint: "..." }` overriding built-in family routing. Valid protocols: `chat-completions`, `messages`, `responses`. |
| `model-metadata-overrides` | `map` | `{}` | Per-model `thinking` (`min`, `max`, `zero-allowed`, `dynamic-allowed`, `levels`), `context-limit`, `output-limit`, `input-modes`, and `output-modes`. Each supplied field overrides the catalog and models.dev; unspecified fields retain their current value. |
| `request-timeout` | `duration` | `5m` | Upstream HTTP request timeout. Must be positive. |
| `max-response-bytes` | `int64` | `67108864` (64 MiB) | Maximum non-streaming response body size in bytes. |
| `allow-http` | `bool` | `false` | When `true`, permits `http://` scheme in `base-url` / `catalog-url` for local testing. |
Expand Down
15 changes: 13 additions & 2 deletions internal/adapter/chatcompletions/request.go
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ func AuthHeaders(key string) http.Header {
func BuildRequest(upstreamModel, sourceFormat string, sourceBody []byte, ts *pluginapi.ThinkingSupport) ([]byte, *errclass.Error) {
switch sourceFormat {
case "openai":
return buildOpenAIRequest(upstreamModel, sourceBody)
return buildOpenAIRequest(upstreamModel, sourceBody, ts)
case "claude":
return claudeToChat(upstreamModel, sourceBody, ts)
case "openai-response":
Expand All @@ -52,14 +52,25 @@ func BuildRequest(upstreamModel, sourceFormat string, sourceBody []byte, ts *plu
// body to upstreamModel, normalizes role:"developer" messages to role:"system",
// and strips any malformed top-level thinking object. DeepSeek models fail if
// thinking lacks a valid string type field or if messages contain role:"developer".
func buildOpenAIRequest(upstreamModel string, body []byte) ([]byte, *errclass.Error) {
func buildOpenAIRequest(upstreamModel string, body []byte, ts *pluginapi.ThinkingSupport) ([]byte, *errclass.Error) {
var req map[string]json.RawMessage
if err := json.Unmarshal(body, &req); err != nil {
return nil, errclass.Translation("malformed openai request JSON: " + err.Error())
}
if req == nil {
return nil, errclass.Translation("malformed request body: JSON null is not a valid request")
}
if raw := req["reasoning_effort"]; len(raw) > 0 && string(raw) != "null" {
var effort string
if json.Unmarshal(raw, &effort) != nil {
return nil, errclass.Translation("reasoning_effort must be a string")
}
if effort != "" {
if eErr := thinking.ValidateEffort(effort, ts); eErr != nil {
return nil, eErr
}
}
}
rawThinking, hasThinking := req["thinking"]
validThinking := hasThinking && isValidThinking(rawThinking)
if hasThinking && !validThinking {
Expand Down
28 changes: 27 additions & 1 deletion internal/adapter/messages/request.go
Original file line number Diff line number Diff line change
Expand Up @@ -56,7 +56,33 @@ type messagesRequest struct {
func BuildRequest(upstreamModel string, sourceFormat string, sourceBody []byte, ts *pluginapi.ThinkingSupport) ([]byte, *errclass.Error) {
switch sourceFormat {
case "claude":
return shared.RewriteModelID(upstreamModel, sourceBody, "claude")
body, eErr := shared.RewriteModelID(upstreamModel, sourceBody, "claude")
if eErr != nil {
return nil, eErr
}
var native struct {
Thinking *shared.ClaudeThinking `json:"thinking"`
}
if json.Unmarshal(body, &native) != nil {
return nil, errclass.Translation("invalid thinking control")
}
if native.Thinking != nil {
switch native.Thinking.Type {
case "disabled":
if eErr := thinking.ValidateEffort("none", ts); eErr != nil {
return nil, eErr
}
case "enabled":
budget := native.Thinking.BudgetTokens
supported := thinking.SupportedLevels(ts)
if budget <= 0 || ts != nil && (ts.Min > 0 && budget < int64(ts.Min) ||
ts.Max > 0 && budget > int64(ts.Max) ||
len(supported) == 1 && supported[0] == "none") {
return nil, &errclass.Error{Class: errclass.ClassUnsupported, Message: "thinking budget_tokens is not supported for this model"}
}
}
}
return body, nil
case "openai":
return fromChatCompletions(upstreamModel, sourceBody, ts)
case "openai-response":
Expand Down
32 changes: 32 additions & 0 deletions internal/adapter/messages/request_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,38 @@ func TestBuildRequestClaudeMalformed(t *testing.T) {
}
}

func TestNativeClaudeThinkingCapability(t *testing.T) {
bounded := &pluginapi.ThinkingSupport{Min: 1024, Max: 4096, Levels: []string{"low", "medium"}}
cases := []struct {
name string
thinking string
ts *pluginapi.ThinkingSupport
wantErr errclass.Class
}{
{"within bounds", `{"type":"enabled","budget_tokens":2048}`, bounded, ""},
{"below minimum", `{"type":"enabled","budget_tokens":512}`, bounded, errclass.ClassUnsupported},
{"above maximum", `{"type":"enabled","budget_tokens":8192}`, bounded, errclass.ClassUnsupported},
{"zero budget", `{"type":"enabled","budget_tokens":0}`, bounded, errclass.ClassUnsupported},
{"missing budget", `{"type":"enabled"}`, bounded, errclass.ClassUnsupported},
{"no enabled level", `{"type":"enabled","budget_tokens":2048}`, &pluginapi.ThinkingSupport{Levels: []string{"none"}}, errclass.ClassUnsupported},
{"unknown capability fallback", `{"type":"enabled","budget_tokens":2048}`, nil, ""},
{"malformed budget", `{"type":"enabled","budget_tokens":"bad"}`, bounded, errclass.ClassTranslation},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
body := []byte(`{"model":"m","messages":[],"max_tokens":10000,"thinking":` + tc.thinking + `}`)
out, eErr := BuildRequest("m", "claude", body, tc.ts)
if tc.wantErr == "" {
if eErr != nil || string(out) != string(body) {
t.Fatalf("native passthrough changed: %s, %v", out, eErr)
}
} else if eErr == nil || eErr.Class != tc.wantErr {
t.Fatalf("error = %v, want %s", eErr, tc.wantErr)
}
})
}
}

func chatReq(t *testing.T, body string) (map[string]any, *errclass.Error) {
return chatReqTS(t, nil, body)
}
Expand Down
41 changes: 35 additions & 6 deletions internal/adapter/responses/request.go
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,6 @@ package responses
import (
"encoding/json"
"fmt"
"slices"
"strings"

"github.com/router-for-me/CLIProxyAPI/v7/sdk/pluginapi"
Expand All @@ -38,7 +37,40 @@ var EndpointPath = catalog.RouteResponses.EndpointPath()
func BuildRequest(upstreamModel string, sourceFormat string, sourceBody []byte, ts *pluginapi.ThinkingSupport) ([]byte, *errclass.Error) {
switch sourceFormat {
case "openai-response":
return shared.RewriteModelID(upstreamModel, sourceBody, "openai-response")
body, eErr := shared.RewriteModelID(upstreamModel, sourceBody, "openai-response")
if eErr != nil {
return nil, eErr
}
var req map[string]json.RawMessage
_ = json.Unmarshal(body, &req)
var reasoning map[string]json.RawMessage
if len(req["reasoning"]) == 0 || string(req["reasoning"]) == "null" {
return body, nil
}
if json.Unmarshal(req["reasoning"], &reasoning) != nil || reasoning == nil {
return nil, errclass.Translation("reasoning must be an object")
}
if raw := reasoning["effort"]; len(raw) > 0 && string(raw) != "null" {
var effort string
if json.Unmarshal(raw, &effort) != nil {
return nil, errclass.Translation("reasoning.effort must be a string")
}
if effort != "" {
if eErr := thinking.ValidateEffort(effort, ts); eErr != nil {
return nil, eErr
}
}
if strings.EqualFold(strings.TrimSpace(effort), "auto") {
delete(reasoning, "effort")
if len(reasoning) == 0 {
delete(req, "reasoning")
} else {
req["reasoning"], _ = json.Marshal(reasoning)
}
body, _ = json.Marshal(req)
}
}
return body, nil
case "openai":
return fromChatCompletions(upstreamModel, sourceBody, ts)
case "claude":
Expand Down Expand Up @@ -276,10 +308,7 @@ func reasoningEffortFor(effort string, ts *pluginapi.ThinkingSupport) (string, b
case effort == "auto":
return "", false
case effort == "none":
if slices.Contains(thinking.SupportedLevels(ts), "none") {
return effort, true
}
return "", false
return effort, true // ValidateEffort admits it only when the model supports off.
default:
return effort, true
}
Expand Down
12 changes: 5 additions & 7 deletions internal/adapter/responses/request_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -683,16 +683,14 @@ func TestFromChatCompletionsEffortCapability(t *testing.T) {
}
}

// Sentinels are capability-gated: a "none" the model does not declare and
// the dynamic "auto" sentinel are omitted — Responses has no off-switch, so
// omission is the no-forced-reasoning policy (matches the Messages-target
// leg); declared levels forward.
func TestFromChatCompletionsEffortSentinelsOmitted(t *testing.T) {
// ZeroAllowed admits an off state even when Levels omits "none"; auto has
// no Responses wire value and is omitted.
func TestFromChatCompletionsEffortSentinels(t *testing.T) {
ts := &pluginapi.ThinkingSupport{ZeroAllowed: true, DynamicAllowed: true}
m := decodeReq(t, mustBuild(t, "m", "openai",
[]byte(`{"messages":[],"reasoning_effort":"none"}`), ts))
if _, has := m["reasoning"]; has {
t.Fatalf("none must omit reasoning: %v", m["reasoning"])
if r := m["reasoning"].(map[string]any); r["effort"] != "none" {
t.Fatalf("none must forward: %v", r)
}

m = decodeReq(t, mustBuild(t, "m", "openai",
Expand Down
82 changes: 82 additions & 0 deletions internal/adapter/responses_parity_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -8,12 +8,94 @@ import (
"strings"
"testing"

"github.com/router-for-me/CLIProxyAPI/v7/sdk/pluginapi"

"opencode-go-cliproxyapi/internal/adapter/chatcompletions"
"opencode-go-cliproxyapi/internal/adapter/messages"
"opencode-go-cliproxyapi/internal/adapter/responses"
"opencode-go-cliproxyapi/internal/errclass"
)

func TestReasoningEffortNativeAndTranslatedParity(t *testing.T) {
ts := &pluginapi.ThinkingSupport{Levels: []string{"minimal", "low", "medium", "high", "xhigh", "max"}, ZeroAllowed: true, DynamicAllowed: true}
for _, effort := range []string{"none", "minimal", "low", "medium", "high", "xhigh", "max", "auto"} {
t.Run(effort, func(t *testing.T) {
cc := []byte(`{"model":"m","messages":[],"reasoning_effort":"` + effort + `"}`)
resp := []byte(`{"model":"m","input":[],"reasoning":{"effort":"` + effort + `"}}`)
for _, tc := range []struct {
name string
build func(string, string, []byte, *pluginapi.ThinkingSupport) ([]byte, *errclass.Error)
format string
body []byte
field string
}{
{"native chat", chatcompletions.BuildRequest, "openai", cc, "reasoning_effort"},
{"translated chat", chatcompletions.BuildRequest, "openai-response", resp, "reasoning_effort"},
{"native responses", responses.BuildRequest, "openai-response", resp, "reasoning"},
{"translated responses", responses.BuildRequest, "openai", cc, "reasoning"},
} {
out, eErr := tc.build("m", tc.format, tc.body, ts)
if eErr != nil {
t.Fatalf("%s: %v", tc.name, eErr)
}
var wire map[string]any
if err := json.Unmarshal(out, &wire); err != nil {
t.Fatal(err)
}
if tc.field == "reasoning_effort" && wire[tc.field] != effort {
t.Errorf("%s: %s = %v", tc.name, tc.field, wire[tc.field])
}
if tc.field == "reasoning" {
if effort == "auto" && wire[tc.field] != nil {
t.Errorf("%s: auto must omit Responses reasoning: %v", tc.name, wire[tc.field])
} else if effort != "auto" {
got, ok := wire[tc.field].(map[string]any)
if !ok || got["effort"] != effort {
t.Errorf("%s: reasoning = %v", tc.name, wire[tc.field])
}
}
}
}
for _, source := range []struct {
format string
body []byte
}{{"openai", cc}, {"openai-response", resp}} {
out, eErr := messages.BuildRequest("m", source.format, source.body, ts)
if eErr != nil {
t.Fatalf("Messages from %s: %v", source.format, eErr)
}
var wire map[string]any
if err := json.Unmarshal(out, &wire); err != nil {
t.Fatal(err)
}
if (wire["thinking"] != nil) != (effort != "none" && effort != "auto") {
t.Errorf("Messages from %s: thinking = %v", source.format, wire["thinking"])
}
}
})
}
limited := &pluginapi.ThinkingSupport{Levels: []string{"high"}}
for _, body := range []struct {
format string
data []byte
build func(string, string, []byte, *pluginapi.ThinkingSupport) ([]byte, *errclass.Error)
}{
{"openai", []byte(`{"model":"m","reasoning_effort":"max"}`), chatcompletions.BuildRequest},
{"openai-response", []byte(`{"model":"m","reasoning":{"effort":"max"}}`), responses.BuildRequest},
} {
if _, eErr := body.build("m", body.format, body.data, limited); eErr == nil || eErr.Class != errclass.ClassUnsupported {
t.Errorf("native %s accepted unsupported max: %v", body.format, eErr)
}
}
claudeOff := []byte(`{"model":"m","max_tokens":1024,"messages":[],"thinking":{"type":"disabled"}}`)
if _, eErr := messages.BuildRequest("m", "claude", claudeOff, limited); eErr == nil || eErr.Class != errclass.ClassUnsupported {
t.Errorf("native Messages accepted unsupported reasoning off: %v", eErr)
}
if _, eErr := messages.BuildRequest("m", "claude", claudeOff, ts); eErr != nil {
t.Errorf("native Messages rejected supported reasoning off: %v", eErr)
}
}

type feeder interface {
Feed(chunk []byte) (events [][]byte, done bool, eErr *errclass.Error)
}
Expand Down
Loading