From fa01081c52b476d9186736c86ea1d8acf5b999fa Mon Sep 17 00:00:00 2001 From: Benno Weinzierl Date: Sat, 19 Sep 2026 12:07:02 +0200 Subject: [PATCH 1/2] feat: enable custom openai endpoints for services like opper.ai --- README.md | 14 +++++++++ internal/llm/gollm/gollm.go | 4 +++ internal/llm/gollm/wire_test.go | 55 +++++++++++++++++++++++++++++++++ internal/model/config.go | 7 +++++ 4 files changed, 80 insertions(+) diff --git a/README.md b/README.md index 479f89d4..e4875791 100644 --- a/README.md +++ b/README.md @@ -420,6 +420,18 @@ llm: timeout: 5m ``` +```yaml +# OpenAI-compatible service, e.g. a gateway +llm: + provider: openai + model: glm-5.3-flash + endpoint: https://eu.gw.opper.ai/openai + api_keys: + openai: op-... + timeout: 5m + rate_limit_rps: 10 +``` + ```yaml # Local Ollama llm: @@ -438,6 +450,8 @@ llm: Remote providers (`anthropic`, `openai`) get a conservative rate limit applied automatically, biased below tier-1 ceilings so bursty operations like `sdd summarize --all` don't trip 429s. Override with `rate_limit_rps` on higher tiers. +The `openai` provider talks to any service that implements the OpenAI chat completions protocol: set `endpoint` to its base URL and the request goes there instead of to OpenAI. The same URL works with or without a trailing `/v1`. Since the automatic rate limit assumes OpenAI's tier-1 ceilings, set `rate_limit_rps` explicitly for a service that allows more. + ### Embedding provider (vector search) Two providers supported: `openai` and `ollama`. Configuring an embedding provider activates `sdd search --query` (semantic mode) and hybrid retrieval. diff --git a/internal/llm/gollm/gollm.go b/internal/llm/gollm/gollm.go index 858de336..61c4fbb1 100644 --- a/internal/llm/gollm/gollm.go +++ b/internal/llm/gollm/gollm.go @@ -87,6 +87,10 @@ func NewRunner(cfg model.LLMConfig) (*Runner, error) { } } + if cfg.Provider == "openai" && cfg.Endpoint != "" { + opts = append(opts, upstream.SetOpenAIEndpoint(cfg.Endpoint)) + } + // Enable Anthropic prompt caching — sends the anthropic-beta header so // cache_control blocks on system prompts are honored server-side. useCache := cfg.Provider == "anthropic" diff --git a/internal/llm/gollm/wire_test.go b/internal/llm/gollm/wire_test.go index 110464f1..06b4dc79 100644 --- a/internal/llm/gollm/wire_test.go +++ b/internal/llm/gollm/wire_test.go @@ -78,6 +78,61 @@ func TestOllamaRequestWire(t *testing.T) { } } +// The openai provider is the one whose target is configurable, so where a +// request actually lands is a property worth asserting on the bytes: a base URL +// only serves an OpenAI-compatible gateway if the path is appended to it. +func TestOpenAIRequestWire(t *testing.T) { + var mu sync.Mutex + var body map[string]any + var path string + + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + raw, _ := io.ReadAll(r.Body) + mu.Lock() + path = r.URL.Path + _ = json.Unmarshal(raw, &body) + mu.Unlock() + _, _ = w.Write([]byte(`{"choices":[{"message":{"content":"SERVED"}}],"usage":{"prompt_tokens":33,"completion_tokens":44}}`)) + })) + defer srv.Close() + + runner, err := gollmrunner.NewRunner(model.LLMConfig{ + Provider: "openai", Model: "m", Endpoint: srv.URL, + APIKeys: map[string]string{"openai": "op-notarealkeyaaaaaaaaaa"}, + }) + if err != nil { + t.Fatalf("NewRunner: %v", err) + } + res, err := runner.Run(context.Background(), llm.Request{ + SystemPrompt: "SYSTEM-BLOCK-MARKER", + UserPrompt: "USER-BLOCK-MARKER", + }) + if err != nil { + t.Fatalf("Run: %v", err) + } + + mu.Lock() + defer mu.Unlock() + t.Logf("request keys: %v", keysOf(body)) + + if path != "/v1/chat/completions" { + t.Errorf("request path = %q, want /v1/chat/completions", path) + } + // A key OpenAI itself would reject must still reach a gateway that issued it. + sent, _ := json.Marshal(body["messages"]) + for _, marker := range []string{"SYSTEM-BLOCK-MARKER", "USER-BLOCK-MARKER"} { + if !strings.Contains(string(sent), marker) { + t.Errorf("%s missing from the messages sent: %s", marker, sent) + } + } + if res.Text != "SERVED" { + t.Errorf("text = %q, want SERVED", res.Text) + } + if res.Usage.InputTokens != 33 || res.Usage.OutputTokens != 44 { + t.Errorf("usage not parsed: %+v", res.Usage) + } +} + func keysOf(m map[string]any) []string { out := make([]string, 0, len(m)) for k := range m { diff --git a/internal/model/config.go b/internal/model/config.go index 80954e87..bd22729f 100644 --- a/internal/model/config.go +++ b/internal/model/config.go @@ -243,6 +243,10 @@ type LLMConfig struct { // Concurrency bounds the worker pool for batch operations. Zero means // "use DefaultLLMConcurrency". Concurrency int `yaml:"concurrency,omitempty"` + // Endpoint overrides the OpenAI-compatible base URL for the openai + // provider. Empty defaults to the OpenAI public endpoint. Useful for a + // gateway or a self-hosted service that implements the same protocol. + Endpoint string `yaml:"endpoint,omitempty"` // OllamaEndpoint overrides the default Ollama URL for the gollm adapter. OllamaEndpoint string `yaml:"ollama_endpoint,omitempty"` // APIKeys maps provider name to API key. Never belongs in the @@ -449,6 +453,9 @@ func mergeLLMConfig(base, overlay LLMConfig) LLMConfig { if overlay.Concurrency != 0 { out.Concurrency = overlay.Concurrency } + if overlay.Endpoint != "" { + out.Endpoint = overlay.Endpoint + } if overlay.OllamaEndpoint != "" { out.OllamaEndpoint = overlay.OllamaEndpoint } From 5eeb2c336c24c1489c6547d1b3bd66c1b8c7fcf8 Mon Sep 17 00:00:00 2001 From: Benno Weinzierl Date: Sat, 19 Sep 2026 17:36:51 +0200 Subject: [PATCH 2/2] chore: pin new compatible version of gollm --- go.mod | 2 +- go.sum | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/go.mod b/go.mod index 81c1e092..c4e4e3b2 100644 --- a/go.mod +++ b/go.mod @@ -92,4 +92,4 @@ require ( // source and honor replace; only `go install pkg@version` (which sdd does not // use) would break. Remove this directive once the change lands upstream and // bump the require above to the released teilomillet/gollm version. -replace github.com/teilomillet/gollm => github.com/networkteam/gollm v0.0.0-20260831213159-f6f84ac83bb5 +replace github.com/teilomillet/gollm => github.com/networkteam/gollm v0.0.0-20260919121334-aa9e2d1faebc diff --git a/go.sum b/go.sum index bd506718..95ef0c47 100644 --- a/go.sum +++ b/go.sum @@ -107,8 +107,8 @@ github.com/modelcontextprotocol/go-sdk v1.7.0 h1:yqjY2dsbKAC0LSuWZVBMrHgiG8ukXv6 github.com/modelcontextprotocol/go-sdk v1.7.0/go.mod h1:dL7u98E/zjJTGzEq+j30jQ8K2k1mb6LeAH4inEcSGts= github.com/muesli/cancelreader v0.2.2 h1:3I4Kt4BQjOR54NavqnDogx/MIoWBFa0StPA8ELUXHmA= github.com/muesli/cancelreader v0.2.2/go.mod h1:3XuTXfFS2VjM+HTLZY9Ak0l6eUKfijIfMUZ4EgX0QYo= -github.com/networkteam/gollm v0.0.0-20260831213159-f6f84ac83bb5 h1:I6UUByvzpT4ExPlk/7RAtI+PnuF2rKEeQMYmJ+FeBXY= -github.com/networkteam/gollm v0.0.0-20260831213159-f6f84ac83bb5/go.mod h1:ba4KWPPrYoc7PbPkcjTD6uMi2s5yXVWVsE5uzP+mvCg= +github.com/networkteam/gollm v0.0.0-20260919121334-aa9e2d1faebc h1:3SID2v1Qy85u5mchX+wlNUctJLa5NjRWMfj8HkeXcAo= +github.com/networkteam/gollm v0.0.0-20260919121334-aa9e2d1faebc/go.mod h1:ba4KWPPrYoc7PbPkcjTD6uMi2s5yXVWVsE5uzP+mvCg= github.com/networkteam/slogutils v0.4.0 h1:tf6Gf//H/a3EUmNszWEw8fhQkscrDdKt7HEdb2wBUhM= github.com/networkteam/slogutils v0.4.0/go.mod h1:j6siLHGfAkUpfhNUXw43EUdZv/HBaGd/RW7ToPwiJZo= github.com/philippgille/chromem-go v0.7.0 h1:4jfvfyKymjKNfGxBUhHUcj1kp7B17NL/I1P+vGh1RvY=