Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -420,6 +420,18 @@ llm:
timeout: 5m
```

```yaml
# OpenAI-compatible service, e.g. a gateway
llm:
provider: openai
model: glm-5.3-flash
endpoint: https://eu.gw.opper.ai/openai
api_keys:
openai: op-...
timeout: 5m
rate_limit_rps: 10
```

```yaml
# Local Ollama
llm:
Expand All @@ -438,6 +450,8 @@ llm:

Remote providers (`anthropic`, `openai`) get a conservative rate limit applied automatically, biased below tier-1 ceilings so bursty operations like `sdd summarize --all` don't trip 429s. Override with `rate_limit_rps` on higher tiers.

The `openai` provider talks to any service that implements the OpenAI chat completions protocol: set `endpoint` to its base URL and the request goes there instead of to OpenAI. The same URL works with or without a trailing `/v1`. Since the automatic rate limit assumes OpenAI's tier-1 ceilings, set `rate_limit_rps` explicitly for a service that allows more.

### Embedding provider (vector search)

Two providers supported: `openai` and `ollama`. Configuring an embedding provider activates `sdd search --query` (semantic mode) and hybrid retrieval.
Expand Down
2 changes: 1 addition & 1 deletion go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -92,4 +92,4 @@ require (
// source and honor replace; only `go install pkg@version` (which sdd does not
// use) would break. Remove this directive once the change lands upstream and
// bump the require above to the released teilomillet/gollm version.
replace github.com/teilomillet/gollm => github.com/networkteam/gollm v0.0.0-20260831213159-f6f84ac83bb5
replace github.com/teilomillet/gollm => github.com/networkteam/gollm v0.0.0-20260919121334-aa9e2d1faebc
4 changes: 2 additions & 2 deletions go.sum
Original file line number Diff line number Diff line change
Expand Up @@ -107,8 +107,8 @@ github.com/modelcontextprotocol/go-sdk v1.7.0 h1:yqjY2dsbKAC0LSuWZVBMrHgiG8ukXv6
github.com/modelcontextprotocol/go-sdk v1.7.0/go.mod h1:dL7u98E/zjJTGzEq+j30jQ8K2k1mb6LeAH4inEcSGts=
github.com/muesli/cancelreader v0.2.2 h1:3I4Kt4BQjOR54NavqnDogx/MIoWBFa0StPA8ELUXHmA=
github.com/muesli/cancelreader v0.2.2/go.mod h1:3XuTXfFS2VjM+HTLZY9Ak0l6eUKfijIfMUZ4EgX0QYo=
github.com/networkteam/gollm v0.0.0-20260831213159-f6f84ac83bb5 h1:I6UUByvzpT4ExPlk/7RAtI+PnuF2rKEeQMYmJ+FeBXY=
github.com/networkteam/gollm v0.0.0-20260831213159-f6f84ac83bb5/go.mod h1:ba4KWPPrYoc7PbPkcjTD6uMi2s5yXVWVsE5uzP+mvCg=
github.com/networkteam/gollm v0.0.0-20260919121334-aa9e2d1faebc h1:3SID2v1Qy85u5mchX+wlNUctJLa5NjRWMfj8HkeXcAo=
github.com/networkteam/gollm v0.0.0-20260919121334-aa9e2d1faebc/go.mod h1:ba4KWPPrYoc7PbPkcjTD6uMi2s5yXVWVsE5uzP+mvCg=
github.com/networkteam/slogutils v0.4.0 h1:tf6Gf//H/a3EUmNszWEw8fhQkscrDdKt7HEdb2wBUhM=
github.com/networkteam/slogutils v0.4.0/go.mod h1:j6siLHGfAkUpfhNUXw43EUdZv/HBaGd/RW7ToPwiJZo=
github.com/philippgille/chromem-go v0.7.0 h1:4jfvfyKymjKNfGxBUhHUcj1kp7B17NL/I1P+vGh1RvY=
Expand Down
4 changes: 4 additions & 0 deletions internal/llm/gollm/gollm.go
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,10 @@ func NewRunner(cfg model.LLMConfig) (*Runner, error) {
}
}

if cfg.Provider == "openai" && cfg.Endpoint != "" {
opts = append(opts, upstream.SetOpenAIEndpoint(cfg.Endpoint))

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Pinned dependency lacks new API

This calls upstream.SetOpenAIEndpoint, but the module replacement remains pinned to the earlier immutable f6f84ac83bb5 revision. The PR description says networkteam/gollm#1 is required, but merging that upstream PR cannot alter this pinned version. Unless go.mod and go.sum are updated to a revision containing the API, the project will fail to compile with an undefined symbol.

Prompt To Fix With AI
This is a comment left during a code review.
Path: internal/llm/gollm/gollm.go
Line: 91

Comment:
**Pinned dependency lacks new API**

This calls `upstream.SetOpenAIEndpoint`, but the module replacement remains pinned to the earlier immutable `f6f84ac83bb5` revision. The PR description says `networkteam/gollm#1` is required, but merging that upstream PR cannot alter this pinned version. Unless `go.mod` and `go.sum` are updated to a revision containing the API, the project will fail to compile with an undefined symbol.

---

For each issue above, determine whether it is valid and should be fixed. If so, fix it directly.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

updated, please review again

}

// Enable Anthropic prompt caching — sends the anthropic-beta header so
// cache_control blocks on system prompts are honored server-side.
useCache := cfg.Provider == "anthropic"
Expand Down
55 changes: 55 additions & 0 deletions internal/llm/gollm/wire_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -78,6 +78,61 @@ func TestOllamaRequestWire(t *testing.T) {
}
}

// The openai provider is the one whose target is configurable, so where a
// request actually lands is a property worth asserting on the bytes: a base URL
// only serves an OpenAI-compatible gateway if the path is appended to it.
func TestOpenAIRequestWire(t *testing.T) {
var mu sync.Mutex
var body map[string]any
var path string

srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
raw, _ := io.ReadAll(r.Body)
mu.Lock()
path = r.URL.Path
_ = json.Unmarshal(raw, &body)
mu.Unlock()
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"SERVED"}}],"usage":{"prompt_tokens":33,"completion_tokens":44}}`))
}))
defer srv.Close()

runner, err := gollmrunner.NewRunner(model.LLMConfig{
Provider: "openai", Model: "m", Endpoint: srv.URL,
APIKeys: map[string]string{"openai": "op-notarealkeyaaaaaaaaaa"},
})
if err != nil {
t.Fatalf("NewRunner: %v", err)
}
res, err := runner.Run(context.Background(), llm.Request{
SystemPrompt: "SYSTEM-BLOCK-MARKER",
UserPrompt: "USER-BLOCK-MARKER",
})
if err != nil {
t.Fatalf("Run: %v", err)
}

mu.Lock()
defer mu.Unlock()
t.Logf("request keys: %v", keysOf(body))

if path != "/v1/chat/completions" {
t.Errorf("request path = %q, want /v1/chat/completions", path)
}
// A key OpenAI itself would reject must still reach a gateway that issued it.
sent, _ := json.Marshal(body["messages"])
for _, marker := range []string{"SYSTEM-BLOCK-MARKER", "USER-BLOCK-MARKER"} {
if !strings.Contains(string(sent), marker) {
t.Errorf("%s missing from the messages sent: %s", marker, sent)
}
}
if res.Text != "SERVED" {
t.Errorf("text = %q, want SERVED", res.Text)
}
if res.Usage.InputTokens != 33 || res.Usage.OutputTokens != 44 {
t.Errorf("usage not parsed: %+v", res.Usage)
}
}

func keysOf(m map[string]any) []string {
out := make([]string, 0, len(m))
for k := range m {
Expand Down
7 changes: 7 additions & 0 deletions internal/model/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -243,6 +243,10 @@ type LLMConfig struct {
// Concurrency bounds the worker pool for batch operations. Zero means
// "use DefaultLLMConcurrency".
Concurrency int `yaml:"concurrency,omitempty"`
// Endpoint overrides the OpenAI-compatible base URL for the openai
// provider. Empty defaults to the OpenAI public endpoint. Useful for a
// gateway or a self-hosted service that implements the same protocol.
Endpoint string `yaml:"endpoint,omitempty"`
// OllamaEndpoint overrides the default Ollama URL for the gollm adapter.
OllamaEndpoint string `yaml:"ollama_endpoint,omitempty"`
// APIKeys maps provider name to API key. Never belongs in the
Expand Down Expand Up @@ -449,6 +453,9 @@ func mergeLLMConfig(base, overlay LLMConfig) LLMConfig {
if overlay.Concurrency != 0 {
out.Concurrency = overlay.Concurrency
}
if overlay.Endpoint != "" {
out.Endpoint = overlay.Endpoint
}
if overlay.OllamaEndpoint != "" {
out.OllamaEndpoint = overlay.OllamaEndpoint
}
Expand Down
Loading