Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,7 @@ by another machine on your network, or by a provider.

- **Third-Party Providers** — integrate OpenAI, DeepSeek, MiMo, Kimi, BigModel, Qianfan, MiniMax, OpenRouter, and any OpenAI-compatible API
- **Coding Agents** — one-click config for Claude Code, Codex, Pi, OpenCode, and Open Code Review
- **Context compression** — optionally shrinks the tool output inside agent requests before any model sees it (Settings → Context compression). Safe mode removes only redundancy and keeps every distinct line; aggressive mode also samples long logs, arrays and search results. File reads and source code are never changed. On captured Claude Code / Codex traffic tool output drops 6% (safe) to 17% (aggressive), search results 24–56%; savings per request show in Observability
- **AI Applications** — one-click setup for Claude Code, OpenCode, Open Code Review, Codex, Codex App, ZCode, Pi, OpenClaw, CSGClaw, Dify, and AnythingLLM

### Dataset Support
Expand Down
36 changes: 36 additions & 0 deletions internal/config/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -150,6 +150,14 @@ type InferenceConfig struct {
// global default decide how many slots a load gets.
LlamaNumParallel int `json:"llama_num_parallel,omitempty"`

// ContextCompression is how far the gateway shrinks the tool output
// inside agent requests before a model sees them: ContextCompressionOff
// (the default when empty), ContextCompressionSafe or
// ContextCompressionAggressive. It is one process-wide policy because the
// same agent traffic reaches local, cluster and provider models alike, and
// no existing setting describes request rewriting.
ContextCompression string `json:"context_compression,omitempty"`

// Models holds the per-model load options keyed by model ID. They live in
// the app config rather than in the model directory so that re-downloading
// a model keeps its settings.
Expand Down Expand Up @@ -507,6 +515,34 @@ func Load() (*Config, error) {
return globalConfig, loadErr
}

const (
ContextCompressionOff = "off"
ContextCompressionSafe = "safe"
ContextCompressionAggressive = "aggressive"
)

// NormalizeContextCompression maps a stored or requested mode to one of the
// ContextCompression constants; anything unknown means off.
func NormalizeContextCompression(value string) string {
switch strings.ToLower(strings.TrimSpace(value)) {
case ContextCompressionSafe:
return ContextCompressionSafe
case ContextCompressionAggressive:
return ContextCompressionAggressive
default:
return ContextCompressionOff
}
}

// IsContextCompressionMode reports whether value names a mode.
func IsContextCompressionMode(value string) bool {
switch strings.ToLower(strings.TrimSpace(value)) {
case ContextCompressionOff, ContextCompressionSafe, ContextCompressionAggressive:
return true
}
return false
}

func NormalizeMarketplaceModelSource(value string) string {
switch strings.ToLower(strings.TrimSpace(value)) {
case "huggingface":
Expand Down
43 changes: 43 additions & 0 deletions internal/config/config_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -323,6 +323,49 @@ func TestInferenceConfigPersistsAcrossSaveAndLoad(t *testing.T) {
}
}

func TestContextCompressionPersistsAndDefaultsOff(t *testing.T) {
var legacy Config
if err := json.Unmarshal([]byte(`{"inference":{"llama_num_parallel":2}}`), &legacy); err != nil {
t.Fatal(err)
}
if got := NormalizeContextCompression(legacy.Inference.ContextCompression); got != ContextCompressionOff {
t.Fatalf("legacy config mode = %q, want off", got)
}

home := t.TempDir()
t.Setenv("HOME", home)
t.Setenv("USERPROFILE", home)
clearCloudServiceEnv(t)
Reset()
t.Cleanup(Reset)

cfg, err := Load()
if err != nil {
t.Fatal(err)
}
cfg.Inference.ContextCompression = ContextCompressionAggressive
if err := Save(cfg); err != nil {
t.Fatal(err)
}
Reset()
loaded, err := Load()
if err != nil {
t.Fatal(err)
}
if loaded.Inference.ContextCompression != ContextCompressionAggressive {
t.Fatalf("mode after reload = %q", loaded.Inference.ContextCompression)
}

for value, want := range map[string]string{"": "off", "OFF": "off", " Safe ": "safe", "aggressive": "aggressive", "max": "off"} {
if got := NormalizeContextCompression(value); got != want {
t.Errorf("NormalizeContextCompression(%q) = %q, want %q", value, got, want)
}
}
if IsContextCompressionMode("max") || !IsContextCompressionMode("Safe") {
t.Fatal("IsContextCompressionMode accepted or rejected the wrong value")
}
}

func TestMarketplaceModelSourcePersistsAndDefaults(t *testing.T) {
home := t.TempDir()
t.Setenv("HOME", home)
Expand Down
84 changes: 84 additions & 0 deletions internal/ctxcompress/corpus_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,84 @@
package ctxcompress

import (
"bufio"
"encoding/json"
"os"
"path/filepath"
"testing"
)

// TestCorpus replays captured traffic through both modes and writes the
// results next to the corpus for token counting. It runs only when
// CTXCOMPRESS_CORPUS_DIR names a directory holding corpus_blocks.jsonl
// ({"tool","text"} per line) and corpus_requests.jsonl ({"id","protocol",
// "body"} per line), such as one exported from the observability store.
func TestCorpus(t *testing.T) {
dir := os.Getenv("CTXCOMPRESS_CORPUS_DIR")
if dir == "" {
t.Skip("CTXCOMPRESS_CORPUS_DIR not set")
}
modes := map[string]Options{"safe": {}, "aggressive": {Sample: true}}

blocksOut, err := os.Create(filepath.Join(dir, "result_blocks.jsonl"))
if err != nil {
t.Fatal(err)
}
defer blocksOut.Close()
encoder := json.NewEncoder(blocksOut)
forEachLine(t, filepath.Join(dir, "corpus_blocks.jsonl"), func(line []byte) {
var block struct{ Tool, Text string }
if err := json.Unmarshal(line, &block); err != nil {
t.Fatal(err)
}
for mode, opts := range modes {
out, kind := Text(block.Text, block.Tool, opts)
if again, _ := Text(out, block.Tool, opts); again != out {
t.Errorf("%s: not idempotent for a %s block from %s (%d → %d → %d bytes)", mode, kind, block.Tool, len(block.Text), len(out), len(again))
}
_ = encoder.Encode(map[string]any{"mode": mode, "tool": block.Tool, "kind": kind, "before": block.Text, "after": out})
}
})

requestsOut, err := os.Create(filepath.Join(dir, "result_requests.jsonl"))
if err != nil {
t.Fatal(err)
}
defer requestsOut.Close()
encoder = json.NewEncoder(requestsOut)
forEachLine(t, filepath.Join(dir, "corpus_requests.jsonl"), func(line []byte) {
var request struct {
ID string
Protocol Protocol
Body string
}
if err := json.Unmarshal(line, &request); err != nil {
t.Fatal(err)
}
for mode, opts := range modes {
result, err := CompressRequest(request.Protocol, []byte(request.Body), opts)
if err != nil {
t.Errorf("%s: %v", request.ID, err)
continue
}
_ = encoder.Encode(map[string]any{"mode": mode, "id": request.ID, "protocol": request.Protocol, "after": string(result.Body), "stats": result.Stats})
}
})
}

func forEachLine(t *testing.T, path string, fn func([]byte)) {
t.Helper()
file, err := os.Open(path)
if err != nil {
t.Fatal(err)
}
defer file.Close()
scanner := bufio.NewScanner(file)
scanner.Buffer(make([]byte, 0, 1<<20), 64<<20)
for scanner.Scan() {
fn(scanner.Bytes())
}
if err := scanner.Err(); err != nil {
t.Fatal(err)
}
}
Loading
Loading