From 3ca9262d7e1b0e20df822f21d56d79e5ef53fcf9 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 10:44:19 +0200 Subject: [PATCH 01/11] chore(bench): add a local-metal vllm-on-tap environment Co-Authored-By: Claude Opus 5.5 (1M context) --- .gitignore | 2 ++ .vot/environments/local-metal.yaml | 5 +++++ 2 files changed, 7 insertions(+) create mode 100644 .vot/environments/local-metal.yaml diff --git a/.gitignore b/.gitignore index 7db74034..4349f7ac 100644 --- a/.gitignore +++ b/.gitignore @@ -223,3 +223,5 @@ __marimo__/ apm.lock.yaml # ...except the generated marketplace plugin, which ships its own MCP config !marketplace/oddyssey/.mcp.json +.vot/config.yaml +.vot/run/ diff --git a/.vot/environments/local-metal.yaml b/.vot/environments/local-metal.yaml new file mode 100644 index 00000000..cb329143 --- /dev/null +++ b/.vot/environments/local-metal.yaml @@ -0,0 +1,5 @@ +name: local-metal +stack: local-vllm-metal +otlp_endpoint: http://localhost:4317 +config: + port: 8000 From 123182f70215f118202841e1aa1ad746288f3640 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 10:48:01 +0200 Subject: [PATCH 02/11] chore(bench): add an azure-sweden vllm-on-tap environment Co-Authored-By: Claude Opus 5.5 (1M context) --- .vot/environments/azure-sweden.yaml | 5 +++++ 1 file changed, 5 insertions(+) create mode 100644 .vot/environments/azure-sweden.yaml diff --git a/.vot/environments/azure-sweden.yaml b/.vot/environments/azure-sweden.yaml new file mode 100644 index 00000000..988c4165 --- /dev/null +++ b/.vot/environments/azure-sweden.yaml @@ -0,0 +1,5 @@ +name: azure-sweden +stack: azure +config: + location: swedencentral + resource_group: rg-vot From ed2c8b544055a43bb169c0b12e4e889320a84021 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 11:06:50 +0200 Subject: [PATCH 03/11] chore(bench): add a gemma4-12b-qat-agent vllm-on-tap preset for A100 80 GB Co-Authored-By: Claude Opus 5.5 (1M context) --- .vot/presets/gemma4-12b-qat-agent.yaml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 .vot/presets/gemma4-12b-qat-agent.yaml diff --git a/.vot/presets/gemma4-12b-qat-agent.yaml b/.vot/presets/gemma4-12b-qat-agent.yaml new file mode 100644 index 00000000..f0ce25d3 --- /dev/null +++ b/.vot/presets/gemma4-12b-qat-agent.yaml @@ -0,0 +1,16 @@ +name: gemma4-12b-qat-agent +description: Gemma 4 12B QAT 4-bit (compressed-tensors, w4a16) tuned for coding-agent CLIs on one A100 80 GB - full 256k context, tool calling, text only; NVIDIA stacks only +model: google/gemma-4-12B-it-qat-w4a16-ct +served_model_name: gemma4-12b-agent +gpu_memory_gb: 48 +vllm_args: + --max-model-len: 262144 + --gpu-memory-utilization: 0.92 + --max-num-seqs: 8 + --max-num-batched-tokens: 16384 + --enable-prefix-caching: true + --language-model-only: true + --enable-auto-tool-choice: true + --tool-call-parser: gemma4 + --reasoning-parser: gemma4 +env: [] From 2506f132ea644714ec7cff847effec90d01d0d10 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 11:06:50 +0200 Subject: [PATCH 04/11] feat(bench): benchmark vllm-on-tap presets served locally in their own table Co-Authored-By: Claude Opus 5.5 (1M context) --- .claude/commands/launch-llms-benchmark.md | 157 +++++++++++++++++++++- .llms-benchmark/README.md | 32 +++++ 2 files changed, 184 insertions(+), 5 deletions(-) diff --git a/.claude/commands/launch-llms-benchmark.md b/.claude/commands/launch-llms-benchmark.md index a2ffb871..54afabd8 100644 --- a/.claude/commands/launch-llms-benchmark.md +++ b/.claude/commands/launch-llms-benchmark.md @@ -1,12 +1,20 @@ --- -description: Benchmark one LLM on the llms-benchmark demo stack - drive it through a coding-agent CLI (opencode, claude or copilot) on the stored scenario, grade the observation report it produced, and propose its row of the results table -argument-hint: " [effort, default medium]" +description: Benchmark one LLM on the llms-benchmark demo stack - drive it through a coding-agent CLI (opencode, claude or copilot) or a vllm-on-tap preset served for the run, on the stored scenario, grade the observation report it produced, and propose its row of the results table +argument-hint: " [effort, default medium] | local [effort, default medium]" --- Run the whole llms-benchmark protocol for one model on one CLI, end to end, and come back with a pull request adding or replacing that row in `.llms-benchmark/README.md`. +Two kinds of run, two tables. **Remote serving**: a provider serves a +model id to the CLI (OpenRouter, Anthropic, Copilot). **Local +serving**: the run serves a vllm-on-tap preset itself, on a vllm-on-tap +environment of this repository (`.vot/environments/`), and drives it +through `opencode` — always opencode, the only CLI that takes a custom +OpenAI-compatible endpoint per launch. Every step below applies to both; +a **Local serving** bullet says what changes for the second. + The question this benchmark answers: **how much of what a model reports, after observing a running stack it has never seen, actually holds up?** The demo stack under `.llms-benchmark/src/` is deliberately defective, @@ -40,6 +48,15 @@ the same way you would grade a colleague's incident report. nothing else changed). A model the CLI cannot run, or cannot run at the requested effort, is a preflight failure, not a row. Below, `` is that argument, passed verbatim to the CLI's flag. +- **Local serving** — `local [effort]`: the + literal `local`; the name of a vllm-on-tap environment + (`.vot/environments/.yaml`); the preset, as vllm-on-tap resolves + it (`.vot/presets/.yaml` first, then the builtins); the effort, + `medium` when omitted, as above. The CLI is `opencode`, never asked. + Preset, effort and **GPU** identify the row; the GPU is read off the + environment, never asked (step 1). Below, `` is the + preset's `served_model_name` (its `name` when absent) and `` + its merged `--max-model-len`. **Never ask for an API key, and never handle one.** Every credential this protocol needs — the OpenRouter provider in opencode, the Claude Code @@ -122,11 +139,36 @@ Steps: `docker compose` reads on its own from that file. Check its presence, never its value, and never print it. The file is gitignored; `.env.example` next to it says what goes in. + - **Local serving**, in place of the CLI-and-model check above (the + opencode binary is still checked, in step 3): + - `.vot/environments/.yaml` exists, and its stack + type is one the vllm-on-tap stack-guide supports; run that type's + *Prerequisites and install* checks (the `vllm-on-tap:vot-config` + skill's step 4), and stop on any failure — installing a tool or + logging in is the user's; + - the preset resolves and validates (the `vllm-on-tap:load-preset` + skill's *Resolve*, *Validate* and *Merge*, for the environment's + stack type) and carries the flags an agent needs: tool calling + (`--enable-auto-tool-choice` with a `--tool-call-parser`) and a + `--max-model-len` of at least 65536 — opencode's system prompt and + tool definitions alone take about 15k tokens, and a report-writing + turn carries the whole observation. A builtin without them is + served through a custom preset of another name in + `.vot/presets/` (the row then names that preset); + - the environment's `otlp_endpoint`, when set, does not point at the + local oddyssey stack: the served model's own spans would land in + the store the run observes; + - **the GPU cell**, read off the environment: on `azure` the serve + profile's GPU, `A100 80 GB`; on `local-vllm-metal` + `sysctl -n machdep.cpu.brand_string` and `hw.memsize` in GB + (`Apple M4 Pro 24 GB`); on `local-vllm` and `local-vllm-docker` + `nvidia-smi --query-gpu=name,memory.total --format=csv,noheader`. 2. **Create the work branch**: `bench/---`, where `` is the model id with `/` and `.` replaced by `-`. Everything the run installs, configures, and produces happens on - this branch, and none of it is what ships. + this branch, and none of it is what ships. **Local serving**: + `bench/local----`. 3. **Install or update the CLI, and the oddyssey package for it.** Record the CLI's version: it goes in the pull request, never in the @@ -195,6 +237,45 @@ Steps: List them before launching: none of the package's nine skills may be there, and a run that lists a skill twice resolves one of them. + **Local serving** — step 3 is opencode's, then the preset is served + **once for both runs**: a serve takes ten to thirty minutes to answer + (image pull, weights, compile), and a second serve between the two runs + would only add that to the bill. + - record `.vot/config.yaml`'s `current`, then set it to + `` (vllm-on-tap serves on the current environment; + step 9 puts the recorded value back); + - run the `vllm-on-tap:vot-serve` skill's steps for ``, up to + its report: the base URL and the served name. **From here on the unit + bills until step 9's destroy, whatever happens in between** — a stop, + a failed smoke, a void run all end in step 9; + - write the provider file into your own scratch directory, never into + the repository nor opencode's user configuration: + + ```json + {"$schema": "https://opencode.ai/config.json", + "provider": {"vot": {"npm": "@ai-sdk/openai-compatible", "name": "vllm-on-tap", + "options": {"baseURL": "/v1", "apiKey": "{env:VOT_API_KEY}"}, + "models": {"": {"name": "", "tool_call": true, + "limit": {"context": , "output": 8192}}}}}} + ``` + + `OPENCODE_CONFIG=` on a launch line adds the `vot` + provider to that launch only: the user's configuration and + `~/.local/share/opencode/auth.json` (the OpenRouter key) are never + written, so a remote run after a local one needs no switch back + (verified on 2026-09-27: `opencode models vot` listed the model, + `opencode models openrouter` still answered, `auth.json` unchanged); + - `VOT_API_KEY` on `azure` is the app's secret, read inline on the + launch line and never printed: + `VOT_API_KEY="$(az containerapp secret show --name vot- --resource-group --secret-name vllm-api-key --query value -o tsv)"`; + on a local stack type vLLM checks no key, and `VOT_API_KEY=none`; + - **smoke the served model through opencode before any run**, from a + scratch directory: `OPENCODE_CONFIG= VOT_API_KEY=... opencode run --model vot/ --format json "run the shell command date and reply with its output" < /dev/null` + must show a `tool_use` event for `bash` and a `text` event: a serve + without vLLM's tool-call parser answers text only, and a mission on + it never drives. A failed smoke is a preflight failure; step 9 still + destroys the unit. + 4. **Select the model.** Nothing to configure: the provider is already set up (preflight), and the model and effort are passed on the command line in step 6, never persisted into a config file — `opencode`: @@ -202,7 +283,10 @@ Steps: `--model --effort `; `copilot`: `--model --effort `. The three flags name the same effort level; that is what makes two rows of one model at one effort - comparable across CLIs. + comparable across CLIs. **Local serving**: `--model vot/`, + with `--variant ` only when `OPENCODE_CONFIG= opencode models vot --verbose` + lists that variant for the model — otherwise no `--variant` and + `default` in the Effort column, step 1's opencode rule. 5. **Clean what the next run must not read — then recreate the demo stack, never reuse a running one.** Before every run, whatever the @@ -298,6 +382,20 @@ Steps: "" < /dev/null ``` + **Local serving** — the same line, with the provider file and the key + in front, and the preset in the title (step 7 selects on it): + + ``` + OPENCODE_CONFIG= VOT_API_KEY="" \ + caffeinate -i opencode run --model vot/ [--variant ] \ + --format json --auto --title "llms-benchmark " \ + "" < /dev/null + ``` + + Between the two runs the unit stays up: step 9's teardown runs in + full except its destroy, and step 5 recreates the demo stack as for + any run. + `claude` — generate the session id yourself and write it down: this CLI takes no title, and step 7 identifies the session by that id: @@ -672,6 +770,15 @@ Steps: If that returns anything other than exactly one row, stop and say so rather than guess. + **Local serving**: the title is `llms-benchmark `, the model + id is `` and `json_extract(model,'$.providerID')` is + `vot`. The tokens are summed exactly as below; **there is no Cost**: + the endpoint bills no tokens (`SUM(cost)` is 0), and the GPU's own + bill runs from the serve to the destroy, across both runs and the + serve's start, so it is no figure of one run. The pull request states + the serve's start and destroy times and, on `azure`, the unit's + GPU-hours; the table has no money column. + **Then sum the whole tree, not the root.** opencode dispatches the observation to a subagent, which gets its own session; on the first run the subagent carried 80% of the spend. Walk `parent_id` @@ -1005,6 +1112,14 @@ Steps: step 10 and let the branch take the file with it; - `git checkout main`, then delete the work branch (`git branch -D`) — it never gets pushed. + - **Local serving, after the last run only** — the second, the one + stopped by step 6's rule, or the first when the preflight or the + run failed: run the `vllm-on-tap:vot-destroy` skill's steps for + `` and check the unit is gone (on `azure`, + `az containerapp show --name vot- ...` fails); put the + `current` recorded in step 3 back into `.vot/config.yaml`; delete + the provider file. Never end the command with the unit up: on + `azure` it bills a GPU by the second until it is destroyed. 10. **Open the results PR from a clean base.** - **Open the run's issue first.** Every PR in this repository @@ -1014,7 +1129,10 @@ Steps: - From `main`, freshly pulled, create `docs/llms-benchmark---` and make **one** change: the row in the results tables of `.llms-benchmark/README.md`. - `## Results` holds the two tables below. **A row is identified by + `## Results` holds two subsections: `### Remote serving`, the two + tables below, and `### Local serving`, the two tables of the local + serving section after them. **Local serving**: the branch is + `docs/llms-benchmark-local---`. **A row is identified by model, effort, CLI and provider together.** That key is not in the table yet → append the row; already there → replace that row in place. The same model driven through two CLIs, at two efforts, or @@ -1122,6 +1240,35 @@ Steps: until it is re-run. Changing a weight or a bound re-scores every row and is the maintainer's decision. + **Local serving — its own two tables, under `### Local serving`.** + The same shape, with three changes: **Model** becomes **Preset** (the + preset's name in backticks, the model it serves is in the pull + request), **Provider** becomes **GPU** (step 1's cell: `A100 80 GB`, + `Apple M4 Pro 24 GB`), and **Cost** and **$/confirmed** are dropped + — the endpoint bills no tokens (step 7). A row is identified by + preset, effort, CLI and GPU; the CLI is always `opencode`. Headline, + twelve columns: + + ```text + | Rank | Preset | Effort | CLI | GPU | oddyssey | Scoring | Confirmed / reported | Telemetry / Perf / Behavior | Total | Accuracy | seconds/confirmed | + ``` + + and detail, the remote detail table's columns with Preset and GPU in + place of Model and Provider: + + ```text + | Preset | Effort | CLI | GPU | oddyssey | Preflight | Drive | Observation | Turns | Median turn | Input | Output | Cache | Signals | + ``` + + Its Scoring keeps the four remaining axes and their bounds, the + weights renormalized so they still sum to one (the maintainer's + decision of 2026-09-27): **seconds/confirmed** 3/7, **Total** 2/7, + **Accuracy** 1/7, **Confirmed** 1/7. With no cost, every tie + that the remote table breaks on the cheaper run is broken on the + shorter: in step 6's choice between the two runs and in the rank. + The two tables are never merged or ranked against each other: the + remote score carries a cost the local one cannot. + Cost per confirmed finding is the column that answers the question in the README's title: cost and duration alone reward whichever model gives up soonest. The breakdown by kind exists because a run can diff --git a/.llms-benchmark/README.md b/.llms-benchmark/README.md index d31e37af..94721c2f 100644 --- a/.llms-benchmark/README.md +++ b/.llms-benchmark/README.md @@ -12,6 +12,13 @@ only variables are the model, its effort, the CLI and the provider. ## Results +Two tables, never ranked against each other: **remote serving**, where a +provider serves the model and bills its tokens, and **local serving**, +where the run serves an open-weight model itself and the only bill is +the GPU's. + +### Remote serving + One row per model, effort, CLI and provider, always its latest run. | Rank | Model | Effort | CLI | Provider | oddyssey | Scoring | Confirmed / reported | Telemetry / Perf / Behavior | Total | Cost | Accuracy | $/confirmed | seconds/confirmed | @@ -94,6 +101,28 @@ design. A row measured under an earlier revision of the protocol is marked ⚠︎ and provisional until re-run. The table keeps no history: one row per model, effort, CLI and provider, its latest run. +### Local serving + +One row per preset, effort, CLI and GPU, always its latest run. The model +is a [vllm-on-tap](https://github.com/using-system/vllm-on-tap) preset +served with vLLM for the run, on a GPU in the cloud or on the machine +itself, and driven through opencode. + +| Rank | Preset | Effort | CLI | GPU | oddyssey | Scoring | Confirmed / reported | Telemetry / Perf / Behavior | Total | Accuracy | seconds/confirmed | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | + +
+Run detail — phases, turns, tokens + +| Preset | Effort | CLI | GPU | oddyssey | Preflight | Drive | Observation | Turns | Median turn | Input | Output | Cache | Signals | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | + +
+ +- **Preset** is the vllm-on-tap preset served, **GPU** the hardware it ran on (`A100 80 GB` for a serverless Azure GPU, the chip and its memory on a local machine). Preset, effort, CLI and GPU identify a row. +- **No Cost, no $/confirmed**: a served model bills no tokens, and the GPU bills by the hour whatever the run does. +- **Scoring** keeps the four other axes and their bounds, weighted **seconds/confirmed** 3/7, **Total** 2/7, **Accuracy** 1/7 and **Confirmed** 1/7; a tie goes to the shorter run. Every other column reads as in the remote table. + ## How a row is produced ```text @@ -101,10 +130,13 @@ A row measured under an earlier revision of the protocol is marked ⚠︎ and pr /launch-llms-benchmark claude anthropic/claude-haiku-4.5 /launch-llms-benchmark copilot openai/gpt-5.6-luna /launch-llms-benchmark copilot openai/gpt-5.6-sol high +/launch-llms-benchmark local azure-sweden gemma4-12b-qat-agent ``` The CLI and the model id, in `vendor/name` form, are required; an optional third argument sets the effort (`medium` by default). Prerequisites, set up once: an OpenRouter provider in opencode, a Claude Code login with the package installed at user scope, or a Copilot CLI login; and `OPENAI_API_KEY` in `docker-compose/llms-benchmark/.env` for the demo agent's own model calls (`.env.example` next to it). +A local serving row takes `local`, a vllm-on-tap environment of this repository (`.vot/environments/`), a preset, and an optional effort; it needs opencode and the vllm-on-tap plugin. The command serves the preset once, runs the mission on it, and destroys the served model at the end. + The command cleans everything a run must not read (stored reports of the three services, leftovers, the local stack's data), recreates the demo stack, drives the model headless at the requested effort through one `/odd-observe` mission naming the three services, the stored scenario and the local stack, grades the report finding by finding on evidence, and opens the pull request carrying the row and the rulings. ## The stack under observation From 23a10c7bcb8f2ddad9efdd88944962bb1a921d27 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 11:31:33 +0200 Subject: [PATCH 05/11] chore(bench): serve odd-prefixed presets, starting with odd-gpt-oss-120b Co-Authored-By: Claude Opus 5.5 (1M context) --- .claude/commands/launch-llms-benchmark.md | 15 ++++++++++++--- .llms-benchmark/README.md | 4 ++-- .vot/presets/gemma4-12b-qat-agent.yaml | 16 ---------------- .vot/presets/odd-gpt-oss-120b.yaml | 15 +++++++++++++++ 4 files changed, 29 insertions(+), 21 deletions(-) delete mode 100644 .vot/presets/gemma4-12b-qat-agent.yaml create mode 100644 .vot/presets/odd-gpt-oss-120b.yaml diff --git a/.claude/commands/launch-llms-benchmark.md b/.claude/commands/launch-llms-benchmark.md index 54afabd8..9adcd771 100644 --- a/.claude/commands/launch-llms-benchmark.md +++ b/.claude/commands/launch-llms-benchmark.md @@ -152,9 +152,11 @@ Steps: (`--enable-auto-tool-choice` with a `--tool-call-parser`) and a `--max-model-len` of at least 65536 — opencode's system prompt and tool definitions alone take about 15k tokens, and a report-writing - turn carries the whole observation. A builtin without them is - served through a custom preset of another name in - `.vot/presets/` (the row then names that preset); + turn carries the whole observation. The benchmark serves its own + presets, `.vot/presets/odd-.yaml`, tuned for the run: a + preset without the `odd-` prefix, a builtin included, is refused + — copy it under an `odd-` name with the flags it lacks (the row + then names that preset); - the environment's `otlp_endpoint`, when set, does not point at the local oddyssey stack: the served model's own spans would land in the store the run observes; @@ -259,6 +261,13 @@ Steps: "limit": {"context": , "output": 8192}}}}}} ``` + A model that takes a reasoning effort (served with a + `--reasoning-parser`, and whose chat template reads + `reasoning_effort`) also gets `"reasoning": true` and + `"variants": {"low": {"reasoningEffort": "low"}, "medium": {"reasoningEffort": "medium"}, "high": {"reasoningEffort": "high"}}` + in its model entry, so step 4's `--variant ` sends the + effort instead of being ignored; + `OPENCODE_CONFIG=` on a launch line adds the `vot` provider to that launch only: the user's configuration and `~/.local/share/opencode/auth.json` (the OpenRouter key) are never diff --git a/.llms-benchmark/README.md b/.llms-benchmark/README.md index 94721c2f..0f72198c 100644 --- a/.llms-benchmark/README.md +++ b/.llms-benchmark/README.md @@ -130,12 +130,12 @@ itself, and driven through opencode. /launch-llms-benchmark claude anthropic/claude-haiku-4.5 /launch-llms-benchmark copilot openai/gpt-5.6-luna /launch-llms-benchmark copilot openai/gpt-5.6-sol high -/launch-llms-benchmark local azure-sweden gemma4-12b-qat-agent +/launch-llms-benchmark local azure-sweden odd-gpt-oss-120b ``` The CLI and the model id, in `vendor/name` form, are required; an optional third argument sets the effort (`medium` by default). Prerequisites, set up once: an OpenRouter provider in opencode, a Claude Code login with the package installed at user scope, or a Copilot CLI login; and `OPENAI_API_KEY` in `docker-compose/llms-benchmark/.env` for the demo agent's own model calls (`.env.example` next to it). -A local serving row takes `local`, a vllm-on-tap environment of this repository (`.vot/environments/`), a preset, and an optional effort; it needs opencode and the vllm-on-tap plugin. The command serves the preset once, runs the mission on it, and destroys the served model at the end. +A local serving row takes `local`, a vllm-on-tap environment of this repository (`.vot/environments/`), one of the benchmark's presets (`.vot/presets/odd-*.yaml`), and an optional effort; it needs opencode and the vllm-on-tap plugin. The command serves the preset once, runs the mission on it, and destroys the served model at the end. The command cleans everything a run must not read (stored reports of the three services, leftovers, the local stack's data), recreates the demo stack, drives the model headless at the requested effort through one `/odd-observe` mission naming the three services, the stored scenario and the local stack, grades the report finding by finding on evidence, and opens the pull request carrying the row and the rulings. diff --git a/.vot/presets/gemma4-12b-qat-agent.yaml b/.vot/presets/gemma4-12b-qat-agent.yaml deleted file mode 100644 index f0ce25d3..00000000 --- a/.vot/presets/gemma4-12b-qat-agent.yaml +++ /dev/null @@ -1,16 +0,0 @@ -name: gemma4-12b-qat-agent -description: Gemma 4 12B QAT 4-bit (compressed-tensors, w4a16) tuned for coding-agent CLIs on one A100 80 GB - full 256k context, tool calling, text only; NVIDIA stacks only -model: google/gemma-4-12B-it-qat-w4a16-ct -served_model_name: gemma4-12b-agent -gpu_memory_gb: 48 -vllm_args: - --max-model-len: 262144 - --gpu-memory-utilization: 0.92 - --max-num-seqs: 8 - --max-num-batched-tokens: 16384 - --enable-prefix-caching: true - --language-model-only: true - --enable-auto-tool-choice: true - --tool-call-parser: gemma4 - --reasoning-parser: gemma4 -env: [] diff --git a/.vot/presets/odd-gpt-oss-120b.yaml b/.vot/presets/odd-gpt-oss-120b.yaml new file mode 100644 index 00000000..19c034a4 --- /dev/null +++ b/.vot/presets/odd-gpt-oss-120b.yaml @@ -0,0 +1,15 @@ +name: odd-gpt-oss-120b +description: OpenAI gpt-oss-120b (MXFP4) tuned for the oddyssey llms-benchmark on one A100 80 GB - full 128k context, tool calling and reasoning parsers; NVIDIA stacks only +model: openai/gpt-oss-120b +served_model_name: gpt-oss-120b +gpu_memory_gb: 80 +vllm_args: + --max-model-len: 131072 + --gpu-memory-utilization: 0.92 + --max-num-seqs: 8 + --max-num-batched-tokens: 16384 + --enable-prefix-caching: true + --enable-auto-tool-choice: true + --tool-call-parser: openai + --reasoning-parser: openai_gptoss +env: [] From 352c96406950aab4f70a8b1dee056414aa126a84 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 18:17:41 +0200 Subject: [PATCH 06/11] feat(bench): record the first local-serving row, odd-qwen3-8-27b on an A100 80 GB Co-Authored-By: Claude Opus 5.5 (1M context) --- .claude/commands/launch-llms-benchmark.md | 50 ++++++++++++++--------- .llms-benchmark/README.md | 2 + .vot/presets/odd-gpt-oss-120b.yaml | 15 ------- .vot/presets/odd-qwen3-8-27b.yaml | 17 ++++++++ 4 files changed, 50 insertions(+), 34 deletions(-) delete mode 100644 .vot/presets/odd-gpt-oss-120b.yaml create mode 100644 .vot/presets/odd-qwen3-8-27b.yaml diff --git a/.claude/commands/launch-llms-benchmark.md b/.claude/commands/launch-llms-benchmark.md index 9adcd771..dddd230b 100644 --- a/.claude/commands/launch-llms-benchmark.md +++ b/.claude/commands/launch-llms-benchmark.md @@ -152,7 +152,8 @@ Steps: (`--enable-auto-tool-choice` with a `--tool-call-parser`) and a `--max-model-len` of at least 65536 — opencode's system prompt and tool definitions alone take about 15k tokens, and a report-writing - turn carries the whole observation. The benchmark serves its own + turn carries the whole observation. Use the model's native + maximum context when the KV cache holds it. The benchmark serves its own presets, `.vot/presets/odd-.yaml`, tuned for the run: a preset without the `odd-` prefix, a builtin included, is refused — copy it under an `odd-` name with the flags it lacks (the row @@ -240,16 +241,19 @@ Steps: be there, and a run that lists a skill twice resolves one of them. **Local serving** — step 3 is opencode's, then the preset is served - **once for both runs**: a serve takes ten to thirty minutes to answer - (image pull, weights, compile), and a second serve between the two runs - would only add that to the bill. + for the run: a serve takes ten to thirty minutes to answer (image + pull, weights, compile), and it bills from then on. - record `.vot/config.yaml`'s `current`, then set it to `` (vllm-on-tap serves on the current environment; step 9 puts the recorded value back); - run the `vllm-on-tap:vot-serve` skill's steps for ``, up to its report: the base URL and the served name. **From here on the unit bills until step 9's destroy, whatever happens in between** — a stop, - a failed smoke, a void run all end in step 9; + a failed smoke, a void run all end in step 9. Collect the unit's log to a file from its creation (on `azure`, + `az containerapp logs show --tail 300` every 20 s, new lines only: + history stops at 300 lines, the environment keeps none, and + `--follow` returned nothing on 2026-09-27). Watch it while it starts: on `EngineCore failed to start` destroy it at + once (it restarts in a loop and bills), fix the preset, serve again; - write the provider file into your own scratch directory, never into the repository nor opencode's user configuration: @@ -258,9 +262,12 @@ Steps: "provider": {"vot": {"npm": "@ai-sdk/openai-compatible", "name": "vllm-on-tap", "options": {"baseURL": "/v1", "apiKey": "{env:VOT_API_KEY}"}, "models": {"": {"name": "", "tool_call": true, - "limit": {"context": , "output": 8192}}}}}} + "limit": {"context": , "output": 32768}}}}}} ``` + Add `"permission": {"bash": {"opencode": "deny", "opencode *": "deny"}}`: + a nested `opencode run` uses the user's default model and voids the run. + A model that takes a reasoning effort (served with a `--reasoning-parser`, and whose chat template reads `reasoning_effort`) also gets `"reasoning": true` and @@ -268,6 +275,8 @@ Steps: in its model entry, so step 4's `--variant ` sends the effort instead of being ignored; + Keep `output` at 32768: it bounds the reasoning too. + `OPENCODE_CONFIG=` on a launch line adds the `vot` provider to that launch only: the user's configuration and `~/.local/share/opencode/auth.json` (the OpenRouter key) are never @@ -401,9 +410,13 @@ Steps: "" < /dev/null ``` - Between the two runs the unit stays up: step 9's teardown runs in - full except its destroy, and step 5 recreates the demo stack as for - any run. + **Local serving runs the mission once, not twice** (no provider varies + between two runs). The row is that single run, and the + two-run rules of this step do not apply to it. A void attempt (step + 8's shapes, a run measuring another model) is re-run once on the + same unit, after step 9's teardown (without its destroy) and step 5; + two void attempts and the preset's result is "did not drive the + scenario", with no row. `claude` — generate the session id yourself and write it down: this CLI takes no title, and step 7 identifies the session by that id: @@ -783,8 +796,8 @@ Steps: id is `` and `json_extract(model,'$.providerID')` is `vot`. The tokens are summed exactly as below; **there is no Cost**: the endpoint bills no tokens (`SUM(cost)` is 0), and the GPU's own - bill runs from the serve to the destroy, across both runs and the - serve's start, so it is no figure of one run. The pull request states + bill runs from the serve to the destroy, the serve's start included, + so it is no figure of the run's. The pull request states the serve's start and destroy times and, on `azure`, the unit's GPU-hours; the table has no money column. @@ -1121,9 +1134,8 @@ Steps: step 10 and let the branch take the file with it; - `git checkout main`, then delete the work branch (`git branch -D`) — it never gets pushed. - - **Local serving, after the last run only** — the second, the one - stopped by step 6's rule, or the first when the preflight or the - run failed: run the `vllm-on-tap:vot-destroy` skill's steps for + - **Local serving, after the last attempt** — the run, or the + preflight or the attempt that failed: run the `vllm-on-tap:vot-destroy` skill's steps for `` and check the unit is gone (on `azure`, `az containerapp show --name vot- ...` fails); put the `current` recorded in step 3 back into `.vot/config.yaml`; delete @@ -1251,8 +1263,8 @@ Steps: **Local serving — its own two tables, under `### Local serving`.** The same shape, with three changes: **Model** becomes **Preset** (the - preset's name in backticks, the model it serves is in the pull - request), **Provider** becomes **GPU** (step 1's cell: `A100 80 GB`, + preset's name in backticks, linked to its YAML: + `` [``](../.vot/presets/.yaml) ``), **Provider** becomes **GPU** (step 1's cell: `A100 80 GB`, `Apple M4 Pro 24 GB`), and **Cost** and **$/confirmed** are dropped — the endpoint bills no tokens (step 7). A row is identified by preset, effort, CLI and GPU; the CLI is always `opencode`. Headline, @@ -1272,9 +1284,9 @@ Steps: Its Scoring keeps the four remaining axes and their bounds, the weights renormalized so they still sum to one (the maintainer's decision of 2026-09-27): **seconds/confirmed** 3/7, **Total** 2/7, - **Accuracy** 1/7, **Confirmed** 1/7. With no cost, every tie - that the remote table breaks on the cheaper run is broken on the - shorter: in step 6's choice between the two runs and in the rank. + **Accuracy** 1/7, **Confirmed** 1/7. With no cost, a tie in the + rank is broken on the shorter run; the row is step 6's single local + run, and the pull request carries its rulings alone. The two tables are never merged or ranked against each other: the remote score carries a cost the local one cannot. diff --git a/.llms-benchmark/README.md b/.llms-benchmark/README.md index 0f72198c..edb30f77 100644 --- a/.llms-benchmark/README.md +++ b/.llms-benchmark/README.md @@ -110,12 +110,14 @@ itself, and driven through opencode. | Rank | Preset | Effort | CLI | GPU | oddyssey | Scoring | Confirmed / reported | Telemetry / Perf / Behavior | Total | Accuracy | seconds/confirmed | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| **#1** | [`odd-qwen3-8-27b`](../.vot/presets/odd-qwen3-8-27b.yaml) | default | opencode | A100 80 GB | 1.13.0 | **9.0** | 7 / 11 | 4 / 3 / 0 | **150m02s** | **64%** | **1286s** |
Run detail — phases, turns, tokens | Preset | Effort | CLI | GPU | oddyssey | Preflight | Drive | Observation | Turns | Median turn | Input | Output | Cache | Signals | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| [`odd-qwen3-8-27b`](../.vot/presets/odd-qwen3-8-27b.yaml) | default | opencode | A100 80 GB | 1.13.0 | 12m13s | 2m04s | 135m45s | 56 | 65.7s | 5.0M | 182k | — | 4/4 |
diff --git a/.vot/presets/odd-gpt-oss-120b.yaml b/.vot/presets/odd-gpt-oss-120b.yaml deleted file mode 100644 index 19c034a4..00000000 --- a/.vot/presets/odd-gpt-oss-120b.yaml +++ /dev/null @@ -1,15 +0,0 @@ -name: odd-gpt-oss-120b -description: OpenAI gpt-oss-120b (MXFP4) tuned for the oddyssey llms-benchmark on one A100 80 GB - full 128k context, tool calling and reasoning parsers; NVIDIA stacks only -model: openai/gpt-oss-120b -served_model_name: gpt-oss-120b -gpu_memory_gb: 80 -vllm_args: - --max-model-len: 131072 - --gpu-memory-utilization: 0.92 - --max-num-seqs: 8 - --max-num-batched-tokens: 16384 - --enable-prefix-caching: true - --enable-auto-tool-choice: true - --tool-call-parser: openai - --reasoning-parser: openai_gptoss -env: [] diff --git a/.vot/presets/odd-qwen3-8-27b.yaml b/.vot/presets/odd-qwen3-8-27b.yaml new file mode 100644 index 00000000..93625e89 --- /dev/null +++ b/.vot/presets/odd-qwen3-8-27b.yaml @@ -0,0 +1,17 @@ +name: odd-qwen3-8-27b +description: Qwen3.8 27B FP8 tuned for the oddyssey llms-benchmark on one A100 80 GB - FP8 weights through Marlin, MTP speculative decoding, 256k context (the model's native maximum), tool calling and reasoning parsers, text only; NVIDIA stacks only +model: Qwen/Qwen3.8-27B-FP8 +served_model_name: qwen3.8-27b +gpu_memory_gb: 40 +vllm_args: + --max-model-len: 262144 + --gpu-memory-utilization: 0.92 + --max-num-seqs: 8 + --max-num-batched-tokens: 32768 + --enable-prefix-caching: true + --language-model-only: true + --speculative-config: '{"method": "mtp", "num_speculative_tokens": 3}' + --enable-auto-tool-choice: true + --tool-call-parser: qwen3_coder + --reasoning-parser: qwen3 +env: [] From a7fd75acbbe7d1ece8524b5d0aa0a3b715d6d348 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 18:17:53 +0200 Subject: [PATCH 07/11] chore(bench): leave the azure-sweden resource group to vllm-on-tap's default Co-Authored-By: Claude Opus 5.5 (1M context) --- .vot/environments/azure-sweden.yaml | 1 - 1 file changed, 1 deletion(-) diff --git a/.vot/environments/azure-sweden.yaml b/.vot/environments/azure-sweden.yaml index 988c4165..761070b9 100644 --- a/.vot/environments/azure-sweden.yaml +++ b/.vot/environments/azure-sweden.yaml @@ -2,4 +2,3 @@ name: azure-sweden stack: azure config: location: swedencentral - resource_group: rg-vot From 237e1828e432ed140d5aae91f93f9c86d5616b3b Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 18:19:30 +0200 Subject: [PATCH 08/11] fix(bench): keep local-metal's traces out of the observed store and point the example at a shipped preset Co-Authored-By: Claude Opus 5.5 (1M context) --- .llms-benchmark/README.md | 2 +- .vot/environments/local-metal.yaml | 1 - 2 files changed, 1 insertion(+), 2 deletions(-) diff --git a/.llms-benchmark/README.md b/.llms-benchmark/README.md index edb30f77..5b13488b 100644 --- a/.llms-benchmark/README.md +++ b/.llms-benchmark/README.md @@ -132,7 +132,7 @@ itself, and driven through opencode. /launch-llms-benchmark claude anthropic/claude-haiku-4.5 /launch-llms-benchmark copilot openai/gpt-5.6-luna /launch-llms-benchmark copilot openai/gpt-5.6-sol high -/launch-llms-benchmark local azure-sweden odd-gpt-oss-120b +/launch-llms-benchmark local azure-sweden odd-qwen3-8-27b ``` The CLI and the model id, in `vendor/name` form, are required; an optional third argument sets the effort (`medium` by default). Prerequisites, set up once: an OpenRouter provider in opencode, a Claude Code login with the package installed at user scope, or a Copilot CLI login; and `OPENAI_API_KEY` in `docker-compose/llms-benchmark/.env` for the demo agent's own model calls (`.env.example` next to it). diff --git a/.vot/environments/local-metal.yaml b/.vot/environments/local-metal.yaml index cb329143..77fc2bfa 100644 --- a/.vot/environments/local-metal.yaml +++ b/.vot/environments/local-metal.yaml @@ -1,5 +1,4 @@ name: local-metal stack: local-vllm-metal -otlp_endpoint: http://localhost:4317 config: port: 8000 From 5be5af69a8296839963be0234eee692a0c465a96 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 20:41:28 +0200 Subject: [PATCH 09/11] docs(bench): record the local-serving limits learned on Gemma 4 Co-Authored-By: Claude Opus 5.5 (1M context) --- .claude/commands/launch-llms-benchmark.md | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/.claude/commands/launch-llms-benchmark.md b/.claude/commands/launch-llms-benchmark.md index dddd230b..28b732cc 100644 --- a/.claude/commands/launch-llms-benchmark.md +++ b/.claude/commands/launch-llms-benchmark.md @@ -293,6 +293,15 @@ Steps: without vLLM's tool-call parser answers text only, and a mission on it never drives. A failed smoke is a preflight failure; step 9 still destroys the unit. + - measure decode at about 80k tokens of context before the run (a + streamed request, time to first token apart): the runs' prompts reach + 200k. A preset whose run cannot end within 40 minutes is not run. + - a model whose chat template gates thinking (Gemma 4: `enable_thinking`, + off by default) gets it through the model entry's + `"options": {"chat_template_kwargs": {"enable_thinking": true}}`, and + its preset through `--default-chat-template-kwargs`. + - on an A100 (Triton attention) an FP8 KV cache is refused (SM89+); + `int8_per_token_head` starts but decodes far slower at long context. 4. **Select the model.** Nothing to configure: the provider is already set up (preflight), and the model and effort are passed on the command @@ -410,7 +419,8 @@ Steps: "" < /dev/null ``` - **Local serving runs the mission once, not twice** (no provider varies + **Local serving runs the mission once, not twice, and stops at 40 + minutes** (no provider varies between two runs). The row is that single run, and the two-run rules of this step do not apply to it. A void attempt (step 8's shapes, a run measuring another model) is re-run once on the From fe2f84909958beeaec840a08c62aac711a68cce7 Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 22:47:20 +0200 Subject: [PATCH 10/11] feat(bench): record odd-qwen3-6-35b-a3b on an A100 80 GB Co-Authored-By: Claude Opus 5.5 (1M context) --- .llms-benchmark/README.md | 4 +++- .vot/presets/odd-qwen3-6-35b-a3b.yaml | 17 +++++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) create mode 100644 .vot/presets/odd-qwen3-6-35b-a3b.yaml diff --git a/.llms-benchmark/README.md b/.llms-benchmark/README.md index 5b13488b..ee5974b6 100644 --- a/.llms-benchmark/README.md +++ b/.llms-benchmark/README.md @@ -110,13 +110,15 @@ itself, and driven through opencode. | Rank | Preset | Effort | CLI | GPU | oddyssey | Scoring | Confirmed / reported | Telemetry / Perf / Behavior | Total | Accuracy | seconds/confirmed | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| **#1** | [`odd-qwen3-8-27b`](../.vot/presets/odd-qwen3-8-27b.yaml) | default | opencode | A100 80 GB | 1.13.0 | **9.0** | 7 / 11 | 4 / 3 / 0 | **150m02s** | **64%** | **1286s** | +| **#1** | [`odd-qwen3-6-35b-a3b`](../.vot/presets/odd-qwen3-6-35b-a3b.yaml) | default | opencode | A100 80 GB | 1.13.0 | **18.5** | 3 / 9 | 1 / 2 / 0 | **21m38s** | 33% | **433s** | +| **#2** | [`odd-qwen3-8-27b`](../.vot/presets/odd-qwen3-8-27b.yaml) | default | opencode | A100 80 GB | 1.13.0 | 9.0 | **7 / 11** | 4 / 3 / 0 | 150m02s | **64%** | 1286s |
Run detail — phases, turns, tokens | Preset | Effort | CLI | GPU | oddyssey | Preflight | Drive | Observation | Turns | Median turn | Input | Output | Cache | Signals | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| [`odd-qwen3-6-35b-a3b`](../.vot/presets/odd-qwen3-6-35b-a3b.yaml) | default | opencode | A100 80 GB | 1.13.0 | 1m20s | 2m00s | 18m18s | 76 | 7.5s | 9.0M | 37k | — | 4/4 | | [`odd-qwen3-8-27b`](../.vot/presets/odd-qwen3-8-27b.yaml) | default | opencode | A100 80 GB | 1.13.0 | 12m13s | 2m04s | 135m45s | 56 | 65.7s | 5.0M | 182k | — | 4/4 |
diff --git a/.vot/presets/odd-qwen3-6-35b-a3b.yaml b/.vot/presets/odd-qwen3-6-35b-a3b.yaml new file mode 100644 index 00000000..32a37a60 --- /dev/null +++ b/.vot/presets/odd-qwen3-6-35b-a3b.yaml @@ -0,0 +1,17 @@ +name: odd-qwen3-6-35b-a3b +description: Qwen3.6 35B-A3B FP8 (MoE, ~3B active) tuned for the oddyssey llms-benchmark on one A100 80 GB - FP8 weights through Marlin, MTP speculative decoding, 256k context (the model's native maximum), tool calling and reasoning parsers, text only; NVIDIA stacks only +model: Qwen/Qwen3.6-35B-A3B-FP8 +served_model_name: qwen3.6-35b-a3b +gpu_memory_gb: 48 +vllm_args: + --max-model-len: 262144 + --gpu-memory-utilization: 0.92 + --max-num-seqs: 8 + --max-num-batched-tokens: 32768 + --enable-prefix-caching: true + --language-model-only: true + --speculative-config: '{"method": "mtp", "num_speculative_tokens": 3}' + --enable-auto-tool-choice: true + --tool-call-parser: qwen3_coder + --reasoning-parser: qwen3 +env: [] From a1b4693cab259f69b77bcebc569feb716962173f Mon Sep 17 00:00:00 2001 From: using-system Date: Sun, 27 Sep 2026 22:56:14 +0200 Subject: [PATCH 11/11] chore(bench): export the azure-sweden serves' traces to its collector Co-Authored-By: Claude Opus 5.5 (1M context) --- .vot/environments/azure-sweden.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/.vot/environments/azure-sweden.yaml b/.vot/environments/azure-sweden.yaml index 761070b9..b30654ff 100644 --- a/.vot/environments/azure-sweden.yaml +++ b/.vot/environments/azure-sweden.yaml @@ -2,3 +2,4 @@ name: azure-sweden stack: azure config: location: swedencentral + telemetry_enabled: true