Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 18 additions & 11 deletions cmd/spinloop/metrics_render.go
Original file line number Diff line number Diff line change
Expand Up @@ -349,15 +349,13 @@ func currentGPUUtil(gpus []metrics.GpuStat, idx int) *float64 {
return nil
}

// currentGPUMem is the GPU's memory ratio from the current reading, 0 where
// the GPU reports no total — the gauge's own rule for that case.
// currentGPUMem is the GPU's memory ratio from the current reading, nil where
// the reading names no such GPU or the GPU reports no memory total (macOS), so
// no memory series is drawn for it unless retained history carries one.
func currentGPUMem(gpus []metrics.GpuStat, idx int) *float64 {
for _, g := range gpus {
if g.Index == idx {
pct := 0.0
if g.MemoryTotal > 0 {
pct = float64(g.MemoryUsed) / float64(g.MemoryTotal) * 100
}
if g.Index == idx && g.MemoryTotal > 0 {
pct := float64(g.MemoryUsed) / float64(g.MemoryTotal) * 100
return &pct
}
}
Expand Down Expand Up @@ -448,8 +446,14 @@ func renderGPUTable(w io.Writer, gpus []metrics.GpuStat) {
}
fmt.Fprintln(w)
for _, g := range gpus {
fmt.Fprintf(w, " GPU %d: %s util=%d%% mem=%s/%s temp=%dC\n",
g.Index, g.Name, g.Utilization, formatBytes(g.MemoryUsed), formatBytes(g.MemoryTotal), g.Temperature)
fmt.Fprintf(w, " GPU %d: %s util=%d%%", g.Index, g.Name, g.Utilization)
if g.MemoryTotal > 0 {
fmt.Fprintf(w, " mem=%s/%s", formatBytes(g.MemoryUsed), formatBytes(g.MemoryTotal))
}
if g.Temperature > 0 {
fmt.Fprintf(w, " temp=%dC", g.Temperature)
}
fmt.Fprintln(w)
}
if len(gpus) > 1 {
var totalUtil, totalMemUsed, totalMemTotal int64
Expand All @@ -459,8 +463,11 @@ func renderGPUTable(w io.Writer, gpus []metrics.GpuStat) {
totalMemTotal += g.MemoryTotal
}
avgUtil := int(totalUtil) / len(gpus)
fmt.Fprintf(w, " avg util: %d%% total mem: %s/%s\n",
avgUtil, formatBytes(totalMemUsed), formatBytes(totalMemTotal))
fmt.Fprintf(w, " avg util: %d%%", avgUtil)
if totalMemTotal > 0 {
fmt.Fprintf(w, " total mem: %s/%s", formatBytes(totalMemUsed), formatBytes(totalMemTotal))
}
fmt.Fprintln(w)
}
}

Expand Down
78 changes: 78 additions & 0 deletions cmd/spinloop/metrics_render_gpu_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,78 @@
package main

import (
"bytes"
"strings"
"testing"

"github.com/spinloop-ai/spinloop/internal/metrics"
)

// A GPU that reports utilisation only (macOS) draws its utilisation series and
// no memory series, on the bar and gauge surfaces alike.
func TestUtilisationOnlyGPUDrawsNoMemorySeries(t *testing.T) {
macGPU := []metrics.GpuStat{{Index: 0, Name: "Apple M5 Max", Utilization: 56}}
history := []metrics.HistorySample{
{GPUs: []metrics.HistoryGPU{{Index: 0, Util: 40}}},
{GPUs: []metrics.HistoryGPU{{Index: 0, Util: 56}}},
}
var bar, gauge bytes.Buffer
renderStatBars(&bar, nil, nil, macGPU, history, barLineW)
renderStatGauges(&gauge, nil, nil, macGPU)
for name, out := range map[string]string{"bar": bar.String(), "gauge": gauge.String()} {
if !strings.Contains(out, "GPU util") || !strings.Contains(out, " 56%") {
t.Errorf("%s: missing GPU util: %q", name, out)
}
if strings.Contains(out, "GPU mem") {
t.Errorf("%s: drew a GPU mem series: %q", name, out)
}
}
}

// A GPU that reports a memory total still draws both series.
func TestGPUWithMemoryTotalDrawsBothSeries(t *testing.T) {
gpus := []metrics.GpuStat{{Index: 0, Name: "NVIDIA L40S", Utilization: 12,
MemoryUsed: 8 << 30, MemoryTotal: 16 << 30, Temperature: 42}}
var bar, gauge bytes.Buffer
renderStatBars(&bar, nil, nil, gpus, nil, barLineW)
renderStatGauges(&gauge, nil, nil, gpus)
for name, out := range map[string]string{"bar": bar.String(), "gauge": gauge.String()} {
if !strings.Contains(out, "GPU util") || !strings.Contains(out, "GPU mem") || !strings.Contains(out, " 50%") {
t.Errorf("%s: %q", name, out)
}
}
}

func TestRenderGPUTableOmitsAbsentFigures(t *testing.T) {
mac := metrics.GpuStat{Index: 0, Name: "Apple M5 Max", Utilization: 56}
nvidia := metrics.GpuStat{Index: 0, Name: "NVIDIA L40S", Utilization: 12,
MemoryUsed: 8 << 30, MemoryTotal: 16 << 30, Temperature: 42}

var b bytes.Buffer
renderGPUTable(&b, []metrics.GpuStat{mac})
if got, want := b.String(), "\n GPU 0: Apple M5 Max util=56%\n"; got != want {
t.Errorf("utilisation-only = %q, want %q", got, want)
}

b.Reset()
renderGPUTable(&b, []metrics.GpuStat{nvidia})
if got, want := b.String(), "\n GPU 0: NVIDIA L40S util=12% mem=8.0 GB/16.0 GB temp=42C\n"; got != want {
t.Errorf("full reading = %q, want %q", got, want)
}

// Two utilisation-only GPUs: the totals line drops its memory figure.
other := mac
other.Index = 1
b.Reset()
renderGPUTable(&b, []metrics.GpuStat{mac, other})
if strings.Contains(b.String(), "mem") || !strings.Contains(b.String(), " avg util: 56%\n") {
t.Errorf("two utilisation-only GPUs = %q", b.String())
}

// A mixed pair keeps the totals line whole.
b.Reset()
renderGPUTable(&b, []metrics.GpuStat{nvidia, other})
if !strings.Contains(b.String(), "avg util: 34% total mem: 8.0 GB/16.0 GB") {
t.Errorf("mixed pair = %q", b.String())
}
}
4 changes: 3 additions & 1 deletion docs/commands/serve.md
Original file line number Diff line number Diff line change
Expand Up @@ -381,7 +381,9 @@ Because the client sets the key, it knows the key — which is what it gives the
agent it launches. `/v1/status` reports only *that* a key is required, never
what it is, and no endpoint returns it. A supervised engine gets its own `/metrics` endpoint switched on
(llama.cpp `--metrics`), which is where the token counters come from; GPU
readings need `nvidia-smi` (no Apple GPU source yet).
readings come from `nvidia-smi` on Linux and from the I/O Kit accelerator
(`ioreg`) on macOS, where only utilisation and the GPU name are reported — no
GPU memory or temperature.

Those counters are also read every 15 seconds in the background, so
`/v1/status` and `/v1/metrics` can both report `lastActiveAt` and
Expand Down
2 changes: 1 addition & 1 deletion docs/http-api.md
Original file line number Diff line number Diff line change
Expand Up @@ -57,7 +57,7 @@ Stops the engine.
### GET `/v1/metrics`
Returns the current metrics:
- Token usage counters (from the engine's Prometheus `/metrics` endpoint)
- Host system metrics (GPU, CPU, RAM)
- Host system metrics (GPU, CPU, RAM). On macOS a GPU carries utilisation and name only; its `memoryUsed`, `memoryTotal` and `temperature` are `0`, meaning the host reports none
- `history`, the daemon's retained system readings — one per sampler tick while an engine ran, each a 0–100% figure per series (`t` time, `c` CPU, `m` memory, `g` per-GPU utilisation and memory), covering at most the last 10 minutes
- `lastActiveAt` and `idleSeconds`, the same pair `/v1/status` reports

Expand Down
20 changes: 17 additions & 3 deletions internal/metrics/collect.go
Original file line number Diff line number Diff line change
Expand Up @@ -14,8 +14,13 @@ var nvidiaSMIArgs = []string{
"--format=csv,noheader,nounits",
}

// ioregAcceleratorArgs dumps the I/O Kit accelerator services. -w 0 turns line
// wrapping off so each property stays on one line when stdout is a pipe.
var ioregAcceleratorArgs = []string{"-rd1", "-c", "IOAccelerator", "-w", "0"}

// Collector gathers system stats with host commands: nvidia-smi/vmstat/free
// on Linux; sysctl, vm_stat and top on macOS (no GPU source there yet). Both
// on Linux; sysctl, vm_stat, top and ioreg on macOS, where the GPU is the I/O
// Kit accelerator service (utilisation and name, no memory or temperature). Both
// the command runner and the platform are injectable so tests feed fixture
// output for either platform without running anything.
type Collector struct {
Expand Down Expand Up @@ -71,8 +76,17 @@ func reportable(err error) bool {
}

func (c *Collector) gpus(ctx context.Context) ([]GpuStat, error) {
if c.goos() != "linux" {
// No GPU source off Linux yet (Apple GPU stats are issue #47).
switch c.goos() {
case "darwin":
// A host with no accelerator service prints nothing, which parses to
// no GPUs.
out, err := c.run(ctx, "ioreg", ioregAcceleratorArgs...)
if err != nil {
return nil, err
}
return ParseIOAcceleratorGPU(out), nil
case "linux":
default:
return nil, nil
}
out, err := c.run(ctx, "nvidia-smi", nvidiaSMIArgs...)
Expand Down
9 changes: 6 additions & 3 deletions internal/metrics/metrics.go
Original file line number Diff line number Diff line change
Expand Up @@ -4,8 +4,9 @@
// the parsers here are ports of that Lambda's, kept value-for-value compatible
// so the remote path can later delegate to a daemon running this code and the
// TypeScript collectors can be deleted. Every stat is optional — a host
// without a source for one (no nvidia-smi, say) simply omits it, which is how
// a macOS node reports engine stats without GPU figures.
// without a source for one (no nvidia-smi, say) simply omits it. A macOS node
// reports GPU utilisation and name from the I/O Kit accelerator service, and
// no GPU memory or temperature.
package metrics

// Stats is the collected state of one serving host: what is running, its
Expand Down Expand Up @@ -75,7 +76,9 @@ type TokenStats struct {
Requests int `json:"requests"`
}

// GpuStat holds per-GPU metrics from nvidia-smi.
// GpuStat holds per-GPU metrics from nvidia-smi, or from the I/O Kit
// accelerator on macOS, where MemoryUsed, MemoryTotal and Temperature are zero
// because the host reports none.
type GpuStat struct {
Index int `json:"index"`
Name string `json:"name"`
Expand Down
98 changes: 96 additions & 2 deletions internal/metrics/metrics_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -203,16 +203,82 @@ func TestCollectorLinux(t *testing.T) {
}
}

const ioregAGXFixture = `+-o AGXAcceleratorG16G <class AGXAcceleratorG16G, id 0x1000003c4, registered, matched, active, busy 0 (503 ms), retain 71>
{
"IOClass" = "AGXAcceleratorG16G"
"PerformanceStatistics" = {"In use system memory (driver)"=0,"Tiler Utilization %"=32,"Renderer Utilization %"=32,"Device Utilization %"=56,"In use system memory"=991494144}
"model" = "Apple M5 Max"
"gpu-core-count" = 40
}

`

const ioregAMDFixture = `+-o AMDRadeonX6000_AMDRadeonAccelerator <class AMDRadeonX6000_AMDRadeonAccelerator, id 0x100000500>
{
"model" = "AMD Radeon Pro 5500M"
"PerformanceStatistics" = {"Device Utilization (%)"=7,"vramFreeBytes"=123}
}
`

const ioregNoUtilFixture = `+-o IntelAccelerator <class IntelAccelerator, id 0x100000600>
{
"model" = "Intel UHD Graphics"
"PerformanceStatistics" = {"Alloc system memory"=1}
}
`

func TestParseIOAcceleratorGPU(t *testing.T) {
t.Run("apple", func(t *testing.T) {
gpus := ParseIOAcceleratorGPU(ioregAGXFixture)
want := []GpuStat{{Index: 0, Name: "Apple M5 Max", Utilization: 56}}
if len(gpus) != 1 || gpus[0] != want[0] {
t.Errorf("gpus = %+v, want %+v", gpus, want)
}
})
t.Run("amd spelling", func(t *testing.T) {
gpus := ParseIOAcceleratorGPU(ioregAMDFixture)
if len(gpus) != 1 || gpus[0].Name != "AMD Radeon Pro 5500M" || gpus[0].Utilization != 7 {
t.Errorf("gpus = %+v", gpus)
}
})
t.Run("two accelerators", func(t *testing.T) {
gpus := ParseIOAcceleratorGPU(ioregAMDFixture + ioregAGXFixture)
if len(gpus) != 2 || gpus[0].Index != 0 || gpus[1].Index != 1 ||
gpus[0].Utilization != 7 || gpus[1].Utilization != 56 {
t.Errorf("gpus = %+v", gpus)
}
})
t.Run("block without utilisation is skipped", func(t *testing.T) {
gpus := ParseIOAcceleratorGPU(ioregNoUtilFixture + ioregAGXFixture)
if len(gpus) != 1 || gpus[0].Index != 0 || gpus[0].Name != "Apple M5 Max" {
t.Errorf("gpus = %+v", gpus)
}
})
t.Run("service name stands in for a missing model", func(t *testing.T) {
out := "+-o SomeAccel <class SomeAccel>\n \"PerformanceStatistics\" = {\"Device Utilization %\"=3}\n"
gpus := ParseIOAcceleratorGPU(out)
if len(gpus) != 1 || gpus[0].Name != "SomeAccel" {
t.Errorf("gpus = %+v", gpus)
}
})
t.Run("empty output", func(t *testing.T) {
if gpus := ParseIOAcceleratorGPU(""); gpus != nil {
t.Errorf("gpus = %+v, want none", gpus)
}
})
}

func TestCollectorDarwin(t *testing.T) {
c := &Collector{GOOS: "darwin", Run: fixtureRunner(map[string]string{
"top": topFixture,
"sysctl": "34359738368\n",
"vm_stat": vmStatFixture,
"ioreg": ioregAGXFixture,
}, nil)}
var stats Stats
c.System(context.Background(), &stats)
if stats.GPUs != nil {
t.Errorf("darwin reported GPUs: %+v", stats.GPUs)
if len(stats.GPUs) != 1 || stats.GPUs[0].Utilization != 56 || stats.GPUs[0].MemoryTotal != 0 {
t.Errorf("darwin GPUs = %+v", stats.GPUs)
}
if stats.CPU == nil || stats.Memory == nil {
t.Errorf("stats = %+v", stats)
Expand All @@ -222,6 +288,34 @@ func TestCollectorDarwin(t *testing.T) {
}
}

func TestCollectorDarwinNoAcceleratorIsSilent(t *testing.T) {
c := &Collector{GOOS: "darwin", Run: fixtureRunner(map[string]string{
"top": topFixture,
"sysctl": "34359738368\n",
"vm_stat": vmStatFixture,
"ioreg": "",
}, nil)}
var stats Stats
c.System(context.Background(), &stats)
if stats.GPUs != nil || len(stats.Errors) != 0 {
t.Errorf("GPUs = %+v, errors = %v; want neither", stats.GPUs, stats.Errors)
}
}

func TestCollectorDarwinIoregFailureIsReported(t *testing.T) {
c := &Collector{GOOS: "darwin", Run: func(_ context.Context, name string, _ ...string) (string, error) {
if name == "ioreg" {
return "", errors.New("exit status 1")
}
return "", exec.ErrNotFound
}}
var stats Stats
c.System(context.Background(), &stats)
if len(stats.Errors) != 1 || !strings.HasPrefix(stats.Errors[0], "gpu: ") {
t.Errorf("errors = %v, want one gpu error", stats.Errors)
}
}

func TestCollectorMissingCommandIsSilent(t *testing.T) {
c := &Collector{GOOS: "linux", Run: fixtureRunner(map[string]string{
"vmstat": vmstatFixture,
Expand Down
56 changes: 56 additions & 0 deletions internal/metrics/parse.go
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,62 @@ func ParseGPUStats(out string) []GpuStat {
return gpus
}

var (
// ioregServiceHeader matches the line that opens one service in
// `ioreg -r` output, e.g. "+-o AGXAcceleratorG16G <class ...>".
ioregServiceHeader = regexp.MustCompile(`^\s*\+-o\s+(\S+)`)
// ioregModel matches the `"model" = "Apple M5 Max"` property.
ioregModel = regexp.MustCompile(`^\s*"model"\s*=\s*"([^"]*)"`)
// ioregUtilisation matches the device-wide utilisation figure inside a
// performance statistics dictionary: Apple's "Device Utilization %" and
// the AMD drivers' "Device Utilization (%)".
ioregUtilisation = regexp.MustCompile(`"Device Utilization (?:%|\(%\))"\s*=\s*(\d+)`)
)

// ParseIOAcceleratorGPU parses the output of
//
// ioreg -rd1 -c IOAccelerator -w 0
//
// into one GpuStat per accelerator service that reports a device utilisation.
// The name is the service's "model" property, falling back to the service
// name; the index is the order of appearance. Memory and temperature are left
// zero: Apple Silicon has no GPU memory total and temperature needs root. A
// service with no utilisation figure contributes no GPU, and empty output
// yields none.
func ParseIOAcceleratorGPU(out string) []GpuStat {
var gpus []GpuStat
var service, model string
util := -1
flush := func() {
if util >= 0 {
name := model
if name == "" {
name = service
}
gpus = append(gpus, GpuStat{Index: len(gpus), Name: name, Utilization: util})
}
service, model, util = "", "", -1
}
for _, line := range strings.Split(out, "\n") {
if m := ioregServiceHeader.FindStringSubmatch(line); m != nil {
flush()
service = m[1]
continue
}
if m := ioregModel.FindStringSubmatch(line); m != nil {
model = m[1]
}
if !strings.Contains(line, "erformance") {
continue
}
if m := ioregUtilisation.FindStringSubmatch(line); m != nil {
util = atoiOrZero(m[1])
}
}
flush()
return gpus
}

// ParseVmstatCPU parses `vmstat 1 2` output, whose last line is the sampled
// interval. The idle column (id) is field 15 of the standard 17-column layout;
// utilization is 100 - idle. Returns nil when the output is not vmstat's.
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
schema: spec-driven
created: 2026-09-26
Loading
Loading