From 86c5ea3d175da809a8e467b9ca5697945206ab93 Mon Sep 17 00:00:00 2001 From: using-system Date: Sat, 26 Sep 2026 20:10:44 +0200 Subject: [PATCH 1/7] fix(skill): name the synthesis baseline from the frontmatter and count a replay's rulings `new` records the recall's first line as a `baseline` field outside a replay (`none` without one), and `show`/`persist` read it instead of the first report filename on section 1's prose line; a report predating the field falls back to that line, where a value opening with none yields no baseline. A re-measure with no check table counts its baseline rulings. The body contract names section 7's Verdict column on a replay. Closes #659 Co-Authored-By: Claude Opus 5.5 (1M context) --- .../references/observe-run-report.md | 6 +- .apm/skills/odd-memory/scripts/odd_recall.py | 23 +--- .apm/skills/odd-memory/scripts/odd_report.py | 87 ++++++++++--- docs/guide/reports.md | 2 + tests/skills/odd-memory/test_odd_report.py | 114 ++++++++++++++++++ 5 files changed, 190 insertions(+), 42 deletions(-) diff --git a/.apm/skills/odd-memory/references/observe-run-report.md b/.apm/skills/odd-memory/references/observe-run-report.md index cf3e58b..eb3f6bc 100644 --- a/.apm/skills/odd-memory/references/observe-run-report.md +++ b/.apm/skills/odd-memory/references/observe-run-report.md @@ -41,7 +41,8 @@ instrumentation` is the other kind's, stated in its own reference. `-observe-` suffix in observe mode, the `verify-` and `remeasure-` prefixes, the next free ordinal when the path is taken), fills `date`, `revision`, `tree_anchor` and `repository` from the - repository itself, writes the frontmatter and the seven-section + repository itself, `baseline` outside a replay from the recall's + first line (`none` without one), writes the frontmatter and the seven-section skeleton — eight with `--custom-stack`, the flag a mission passes when the handoff names a custom stack: the frontmatter then carries `stack_friction: 0` and the skeleton the `## 8. Stack friction` @@ -278,7 +279,8 @@ heading: way: a check keyed more coarsely than the operations it rules can never be re-read per operation later. In a verify or re-measure, this table rules the baseline's **checks**, each under the key the baseline - gave it; a check key is never a finding id, and a check ruled here + gave it, as `| Check | Before | After | Verdict |` — a ruling + outside a Verdict column is one no script reads; a check key is never a finding id, and a check ruled here never stands in for section 3's ruling on a baseline finding — the two tables answer to different keys. A baseline check grouped more coarsely than the operations it rules — by the route alone, its verbs diff --git a/.apm/skills/odd-memory/scripts/odd_recall.py b/.apm/skills/odd-memory/scripts/odd_recall.py index ca25a72..a1d48e1 100755 --- a/.apm/skills/odd-memory/scripts/odd_recall.py +++ b/.apm/skills/odd-memory/scripts/odd_recall.py @@ -58,6 +58,7 @@ check_report, parse_value, read_report, + recall_matches, ) from odd_report import ( git_root as _git_root, @@ -109,25 +110,7 @@ def check(report: dict, stored_names: set[str], root: Path) -> list[str]: # --- the contract's checks, as the memory invariant applies them --------------------- -# --- the matching rules, as the references state them -------------------------------- - - -def matches(report: dict, scope: dict) -> bool: - if "unreadable" in report: - return False - fm = report["frontmatter"] - if scope["stack"] and str(fm.get("stack")) != scope["stack"]: - return False - if report["kind"] == "instrumentation": - project = str(fm.get("project") or "") - target = scope["project"] - return not target or target == project or target.startswith(project + "/") - # the same service set: the lineage get-status keys a report by - if scope["services"] and set(scope["services"]) != set(as_list(fm.get("services"))): - return False - if scope["environment"] and str(fm.get("environment")) != scope["environment"]: - return False - return not scope["modes"] or str(fm.get("mode")) in scope["modes"] +# --- the matching rules, as the references state them: recall_matches, in odd_report -- def cell(value: Any) -> str: @@ -185,7 +168,7 @@ def recall(root: Path, kind: str, scope: dict) -> tuple[list[str], list[str]]: # every stored report is checked, matched or not: a flaw in the very # field the scope matches on must never hide the report silently problems = {r["name"]: check(r, stored, root) for r in reports} - matched = [r for r in reports if matches(r, scope)] + matched = [r for r in reports if recall_matches(r, scope)] matched_names = {r["name"] for r in matched} for r in matched: out.append(line_of(r)) diff --git a/.apm/skills/odd-memory/scripts/odd_report.py b/.apm/skills/odd-memory/scripts/odd_report.py index ebf0289..880675a 100755 --- a/.apm/skills/odd-memory/scripts/odd_report.py +++ b/.apm/skills/odd-memory/scripts/odd_report.py @@ -206,6 +206,7 @@ def body_contract() -> str: "window", "run_name", "date", + "baseline", "verifies", "revision", "tree_anchor", @@ -1179,6 +1180,42 @@ def scalar(key: str) -> str | None: return problems +def recall_matches(report: dict, scope: dict) -> bool: + """The recall's matching rules (``odd_recall.py`` applies them too).""" + if "unreadable" in report: + return False + fm = report["frontmatter"] + if scope["stack"] and str(fm.get("stack")) != scope["stack"]: + return False + if report["kind"] == "instrumentation": + project = str(fm.get("project") or "") + target = scope["project"] + return not target or target == project or target.startswith(project + "/") + # the same service set: the lineage get-status keys a report by + if scope["services"] and set(scope["services"]) != set(as_list(fm.get("services"))): + return False + if scope["environment"] and str(fm.get("environment")) != scope["environment"]: + return False + return not scope["modes"] or str(fm.get("mode")) in scope["modes"] + + +def recalled_baseline(root: Path, services: list[str], stack: str, env: str) -> str: + """The recall's first line - the baseline a run that is no replay + diffs against - or ``none``: the value of the ``baseline`` field.""" + store = root / OBSERVATION_DIR + scope = { + "services": services, + "stack": stack, + "environment": env, + "modes": [], + "project": None, + } + for path in sorted(store.glob("*.md"), reverse=True) if store.is_dir() else []: + if recall_matches(read_report(path, "observation"), scope): + return path.name + return "none" + + def baseline_path(root: Path, verifies: str) -> Path: """Where a replay's baseline lives: a bare filename is a sibling observation report, a repo-relative path names the directory itself.""" @@ -1813,6 +1850,8 @@ def new_report(args: argparse.Namespace) -> tuple[Path, str, list[str]]: } if replay: fields["verifies"] = args.verifies + else: + fields["baseline"] = recalled_baseline(root, services, args.stack, args.env) if repo_root is not None: fields["revision"] = git(repo_root, "rev-parse", "--short", "HEAD") fields["tree_anchor"] = ls_tree(repo_root, "HEAD") @@ -1918,6 +1957,16 @@ def pinned_packages(cell: str) -> list[str]: return [p.strip(" `*") for p in parts if p.strip() and re.search(r"\d", p)] +def prose_baseline(line: str) -> tuple[str | None, bool]: + """A report predating the ``baseline`` field: the report its section 1 + baseline line names, or none when that line's value opens with none.""" + value = line.split(":", 1)[-1].lstrip("*` ") + if NONE_RE.match(value) or re.match(r"\W*no previous", value, re.IGNORECASE): + return None, True + match = REPORT_FILE_RE.search(line) + return (match.group(0), False) if match else (None, False) + + def instrumentation_data(fm: dict, text: str, sections: list[dict]) -> dict: data: dict[str, Any] = { "kind": "instrumentation", @@ -1949,11 +1998,7 @@ def instrumentation_data(fm: dict, text: str, sections: list[dict]) -> dict: ] data["baseline_lines"] = found[:1] for line in found[:1]: - match = REPORT_FILE_RE.search(line) - if match: - data["baseline_name"] = match.group(0) - elif re.search(r"no previous", line, re.IGNORECASE): - data["no_baseline"] = True + data["baseline_name"], data["no_baseline"] = prose_baseline(line) two = section(sections, 2) if two is not None: table, _ = table_with(two["tables"], SUMMARY_PATTERNS) @@ -2055,16 +2100,12 @@ def synthesis_data(text: str, kind: str | None = None) -> dict: ] notes = [i for i in listed if BASELINE_NOTE_RE.search(i) and i not in found[:1]] data["baseline_lines"] = found[:1] + notes - for line in data["baseline_lines"]: - match = REPORT_FILE_RE.search(line) - if match and data["baseline_name"] is None: - data["baseline_name"] = match.group(0) - if ( - found - and data["baseline_name"] is None - and re.search(r"no previous", found[0], re.IGNORECASE) - ): - data["no_baseline"] = True + if found and "baseline" not in fm: + data["baseline_name"], data["no_baseline"] = prose_baseline(found[0]) + if "baseline" in fm: + named = str(fm["baseline"]) + data["baseline_name"] = None if named == "none" else named + data["no_baseline"] = named == "none" two = section(sections, 2) if two is not None and not replay: data["deltas"] = [i for i in items(two["lines"]) if DELTA_RE.search(i)][:20] @@ -2970,11 +3011,17 @@ def render_headline(data: dict) -> str: elif mode == "re-measure": passed, failed = verdict_counts(data["checks"]) total = len(data["checks"]) - text = ( - f"{'drift' if failed else 'no drift'} — {passed}/{total} checks within range" - if total - else f"re-measure — {plural(len(data['findings']), 'finding')} re-measured" - ) + if total: + text = f"{'drift' if failed else 'no drift'} — {passed}/{total} checks within range" + elif data["rulings"]: + text = ( + f"re-measure — {plural(len(data['rulings']), 'baseline finding')} " + "ruled, no check table" + ) + else: + text = ( + f"re-measure — {plural(len(data['findings']), 'finding')} re-measured" + ) else: text = ( f"{plural(len(findings), 'anomaly', 'anomalies')} ({high} high, " diff --git a/docs/guide/reports.md b/docs/guide/reports.md index 94be6ce..fe370fe 100644 --- a/docs/guide/reports.md +++ b/docs/guide/reports.md @@ -48,6 +48,7 @@ mode: drive window: 2026-08-22T10:04:12Z/2026-08-22T10:05:03Z run_name: checkout-latency-sweep date: 2026-08-22 +baseline: none revision: 2299d4c workload: repo-under-analysis instance: {checkout: af6070c1} @@ -64,6 +65,7 @@ process_restarted: true | `window` | yes | The observed interval, UTC — the run's own span, not the time a mission spent waiting for it | `start/end` | | `run_name` | yes | The filename's slug | kebab-case | | `date` | yes | The run's UTC date | `YYYY-MM-DD` | +| `baseline` | drive, observe, post-hoc | The report the run diffs against: the recall's first match | its exact filename, or `none` | | `verifies` | verify, re-measure | The report whose protocol was replayed | its exact filename; the repo-relative path for an instrumentation report | | `revision` | optional | The observed repo's commit at run time | short SHA | | `tree_anchor` | optional | The top-level tree hashes at `revision`, so "code unchanged since" survives squash merges and fresh clones | entry name to hash | diff --git a/tests/skills/odd-memory/test_odd_report.py b/tests/skills/odd-memory/test_odd_report.py index 9502206..8b9fba7 100644 --- a/tests/skills/odd-memory/test_odd_report.py +++ b/tests/skills/odd-memory/test_odd_report.py @@ -2028,3 +2028,117 @@ def test_baseline_asks_on_the_newest_day_only_and_takes_a_dotted_path(repo): f"./{OBS}/2026-08-08-1000-checkout-sweep.md", ) assert proc.returncode == 0 and "checkout-sweep" in lines_of(proc)["report"] + + +# --- the synthesis: the baseline from a structured source, a replay's rulings ------ + + +def drafted(report, section_one: str, three: str = "", seven: str = "") -> str: + parts = ["# Observation report — checkout-sweep", "**Fine.**"] + for n, title in enumerate(report.SECTION_TITLES, 1): + text = {1: section_one, 3: three, 7: seven}.get(n) or f"text {n}" + parts.append(f"## {n}. {title}\n\n{text}") + return "\n\n".join(parts) + "\n" + + +def test_new_records_the_recalls_first_line_as_the_baseline(repo, report): + older = stored(repo, "2026-08-07-1000-checkout-sweep.md") + newer = stored(repo, "2026-08-08-1000-checkout-sweep.md") + other_env = stored( + repo, + "2026-08-09-1000-checkout-sweep.md", + BASELINE.replace("environment: local", "environment: dev"), + ) + assert other_env and older + assert frontmatter(report, new(repo))["baseline"] == newer + + +def test_new_records_no_baseline_when_the_recall_finds_none(repo, report): + stored( + repo, + "2026-08-08-1000-checkout-sweep.md", + BASELINE.replace("environment: local", "environment: unknown"), + ) + assert frontmatter(report, new(repo))["baseline"] == "none" + + +def test_a_replay_names_its_baseline_in_verifies_alone(repo, report): + name = baseline(repo) + path = new(repo, "--mode", "verify", "--verifies", name, run_name=None) + assert "baseline" not in frontmatter(report, path) + + +def test_show_never_names_a_baseline_the_recall_did_not_find(repo, report): + path = new(repo) + draft = repo.write( + "scratch/draft.md", + drafted( + report, + "- **Baseline recalled**: none. `odd_recall.py --env dev` -> no stored " + "report matches. The only report, `2026-08-01-1000-checkout-sweep.md`, " + "records `environment: unknown`.", + ), + ) + proc = run(repo, "persist", str(path), "--body", str(draft)) + assert proc.returncode == 0, proc.stderr + assert "no previous report" in proc.stdout and "vs baseline" not in proc.stdout + proc = run(repo, "show", str(path)) + assert "baseline: none" in proc.stdout + + +def test_a_legacy_report_whose_baseline_line_says_none_names_none(repo, report): + text = BASELINE.replace( + "- **Recalled baseline:** no previous report.", + "- **Baseline recalled**: none. The only report, " + "`2026-08-01-1000-checkout-sweep.md`, records `environment: unknown`.", + ) + data = report.synthesis_data(text) + assert data["baseline_name"] is None and data["no_baseline"] + assert "vs baseline" not in report.render_headline(data) + + +def test_a_re_measure_headline_counts_its_baseline_rulings(repo, report): + name = baseline(repo) + path = new(repo, "--mode", "re-measure", "--verifies", name, run_name=None) + three = ( + "| # | Baseline finding | Verdict | Evidence |\n|---|---|---|---|\n" + "| F1 | N+1 | still present | trace abc |\n" + "| F2 | Missing db spans | still present | none |" + ) + draft = repo.write("scratch/draft.md", drafted(report, "- baseline", three)) + proc = run(repo, "persist", str(path), "--body", str(draft)) + assert proc.returncode == 0, proc.stderr + assert "re-measure — 2 baseline findings ruled" in proc.stdout + + +def test_the_body_contract_names_section_7s_verdict_column_and_synthesis_reads_it( + repo, report +): + name = baseline(repo) + proc = run( + repo, + *NEW[:-4], + "--mode", + "re-measure", + "--window", + WINDOW, + "--verifies", + name, + "--repo", + str(repo.root), + ) + assert proc.returncode == 0, proc.stderr + header = "| Check | Before | After | Verdict |" + assert header in proc.stdout.split(BODY_CONTRACT_MARK, 1)[1] + path = Path(proc.stdout.splitlines()[0]) + three = ( + "| # | Baseline finding | Verdict | Evidence |\n|---|---|---|---|\n" + "| F1 | N+1 | still present | a |\n| F2 | spans | still present | b |" + ) + seven = ( + f"{header}\n|---|---|---|---|\n| 1. p95 of GET /products | 9 ms | 9 ms | fail |" + ) + draft = repo.write("scratch/draft.md", drafted(report, "- baseline", three, seven)) + assert run(repo, "persist", str(path), "--body", str(draft)).returncode == 0 + out = run(repo, "synthesis", str(path)).stdout + assert "| 1. p95 of GET /products | 9 ms | 9 ms | fail |" in out From b5e7a599a37a46fb10fd799c66b1bdeec707b355 Mon Sep 17 00:00:00 2001 From: using-system Date: Sat, 26 Sep 2026 20:15:14 +0200 Subject: [PATCH 2/7] fix(skill): rank azure-monitor exemplars per operation and pick a p50 one `azure-monitor-traces.py exemplars` took the slowest requests over the union of the operations, so one slow operation filled every slot, and had no p50 pick. It now partitions the slowest by operation (`--slow` per operation, default 1) and adds, per operation, the request nearest its p50. Fixtures re-captured live, masked. Closes #659 Co-Authored-By: Claude Opus 5.5 (1M context) --- .../references/azure-monitor.md | 11 +++-- .../scripts/azure-monitor-traces.py | 44 ++++++++++++++----- .../fixtures/azure-monitor/1458b21ebbe6.json | 20 +++++++++ .../fixtures/azure-monitor/44a64de3a35e.json | 20 +++++++++ .../fixtures/azure-monitor/5a33d2716413.json | 20 --------- .../fixtures/azure-monitor/7bf86e597863.json | 6 +-- .../fixtures/azure-monitor/d608a619adfc.json | 20 +++++++++ .../fixtures/azure-monitor/f0f111e20734.json | 20 --------- .../test_azure_monitor_scripts.py | 21 ++++++--- 9 files changed, 119 insertions(+), 63 deletions(-) create mode 100644 tests/skills/observability-cli-guides/fixtures/azure-monitor/1458b21ebbe6.json create mode 100644 tests/skills/observability-cli-guides/fixtures/azure-monitor/44a64de3a35e.json delete mode 100644 tests/skills/observability-cli-guides/fixtures/azure-monitor/5a33d2716413.json create mode 100644 tests/skills/observability-cli-guides/fixtures/azure-monitor/d608a619adfc.json delete mode 100644 tests/skills/observability-cli-guides/fixtures/azure-monitor/f0f111e20734.json diff --git a/.apm/skills/observability-cli-guides/references/azure-monitor.md b/.apm/skills/observability-cli-guides/references/azure-monitor.md index cad87a5..1f72a54 100644 --- a/.apm/skills/observability-cli-guides/references/azure-monitor.md +++ b/.apm/skills/observability-cli-guides/references/azure-monitor.md @@ -152,7 +152,7 @@ these scripts do. ```bash python3 /observability-cli-guides/scripts/azure-monitor-traces.py operations --app --service --from --to python3 /observability-cli-guides/scripts/azure-monitor-traces.py dependencies --app --service --since 30m -python3 /observability-cli-guides/scripts/azure-monitor-traces.py exemplars --app --service --slow 3 --failed 3 --since 30m +python3 /observability-cli-guides/scripts/azure-monitor-traces.py exemplars --app --service --since 30m python3 /observability-cli-guides/scripts/azure-monitor-traces.py trace --app --from --to python3 /observability-cli-guides/scripts/azure-monitor-traces.py watch --app --identity --from --state /-watch.json --length --expect [--to ] [--bin 30s] [--ended-after 4] [--settle auto] [--every 5s] [--max 8m] [--service ]... [--dimension ]... [--json] ``` @@ -163,7 +163,7 @@ optional `--to`, the deadline), `--json`; `operations`, `dependencies`, `exemplars` and `watch` take `--service`; `operations` adds `--top` (rows, default 20) and `--bin ` (adds the request count, failures and p95 per time bucket); `exemplars` adds `--operation ` (repeatable), `--slow N` (the slowest requests, default 3), +name>` (repeatable), `--slow N` (the slowest requests per operation, default 1), `--failed N` (the newest failed requests, default 3); `trace` takes the `operation_Id`; `watch` takes `--identity` (the prefix, matched with `startswith`), `--state` (its state file), `--length` (the manifest's @@ -193,13 +193,16 @@ non-empty one wins; default `user_agent.original` then join leaves out - a client span whose request is outside the window). Verified 2026-09-11: seven joined rows (the load generator's client spans and the service's own outgoing calls), five in `all`. -- `exemplars` - Output: `slow` and `failed_requests`, each request with +- `exemplars` - Output: `p50` (per operation, the request nearest its + p50, with that `p50`), `slow` (ranked per operation) and + `failed_requests`, each request with `timestamp`, `name`, `duration`, `resultCode`, `operation_Id`, `id` and, from one union over the picked ids, its `dependencies`, `exceptions` and `logs` (the `traces` rows of warning level and above). Verified 2026-09-11: the slowest carried its `payment.authorize` dependency, a failed one its failed `storage.delete` and the warning - line explaining it. + line explaining it. Verified 2026-09-26: five operations, a `p50` + and two `slow` each. - `trace` - Output: `summary` (`root`, `duration_ms`, `spans`, `failed_spans`, `logs`, `exceptions`, `services`) and `nodes` (depth-first: `request`/`dependency` spans with `duration`, `success`, diff --git a/.apm/skills/observability-cli-guides/scripts/azure-monitor-traces.py b/.apm/skills/observability-cli-guides/scripts/azure-monitor-traces.py index 0d59801..f969991 100755 --- a/.apm/skills/observability-cli-guides/scripts/azure-monitor-traces.py +++ b/.apm/skills/observability-cli-guides/scripts/azure-monitor-traces.py @@ -3,7 +3,7 @@ azure-monitor-traces.py operations --app --service orders-api --from ... --to ... azure-monitor-traces.py dependencies --app --service orders-api --since 30m - azure-monitor-traces.py exemplars --app --service orders-api --slow 3 --failed 3 --since 30m + azure-monitor-traces.py exemplars --app --service orders-api --since 30m azure-monitor-traces.py trace --app [--since 24h] azure-monitor-traces.py watch --app --identity odd-bench/ --from --state /-watch.json [--to ] @@ -13,9 +13,10 @@ dependencies, exemplars and watch take --service (repeatable, a cloud_RoleName; none = every service). operations adds --top (rows, default 20) and --bin (a duration: adds the request count, failures and p95 per -time bucket). exemplars adds --operation (the request name, repeatable), ---slow N (the slowest requests, default 3), --failed N (the newest failed -requests, default 3). trace takes the operation_Id. watch takes --identity +time bucket). exemplars picks, per operation, the request nearest its p50 +and the slowest; it adds --operation (the request name, repeatable), +--slow N (the slowest requests per operation, default 1), --failed N (the +newest failed requests, default 3). trace takes the operation_Id. watch takes --identity (the run's User-Agent prefix, matched with startswith), --state (its state file; the same invocation again resumes it), --bin (default 30s), --ended-after (empty closed bins that end a started run with no schedule @@ -334,7 +335,19 @@ def cmd_exemplars(ns) -> tuple[int, dict]: calls = [ ai_call( ns.app, - f"requests {svc}{op}| top {ns.slow} by duration desc | project {EX_COLS}", + f"requests {svc}{op}| summarize p50=percentile(duration, 50) by cloud_RoleName, name" + f" | join kind=inner (requests {svc}{op}) on cloud_RoleName, name" + " | extend odd_gap = abs(duration - p50)" + " | summarize arg_min(odd_gap, timestamp, duration, success, resultCode, operation_Id, id, p50) by cloud_RoleName, name" + f" | order by p50 desc | project {EX_COLS}, p50", + frm, + to, + ), + ai_call( + ns.app, + f"requests {svc}{op}| extend odd_key = strcat(cloud_RoleName, '/', name)" + f" | partition hint.strategy=native by odd_key (top {ns.slow} by duration desc)" + f" | order by duration desc | project {EX_COLS}", frm, to, ), @@ -346,9 +359,10 @@ def cmd_exemplars(ns) -> tuple[int, dict]: ), ] res = run_many(calls) - slow = ai_rows(res[0].data) if res[0].ok else [] - failed = ai_rows(res[1].data) if res[1].ok else [] - ids = list(dict.fromkeys([r["operation_Id"] for r in slow + failed])) + p50 = ai_rows(res[0].data) if res[0].ok else [] + slow = ai_rows(res[1].data) if res[1].ok else [] + failed = ai_rows(res[2].data) if res[2].ok else [] + ids = list(dict.fromkeys([r["operation_Id"] for r in p50 + slow + failed])) detail: dict[str, dict] = { i: {"dependencies": [], "exceptions": [], "logs": []} for i in ids } @@ -392,10 +406,11 @@ def cmd_exemplars(ns) -> tuple[int, dict]: "message": row["message"], } ) - for r in slow + failed: + for r in p50 + slow + failed: r.update(detail.get(r["operation_Id"], {})) out = { "window": [frm, to], + "p50": p50, "slow": slow, "failed_requests": failed, "failed": failures(res), @@ -420,7 +435,14 @@ def _render_ex(r: dict) -> list[str]: def render_exemplars(o: dict) -> str: - out = [f"slowest requests, {o['window'][0]}..{o['window'][1]}:"] + out = [f"p50 request per operation, {o['window'][0]}..{o['window'][1]}:"] + for r in o["p50"]: + lines = _render_ex(r) + lines[0] += f" (operation p50 {r['p50']:.4g} ms)" + out += lines + if not o["p50"]: + out.append(" (none)") + out.append("slowest requests per operation:") for r in o["slow"]: out += _render_ex(r) if not o["slow"]: @@ -1039,7 +1061,7 @@ def main() -> int: c = sub.add_parser("exemplars") c.add_argument("--service", action="append") c.add_argument("--operation", action="append") - c.add_argument("--slow", type=int, default=3) + c.add_argument("--slow", type=int, default=1) c.add_argument("--failed", type=int, default=3) d = sub.add_parser("trace") d.add_argument("operation_id") diff --git a/tests/skills/observability-cli-guides/fixtures/azure-monitor/1458b21ebbe6.json b/tests/skills/observability-cli-guides/fixtures/azure-monitor/1458b21ebbe6.json new file mode 100644 index 0000000..fa14e4c --- /dev/null +++ b/tests/skills/observability-cli-guides/fixtures/azure-monitor/1458b21ebbe6.json @@ -0,0 +1,20 @@ +{ + "args": [ + "monitor", + "app-insights", + "query", + "--app", + "aaaaaaaa-0000-0000-0000-00000000000a", + "--analytics-query", + "requests | where cloud_RoleName in ('orders-api') | summarize p50=percentile(duration, 50) by cloud_RoleName, name | join kind=inner (requests | where cloud_RoleName in ('orders-api') ) on cloud_RoleName, name | extend odd_gap = abs(duration - p50) | summarize arg_min(odd_gap, timestamp, duration, success, resultCode, operation_Id, id, p50) by cloud_RoleName, name | order by p50 desc | project timestamp, cloud_RoleName, name, duration, success, resultCode, operation_Id, id, p50", + "--start-time", + "2026-09-11T11:05:27Z", + "--end-time", + "2026-09-11T11:20:27Z", + "-o", + "json" + ], + "code": 0, + "stdout": "{\n \"tables\": [\n {\n \"columns\": [\n {\n \"name\": \"timestamp\",\n \"type\": \"datetime\"\n },\n {\n \"name\": \"cloud_RoleName\",\n \"type\": \"string\"\n },\n {\n \"name\": \"name\",\n \"type\": \"string\"\n },\n {\n \"name\": \"duration\",\n \"type\": \"real\"\n },\n {\n \"name\": \"success\",\n \"type\": \"string\"\n },\n {\n \"name\": \"resultCode\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_Id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"p50\",\n \"type\": \"real\"\n }\n ],\n \"name\": \"PrimaryResult\",\n \"rows\": [\n [\n \"2026-09-11T11:16:36.3828658Z\",\n \"orders-api\",\n \"POST /orders/{order_id}/checkout\",\n 550.777,\n \"True\",\n \"200\",\n \"7d0c7db57c2c1649766e32f97c74614b\",\n \"081b5ba8141a49fd\",\n 550.777\n ],\n [\n \"2026-09-11T11:11:12.0080544Z\",\n \"orders-api\",\n \"DELETE /orders/{order_id}\",\n 51.401,\n \"True\",\n \"204\",\n \"4f4c53fcfdc361ff94e6f1be8678f8cb\",\n \"f56946242f99a941\",\n 51.401\n ],\n [\n \"2026-09-11T11:10:49.9473165Z\",\n \"orders-api\",\n \"POST /orders\",\n 1.023,\n \"True\",\n \"201\",\n \"85ee8c30955d03c3bcb331b53b3cc528\",\n \"a29e3f67314ccc8b\",\n 1.0234444444444444\n ],\n [\n \"2026-09-11T11:05:43.2542094Z\",\n \"orders-api\",\n \"GET /products\",\n 0.887,\n \"True\",\n \"200\",\n \"68c20d7c133210ceea71fc01ac1c2801\",\n \"cb2256cd133fce61\",\n 0.8866904761904763\n ],\n [\n \"2026-09-11T11:18:56.3300895Z\",\n \"orders-api\",\n \"GET /orders/{order_id}\",\n 0.8,\n \"True\",\n \"200\",\n \"c84a1aaccb19afdeb0da9689fc5997a9\",\n \"157697b0597196e2\",\n 0.7995000000000001\n ]\n ]\n }\n ]\n}\n", + "stderr": "" +} \ No newline at end of file diff --git a/tests/skills/observability-cli-guides/fixtures/azure-monitor/44a64de3a35e.json b/tests/skills/observability-cli-guides/fixtures/azure-monitor/44a64de3a35e.json new file mode 100644 index 0000000..2cca210 --- /dev/null +++ b/tests/skills/observability-cli-guides/fixtures/azure-monitor/44a64de3a35e.json @@ -0,0 +1,20 @@ +{ + "args": [ + "monitor", + "app-insights", + "query", + "--app", + "aaaaaaaa-0000-0000-0000-00000000000a", + "--analytics-query", + "union dependencies, exceptions, traces | where operation_Id in ('7d0c7db57c2c1649766e32f97c74614b', '4f4c53fcfdc361ff94e6f1be8678f8cb', '85ee8c30955d03c3bcb331b53b3cc528', '68c20d7c133210ceea71fc01ac1c2801', 'c84a1aaccb19afdeb0da9689fc5997a9', '51fe548d6c6cdcdd8fa2a68820f34905', 'd460e57292ad08c472c14df9eee4ff82', '7494d8b86da39ef807b0a5bc76b3730b', 'f37dfaa32ecf7f774b6179515c064edc', '634120ac7688bd25a454d71172733a14', '8a658d9ffe4cf1169e3d2e86f49c70e1', '297c633203270be51ebe0701459f5493', '0ba4d75dd75457fa480dda6acb203fef', 'ccc38a8db02bbaf03c305d62e7b2d47c', 'f0633b94901c3ac17d8170cf85b015db', '31ca424f368f565b68c332fb4f236c48', 'cdf11d86efff6370a1a08774c20b8c4c') | where itemType != 'trace' or severityLevel >= 2 | project itemType, timestamp, id, operation_Id, operation_ParentId, cloud_RoleName, name, duration, success, resultCode, type, target, message, severityLevel, outerMessage, problemId | order by timestamp asc", + "--start-time", + "2026-09-11T11:05:27Z", + "--end-time", + "2026-09-11T11:20:27Z", + "-o", + "json" + ], + "code": 0, + "stdout": "{\n \"tables\": [\n {\n \"columns\": [\n {\n \"name\": \"itemType\",\n \"type\": \"string\"\n },\n {\n \"name\": \"timestamp\",\n \"type\": \"datetime\"\n },\n {\n \"name\": \"id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_Id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_ParentId\",\n \"type\": \"string\"\n },\n {\n \"name\": \"cloud_RoleName\",\n \"type\": \"string\"\n },\n {\n \"name\": \"name\",\n \"type\": \"string\"\n },\n {\n \"name\": \"duration\",\n \"type\": \"real\"\n },\n {\n \"name\": \"success\",\n \"type\": \"string\"\n },\n {\n \"name\": \"resultCode\",\n \"type\": \"string\"\n },\n {\n \"name\": \"type\",\n \"type\": \"string\"\n },\n {\n \"name\": \"target\",\n \"type\": \"string\"\n },\n {\n \"name\": \"message\",\n \"type\": \"string\"\n },\n {\n \"name\": \"severityLevel\",\n \"type\": \"int\"\n },\n {\n \"name\": \"outerMessage\",\n \"type\": \"string\"\n },\n {\n \"name\": \"problemId\",\n \"type\": \"string\"\n }\n ],\n \"name\": \"PrimaryResult\",\n \"rows\": [\n [\n \"dependency\",\n \"2026-09-11T11:05:43.2532372Z\",\n \"b80faa4b7c69e7f7\",\n \"68c20d7c133210ceea71fc01ac1c2801\",\n \"68c20d7c133210ceea71fc01ac1c2801\",\n \"load-generator\",\n \"GET\",\n 2.056,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:06:16.7848166Z\",\n \"c57f58070c1ee38a\",\n \"ccc38a8db02bbaf03c305d62e7b2d47c\",\n \"ccc38a8db02bbaf03c305d62e7b2d47c\",\n \"load-generator\",\n \"POST\",\n 3.349,\n \"True\",\n \"201\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:06:26.7500771Z\",\n \"a060d545d03e75e3\",\n \"7494d8b86da39ef807b0a5bc76b3730b\",\n \"7494d8b86da39ef807b0a5bc76b3730b\",\n \"load-generator\",\n \"DELETE\",\n 54.487,\n \"False\",\n \"500\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:06:26.7511392Z\",\n \"f78a7a5c9fb2c869\",\n \"7494d8b86da39ef807b0a5bc76b3730b\",\n \"7fa00349870aa010\",\n \"orders-api\",\n \"storage.delete\",\n 50.996,\n \"False\",\n \"2\",\n \"Other\",\n \"storage.delete\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"trace\",\n \"2026-09-11T11:06:26.8017751Z\",\n \"\",\n \"7494d8b86da39ef807b0a5bc76b3730b\",\n \"f78a7a5c9fb2c869\",\n \"orders-api\",\n \"\",\n null,\n \"\",\n \"\",\n \"\",\n \"\",\n \"storage flake, delete failed for order 2228\",\n 2,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:06:26.8084296Z\",\n \"38e9378d6043c220\",\n \"f0633b94901c3ac17d8170cf85b015db\",\n \"f0633b94901c3ac17d8170cf85b015db\",\n \"load-generator\",\n \"GET\",\n 6.651,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:06:39.3855713Z\",\n \"0c5f4c982fcfac83\",\n \"8a658d9ffe4cf1169e3d2e86f49c70e1\",\n \"8a658d9ffe4cf1169e3d2e86f49c70e1\",\n \"load-generator\",\n \"GET\",\n 3.588,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:08:05.4416818Z\",\n \"2a8214a6f292bc3c\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"load-generator\",\n \"POST\",\n 792.252,\n \"False\",\n \"502\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:08:05.4428362Z\",\n \"55ec7287db9078b2\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"f8c992d680666eb2\",\n \"orders-api\",\n \"payment.authorize\",\n 789.694,\n \"False\",\n \"2\",\n \"Other\",\n \"payment.authorize\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"trace\",\n \"2026-09-11T11:08:06.2322163Z\",\n \"\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"55ec7287db9078b2\",\n \"orders-api\",\n \"\",\n null,\n \"\",\n \"\",\n \"\",\n \"\",\n \"payment upstream unavailable, checkout failed for order 2273\",\n 2,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:08:12.8569619Z\",\n \"6fc65c38d4463be4\",\n \"634120ac7688bd25a454d71172733a14\",\n \"634120ac7688bd25a454d71172733a14\",\n \"load-generator\",\n \"GET\",\n 6.293,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:10:46.9513246Z\",\n \"e1c1d781b06f68d7\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"load-generator\",\n \"POST\",\n 798.255,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:10:46.9526406Z\",\n \"fb4dfc436ae2f81a\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"84192512f6ba130b\",\n \"orders-api\",\n \"payment.authorize\",\n 795.348,\n \"True\",\n \"0\",\n \"Other\",\n \"payment.authorize\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:10:49.9461317Z\",\n \"7a20b810fc51c229\",\n \"85ee8c30955d03c3bcb331b53b3cc528\",\n \"85ee8c30955d03c3bcb331b53b3cc528\",\n \"load-generator\",\n \"POST\",\n 2.5,\n \"True\",\n \"201\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:11:12.0070671Z\",\n \"814e6e9890421267\",\n \"4f4c53fcfdc361ff94e6f1be8678f8cb\",\n \"4f4c53fcfdc361ff94e6f1be8678f8cb\",\n \"load-generator\",\n \"DELETE\",\n 52.864,\n \"True\",\n \"204\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:11:12.0083596Z\",\n \"4e830bbf6e8bfbe1\",\n \"4f4c53fcfdc361ff94e6f1be8678f8cb\",\n \"f56946242f99a941\",\n \"orders-api\",\n \"storage.delete\",\n 50.402,\n \"True\",\n \"0\",\n \"Other\",\n \"storage.delete\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:11:33.6708011Z\",\n \"f1313d3384ee79b5\",\n \"f37dfaa32ecf7f774b6179515c064edc\",\n \"f37dfaa32ecf7f774b6179515c064edc\",\n \"load-generator\",\n \"DELETE\",\n 54.459,\n \"True\",\n \"204\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:11:33.672266Z\",\n \"7e6527507f33f624\",\n \"f37dfaa32ecf7f774b6179515c064edc\",\n \"f3c59a650dc20dfb\",\n \"orders-api\",\n \"storage.delete\",\n 51.473,\n \"True\",\n \"0\",\n \"Other\",\n \"storage.delete\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:13:46.2036393Z\",\n \"7cb2640f37b95a09\",\n \"0ba4d75dd75457fa480dda6acb203fef\",\n \"0ba4d75dd75457fa480dda6acb203fef\",\n \"load-generator\",\n \"POST\",\n 4.953,\n \"True\",\n \"201\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:16:36.3818655Z\",\n \"551cd4183046ce79\",\n \"7d0c7db57c2c1649766e32f97c74614b\",\n \"7d0c7db57c2c1649766e32f97c74614b\",\n \"load-generator\",\n \"POST\",\n 552.145,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:16:36.3830957Z\",\n \"10f9076166ac6320\",\n \"7d0c7db57c2c1649766e32f97c74614b\",\n \"081b5ba8141a49fd\",\n \"orders-api\",\n \"payment.authorize\",\n 549.676,\n \"True\",\n \"0\",\n \"Other\",\n \"payment.authorize\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:17:42.8635317Z\",\n \"fdd61856a01e9bfd\",\n \"297c633203270be51ebe0701459f5493\",\n \"297c633203270be51ebe0701459f5493\",\n \"load-generator\",\n \"GET\",\n 6.106,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:18:56.3289092Z\",\n \"fbf62100fbb51ad3\",\n \"c84a1aaccb19afdeb0da9689fc5997a9\",\n \"c84a1aaccb19afdeb0da9689fc5997a9\",\n \"load-generator\",\n \"GET\",\n 2.272,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:19:48.7009978Z\",\n \"b3a5af6a942e6bfc\",\n \"cdf11d86efff6370a1a08774c20b8c4c\",\n \"cdf11d86efff6370a1a08774c20b8c4c\",\n \"load-generator\",\n \"GET\",\n 3.224,\n \"False\",\n \"404\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:19:58.7623074Z\",\n \"311579e7dd10e76b\",\n \"31ca424f368f565b68c332fb4f236c48\",\n \"31ca424f368f565b68c332fb4f236c48\",\n \"load-generator\",\n \"GET\",\n 2.582,\n \"False\",\n \"404\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ]\n ]\n }\n ]\n}\n", + "stderr": "" +} \ No newline at end of file diff --git a/tests/skills/observability-cli-guides/fixtures/azure-monitor/5a33d2716413.json b/tests/skills/observability-cli-guides/fixtures/azure-monitor/5a33d2716413.json deleted file mode 100644 index 41bc4ad..0000000 --- a/tests/skills/observability-cli-guides/fixtures/azure-monitor/5a33d2716413.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "args": [ - "monitor", - "app-insights", - "query", - "--app", - "aaaaaaaa-0000-0000-0000-00000000000a", - "--analytics-query", - "requests | where cloud_RoleName in ('orders-api') | top 2 by duration desc | project timestamp, cloud_RoleName, name, duration, success, resultCode, operation_Id, id", - "--start-time", - "2026-09-11T11:07:57Z", - "--end-time", - "2026-09-11T11:22:57Z", - "-o", - "json" - ], - "code": 0, - "stdout": "{\n \"tables\": [\n {\n \"columns\": [\n {\n \"name\": \"timestamp\",\n \"type\": \"datetime\"\n },\n {\n \"name\": \"cloud_RoleName\",\n \"type\": \"string\"\n },\n {\n \"name\": \"name\",\n \"type\": \"string\"\n },\n {\n \"name\": \"duration\",\n \"type\": \"real\"\n },\n {\n \"name\": \"success\",\n \"type\": \"string\"\n },\n {\n \"name\": \"resultCode\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_Id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"id\",\n \"type\": \"string\"\n }\n ],\n \"name\": \"PrimaryResult\",\n \"rows\": [\n [\n \"2026-09-11T11:10:46.9523392Z\",\n \"orders-api\",\n \"POST /orders/{order_id}/checkout\",\n 796.709,\n \"True\",\n \"200\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"84192512f6ba130b\"\n ],\n [\n \"2026-09-11T11:08:05.4425919Z\",\n \"orders-api\",\n \"POST /orders/{order_id}/checkout\",\n 790.852,\n \"False\",\n \"502\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"f8c992d680666eb2\"\n ]\n ]\n }\n ]\n}\n", - "stderr": "" -} \ No newline at end of file diff --git a/tests/skills/observability-cli-guides/fixtures/azure-monitor/7bf86e597863.json b/tests/skills/observability-cli-guides/fixtures/azure-monitor/7bf86e597863.json index a1b5e27..c7edca3 100644 --- a/tests/skills/observability-cli-guides/fixtures/azure-monitor/7bf86e597863.json +++ b/tests/skills/observability-cli-guides/fixtures/azure-monitor/7bf86e597863.json @@ -8,13 +8,13 @@ "--analytics-query", "requests | where cloud_RoleName in ('orders-api') | where success == false | top 2 by timestamp desc | project timestamp, cloud_RoleName, name, duration, success, resultCode, operation_Id, id", "--start-time", - "2026-09-11T11:07:57Z", + "2026-09-11T11:05:27Z", "--end-time", - "2026-09-11T11:22:57Z", + "2026-09-11T11:20:27Z", "-o", "json" ], "code": 0, - "stdout": "{\n \"tables\": [\n {\n \"columns\": [\n {\n \"name\": \"timestamp\",\n \"type\": \"datetime\"\n },\n {\n \"name\": \"cloud_RoleName\",\n \"type\": \"string\"\n },\n {\n \"name\": \"name\",\n \"type\": \"string\"\n },\n {\n \"name\": \"duration\",\n \"type\": \"real\"\n },\n {\n \"name\": \"success\",\n \"type\": \"string\"\n },\n {\n \"name\": \"resultCode\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_Id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"id\",\n \"type\": \"string\"\n }\n ],\n \"name\": \"PrimaryResult\",\n \"rows\": [\n [\n \"2026-09-11T11:22:15.2442219Z\",\n \"orders-api\",\n \"GET /orders/{order_id}\",\n 0.954,\n \"False\",\n \"404\",\n \"8015164d4f2eb710f78badccf13027f2\",\n \"c146aba9f2912773\"\n ],\n [\n \"2026-09-11T11:22:09.3291453Z\",\n \"orders-api\",\n \"GET /orders/{order_id}\",\n 0.909,\n \"False\",\n \"404\",\n \"4875678a59d063838e93a7d906aba956\",\n \"77e90dd20d7aafe1\"\n ]\n ]\n }\n ]\n}\n", + "stdout": "{\n \"tables\": [\n {\n \"columns\": [\n {\n \"name\": \"timestamp\",\n \"type\": \"datetime\"\n },\n {\n \"name\": \"cloud_RoleName\",\n \"type\": \"string\"\n },\n {\n \"name\": \"name\",\n \"type\": \"string\"\n },\n {\n \"name\": \"duration\",\n \"type\": \"real\"\n },\n {\n \"name\": \"success\",\n \"type\": \"string\"\n },\n {\n \"name\": \"resultCode\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_Id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"id\",\n \"type\": \"string\"\n }\n ],\n \"name\": \"PrimaryResult\",\n \"rows\": [\n [\n \"2026-09-11T11:19:58.7637159Z\",\n \"orders-api\",\n \"GET /orders/{order_id}\",\n 0.896,\n \"False\",\n \"404\",\n \"31ca424f368f565b68c332fb4f236c48\",\n \"5cae9fbe91cc5abf\"\n ],\n [\n \"2026-09-11T11:19:48.7022857Z\",\n \"orders-api\",\n \"GET /orders/{order_id}\",\n 1.084,\n \"False\",\n \"404\",\n \"cdf11d86efff6370a1a08774c20b8c4c\",\n \"dd3252eefe53f074\"\n ]\n ]\n }\n ]\n}\n", "stderr": "" } \ No newline at end of file diff --git a/tests/skills/observability-cli-guides/fixtures/azure-monitor/d608a619adfc.json b/tests/skills/observability-cli-guides/fixtures/azure-monitor/d608a619adfc.json new file mode 100644 index 0000000..07b56af --- /dev/null +++ b/tests/skills/observability-cli-guides/fixtures/azure-monitor/d608a619adfc.json @@ -0,0 +1,20 @@ +{ + "args": [ + "monitor", + "app-insights", + "query", + "--app", + "aaaaaaaa-0000-0000-0000-00000000000a", + "--analytics-query", + "requests | where cloud_RoleName in ('orders-api') | extend odd_key = strcat(cloud_RoleName, '/', name) | partition hint.strategy=native by odd_key (top 2 by duration desc) | order by duration desc | project timestamp, cloud_RoleName, name, duration, success, resultCode, operation_Id, id", + "--start-time", + "2026-09-11T11:05:27Z", + "--end-time", + "2026-09-11T11:20:27Z", + "-o", + "json" + ], + "code": 0, + "stdout": "{\n \"tables\": [\n {\n \"columns\": [\n {\n \"name\": \"timestamp\",\n \"type\": \"datetime\"\n },\n {\n \"name\": \"cloud_RoleName\",\n \"type\": \"string\"\n },\n {\n \"name\": \"name\",\n \"type\": \"string\"\n },\n {\n \"name\": \"duration\",\n \"type\": \"real\"\n },\n {\n \"name\": \"success\",\n \"type\": \"string\"\n },\n {\n \"name\": \"resultCode\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_Id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"id\",\n \"type\": \"string\"\n }\n ],\n \"name\": \"PrimaryResult\",\n \"rows\": [\n [\n \"2026-09-11T11:10:46.9523392Z\",\n \"orders-api\",\n \"POST /orders/{order_id}/checkout\",\n 796.709,\n \"True\",\n \"200\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"84192512f6ba130b\"\n ],\n [\n \"2026-09-11T11:08:05.4425919Z\",\n \"orders-api\",\n \"POST /orders/{order_id}/checkout\",\n 790.852,\n \"False\",\n \"502\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"f8c992d680666eb2\"\n ],\n [\n \"2026-09-11T11:06:26.750922Z\",\n \"orders-api\",\n \"DELETE /orders/{order_id}\",\n 54.3,\n \"False\",\n \"500\",\n \"7494d8b86da39ef807b0a5bc76b3730b\",\n \"7fa00349870aa010\"\n ],\n [\n \"2026-09-11T11:11:33.6719253Z\",\n \"orders-api\",\n \"DELETE /orders/{order_id}\",\n 52.629,\n \"True\",\n \"204\",\n \"f37dfaa32ecf7f774b6179515c064edc\",\n \"f3c59a650dc20dfb\"\n ],\n [\n \"2026-09-11T11:08:12.8579144Z\",\n \"orders-api\",\n \"GET /orders/{order_id}\",\n 4.451,\n \"True\",\n \"200\",\n \"634120ac7688bd25a454d71172733a14\",\n \"0ed5e8034381c6bd\"\n ],\n [\n \"2026-09-11T11:06:39.3875242Z\",\n \"orders-api\",\n \"GET /products\",\n 3.608,\n \"True\",\n \"200\",\n \"8a658d9ffe4cf1169e3d2e86f49c70e1\",\n \"3434d76768fd17c7\"\n ],\n [\n \"2026-09-11T11:17:42.8673335Z\",\n \"orders-api\",\n \"GET /products\",\n 3.201,\n \"True\",\n \"200\",\n \"297c633203270be51ebe0701459f5493\",\n \"da056b805c99e8d3\"\n ],\n [\n \"2026-09-11T11:13:46.205267Z\",\n \"orders-api\",\n \"POST /orders\",\n 2.979,\n \"True\",\n \"201\",\n \"0ba4d75dd75457fa480dda6acb203fef\",\n \"0ea70d9efb51f2bf\"\n ],\n [\n \"2026-09-11T11:06:16.7861759Z\",\n \"orders-api\",\n \"POST /orders\",\n 2.805,\n \"True\",\n \"201\",\n \"ccc38a8db02bbaf03c305d62e7b2d47c\",\n \"7ba9e50bb80e645d\"\n ],\n [\n \"2026-09-11T11:06:26.810381Z\",\n \"orders-api\",\n \"GET /orders/{order_id}\",\n 2.279,\n \"True\",\n \"200\",\n \"f0633b94901c3ac17d8170cf85b015db\",\n \"c5877ca1565f6e0e\"\n ]\n ]\n }\n ]\n}\n", + "stderr": "" +} \ No newline at end of file diff --git a/tests/skills/observability-cli-guides/fixtures/azure-monitor/f0f111e20734.json b/tests/skills/observability-cli-guides/fixtures/azure-monitor/f0f111e20734.json deleted file mode 100644 index fe7d853..0000000 --- a/tests/skills/observability-cli-guides/fixtures/azure-monitor/f0f111e20734.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "args": [ - "monitor", - "app-insights", - "query", - "--app", - "aaaaaaaa-0000-0000-0000-00000000000a", - "--analytics-query", - "union dependencies, exceptions, traces | where operation_Id in ('51fe548d6c6cdcdd8fa2a68820f34905', 'd460e57292ad08c472c14df9eee4ff82', '8015164d4f2eb710f78badccf13027f2', '4875678a59d063838e93a7d906aba956') | where itemType != 'trace' or severityLevel >= 2 | project itemType, timestamp, id, operation_Id, operation_ParentId, cloud_RoleName, name, duration, success, resultCode, type, target, message, severityLevel, outerMessage, problemId | order by timestamp asc", - "--start-time", - "2026-09-11T11:07:57Z", - "--end-time", - "2026-09-11T11:22:57Z", - "-o", - "json" - ], - "code": 0, - "stdout": "{\n \"tables\": [\n {\n \"columns\": [\n {\n \"name\": \"itemType\",\n \"type\": \"string\"\n },\n {\n \"name\": \"timestamp\",\n \"type\": \"datetime\"\n },\n {\n \"name\": \"id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_Id\",\n \"type\": \"string\"\n },\n {\n \"name\": \"operation_ParentId\",\n \"type\": \"string\"\n },\n {\n \"name\": \"cloud_RoleName\",\n \"type\": \"string\"\n },\n {\n \"name\": \"name\",\n \"type\": \"string\"\n },\n {\n \"name\": \"duration\",\n \"type\": \"real\"\n },\n {\n \"name\": \"success\",\n \"type\": \"string\"\n },\n {\n \"name\": \"resultCode\",\n \"type\": \"string\"\n },\n {\n \"name\": \"type\",\n \"type\": \"string\"\n },\n {\n \"name\": \"target\",\n \"type\": \"string\"\n },\n {\n \"name\": \"message\",\n \"type\": \"string\"\n },\n {\n \"name\": \"severityLevel\",\n \"type\": \"int\"\n },\n {\n \"name\": \"outerMessage\",\n \"type\": \"string\"\n },\n {\n \"name\": \"problemId\",\n \"type\": \"string\"\n }\n ],\n \"name\": \"PrimaryResult\",\n \"rows\": [\n [\n \"dependency\",\n \"2026-09-11T11:08:05.4416818Z\",\n \"2a8214a6f292bc3c\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"load-generator\",\n \"POST\",\n 792.252,\n \"False\",\n \"502\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:08:05.4428362Z\",\n \"55ec7287db9078b2\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"f8c992d680666eb2\",\n \"orders-api\",\n \"payment.authorize\",\n 789.694,\n \"False\",\n \"2\",\n \"Other\",\n \"payment.authorize\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"trace\",\n \"2026-09-11T11:08:06.2322163Z\",\n \"\",\n \"d460e57292ad08c472c14df9eee4ff82\",\n \"55ec7287db9078b2\",\n \"orders-api\",\n \"\",\n null,\n \"\",\n \"\",\n \"\",\n \"\",\n \"payment upstream unavailable, checkout failed for order 2273\",\n 2,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:10:46.9513246Z\",\n \"e1c1d781b06f68d7\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"load-generator\",\n \"POST\",\n 798.255,\n \"True\",\n \"200\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:10:46.9526406Z\",\n \"fb4dfc436ae2f81a\",\n \"51fe548d6c6cdcdd8fa2a68820f34905\",\n \"84192512f6ba130b\",\n \"orders-api\",\n \"payment.authorize\",\n 795.348,\n \"True\",\n \"0\",\n \"Other\",\n \"payment.authorize\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:22:09.3276452Z\",\n \"712c76f095238427\",\n \"4875678a59d063838e93a7d906aba956\",\n \"4875678a59d063838e93a7d906aba956\",\n \"load-generator\",\n \"GET\",\n 2.759,\n \"False\",\n \"404\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ],\n [\n \"dependency\",\n \"2026-09-11T11:22:15.2430192Z\",\n \"8fe8a41b01cc8660\",\n \"8015164d4f2eb710f78badccf13027f2\",\n \"8015164d4f2eb710f78badccf13027f2\",\n \"load-generator\",\n \"GET\",\n 2.491,\n \"False\",\n \"404\",\n \"HTTP\",\n \"localhost:8000\",\n \"\",\n null,\n \"\",\n \"\"\n ]\n ]\n }\n ]\n}\n", - "stderr": "" -} \ No newline at end of file diff --git a/tests/skills/observability-cli-guides/test_azure_monitor_scripts.py b/tests/skills/observability-cli-guides/test_azure_monitor_scripts.py index 9cf8dad..2ea4efd 100644 --- a/tests/skills/observability-cli-guides/test_azure_monitor_scripts.py +++ b/tests/skills/observability-cli-guides/test_azure_monitor_scripts.py @@ -420,7 +420,7 @@ def test_traces_dependencies_are_joined_on_operation_id(fake): assert "join kind=inner (requests" in result.stdout -def test_traces_exemplars_carry_their_dependencies_and_logs(fake): +def test_traces_exemplars_pick_a_p50_and_the_slowest_per_operation(fake): code, out = fake.json( "traces", "exemplars", @@ -435,9 +435,19 @@ def test_traces_exemplars_carry_their_dependencies_and_logs(fake): *WINDOW, ) assert code == 0 + # one request per operation nearest its own p50 + p50 = out["p50"] + assert len({r["name"] for r in p50}) == len(p50) == 5 + checkout = p50[0] + assert checkout["name"] == "POST /orders/{order_id}/checkout" + assert abs(checkout["duration"] - checkout["p50"]) < 5 + # the slowest are ranked per operation: every operation gets its two slow = out["slow"] - assert len(slow) == 2 and slow[0]["duration"] == 796.709 - assert slow[0]["operation_Id"] == OPID + assert len(slow) == 10 + assert all( + sum(r["name"] == name for r in slow) == 2 for name in {r["name"] for r in p50} + ) + assert slow[0]["duration"] == 796.709 and slow[0]["operation_Id"] == OPID assert [d["name"] for d in slow[0]["dependencies"]] == ["POST", "payment.authorize"] failed = slow[1] assert failed["resultCode"] == "502" @@ -448,8 +458,9 @@ def test_traces_exemplars_carry_their_dependencies_and_logs(fake): } ] assert [r["resultCode"] for r in out["failed_requests"]] == ["404", "404"] - assert "union dependencies, exceptions, traces" in out["commands"][2] - assert len(out["commands"]) == 3, ( + assert "partition hint.strategy=native by odd_key" in out["commands"][1] + assert "union dependencies, exceptions, traces" in out["commands"][3] + assert len(out["commands"]) == 4, ( "one union over the picked ids, never one call per exemplar" ) From bd2b11164b1bb93d35d9047e24838305dee33d61 Mon Sep 17 00:00:00 2001 From: using-system Date: Sat, 26 Sep 2026 20:18:13 +0200 Subject: [PATCH 3/7] fix(skill): count grafana label values past an adaptive metrics aggregation `grafana-metrics.py labels --label` exited 1 on Grafana Cloud as soon as one series under the selector was aggregated by an Adaptive Metrics rule. It now retries once without those series (`__aggregation__=""`) and says so in its output. Closes #659 Co-Authored-By: Claude Opus 5.5 (1M context) --- .../references/grafana.md | 4 ++- .../scripts/grafana-metrics.py | 26 +++++++++++++- .../fixtures/grafana/m_aggregated.json | 1 + .../test_grafana_scripts.py | 35 +++++++++++++++++++ 4 files changed, 64 insertions(+), 2 deletions(-) create mode 100644 tests/skills/observability-cli-guides/fixtures/grafana/m_aggregated.json diff --git a/.apm/skills/observability-cli-guides/references/grafana.md b/.apm/skills/observability-cli-guides/references/grafana.md index 391840a..e0c72c4 100644 --- a/.apm/skills/observability-cli-guides/references/grafana.md +++ b/.apm/skills/observability-cli-guides/references/grafana.md @@ -192,7 +192,9 @@ lists the distinct metric names behind one or more `--match` selectors not evidence of an absent metric). `labels` takes one `--match` selector and a window: with `--label ` it lists that label's values across the series the selector matches in the window (each with its series -count), without it the label *names* those series carry. `instant` and +count; on Grafana Cloud, the series an Adaptive Metrics rule aggregated +left out and said so, verified 2026-09-26), without it the label *names* +those series carry. `instant` and `range` take a raw PromQL expression for anything the first four do not shape — always with a selector or an aggregation, since a bare metric name lists every series it has; `instant` takes `--at` (default now), diff --git a/.apm/skills/observability-cli-guides/scripts/grafana-metrics.py b/.apm/skills/observability-cli-guides/scripts/grafana-metrics.py index 6db9cce..5e598a1 100755 --- a/.apm/skills/observability-cli-guides/scripts/grafana-metrics.py +++ b/.apm/skills/observability-cli-guides/scripts/grafana-metrics.py @@ -142,6 +142,19 @@ def cmd_names(ns) -> tuple[int, dict]: } +AGGREGATED = "Can't query aggregated metric" + + +def _unaggregated(match: str) -> str: + """The selector without the series an Adaptive Metrics rule aggregated.""" + if match.rstrip().endswith("}"): + head = match.rstrip()[:-1] + return ( + head + (", " if head.rstrip()[-1:] != "{" else "") + '__aggregation__=""}' + ) + return match + '{__aggregation__=""}' + + def cmd_labels(ns) -> tuple[int, dict]: """A label's values (--label) or the label names behind a selector.""" frm, to = resolve_window(ns) @@ -152,6 +165,14 @@ def cmd_labels(ns) -> tuple[int, dict]: at, win = _settled(frm, to, "0s") expr = f"count by ({ns.label}) (last_over_time({ns.match}{win}))" r = run_gcx(["metrics", "query", expr, "--time", at]) + cmds, note = [r.command], None + if not r.ok and AGGREGATED in (r.error or ""): + # Grafana Cloud refuses the whole selector when one series under + # it is aggregated: count the others, and say so + expr = f"count by ({ns.label}) (last_over_time({_unaggregated(ns.match)}{win}))" + r = run_gcx(["metrics", "query", expr, "--time", at]) + cmds.append(r.command) + note = "an Adaptive Metrics rule aggregates series under the selector: counted without them" values: dict[str, int] = {} for x in prom_result(r.data) if r.ok else []: try: @@ -163,7 +184,8 @@ def cmd_labels(ns) -> tuple[int, dict]: "error": r.error, "label": ns.label, "values": dict(sorted(values.items(), key=lambda kv: -kv[1])), - "commands": [r.command], + "note": note, + "commands": cmds, } r = run_gcx(["metrics", "series", ns.match, "--from", frm, "--to", to]) names: dict[str, int] = {} @@ -311,6 +333,8 @@ def render(o: dict) -> str: out += [f"{v or '(unset)'} ({c} {unit})" for v, c in o["values"].items()] or [ "(no series)" ] + if o.get("note"): + out.append(o["note"]) elif isinstance(o.get("rows"), dict): cols = [ c diff --git a/tests/skills/observability-cli-guides/fixtures/grafana/m_aggregated.json b/tests/skills/observability-cli-guides/fixtures/grafana/m_aggregated.json new file mode 100644 index 0000000..e1de61c --- /dev/null +++ b/tests/skills/observability-cli-guides/fixtures/grafana/m_aggregated.json @@ -0,0 +1 @@ +{"type":"gcx.error","schema_version":"1","error":{"summary":"Invalid Prometheus query","exitCode":1,"details":"execution: Can't query aggregated metric http_server_response_body_size_bytes_bucket without aggregation because the following labels are aggregated: error_type, http_response_status_code. Use an appropriate aggregation function to query these series, such as sum by (...) (rate(http_server_response_body_size_bytes_bucket[5m])). If you need to combine aggregated and non-aggregated series, query each metric name in its own aggregation, such as sum by (...) (rate(http_server_response_body_size_bytes_bucket[5m])) + sum by (...) (rate(target_info[5m])).","suggestions":["Run 'gcx metrics query --help' for usage and examples","If the cause isn't clear from the details, fetch the documentation at https://grafana.com/docs/grafana/latest/datasources/prometheus/query-editor.md for guidance before retrying."],"docsLink":"https://grafana.com/docs/grafana/latest/datasources/prometheus/query-editor.md"}} diff --git a/tests/skills/observability-cli-guides/test_grafana_scripts.py b/tests/skills/observability-cli-guides/test_grafana_scripts.py index 5a9349d..5007c89 100644 --- a/tests/skills/observability-cli-guides/test_grafana_scripts.py +++ b/tests/skills/observability-cli-guides/test_grafana_scripts.py @@ -72,6 +72,9 @@ def flag(name): if a[:2] == ["metrics", "query"]: q = a[2] if "no_such" in q: out(fx("m_empty")) + if os.environ.get("FAKE_AGGREGATED") == "1" and "count by" in q and "__aggregation__" not in q: + # an Adaptive Metrics rule aggregates a series under the selector + print(open(os.path.join(F, "m_aggregated.json")).read()); sys.exit(1) if "sum by (" == q: err() if "--step" in a: out(fx("m_range")) if "histogram_quantile" in q: out(fx("m_hist")) @@ -187,6 +190,7 @@ def fake_gcx(tmp_path, monkeypatch): "FAKE_T_EDGE", "FAKE_M_SERIES_ERR", "FAKE_L_ENV", + "FAKE_AGGREGATED", ): monkeypatch.delenv(var, raising=False) return log @@ -568,6 +572,37 @@ def test_metrics_labels_lists_values_and_names_from_verified_queries(fake_gcx): assert "count by (http_route) (last_over_time(" in o["commands"][0] +def test_metrics_labels_skips_the_series_an_adaptive_rule_aggregated( + fake_gcx, monkeypatch +): + monkeypatch.setenv("FAKE_AGGREGATED", "1") + r = run( + "grafana-metrics", + "labels", + "--match", + '{service_name="svc"}', + "--label", + "http_route", + *WIN, + "--json", + ) + assert r.returncode == 0, r.stdout + o = json.loads(r.stdout) + assert o["values"] and not o["error"] + assert '{service_name="svc", __aggregation__=""}' in o["commands"][-1] + assert "Adaptive Metrics" in o["note"] + r = run( + "grafana-metrics", + "labels", + "--match", + '{service_name="svc"}', + "--label", + "http_route", + *WIN, + ) + assert "Adaptive Metrics" in r.stdout + + def test_errors_are_one_per_fact_and_commands_fold_losslessly(): gcx = load("grafana_gcx") bad = [ From f062a4dbee31cc01bdf4bf1c1067bf353641795b Mon Sep 17 00:00:00 2001 From: using-system Date: Sat, 26 Sep 2026 20:20:55 +0200 Subject: [PATCH 4/7] fix(skill): fall back to increase() for grafana ops calls after a reset `grafana-traces.py ops` withheld every operation's calls when the span metrics counter reset inside the window. It now reads the counter's increase() over the window for the reset operations, as `histogram` and `counter` already do, and marks the row. Closes #659 Co-Authored-By: Claude Opus 5.5 (1M context) --- .../references/grafana.md | 4 +-- .../scripts/grafana-traces.py | 31 ++++++++++++++++--- .../test_grafana_scripts.py | 18 ++++++----- 3 files changed, 40 insertions(+), 13 deletions(-) diff --git a/.apm/skills/observability-cli-guides/references/grafana.md b/.apm/skills/observability-cli-guides/references/grafana.md index e0c72c4..41d3a0b 100644 --- a/.apm/skills/observability-cli-guides/references/grafana.md +++ b/.apm/skills/observability-cli-guides/references/grafana.md @@ -232,8 +232,8 @@ Six subcommands, the whole surface above (`--since ` replaces - `ops` — per operation of each service: rooted and containing trace counts, span-level p50/p95/p99 and calls (span metrics, settled, - bucket-interpolated; `RESET` and calls withheld when the counter fell - inside the window; absence said), trace-level p50/p95/max over the + bucket-interpolated; `RESET` and calls from `increase()` when the + counter fell inside the window, verified 2026-09-26; absence said), trace-level p50/p95/max over the rooted traces (integer ms), the worst containing trace. `--fetch` adds each operation's p50, worst-rooted and worst-containing exemplar with its summary. `--name` adds an operation the roots do not show. diff --git a/.apm/skills/observability-cli-guides/scripts/grafana-traces.py b/.apm/skills/observability-cli-guides/scripts/grafana-traces.py index 133132c..2fe5dd1 100755 --- a/.apm/skills/observability-cli-guides/scripts/grafana-traces.py +++ b/.apm/skills/observability-cli-guides/scripts/grafana-traces.py @@ -247,7 +247,30 @@ def _span_quantiles(keys, frm: str, to: str, settle: str) -> tuple[dict, list, b else: e["span_calls"] = None out[key] = e - return out, results, present + # a reset withholds the subtraction: the counter's increase() over the + # window, which counts across resets, stands in (as histogram and counter do) + reset = [k for k in keys if out[k].get("span_calls_reset")] + extra = run_many( + [ + [ + "metrics", + "query", + f'sum(increase(traces_spanmetrics_calls_total{{service="{s}", span_name="{n}"}}{win}))', + "--time", + at, + ] + for s, n in reset + ] + ) + for key, r in zip(reset, extra): + res = prom_result(r.data) if r.ok else [] + try: + out[key]["span_calls_increase"] = ( + round(float(res[0]["value"][1])) if res else None + ) + except (KeyError, IndexError, TypeError, ValueError): + out[key]["span_calls_increase"] = None + return out, results + list(extra), present def cmd_ops(ns) -> tuple[int, dict]: @@ -1256,10 +1279,10 @@ def render(o: dict) -> str: ) for k, e in o["operations"].items(): out.append( - f"{k:44s} {_f(e['rooted_traces']):>6} {_f(e['containing_traces']):>7} {_f(e.get('span_p50_ms')):>8} {_f(e.get('span_p95_ms')):>7} {_f(e.get('span_p99_ms')):>7} {_f(e.get('span_calls')):>6} | " + f"{k:44s} {_f(e['rooted_traces']):>6} {_f(e['containing_traces']):>7} {_f(e.get('span_p50_ms')):>8} {_f(e.get('span_p95_ms')):>7} {_f(e.get('span_p99_ms')):>7} {_f(e.get('span_calls') if e.get('span_calls') is not None else e.get('span_calls_increase')):>6} | " f"{_f(e['trace_p50_ms']):>9} {_f(e['trace_p95_ms']):>7} {_f(e['trace_max_ms']):>7} | {e['worst_containing_trace']} ({_f(e['worst_containing_ms'])} ms)" f"{' TRUNCATED' if e['truncated'] else ''}" - f"{' RESET inside the window (calls withheld)' if e.get('span_calls_reset') else ''}" + f"{' RESET inside the window (calls = increase())' if e.get('span_calls_reset') else ''}" ) for s, nr in (o.get("never_rooted") or {}).items(): out.append( @@ -1271,7 +1294,7 @@ def render(o: dict) -> str: ) ) out.append( - " span p50/p95/p99 and calls: span metrics, settled (bucket-interpolated latency; calls = raw settled - raw start, withheld on a reset)" + " span p50/p95/p99 and calls: span metrics, settled (bucket-interpolated latency; calls = raw settled - raw start, increase() on a reset)" + ( "" if o.get("span_metrics_present") diff --git a/tests/skills/observability-cli-guides/test_grafana_scripts.py b/tests/skills/observability-cli-guides/test_grafana_scripts.py index 5007c89..920a1ae 100644 --- a/tests/skills/observability-cli-guides/test_grafana_scripts.py +++ b/tests/skills/observability-cli-guides/test_grafana_scripts.py @@ -525,18 +525,22 @@ def test_counter_reset_inside_the_window_withholds_the_delta(fake_gcx, monkeypat assert row["count_increase"] > 0 text = run("grafana-metrics", "histogram", "h", *WIN).stdout assert "RESET" in text and "count=-" not in text and "sum=-" not in text - # the span-metrics counter behind `ops` is guarded the same way + # the span-metrics counter behind `ops` is guarded the same way, and + # falls back to its increase() instead of withholding the calls r = run("grafana-traces", "ops", "--service", "llmbench-api", *WIN, "--json") - ops = json.loads(r.stdout)["operations"] + o = json.loads(r.stdout) + ops = o["operations"] assert r.returncode == 0 and ops assert all( - e["span_calls"] is None and e.get("span_calls_reset") is True + e["span_calls"] is None + and e.get("span_calls_reset") is True + and e["span_calls_increase"] > 0 for e in ops.values() ) - assert ( - "RESET" - in run("grafana-traces", "ops", "--service", "llmbench-api", *WIN).stdout - ) + assert any("increase(traces_spanmetrics_calls_total" in c for c in o["commands"]) + text = run("grafana-traces", "ops", "--service", "llmbench-api", *WIN).stdout + assert "RESET inside the window (calls = increase())" in text + assert "withheld" not in text def test_histogram_rows_sort_busiest_first_even_when_a_row_reset(): From 974aec638b7a89acd5cbe7c6641631975cd8aea5 Mon Sep 17 00:00:00 2001 From: using-system Date: Sat, 26 Sep 2026 20:21:24 +0200 Subject: [PATCH 5/7] fix(agents): give observe-run one zsh-safe form for reading several ranges Every observe-run agent of the field campaign tripped on `echo ====` as a separator between reads. The warning is replaced by the form to use: `sed -n 'A,Bp;C,Dp' `, no separator line. Closes #659 Co-Authored-By: Claude Opus 5.5 (1M context) --- .apm/agents/observe-run.agent.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.apm/agents/observe-run.agent.md b/.apm/agents/observe-run.agent.md index 6f88bbf..c8292dc 100644 --- a/.apm/agents/observe-run.agent.md +++ b/.apm/agents/observe-run.agent.md @@ -570,9 +570,9 @@ subscript: `CD=customDimensions; echo math expression: operand expected`, exit 1, so the CLI never runs, and `"tostring($CD[1])"` prints `tostring(c)`, one character of the scalar, where bash prints both as written — write `${CD}[...]`, or the literal -name; no word starting with `=` — zsh looks up a command named `===` for -`echo ====` and fails with `=== not found` where bash prints it — write -`echo "----- $f"`. +name; no separator line between reads — zsh runs `echo ====` as a +command lookup, `=== not found`, and aborts the chain — read several +ranges of one file as `sed -n 'A,Bp;C,Dp' `. Then query per signal from what came back — keyed by **operation**, the smallest unit the service serves distinctly: on an HTTP server the From d3313013d71ce8863944674ba0d640f794f5a66c Mon Sep 17 00:00:00 2001 From: using-system Date: Sat, 26 Sep 2026 20:43:39 +0200 Subject: [PATCH 6/7] fix(skill): record the baseline a mission names instead of the recall's `new` took the `baseline` field from the recall's first line even when the mission named another report, which the agent then diffed against, so `show` and `persist` named the wrong baseline. `new --baseline ` records the named report as given (refused on a replay, whose baseline is `--verifies`, and when nothing stored carries that name); the recall fills the field only without it. Closes #659 Co-Authored-By: Claude Opus 5.5 (1M context) --- .../references/observe-run-report.md | 7 ++--- .apm/skills/odd-memory/scripts/odd_report.py | 13 ++++++++- docs/guide/reports.md | 2 +- tests/skills/odd-memory/test_odd_report.py | 27 +++++++++++++++++++ 4 files changed, 44 insertions(+), 5 deletions(-) diff --git a/.apm/skills/odd-memory/references/observe-run-report.md b/.apm/skills/odd-memory/references/observe-run-report.md index eb3f6bc..b82f507 100644 --- a/.apm/skills/odd-memory/references/observe-run-report.md +++ b/.apm/skills/odd-memory/references/observe-run-report.md @@ -15,7 +15,7 @@ python3 /scripts/odd_report.py new [--repo [--service ...] --stack --env \ --mode \ --window / | --from --to --run-name \ - [--verifies ] [--workload ] [--instance = ...] \ + [--verifies ] [--baseline ] [--workload ] [--instance = ...] \ [--process-restarted ...] [--repository ] \ [--at ] [--no-revision] [--custom-stack] python3 /scripts/odd_report.py check @@ -41,8 +41,9 @@ instrumentation` is the other kind's, stated in its own reference. `-observe-` suffix in observe mode, the `verify-` and `remeasure-` prefixes, the next free ordinal when the path is taken), fills `date`, `revision`, `tree_anchor` and `repository` from the - repository itself, `baseline` outside a replay from the recall's - first line (`none` without one), writes the frontmatter and the seven-section + repository itself, `baseline` outside a replay — the mission's named + baseline when it names one (`--baseline`), else the recall's first + line, or `none`, writes the frontmatter and the seven-section skeleton — eight with `--custom-stack`, the flag a mission passes when the handoff names a custom stack: the frontmatter then carries `stack_friction: 0` and the skeleton the `## 8. Stack friction` diff --git a/.apm/skills/odd-memory/scripts/odd_report.py b/.apm/skills/odd-memory/scripts/odd_report.py index 880675a..2f49e3a 100755 --- a/.apm/skills/odd-memory/scripts/odd_report.py +++ b/.apm/skills/odd-memory/scripts/odd_report.py @@ -1661,6 +1661,7 @@ def new_instrumentation_report(args: argparse.Namespace) -> tuple[Path, str, lis ("--window", args.window), ("--from/--to", args.start or args.end), ("--verifies", args.verifies), + ("--baseline", args.baseline), ("--workload", args.workload), ("--instance", args.instance), ("--process-restarted", args.process_restarted), @@ -1778,6 +1779,8 @@ def new_report(args: argparse.Namespace) -> tuple[Path, str, list[str]]: raise Refusal(f"--verifies is required on a {args.mode} run") if args.verifies and not replay: raise Refusal("--verifies applies to a verify or re-measure run only") + if args.baseline and replay: + raise Refusal("a replay's baseline is its --verifies; --baseline is not taken") if args.no_revision: root = Path(args.repo).resolve() @@ -1851,7 +1854,12 @@ def new_report(args: argparse.Namespace) -> tuple[Path, str, list[str]]: if replay: fields["verifies"] = args.verifies else: - fields["baseline"] = recalled_baseline(root, services, args.stack, args.env) + if args.baseline: + if not (root / OBSERVATION_DIR / args.baseline).is_file(): + raise Refusal(f"--baseline names no stored report: {args.baseline}") + fields["baseline"] = args.baseline + else: + fields["baseline"] = recalled_baseline(root, services, args.stack, args.env) if repo_root is not None: fields["revision"] = git(repo_root, "rev-parse", "--short", "HEAD") fields["tree_anchor"] = ls_tree(repo_root, "HEAD") @@ -3339,6 +3347,9 @@ def main(argv: list[str] | None = None) -> int: p.add_argument( "--verifies", help="the replayed report: a filename, or a repo-relative path" ) + p.add_argument( + "--baseline", help="the report the mission named as the baseline (a filename)" + ) p.add_argument("--workload") p.add_argument( "--instance", action="append", default=[], help="SERVICE=IDENTITY, repeatable" diff --git a/docs/guide/reports.md b/docs/guide/reports.md index fe370fe..743f058 100644 --- a/docs/guide/reports.md +++ b/docs/guide/reports.md @@ -65,7 +65,7 @@ process_restarted: true | `window` | yes | The observed interval, UTC — the run's own span, not the time a mission spent waiting for it | `start/end` | | `run_name` | yes | The filename's slug | kebab-case | | `date` | yes | The run's UTC date | `YYYY-MM-DD` | -| `baseline` | drive, observe, post-hoc | The report the run diffs against: the recall's first match | its exact filename, or `none` | +| `baseline` | drive, observe, post-hoc | The report the run diffs against: the one the mission named, else the recall's first match | its exact filename, or `none` | | `verifies` | verify, re-measure | The report whose protocol was replayed | its exact filename; the repo-relative path for an instrumentation report | | `revision` | optional | The observed repo's commit at run time | short SHA | | `tree_anchor` | optional | The top-level tree hashes at `revision`, so "code unchanged since" survives squash merges and fresh clones | entry name to hash | diff --git a/tests/skills/odd-memory/test_odd_report.py b/tests/skills/odd-memory/test_odd_report.py index 8b9fba7..f94ecf1 100644 --- a/tests/skills/odd-memory/test_odd_report.py +++ b/tests/skills/odd-memory/test_odd_report.py @@ -2062,6 +2062,33 @@ def test_new_records_no_baseline_when_the_recall_finds_none(repo, report): assert frontmatter(report, new(repo))["baseline"] == "none" +def test_a_baseline_the_mission_named_is_recorded_as_given(repo, report): + named = stored(repo, "2026-08-07-1000-checkout-sweep.md") + stored(repo, "2026-08-08-1000-checkout-sweep.md") + path = new(repo, "--baseline", named) + assert frontmatter(report, path)["baseline"] == named + proc = run( + repo, *NEW, "--repo", str(repo.root), "--run-name", "x", "--baseline", "nope.md" + ) + assert proc.returncode == 2 and "--baseline names no stored report" in proc.stderr + name = baseline(repo) + proc = run( + repo, + *NEW[:-4], + "--mode", + "verify", + "--window", + WINDOW, + "--verifies", + name, + "--baseline", + named, + "--repo", + str(repo.root), + ) + assert proc.returncode == 2 and "--verifies" in proc.stderr + + def test_a_replay_names_its_baseline_in_verifies_alone(repo, report): name = baseline(repo) path = new(repo, "--mode", "verify", "--verifies", name, run_name=None) From 7c30a43bcb919591f48198cccfd55738eb144d71 Mon Sep 17 00:00:00 2001 From: using-system Date: Sat, 26 Sep 2026 21:34:55 +0200 Subject: [PATCH 7/7] test(skill): pin the report baseline field to the recall's same-service-set rule Closes #659 Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/skills/odd-memory/test_odd_report.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tests/skills/odd-memory/test_odd_report.py b/tests/skills/odd-memory/test_odd_report.py index f94ecf1..f9809fc 100644 --- a/tests/skills/odd-memory/test_odd_report.py +++ b/tests/skills/odd-memory/test_odd_report.py @@ -2053,6 +2053,16 @@ def test_new_records_the_recalls_first_line_as_the_baseline(repo, report): assert frontmatter(report, new(repo))["baseline"] == newer +def test_new_takes_a_baseline_of_the_same_service_set_only(repo, report): + same = stored(repo, "2026-08-07-1000-checkout-sweep.md") + stored( + repo, + "2026-08-08-1000-checkout-sweep.md", + BASELINE.replace("services: [checkout]", "services: [checkout, payment]"), + ) + assert frontmatter(report, new(repo))["baseline"] == same + + def test_new_records_no_baseline_when_the_recall_finds_none(repo, report): stored( repo,