diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index de4138b..d2609af 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -2,7 +2,7 @@ "name": "linearb-ai", "metadata": { "description": "LinearB plugins for AI coding agents: engineering context from LinearB, right inside your agent.", - "version": "1.1.0" + "version": "1.1.1" }, "owner": { "name": "LinearB", @@ -12,7 +12,7 @@ { "name": "agentic-advisor", "description": "Before writing code, grades how fragile the target is from LinearB health signals + local git history and holds the agent to a matching effort level (LOW/MEDIUM/HIGH) — lean on calm repos, defensive on fragile ones.", - "version": "1.1.0", + "version": "1.1.1", "author": { "name": "LinearB", "email": "support@linearb.io" diff --git a/plugins/agentic-advisor/.claude-plugin/plugin.json b/plugins/agentic-advisor/.claude-plugin/plugin.json index 9eabe93..386afcc 100644 --- a/plugins/agentic-advisor/.claude-plugin/plugin.json +++ b/plugins/agentic-advisor/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "agentic-advisor", - "version": "1.1.0", + "version": "1.1.1", "description": "Before writing code, grades how fragile the target is from LinearB health signals + local git history (rework, incidents, fix/revert density, code ownership) and holds the agent to a matching effort level (LOW/MEDIUM/HIGH) — lean on calm repos, defensive on fragile ones. Auto-triggers via a bundled hook and can report each effort decision to your LinearB org as usage telemetry.", "author": { "name": "LinearB", diff --git a/plugins/agentic-advisor/README.md b/plugins/agentic-advisor/README.md index e359cd5..f535df0 100644 --- a/plugins/agentic-advisor/README.md +++ b/plugins/agentic-advisor/README.md @@ -40,7 +40,7 @@ Bundled hooks can report each effort decision to LinearB's reported-metrics API | event | when | `value` | token tag | | --- | --- | --- | --- | -| **decision** (`phase=decision`) | on `Stop`, right after the verdict — once per repo + effort level per session | `1` / `2` / `3` = LOW / MEDIUM / HIGH | `grading_tokens` — output tokens spent producing the verdict | +| **decision** (`phase=decision`) | on `Stop`, right after the verdict (or at `SessionEnd`, tagged `backfill: session_end`, if that turn was interrupted) — once per repo + effort level per session | `1` / `2` / `3` = LOW / MEDIUM / HIGH | `grading_tokens` — output tokens spent producing the verdict | | **tokens** (`phase=tokens`) | on `SessionEnd`, best-effort | `0` (not a grade — exclude from grade aggregates) | `coding_tokens` — output tokens spent on the work after the verdict | Count adoption and grades from **`phase=decision`** rows; use **`phase=tokens`** rows only for the coding-token measurement. (If you opt into the `LINEARB_BASELINE_HOLDOUT_PCT` experiment, held-out sessions also send a `value=0`, `label=baseline` event with `coding_tokens`; it's off by default.) diff --git a/plugins/agentic-advisor/hooks/agentic-advisor-report.sh b/plugins/agentic-advisor/hooks/agentic-advisor-report.sh index b68b8fe..0db5cc1 100755 --- a/plugins/agentic-advisor/hooks/agentic-advisor-report.sh +++ b/plugins/agentic-advisor/hooks/agentic-advisor-report.sh @@ -8,8 +8,9 @@ # session that never ends cleanly (closed terminal, Ctrl-C, crash). coding_tokens # is intentionally omitted here: it isn't complete until the session is over. # -# SessionEnd (once, best-effort) -> coding_tokens. -# graded -> a TOKENS event (value 0, effort_level kept, phase=tokens) carrying +# SessionEnd (once, best-effort) -> backfill + coding_tokens. +# graded -> first, any DECISION Stop never sent (an interrupted turn never fires +# Stop), tagged backfill=session_end; then a TOKENS event (value 0, effort_level kept, phase=tokens) carrying # the whole-session coding_tokens for the efficiency experiment. # baseline -> a beacon (value 0, label=baseline) carrying coding_tokens; the skill # was withheld so there is no verdict to report earlier. @@ -110,7 +111,6 @@ if [ "$label" = "graded" ]; then vcount="$(grep -cE 'LinearB: .+ (LOW|MEDIUM|HIGH) effort' "$transcript" 2>/dev/null)"; vcount="${vcount:-0}" vseen="$marker_dir/${session_id:-default}.vseen" [ -f "$vseen" ] && [ "$(cat "$vseen" 2>/dev/null)" = "$vcount" ] && exit 0 - printf '%s' "$vcount" > "$vseen" 2>/dev/null || true fi fi @@ -293,8 +293,10 @@ fi # ---- graded path ---- # (GATE 1 — skill actually ran — was already checked in the cheap fast-exit block above.) -# Don't lock in a repo-less event (needs repo_url to join to the PR). +# Don't lock in a repo-less event (needs repo_url to join to the PR). Mark this verdict +# count seen only after that, so a Stop before the repo resolves retries on the next Stop. [ -n "$repo_url" ] || exit 0 +[ "$event" = "Stop" ] && [ -n "${vseen:-}" ] && { printf '%s' "$vcount" > "$vseen" 2>/dev/null || true; } # GATE 2 — a real assistant-authored verdict line exists, as text or as the hold's # `printf` Bash call (grep, not python, so it still works without python3). @@ -316,9 +318,11 @@ parse_verdict() { # $1 = line } effort_val() { case "$1" in LOW) echo 1 ;; MEDIUM) echo 2 ;; HIGH) echo 3 ;; *) echo 0 ;; esac; } -if [ "$event" = "Stop" ]; then - # DECISION event(s): report the grade the moment it exists — reliable, per turn, - # once per (repo|effort). No coding_tokens (not complete until the session ends). +# DECISION event(s), once per (repo|effort) per session. Sent on Stop the moment the grade +# exists; SessionEnd re-runs it to backfill any grade Stop missed (e.g. an interrupted turn +# never fires Stop). No coding_tokens here (not complete until the session ends). +report_decisions() { + local backfill=""; [ "$event" = "SessionEnd" ] && backfill="session_end" while IFS= read -r line; do [ -n "$line" ] || continue parse_verdict "$line" || continue @@ -331,7 +335,8 @@ if [ "$event" = "Stop" ]; then --arg effort "$veffort" --arg repo "$vrepo" --arg ev "$vev" \ --arg email "$email" --arg repourl "$repo_url" --arg branch "$branch" \ --arg actual "$effort_actual" --arg ticket "$ticket" --arg sname "$session_name" \ - --arg gdur "$grading_duration" --arg model "$model" --arg gtok "$grading_tokens" --arg pver "$plugin_version" --arg idsrc "$identity_source" ' + --arg gdur "$grading_duration" --arg model "$model" --arg gtok "$grading_tokens" --arg pver "$plugin_version" --arg idsrc "$identity_source" \ + --arg backfill "$backfill" ' { metric_name: $m, source: $src, timestamp: $ts, value: $val, entity: ( @@ -348,15 +353,18 @@ if [ "$event" = "Stop" ]; then (if $gdur != "" then {grading_duration_s: $gdur} else {} end) + (if $model != "" then {model: $model} else {} end) + (if $gtok != "" then {grading_tokens: $gtok} else {} end) + - (if $pver != "" then {plugin_version: $pver} else {} end) + (if $pver != "" then {plugin_version: $pver} else {} end) + + (if $backfill != "" then {backfill: $backfill} else {} end) ) }' 2>/dev/null)" [ -n "$payload" ] || continue post "$payload" printf '%s\n' "$key" >> "$state" 2>/dev/null || true done <<< "$verdicts" - exit 0 -fi +} + +report_decisions +[ "$event" = "Stop" ] && exit 0 # event == SessionEnd, graded: TOKENS event — the whole-session coding_tokens, attributed # to the final (last) verdict. value 0 so it doesn't double-count on grade charts; the