diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 1d0eed7..de4138b 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -2,7 +2,7 @@ "name": "linearb-ai", "metadata": { "description": "LinearB plugins for AI coding agents: engineering context from LinearB, right inside your agent.", - "version": "1.0.0" + "version": "1.1.0" }, "owner": { "name": "LinearB", @@ -12,7 +12,7 @@ { "name": "agentic-advisor", "description": "Before writing code, grades how fragile the target is from LinearB health signals + local git history and holds the agent to a matching effort level (LOW/MEDIUM/HIGH) — lean on calm repos, defensive on fragile ones.", - "version": "1.0.0", + "version": "1.1.0", "author": { "name": "LinearB", "email": "support@linearb.io" diff --git a/plugins/agentic-advisor/.claude-plugin/plugin.json b/plugins/agentic-advisor/.claude-plugin/plugin.json index 4299467..9eabe93 100644 --- a/plugins/agentic-advisor/.claude-plugin/plugin.json +++ b/plugins/agentic-advisor/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "agentic-advisor", - "version": "1.0.0", + "version": "1.1.0", "description": "Before writing code, grades how fragile the target is from LinearB health signals + local git history (rework, incidents, fix/revert density, code ownership) and holds the agent to a matching effort level (LOW/MEDIUM/HIGH) — lean on calm repos, defensive on fragile ones. Auto-triggers via a bundled hook and can report each effort decision to your LinearB org as usage telemetry.", "author": { "name": "LinearB", diff --git a/plugins/agentic-advisor/README.md b/plugins/agentic-advisor/README.md index 882b654..e359cd5 100644 --- a/plugins/agentic-advisor/README.md +++ b/plugins/agentic-advisor/README.md @@ -24,12 +24,13 @@ The agent prints a one-line verdict before writing code, e.g.: ## How it decides - **Effort bands** match LinearB's own rework benchmark: ELITE <3% / STRONG 3–6% → **LOW**; FAIR 6–7% → **MEDIUM**; NEEDS FOCUS >7% → **HIGH**. Effort = the highest-firing signal (rework, unreviewed merges, or a recent serious incident). +- **Task complexity:** the request itself is graded too (scope, breadth, tricky logic, blast radius), so a demanding change in a calm repo still gets real care. The final effort is the highest of repo health, task complexity and (when it runs) phase 2. - **Phase 2 (file-grained):** on non-trivial changes to a fragile/sensitive area, it also checks local git history on the exact files — how much existing code was rewritten over the last 90 days (the main signal) and who owns it — to target where to concentrate care. Commit messages saying "fix"/"revert" count only as a weak hint. - **Stays lean:** trivial edits (comments, docs, formatting, pure renames) skip everything and cap at LOW; the repo-health part of the verdict is cached per repo for 24 hours; the task and file checks are never cached and are redone whenever the skill runs. ## Triggers -Auto-fires (once per repo, per session) on a Jira ticket in the prompt, code-task keywords (fix / implement / refactor…), or the first source-file edit in a git repo. Skips questions, doc/text edits, and files outside a git repo. +Auto-fires (once per repo, per session) on a Jira ticket in the prompt, code-task keywords (fix / implement / refactor…), or the first source-file edit in a git repo. Skips questions, doc/text edits, and files outside a git repo. If the grade was decided but never printed (common at lower effort levels, where the model keeps it in its thinking), the next file read or edit is held once until the agent records the verdict line. ## Usage telemetry (on by default) diff --git a/plugins/agentic-advisor/hooks/agentic-advisor-report.sh b/plugins/agentic-advisor/hooks/agentic-advisor-report.sh index 3e415e5..b68b8fe 100755 --- a/plugins/agentic-advisor/hooks/agentic-advisor-report.sh +++ b/plugins/agentic-advisor/hooks/agentic-advisor-report.sh @@ -225,8 +225,10 @@ for line in open(sys.argv[1]): first_edit = te # verdict must be ASSISTANT-authored text AFTER the Skill call — not the # SKILL.md examples that load on invocation (they also match the pattern). - elif typ == "text" and role == "assistant" and skill is not None and verdict is None \ - and re.search(r"LinearB: .+ (LOW|MEDIUM|HIGH) effort", x.get("text") or ""): + elif role == "assistant" and skill is not None and verdict is None \ + and re.search(r"LinearB: .+ (LOW|MEDIUM|HIGH) effort", (x.get("text") or "") if typ == "text" else + str((x.get("input") or {}).get("command", "")) if typ == "tool_use" and x.get("name") == "Bash" + and re.search(r"(^|\n|;|&&)\s*printf\s[^\n]*LinearB: ", str((x.get("input") or {}).get("command", ""))) else ""): verdict = te out = {"grading_tokens": "", "coding_tokens": "", "grading_duration_s": "", "has_edit": first_edit is not None} @@ -294,9 +296,9 @@ fi # Don't lock in a repo-less event (needs repo_url to join to the PR). [ -n "$repo_url" ] || exit 0 -# GATE 2 — a real assistant-authored verdict line exists (grep, not python, so the -# decision can still be reported when python3 is unavailable). -verdicts="$(jq -r 'select(.message.role? == "assistant") | .message.content? // [] | .[]? | select(.type? == "text") | .text? // empty' "$transcript" 2>/dev/null \ +# GATE 2 — a real assistant-authored verdict line exists, as text or as the hold's +# `printf` Bash call (grep, not python, so it still works without python3). +verdicts="$(jq -r 'select(.message.role? == "assistant") | .message.content? // [] | .[]? | if .type? == "text" then .text? // empty elif .type? == "tool_use" and .name? == "Bash" and ((.input.command? // "") | test("(^|\\n|;|&&)\\s*printf\\s[^\\n]*LinearB: ")) then .input.command else empty end' "$transcript" 2>/dev/null \ | grep -E 'LinearB: .+ (LOW|MEDIUM|HIGH) effort' || true)" [ -n "$verdicts" ] || exit 0 diff --git a/plugins/agentic-advisor/hooks/agentic-advisor-trigger.sh b/plugins/agentic-advisor/hooks/agentic-advisor-trigger.sh index d5de0d1..3721825 100755 --- a/plugins/agentic-advisor/hooks/agentic-advisor-trigger.sh +++ b/plugins/agentic-advisor/hooks/agentic-advisor-trigger.sh @@ -11,7 +11,9 @@ # High precision: a ticket almost always means "starting work". # 2. FALLBACK - code-task keywords, for ad-hoc fixes with no ticket. # -# PreToolUse (matcher Edit|Write) +# PreToolUse (matcher Edit|Write|Read|Grep|Glob|Task|Agent) +# VERDICT - the skill ran but no verdict line was printed yet (at low effort the +# model keeps it in thinking): hold the next tool once so it prints it. # BACKSTOP - a code edit is starting and this repo wasn't nudged yet this # session. Catches tasks that slipped past the prompt gate. # @@ -41,6 +43,27 @@ skill_ran() { grep -qE '"skill"[[:space:]]*:[[:space:]]*"[^"]*agentic-advisor[^"]*"' "$transcript" 2>/dev/null } +# Is the LATEST advisor Skill call still missing its verdict (assistant text, or the +# hold's `printf` Bash call)? Prints "pending" or "ok"; prints nothing on parse failure. +verdict_state() { + jq -rs --arg n "agentic-advisor" --arg re 'LinearB: .+ (LOW|MEDIUM|HIGH) effort' ' + [.[] | select(.message.role? == "assistant") | .message.content? // [] | .[]?] as $c + | [$c | to_entries[] | select(.value.type? == "tool_use" and .value.name? == "Skill" + and ((.value.input.skill? // "") | tostring | contains($n))) | .key] as $sk + | if ($sk | length) == 0 then "ok" + elif ($c[($sk[-1] + 1):] | any( + (.type? == "text" and ((.text? // "") | test($re))) + or (.type? == "tool_use" and .name? == "Bash" + and ((.input.command? // "") | test("(^|\\n|;|&&)\\s*printf\\s[^\\n]*" + $re))))) + then "ok" else "pending" end' "$transcript" 2>/dev/null +} + +deny() { # $1 = reason fed back to the model + printf '{"hookSpecificOutput":{"hookEventName":"PreToolUse","permissionDecision":"deny","permissionDecisionReason":%s}}\n' \ + "$(printf '%s' "$1" | jq -Rs .)" + exit 0 +} + # Per-user state dir (not shared /tmp): another local user can't plant or read markers. marker_dir="${XDG_CACHE_HOME:-$HOME/.cache}/agentic-advisor" (umask 077; mkdir -p "$marker_dir") 2>/dev/null || true @@ -58,7 +81,7 @@ set_marker() { # $1 = a directory inside the repo } # Imperative, silent guidance — no permission question, no announcement. -guidance="MANDATORY FIRST STEP for this code task: before you read files, search, explore, spawn any subagent, or edit anything, you MUST invoke the \`agentic-advisor\` skill (via the Skill tool) to grade this repo's effort level. Do not start the task, do not touch code, until you have run it and printed its verdict line. This is required, not optional — do not skip it even if the task seems obvious. Invoke it directly (no need to ask the user or announce it). It degrades gracefully: if LinearB is unavailable, default to MEDIUM effort and continue." +guidance="MANDATORY FIRST STEP for this code task: before you read files, search, explore, spawn any subagent, or edit anything, you MUST invoke the \`agentic-advisor\` skill (via the Skill tool) to grade this repo's effort level. Do not start the task, do not touch code, until you have run it and printed its verdict line verbatim in this shape: > LinearB: — effort () — (a hook parses that line; a prose summary is not recorded). This is required, not optional — do not skip it even if the task seems obvious. Invoke it directly (no need to ask the user or announce it). It degrades gracefully: if LinearB is unavailable, default to MEDIUM effort and continue." # Mark handled (this session+repo) and emit additionalContext for the event. emit() { # $1 = hookEventName, $2 = context text @@ -106,7 +129,22 @@ case "$event" in ;; PreToolUse) - # Reached only for Edit|Write (matcher in settings). ONE-TIME grade-before-edit + # Verdict gate: once per advisor run, hold the first non-shell tool until that run's + # verdict line exists. Keyed by the run count so a second repo's run is gated too, and + # the transcript is parsed only until that run is settled (marker short-circuits). + if skill_ran; then + # Count only assistant Skill calls (same thing verdict_state indexes), not echoes elsewhere. + runs="$(grep '"role":"assistant"' "$transcript" 2>/dev/null | grep -cE '"skill"[[:space:]]*:[[:space:]]*"[^"]*'"agentic-advisor"'[^"]*"')" + vdone="$marker_dir/${session_id:-default}.${runs:-0}.vdone" + if [ ! -f "$vdone" ]; then + state="$(verdict_state)" + [ -n "$state" ] && : > "$vdone" 2>/dev/null && [ "$state" = "pending" ] && deny "LinearB: one quick step first — record this session's effort grade by running: printf '%s\\n' '> LinearB: — effort () — ' (the grade is only recorded from that line, not from thinking). Then retry this same call; it won't pause again." + fi + fi + tool="$(jq -r '.tool_name // empty' <<<"$input" 2>/dev/null || true)" + case "$tool" in Read|Grep|Glob|Task|Agent) exit 0 ;; esac + + # Edit|Write only from here. ONE-TIME grade-before-edit # HOLD: the FIRST source edit in a repo this session is denied (permissionDecision:deny) so the # skill grades before code is written. Held at most ONCE per repo/session (via # its own dedicated .held marker — NOT the UserPromptSubmit nudge (.done) marker, @@ -144,10 +182,8 @@ case "$event" in # prefix) and feeds that reason to the model. FAIL OPEN if we can't record the # marker (don't wedge). : > "$held" 2>/dev/null || exit 0 - reason="One quick first step before this edit: run the \`agentic-advisor\` skill to grade this repo's effort (just this once per repo, this session). Print its verdict, then make the same edit again — it'll go right through, and later edits won't pause. If LinearB isn't available it just defaults to MEDIUM." - printf '{"hookSpecificOutput":{"hookEventName":"PreToolUse","permissionDecision":"deny","permissionDecisionReason":%s}}\n' \ - "$(printf '%s' "$reason" | jq -Rs .)" - exit 0 + reason="One quick first step before this edit: run the \`agentic-advisor\` skill to grade this repo's effort (just this once per repo, this session). Print its verdict line (> LinearB: — effort (…) — …), then make the same edit again — it'll go right through, and later edits won't pause. If LinearB isn't available it just defaults to MEDIUM." + deny "$reason" ;; esac diff --git a/plugins/agentic-advisor/hooks/hooks.json b/plugins/agentic-advisor/hooks/hooks.json index 75f76a7..4c6a213 100644 --- a/plugins/agentic-advisor/hooks/hooks.json +++ b/plugins/agentic-advisor/hooks/hooks.json @@ -12,7 +12,7 @@ ], "PreToolUse": [ { - "matcher": "Edit|Write", + "matcher": "Edit|Write|Read|Grep|Glob|Task|Agent", "hooks": [ { "type": "command", diff --git a/plugins/agentic-advisor/skills/agentic-advisor/SKILL.md b/plugins/agentic-advisor/skills/agentic-advisor/SKILL.md index 2b3b392..7039f8d 100644 --- a/plugins/agentic-advisor/skills/agentic-advisor/SKILL.md +++ b/plugins/agentic-advisor/skills/agentic-advisor/SKILL.md @@ -9,6 +9,19 @@ description: Pull LinearB health signals for a repository before and during code Run this at the start of a code task to load the repo's current health signals, then let them set how much effort to spend writing the code. This is writing-phase guidance only — it does not gate merges, pushes, PRs, or deploys. +## Output contract (read this first) + +Right after the repo sweep, your first assistant text **must be this verdict line, verbatim in shape**, as a markdown blockquote: + +> LinearB: — effort () — + +- **A hook parses this exact line** to record the decision. Prose such as "repo health looks solid… I'll treat it as HIGH effort" is not recognized: the grade is lost and nothing is reported. +- The level is one word in capitals (`LOW`, `MEDIUM` or `HIGH`) followed immediately by the word `effort`. +- Explanation may follow on later lines, but the verdict line comes first and is never reworded. +- If you aren't writing visible text at this point, record it with one Bash call instead: `printf '%s\n' '> LinearB: — effort () — '`. A hook holds your next file read or edit until one of the two exists. + +Example: `> LinearB: api-service — HIGH effort (rework 3.4%, calm repo — but auth refactor with wide blast radius) — small behavior-preserving steps, tests green, confirm scope first.` + ## Access (read first) All signals come from LinearB's **public API** over `curl` — no MCP connector needed. The base URL defaults to `https://public-api.linearb.io`; on-prem or regional deployments override it with `LINEARB_API_URL`. Authentication is a single org-scoped token in the `LINEARB_API_TOKEN` environment variable (created in the LinearB UI: **Settings → API Tokens → Create API Token**). Set up shared values once (`Content-Type` is required even on GET). The token is fed to `curl` on stdin (`-H @-` plus `<<<"x-api-key: …"` on each call), never as an argument, so it can't show up in process listings: @@ -248,7 +261,7 @@ HIGH effort — repo looks fragile, be careful: ## Human-Facing Note -The moment the **repo sweep** returns — and **before any other tool call, and before spawning any Explore/subagent, reading, searching, `ls`/`find`, or editing code** — your next output MUST be this exact verdict line, **formatted as a markdown blockquote** (start it with `> ` so the terminal renders it with the left bar / emphasis and it stands out). Emit it as assistant text — do NOT wrap it in an `echo`/shell command or code fence (that hides it in a collapsed tool block). Do not paraphrase it into prose like "LOW effort signals confirmed". This is the **repo-level verdict**; it is fully knowable from the sweep alone, so nothing else may happen first. Format: +The moment the **repo sweep** returns — and **before any other tool call, and before spawning any Explore/subagent, reading, searching, `ls`/`find`, or editing code** — your next output MUST be this exact verdict line, **formatted as a markdown blockquote** (start it with `> ` so the terminal renders it with the left bar / emphasis and it stands out). Emit it as assistant text — do NOT wrap it in an `echo`/shell command or code fence (that hides it in a collapsed tool block). The one exception is the exact `printf '%s\n' '> LinearB: …'` Bash call from the Output contract, when you aren't writing visible text or the hook asks for it. Do not paraphrase it into prose like "LOW effort signals confirmed". This is the **repo-level verdict**; it is fully knowable from the sweep alone, so nothing else may happen first. Format: ```text > LinearB: — effort () —