diff --git a/cmd/spinloop/work.go b/cmd/spinloop/work.go index bec3f064..7739e4c1 100644 --- a/cmd/spinloop/work.go +++ b/cmd/spinloop/work.go @@ -40,7 +40,7 @@ func workCmd() *cobra.Command { Long: `works the orchestrator's work list — the backlog it works — from the shell, as a client of the work list API the orchestrator serves: add an item, read the work, read an item's kept output, stop a running item, -remove an item. +re-queue a failed item, remove an item. Each subcommand takes --url, the API's base address, and presents the API's token — from --api-token, else --api-token-file, else the SPINLOOP_API_TOKEN @@ -53,6 +53,7 @@ fails before it calls the API, naming the flag.`, workAddCmd(), workListCmd(), workAbortCmd(), + workRetryCmd(), workRemoveCmd(), workLogsCmd(), workBoardCmd(), @@ -384,6 +385,44 @@ the command reports its answer: a refusal reads the way the API states it.`, return c } +// workRetryCmd builds `work retry`. +func workRetryCmd() *cobra.Command { + var base, apiToken, apiTokenFile string + c := &cobra.Command{ + Use: "retry ", + Short: "put a failed item back in the backlog", + Long: `puts a failed item back in the backlog, through the work list API +the orchestrator serves: the item's record removed, and the run's next pass +admits it again. The item's fields are unchanged, and its kept output stays +until the new attempt writes over it. + +Only a failed item can be retried: an item the run records backlog, running +or done is refused, naming the item and its state, and an id the file does +not carry is refused, naming it. The API answers once the item is back in +the backlog, and the command reports its answer: a refusal reads the way the +API states it.`, + Args: cobra.ExactArgs(1), + SilenceErrors: true, + SilenceUsage: true, + RunE: func(_ *cobra.Command, args []string) error { + id := args[0] + b, token, err := workTarget("work retry", base, apiToken, apiTokenFile) + if err != nil { + return err + } + if _, err := workRequest(b, token, http.MethodPost, "/v1/items/"+url.PathEscape(id)+"/retry", nil); err != nil { + return err + } + fmt.Printf("item %q is back in the backlog\n", id) + return nil + }, + } + fs := c.Flags() + workAPIFlags(fs, &base, &apiToken, &apiTokenFile) + c.ValidArgsFunction = itemIDSlot + return c +} + // workRemoveCmd builds `work remove`. func workRemoveCmd() *cobra.Command { var base, apiToken, apiTokenFile string diff --git a/cmd/spinloop/work_board_model.go b/cmd/spinloop/work_board_model.go index 00e7f6ef..30c31f06 100644 --- a/cmd/spinloop/work_board_model.go +++ b/cmd/spinloop/work_board_model.go @@ -56,6 +56,7 @@ type workBoardVerb string const ( workAbort workBoardVerb = "abort" + workRetry workBoardVerb = "retry" workRemove workBoardVerb = "remove" workAdd workBoardVerb = "add" ) @@ -209,7 +210,7 @@ type workBoardModel struct { formAsk bool // the discard question stands in front of the form form workBoardForm - confirm bool // a removal stands in front of the board, waiting on its yes + confirm workBoardVerb // the action (remove or retry) standing in front of the board, waiting on its yes; empty when none width, height int } @@ -381,20 +382,25 @@ func (m *workBoardModel) updateKey(msg tea.KeyMsg) (tea.Model, tea.Cmd) { } return m, m.updateFormKey(msg) } - if m.confirm { + if m.confirm != "" { switch msg.String() { case "y": v := m.selectedItem() - m.confirm = false + verb := m.confirm + m.confirm = "" if v == nil { return m, nil } - return m, m.beginAction(workRemove, v.ID) + return m, m.beginAction(verb, v.ID) case "n", "esc": - m.confirm = false - m.statusLine = "declined — nothing removed" + declined := "nothing removed" + if m.confirm == workRetry { + declined = "nothing retried" + } + m.confirm = "" + m.statusLine = "declined — " + declined case "q", "ctrl+c": - m.confirm = false + m.confirm = "" return m, tea.Quit } return m, nil @@ -431,9 +437,14 @@ func (m *workBoardModel) updateBoardKey(msg tea.KeyMsg) tea.Cmd { if v := m.selectedItem(); v != nil { return m.beginAction(workAbort, v.ID) } + case "t": + // As with abort, the API refuses a retry of what has not failed. + if m.selectedItem() != nil { + m.confirm = workRetry + } case "x": if m.selectedItem() != nil { - m.confirm = true + m.confirm = workRemove } case "n": m.formOpen = true @@ -573,6 +584,11 @@ func (m *workBoardModel) beginAction(verb workBoardVerb, id string) tea.Cmd { _, err := workRequest(base, token, "POST", "/v1/items/"+url.PathEscape(id)+"/abort", nil) return workBoardActionMsg{verb: verb, id: id, err: err} } + case workRetry: + run = func() tea.Msg { + _, err := workRequest(base, token, "POST", "/v1/items/"+url.PathEscape(id)+"/retry", nil) + return workBoardActionMsg{verb: verb, id: id, err: err} + } case workRemove: run = func() tea.Msg { _, err := workRequest(base, token, "DELETE", "/v1/items/"+url.PathEscape(id), nil) @@ -610,6 +626,8 @@ func workBoardActionLine(msg workBoardActionMsg) string { switch msg.verb { case workAbort: return fmt.Sprintf("item %q stopped: it is back in the backlog", msg.id) + case workRetry: + return fmt.Sprintf("item %q is back in the backlog", msg.id) case workRemove: return fmt.Sprintf("item %q removed", msg.id) } diff --git a/cmd/spinloop/work_board_render.go b/cmd/spinloop/work_board_render.go index 4d8504fc..1d54c95f 100644 --- a/cmd/spinloop/work_board_render.go +++ b/cmd/spinloop/work_board_render.go @@ -379,7 +379,7 @@ func padTo(line string, w int) string { } // footerLine is the board's bottom line: the keys that would do something -// where the cursor stands, replaced by the removal question while one is +// where the cursor stands, replaced by the removal or retry question while one is // pending and by an in-flight action's progress while a call is out; the // status line rides at the end. func (m workBoardModel) footerLine(w int, keys string) string { @@ -387,13 +387,13 @@ func (m workBoardModel) footerLine(w int, keys string) string { if m.action.verb != "" { line = m.action.progress(workBoardNow()) } - if m.confirm { + if m.confirm != "" { v := m.selectedItem() id := "" if v != nil { id = fmt.Sprintf(" %q", v.ID) } - line = "remove item" + id + "?" + dashHintGap + + line = string(m.confirm) + " item" + id + "?" + dashHintGap + dashKeyHints("y yes"+dashHintGap+"n no") } if m.statusLine != "" { @@ -406,8 +406,8 @@ func (m workBoardModel) footerLine(w int, keys string) string { } // boardKeys names the keys that would do something for what is selected: -// abort is named only on a running card, remove only on one that is not, -// detail only where there is an item to open. +// abort is named only on a running card, retry only on a failed one, remove +// only on one that is not running, detail only where there is an item to open. func (m workBoardModel) boardKeys() string { parts := []string{} cols := m.columnIndexes() @@ -420,6 +420,9 @@ func (m workBoardModel) boardKeys() string { if v.State == orchestrator.StateRunning { parts = append(parts, "a abort") } else { + if v.State == orchestrator.StateFailed { + parts = append(parts, "t retry") + } parts = append(parts, "x remove") } } diff --git a/cmd/spinloop/work_board_test.go b/cmd/spinloop/work_board_test.go index 8a3d2b14..d1be7309 100644 --- a/cmd/spinloop/work_board_test.go +++ b/cmd/spinloop/work_board_test.go @@ -63,6 +63,7 @@ func (a *wbAPI) ServeHTTP(w http.ResponseWriter, r *http.Request) { list := func(v string) string { tail := strings.TrimPrefix(r.URL.Path, "/v1/items/") tail = strings.TrimSuffix(tail, "/log") + tail = strings.TrimSuffix(tail, "/retry") return strings.TrimSuffix(tail, "/abort") } switch { @@ -93,6 +94,25 @@ func (a *wbAPI) ServeHTTP(w http.ResponseWriter, r *http.Request) { Tags: body.Tags, Priority: body.Priority, State: orchestrator.StateBacklog, }) out(http.StatusOK, map[string]any{"ok": true}) + case r.Method == http.MethodPost && strings.HasSuffix(r.URL.Path, "/retry"): + id := list(r.URL.Path) + for i := range a.items { + if a.items[i].ID == id { + if a.items[i].State != orchestrator.StateFailed { + out(http.StatusConflict, map[string]any{"error": map[string]string{ + "message": fmt.Sprintf("item %q is not failed: it is %s", id, a.items[i].State), + }}) + return + } + a.items[i].State = orchestrator.StateBacklog + a.items[i].Node, a.items[i].EndedAt, a.items[i].Why = "", "", "" + out(http.StatusOK, map[string]any{"ok": true}) + return + } + } + out(http.StatusNotFound, map[string]any{"error": map[string]string{ + "message": fmt.Sprintf("the work list does not carry item %q", id), + }}) case r.Method == http.MethodPost && strings.HasSuffix(r.URL.Path, "/abort"): id := list(r.URL.Path) for i := range a.items { @@ -1316,3 +1336,88 @@ func TestWorkBoard_ProgramSmoke(t *testing.T) { tm.Send(tea.KeyMsg{Type: tea.KeyRunes, Runes: []rune("q")}) tm.WaitFinished(t, teatest.WithFinalTimeout(3*time.Second)) } + +func TestWorkBoard_RetryMovesAFailedCardBackToBacklog(t *testing.T) { + a := newWBAPI(t, []orchestrator.ItemView{wbItem("crank", orchestrator.StateFailed)}, nil) + m := newWBTestModel(t, a) + wbRound(t, m) + wbAct(t, m, "right", "t", "y") + if a.callCount("POST /v1/items/crank/retry") != 1 { + t.Fatal("the retry did not reach the API") + } + if !strings.Contains(m.statusLine, `item "crank" is back in the backlog`) { + t.Errorf("status = %q, want the retried line", m.statusLine) + } + view := wbPlain(m.View()) + if !strings.Contains(view, "Failed 0") || !strings.Contains(view, "Backlog 1") { + t.Errorf("the card did not move back:\n%s", view) + } +} + +func TestWorkBoard_RetryOfAnItemThatHasNotFailedIsRefusedTheAPISWay(t *testing.T) { + a := newWBAPI(t, []orchestrator.ItemView{wbItem("solo", orchestrator.StateBacklog)}, nil) + m := newWBTestModel(t, a) + wbRound(t, m) + wbAct(t, m, "t", "y") + if !strings.Contains(m.statusLine, `item "solo" is not failed`) { + t.Errorf("status = %q, want the API's own refusal", m.statusLine) + } + if !strings.Contains(wbPlain(m.View()), "solo") { + t.Error("the refusal took the board down with it") + } +} + +func TestWorkBoard_RetryIsNamedOnlyOnAFailedCard(t *testing.T) { + cases := map[string]bool{ + orchestrator.StateBacklog: false, + orchestrator.StateRunning: false, + orchestrator.StateDone: false, + orchestrator.StateFailed: true, + } + for state, want := range cases { + a := newWBAPI(t, []orchestrator.ItemView{wbItem("solo", state)}, nil) + m := newWBTestModel(t, a) + wbRound(t, m) + keys := m.boardKeys() + if m.selectedItem() == nil { + wbKeys(t, m, "right") + keys = m.boardKeys() + } + for i := 0; i < 3 && m.selectedItem() == nil; i++ { + wbKeys(t, m, "right") + keys = m.boardKeys() + } + if got := strings.Contains(keys, "t retry"); got != want { + t.Errorf("%s card: keys %q, retry named = %v, want %v", state, keys, got, want) + } + } +} + +func TestWorkBoard_RetryAsksFirst(t *testing.T) { + a := newWBAPI(t, []orchestrator.ItemView{wbItem("crank", orchestrator.StateFailed)}, nil) + m := newWBTestModel(t, a) + wbRound(t, m) + wbKeys(t, m, "right", "t") + footer := wbPlain(m.footerLine(m.effWidth(), m.boardKeys())) + if !strings.Contains(footer, `retry item "crank"?`) { + t.Errorf("the question did not stand: %q", footer) + } + if a.callCount("/retry") != 0 { + t.Error("the retry was sent before the yes") + } + wbKeys(t, m, "n") + if a.callCount("/retry") != 0 { + t.Error("a declined retry was sent anyway") + } + if !strings.Contains(m.statusLine, "nothing retried") { + t.Errorf("status = %q, want the declined line", m.statusLine) + } + if !strings.Contains(wbPlain(m.View()), "Failed 1") { + t.Error("the declined card left the Failed column") + } + // Escape abandons the question the same way. + wbKeys(t, m, "t", "esc") + if a.callCount("/retry") != 0 { + t.Error("an abandoned retry was sent anyway") + } +} diff --git a/cmd/spinloop/work_test.go b/cmd/spinloop/work_test.go index 74af224b..e7fb9854 100644 --- a/cmd/spinloop/work_test.go +++ b/cmd/spinloop/work_test.go @@ -207,6 +207,49 @@ func TestWorkAbort_TheRefusalsReadTheWayTheAPIStatesThem(t *testing.T) { } } +func TestWorkRetry_CallsTheRetryPathAndReports(t *testing.T) { + base, got := workAPIStub(t, http.StatusOK, map[string]any{"ok": true, "id": "a"}) + out, err := runWork(t, "retry", "a", "--url", base) + if err != nil { + t.Fatalf("the retry: %v (out %s)", err, out) + } + if !strings.Contains(out, `item "a" is back in the backlog`) { + t.Errorf("the retry reports the item back in the backlog: %s", out) + } + if got.method != "POST" || got.path != "/v1/items/a/retry" { + t.Errorf("the retry calls the API's retry path for the id: %s %s", got.method, got.path) + } +} + +func TestWorkRetry_TheRefusalsReadTheWayTheAPIStatesThem(t *testing.T) { + base, _ := workAPIStub(t, http.StatusConflict, + workAPIErrorReply(`item "a" is not failed: it is running`)) + _, err := runWork(t, "retry", "a", "--url", base) + if err == nil || !strings.Contains(err.Error(), "it is running") { + t.Errorf("a not-failed item is refused, naming its state: %v", err) + } + + base, _ = workAPIStub(t, http.StatusNotFound, + workAPIErrorReply(`the items file carries no item with id "b"`)) + _, err = runWork(t, "retry", "b", "--url", base) + if err == nil || !strings.Contains(err.Error(), `no item with id "b"`) { + t.Errorf("an id the file does not carry is refused, naming it: %v", err) + } +} + +func TestWorkRetry_NeedsOneIDAndAnAddress(t *testing.T) { + if _, err := runWork(t, "retry", "--url", "http://127.0.0.1:1"); err == nil { + t.Error("retry with no id is refused") + } + if _, err := runWork(t, "retry", "a", "b", "--url", "http://127.0.0.1:1"); err == nil { + t.Error("retry with two ids is refused") + } + _, err := runWork(t, "retry", "a") + if err == nil || !strings.Contains(err.Error(), "--url") { + t.Errorf("retry with no address names the flag: %v", err) + } +} + func TestWorkList_PlainLinesInFileOrder(t *testing.T) { reply := map[string]any{"object": "list", "data": []any{ map[string]any{"id": "a", "instructions": "do a", "dir": ".", "state": "backlog"}, diff --git a/docs/commands/orchestrator.md b/docs/commands/orchestrator.md index 3325a822..a90a778c 100644 --- a/docs/commands/orchestrator.md +++ b/docs/commands/orchestrator.md @@ -143,6 +143,7 @@ shows in that state, not `backlog`. | `POST /v1/items` | Add an item to the file and the backlog: the file's validation on its fields, and the file stays a valid items file after the write. | | `DELETE /v1/items/{id}` | Take an item out of the work list: the items file, its record, and its kept output, all of it. | | `POST /v1/items/{id}/abort` | Stop a running item's agent the way a clean interrupt stops it — the polite signal, the grace, then the hard end — and put the item back in the backlog, where the run admits it again on a later pass. | +| `POST /v1/items/{id}/retry` | Put a failed item back in the backlog: its record is removed, so the run admits it again on a later pass. The items file is unchanged and the failed attempt's output stays until the new attempt writes over it. | | Any other path or method | A `404` naming the paths the API serves. | The mutations refuse rather than force: @@ -156,6 +157,9 @@ The mutations refuse rather than force: - An **abort** is refused a `409` where the item is not running — naming the item and its state — and a `404` where the file does not carry the id, naming it. +- A **retry** is refused a `409` where the item is not failed — naming the + item and its state — and a `404` where the file does not carry the id, + naming it. ### The API's token diff --git a/docs/commands/work.md b/docs/commands/work.md index 2c2f6174..ec578a09 100644 --- a/docs/commands/work.md +++ b/docs/commands/work.md @@ -3,7 +3,7 @@ Work the [orchestrator](orchestrator.md)'s work list — the backlog it works — from the shell, as a client of the [work list API](orchestrator.md#the-work-list-api) the orchestrator serves: add an item, read the work, read an item's kept -output, stop a running item, remove an item — or watch the whole run on a +output, stop a running item, re-queue a failed one, remove an item — or watch the whole run on a live board. ```sh @@ -11,6 +11,7 @@ spinloop work add --url http://127.0.0.1:4010 --id fix-parser --instructions "fi spinloop work list --url http://127.0.0.1:4010 spinloop work logs --url http://127.0.0.1:4010 fix-parser -f spinloop work abort --url http://127.0.0.1:4010 fix-parser +spinloop work retry --url http://127.0.0.1:4010 fix-parser spinloop work remove --url http://127.0.0.1:4010 docs-refresh spinloop work board --url http://127.0.0.1:4010 ``` @@ -117,6 +118,23 @@ naming it, and an item the run records `backlog`, `done` or `failed` is refused too, naming the item and its state. The API answers once the item is stopped, and the command reports its answer: a refusal reads the way the API states it. +## Retrying a failed item + +```sh +spinloop work retry --url http://127.0.0.1:4010 fix-parser +``` + +Puts a failed item back in the backlog through the API's retry path: its record +is removed, and the run's next pass admits it again. The item's fields in the +items file are unchanged, and its kept output from the failed attempt stays +until the new attempt writes over it. + +Only a failed item can be retried: an id the file does not carry is refused, +naming it, and an item the run records `backlog`, `running` or `done` is refused +too, naming the item and its state. The API answers once the item is back in +the backlog, and the command reports its answer: a refusal reads the way the +API states it. + ## Removing an item ```sh @@ -158,8 +176,9 @@ what the cursor stands on: timings and failure reason, with its kept output tailed beneath as `work logs -f` tails it, ending when the item ends or drops out. `esc` returns; the board cannot be quit from inside the detail. -- `a` aborts a running item, `x` removes one that is not — the removal - asks first, and declining sends nothing. A refusal from the API +- `a` aborts a running item, `t` retries a failed one, `x` removes one that + is not running — the retry and the removal each ask first, and declining + sends nothing. A refusal from the API reads on the status line the way the API states it. - `n` opens the add form — the same add `work add` sends, through the API's add path. Its five fields stand before you at once (id, @@ -185,8 +204,9 @@ work into a pipe. run owns. - It never runs an agent, and never starts or stops one: an abort asks the run to stop its agent, and the run's own grace bounds the stop. -- It does not re-run an ended item. A `done` or `failed` record stands against - the id's re-add, the way the orchestrator's does. +- It does not re-run a `done` item, and a `failed` one only when asked with + `work retry`. A `done` or `failed` record stands against the id's re-add, + the way the orchestrator's does. - An abort is a stop, not a cancel of the work: the item goes back to the backlog and is worked again on a later pass. To keep it out, remove it. @@ -194,7 +214,7 @@ work into a pipe. | Flag | Meaning | | ---- | ------- | -| `--url
` | The work list API's base address — `add`, `list`, `logs`, `abort`, `remove`, `board` | +| `--url
` | The work list API's base address — `add`, `list`, `logs`, `abort`, `retry`, `remove`, `board` | | `--api-token ` | The work list API's bearer token — every subcommand | | `--api-token-file ` | The file the work list API's bearer token stands in — every subcommand | | `--id ` | The item's id — `add` | diff --git a/docs/guides/work-items.md b/docs/guides/work-items.md index 161f0f6f..c08049ee 100644 --- a/docs/guides/work-items.md +++ b/docs/guides/work-items.md @@ -121,9 +121,9 @@ backlog through the run's [work list API](../commands/orchestrator.md#the-work-l `work list` reports every item with its state — one plain line per item, a dash where a value is absent; `work logs ` (`-f` to follow) prints an item's kept agent output; `work abort ` stops a running item and puts -it back in the backlog; `work remove ` takes an item out of the file, its +it back in the backlog; `work retry ` does the same for a failed item; `work remove ` takes an item out of the file, its state, and its log. `work board` draws the same list as a live kanban — a -column per state, with keys to add, abort, remove, and tail an item's +column per state, with keys to add, abort, retry, remove, and tail an item's log without leaving the screen. The commands name the API's address with `--url` and present its token, and a refusal reads the way the API states it. @@ -148,7 +148,7 @@ giving only the gateway's address. - [`spinloop orchestrator`](../commands/orchestrator.md) — the full command reference, and the [work list API](../commands/orchestrator.md#the-work-list-api) a client works the backlog through - [`spinloop work`](../commands/work.md) — the backlog driven from the shell, - through the run's work list API: add, list, logs, abort, remove, and the + through the run's work list API: add, list, logs, abort, retry, remove, and the live [`board`](../commands/work.md#watching-the-board) - [The fleet file](../fleet-file.md) — tags, concurrency, and waking - [The gateway](../commands/gateway.md) — the front door the orchestrator reads and routes through diff --git a/internal/orchestrator/api.go b/internal/orchestrator/api.go index 7d702695..b6059638 100644 --- a/internal/orchestrator/api.go +++ b/internal/orchestrator/api.go @@ -42,7 +42,7 @@ const LoopbackListen = "127.0.0.1:4010" // pathsServed is the surface the work list API answers, for the 404 that // names it. var pathsServed = []string{ - "/v1/items", "/v1/items/{id}", "/v1/items/{id}/log", "/v1/items/{id}/abort", "/health", + "/v1/items", "/v1/items/{id}", "/v1/items/{id}/log", "/v1/items/{id}/abort", "/v1/items/{id}/retry", "/health", } // Handler is the work list API: the work list it serves, the token its @@ -116,6 +116,8 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) { handler = h.handleLog(id) case ok && action == "abort" && r.Method == http.MethodPost: handler = h.handleAbort(id) + case ok && action == "retry" && r.Method == http.MethodPost: + handler = h.handleRetry(id) default: handler = h.notFound(r) } @@ -314,6 +316,18 @@ func (h *Handler) handleAbort(id string) http.HandlerFunc { } } +// handleRetry puts a failed item back in the backlog; an item that has not +// failed is refused, naming the item and its state. +func (h *Handler) handleRetry(id string) http.HandlerFunc { + return func(w http.ResponseWriter, _ *http.Request) { + if err := h.wl.Retry(id); err != nil { + writeError(w, apiStatus(err), err) + return + } + writeJSON(w, http.StatusOK, map[string]any{"ok": true, "id": id}) + } +} + // writeJSON sends a JSON reply. func writeJSON(w http.ResponseWriter, status int, v any) { w.Header().Set("Content-Type", "application/json") diff --git a/internal/orchestrator/api_test.go b/internal/orchestrator/api_test.go index 5f9aadfa..03253b3f 100644 --- a/internal/orchestrator/api_test.go +++ b/internal/orchestrator/api_test.go @@ -4,6 +4,7 @@ import ( "bytes" "context" "encoding/json" + "errors" "fmt" "io" "log/slog" @@ -205,7 +206,7 @@ func TestAPI_AnUnknownPathIsNamedAsSuch(t *testing.T) { for _, p := range []struct{ method, path string }{ {http.MethodGet, "/nope"}, {http.MethodDelete, "/v1/items"}, - {http.MethodPost, "/v1/items/a/retry"}, + {http.MethodPost, "/v1/items/a/restart"}, {http.MethodPut, "/v1/items/a/log"}, {http.MethodGet, "/v1/items/a"}, } { @@ -252,6 +253,7 @@ func TestAPI_ABearerTokenGatesEveryPath(t *testing.T) { {http.MethodGet, "/v1/items/a/log", ""}, {http.MethodDelete, "/v1/items/a", ""}, {http.MethodPost, "/v1/items/a/abort", ""}, + {http.MethodPost, "/v1/items/a/retry", ""}, {http.MethodGet, "/nope", ""}, } for token, wantStatus := range map[string]int{"": http.StatusUnauthorized, "the-wrong-token": http.StatusUnauthorized} { @@ -614,6 +616,135 @@ func TestAPI_AnAbortOfANonRunningItemIsRefused(t *testing.T) { } } +func TestAPI_ARetryReturnsAFailedItemToTheBacklog(t *testing.T) { + wl, store, path := testWorkList(t, itemsFile(itemSpec{id: "a", instr: "do a", dir: "./a"}), + func(s *memStore) { + s.seedRecord("a", ItemState{State: StateFailed, Why: "it failed", Node: "n1"}) + }) + before, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + srv := workListServer(t, wl, "") + defer srv.Close() + + code, raw := apiDo(t, srv, "", http.MethodPost, "/v1/items/a/retry", "") + if code != http.StatusOK { + t.Fatalf("a retry the work list accepts is answered, got %d: %s", code, raw) + } + if st, recorded := store.records["a"]; recorded { + t.Errorf("the record is gone from the state, got %+v", st) + } + for _, v := range wl.List() { + if v.ID == "a" && v.State != StateBacklog { + t.Errorf("the item shows backlog, got %s", v.State) + } + } + after, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + if string(before) != string(after) { + t.Error("the items file is unchanged by a retry") + } +} + +func TestAPI_ARetryOfAnItemThatHasNotFailedIsRefused(t *testing.T) { + wl, _, _ := testWorkList(t, itemsFile(itemSpec{id: "a", instr: "do a", dir: "./a"}), nil) + srv := workListServer(t, wl, "") + defer srv.Close() + + // In the backlog: the refusal names the item and its state. + code, raw := apiDo(t, srv, "", http.MethodPost, "/v1/items/a/retry", "") + if code != http.StatusConflict || !strings.Contains(raw, `\"a\"`) || !strings.Contains(raw, StateBacklog) { + t.Fatalf("retrying a backlog item is refused, naming the item and its state, got %d: %s", code, raw) + } + for _, state := range []string{StateDone, StateRunning} { + wl.mu.Lock() + wl.records["a"] = ItemState{State: state, Node: "n1"} + wl.mu.Unlock() + code, raw = apiDo(t, srv, "", http.MethodPost, "/v1/items/a/retry", "") + if code != http.StatusConflict || !strings.Contains(raw, state) { + t.Fatalf("retrying a %s item is refused, naming the state, got %d: %s", state, code, raw) + } + wl.mu.Lock() + _, kept := wl.records["a"] + wl.mu.Unlock() + if !kept { + t.Errorf("a refused retry leaves the %s record in place", state) + } + } + code, raw = apiDo(t, srv, "", http.MethodPost, "/v1/items/ghost/retry", "") + if code != http.StatusNotFound || !strings.Contains(raw, "ghost") { + t.Fatalf("an id the file does not carry is refused, naming it, got %d: %s", code, raw) + } + code, _ = apiDo(t, srv, "", http.MethodGet, "/v1/items/a/retry", "") + if code != http.StatusNotFound { + t.Errorf("a retry is a POST only, got %d for a GET", code) + } +} + +// A retried item is admitted again by the run's loop, like any backlog item. +func TestWorkList_ARetriedItemIsAdmittedAgain(t *testing.T) { + topo := &fakeTopo{} + topo.set(Topology{Wake: true, Nodes: []Node{runningNode("n", "org/m", nil)}}) + h := &fakeHarness{name: "opencode", bin: "unused"} + var mu sync.Mutex + launches := 0 + rec := &launchRecorder{factory: func(bin string, args []string, dir, logPath string, env []string) (Child, error) { + mu.Lock() + launches++ + n := launches + mu.Unlock() + if n == 1 { + return newFakeChild(errors.New("boom"), nil), nil + } + return newFakeChild(nil, nil), nil + }} + d := NewDispatcher(h, "http://gateway:4000", "the-token", false) + d.start = rec.start + dir := t.TempDir() + path := writeItems(t, itemsFile(itemSpec{id: "a", dir: dir})) + store, err := OpenStore(path) + if err != nil { + t.Fatal(err) + } + wl, err := NewWorkList(path, store, d, "http://gateway:4000", nil) + if err != nil { + t.Fatal(err) + } + defer wl.Close() + srv := workListServer(t, wl, "") + defer srv.Close() + + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error, 1) + go func() { + done <- Run(ctx, Config{ + Gateway: "http://gateway:4000", + ItemsPath: path, + Topologist: topo, + Dispatcher: d, + Tick: 2 * time.Millisecond, + WorkList: wl, + }) + }() + if err := waitForState(t, path, "a", StateFailed); err != nil { + t.Fatal(err) + } + code, raw := apiDo(t, srv, "", http.MethodPost, "/v1/items/a/retry", "") + if code != http.StatusOK { + t.Fatalf("the retry is answered: %d: %s", code, raw) + } + if err := waitForState(t, path, "a", StateDone); err != nil { + t.Fatal(err) + } + cancel() + if err := <-done; err != nil { + t.Fatalf("the interrupt ends the run without an error: %v", err) + } +} + // --- 1.2: the loop and the API share one state -------------------------------- // The two callers the design names: the loop, and a handler-shaped caller diff --git a/internal/orchestrator/worklist.go b/internal/orchestrator/worklist.go index 175153e7..2921b33f 100644 --- a/internal/orchestrator/worklist.go +++ b/internal/orchestrator/worklist.go @@ -561,6 +561,30 @@ func (w *WorkList) Abort(id string) error { return nil } +// Retry puts a failed item back in the backlog: its record is removed from +// the state, so the run admits it again on a later pass. The items file and +// the item's kept output are left as they are; the output is replaced when +// the new attempt starts. An item that has not failed is refused, naming the +// item and its state, and an id the file does not carry is refused, naming it. +func (w *WorkList) Retry(id string) error { + w.mu.Lock() + defer w.mu.Unlock() + if !w.carries(id) { + return &errMissing{msg: fmt.Sprintf("the items file carries no item with id %q", id)} + } + state := StateBacklog + if r, recorded := w.records[id]; recorded { + state = r.State + } + if state != StateFailed { + return &errConflict{msg: fmt.Sprintf("item %q is not failed: it is %s", id, state)} + } + delete(w.records, id) + w.saveLocked() + w.log.Info("item retried", slog.String("item", id)) + return nil +} + // detachLocked removes the in-flight flight and its record and saves the // state, the caller holding the lock: ok false where the id is not in // flight. The API's abort and the marker's consumption both take the item diff --git a/openspec/changes/archive/2026-10-05-add-work-retry/.openspec.yaml b/openspec/changes/archive/2026-10-05-add-work-retry/.openspec.yaml new file mode 100644 index 00000000..e3966d7a --- /dev/null +++ b/openspec/changes/archive/2026-10-05-add-work-retry/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-10-05 diff --git a/openspec/changes/archive/2026-10-05-add-work-retry/design.md b/openspec/changes/archive/2026-10-05-add-work-retry/design.md new file mode 100644 index 00000000..0fe4ede9 --- /dev/null +++ b/openspec/changes/archive/2026-10-05-add-work-retry/design.md @@ -0,0 +1,65 @@ +## Context + +A finished item's record (`done` or `failed`) sits in the state file beside +the items file. The loop's pass skips an item with such a record +(`WorkList.pass`), and `Add` refuses its id. `Abort` already puts a running +item back in the backlog by deleting its record, under the work list's lock, +and saving. See proposal.md for the motivation. + +## Goals / Non-Goals + +**Goals:** +- Retry reuses the record-removal route that abort already takes, so the + loop needs no new state. +- The command and the board are thin clients of one API path. + +**Non-Goals:** +- Retrying a `done` item. The issue asks for failed items only. +- Retry counts, back-off, or automatic retry. +- Clearing the failed attempt's kept output on retry. + +## Decisions + +**A new `WorkList.Retry(id)` removes the record and saves.** It holds the +lock, refuses with `errMissing` where the file does not carry the id and +with `errConflict` where the record's state is not `failed` (naming the item +and its state, backlog where there is no record), then deletes the record +and calls `saveLocked`. Alternative: a `state` field on the request that +sets the record to backlog. Rejected: a backlog item has no record +elsewhere, so a stored backlog record would be a second shape the loop and +`Join` would have to handle. + +**Route: `POST /v1/items/{id}/retry`, beside `abort`.** `itemPath` already +splits an action segment, so it is one new case in `ServeHTTP` and a new +entry in `pathsServed`. The handler is the same shape as `handleAbort`. +Alternative: `PATCH` the item. Rejected: nothing else in the API edits an +item in place. + +**The kept log stays until the next launch.** Remove deletes the log +because the item is gone. Retry keeps the item, so the old output stays +readable on the detail view and `work logs` until the agent for the new +attempt starts and replaces it. Dropping it at retry would lose the failure +output at the moment someone may still want it. + +**No refusal for the pass in progress.** The failed record is only ever +held by the work list under its lock, and a failed item has no child in +flight, so no stopping-set guard like abort's is needed. Retry takes effect +for the next pass. + +**CLI:** `work retry ` mirrors `workAbortCmd`: `cobra.ExactArgs(1)`, +`workTarget`, `workRequest` with `POST /v1/items/{id}/retry`, the same item +id completion slot, and the message `item %q is back in the backlog`. + +**Board:** a `workRetry` verb beside `workAbort`/`workRemove`, bound to the +`t` key (`r` is refresh). As with abort, there is no client-side state guard: +the API's refusal is shown on the status line. The key asks for a yes first, through the same on-screen question the +removal uses (`y` sends, `n` or `esc` declines). The key hint adds `t retry` +on a failed card. A failed card keeps `x remove`, so the hint code shows +both there. + +## Risks / Trade-offs + +- A retry of an item that failed at launch (for example no matching node) + will fail again → the failure reason is shown on the card as it is today. +- Two operators retrying at once → the second is refused as not failed, + naming the state it now has. diff --git a/openspec/changes/archive/2026-10-05-add-work-retry/proposal.md b/openspec/changes/archive/2026-10-05-add-work-retry/proposal.md new file mode 100644 index 00000000..ee66899d --- /dev/null +++ b/openspec/changes/archive/2026-10-05-add-work-retry/proposal.md @@ -0,0 +1,42 @@ +## Why + +A failed item stays failed: the run records it `failed` and never admits it +again, and `work add` refuses its id. The only ways to work it again are to +remove it and add it back with every field retyped, or to edit the state file +by hand. Issue 250 asks for a retry that puts a failed item back in the +backlog, from the shell and from the work board. + +## What Changes + +- The work list API gains `POST /v1/items/{id}/retry`: it clears a failed + item's record so the item is backlog again and the run admits it on a later + pass. An item that is not failed is refused (`409`, naming the item and its + state) and an id the file does not carry is refused (`404`). +- New `spinloop work retry ` subcommand, a client of that path, worded + and refused the way `work abort` is. +- The work board gains a key that retries the selected failed card. Its key + hint shows only on a failed card. +- The item's kept output from the failed attempt is left in place until the + retried agent starts and writes over it. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `fleet-orchestrator`: the work list API serves a retry path. +- `work-commands`: `work` gains a `retry` subcommand. +- `work-board`: the board offers retry on a failed card. + +## Impact + +- `internal/orchestrator/worklist.go`, `internal/orchestrator/api.go`: a + `Retry` operation and its route. +- `cmd/spinloop/work.go`: the `retry` subcommand; `cmd/spinloop/work_board_model.go` + and `work_board_render.go`: the key, the verb and the hint. +- Shell completion for the item id slot. +- `docs/commands/orchestrator.md` and the work command docs. +- `docs/openapi.yaml` does not cover this API, so it is unchanged. diff --git a/openspec/changes/archive/2026-10-05-add-work-retry/specs/fleet-orchestrator/spec.md b/openspec/changes/archive/2026-10-05-add-work-retry/specs/fleet-orchestrator/spec.md new file mode 100644 index 00000000..f7eef4ac --- /dev/null +++ b/openspec/changes/archive/2026-10-05-add-work-retry/specs/fleet-orchestrator/spec.md @@ -0,0 +1,36 @@ +## ADDED Requirements + +### Requirement: Retrying a failed item through the work list API + +The work list API SHALL serve `POST /v1/items/{id}/retry`, which puts a +failed item back in the backlog: the item's record is removed from the run's +view and the state, so the item is backlog, and the run's loop admits it on +a later pass the way it admits any backlog item. Only a failed item SHALL be +retried: an item recorded done, running, or with no record SHALL be refused +with a conflict naming the item and its state, and an id the items file does +not carry SHALL be refused as not found, naming it. The item's fields in the +items file SHALL NOT change. The pass that handles the request SHALL NOT +admit the item before the record is gone, and a retried item SHALL be +subject to the same matching and limits as any backlog item. + +#### Scenario: A failed item returns to the backlog + +- **WHEN** a request retries an item the run records failed +- **THEN** the item shows backlog to a request that reads the list, and the + loop's next pass may admit it + +#### Scenario: An item that has not failed is refused + +- **WHEN** a request retries an item that is running, done, or backlog +- **THEN** the API answers a conflict naming the item and its state, and + the run's view is unchanged + +#### Scenario: An unknown id is refused + +- **WHEN** a request retries an id the items file does not carry +- **THEN** the API answers not found, naming the id + +#### Scenario: The items file is untouched + +- **WHEN** a failed item is retried +- **THEN** the items file is byte-for-byte what it was before diff --git a/openspec/changes/archive/2026-10-05-add-work-retry/specs/work-board/spec.md b/openspec/changes/archive/2026-10-05-add-work-retry/specs/work-board/spec.md new file mode 100644 index 00000000..5d292249 --- /dev/null +++ b/openspec/changes/archive/2026-10-05-add-work-retry/specs/work-board/spec.md @@ -0,0 +1,60 @@ +## MODIFIED Requirements + +### Requirement: The board acts through the work list API + +The board SHALL offer the work list API's actions on the selected +item: a key that aborts a running item, a key that retries a failed one, and a +key that removes one that is not running. A removal and a retry SHALL each ask on screen for confirmation before +they are sent, defaulting to not proceeding, a declined or abandoned question +sending nothing and saying so. An accepted action SHALL be shown underway +on the status line until the API answers, and the answered action's effect +— the card moving, the card leaving — SHALL follow on the next read that +sees it. Where the API refuses — aborting what is not running, retrying what has not failed, removing a +running item, an id the list no longer carries — the board SHALL show the +refusal on its status line, worded the way the API states it, and keep +drawing. While an action is underway the board SHALL keep a spinner moving, +and the keys it names on screen SHALL be only those that would do +something to what is selected: abort on a running card, retry on a failed +one, and remove on any card that is not running. + +#### Scenario: A running card is aborted + +- **WHEN** the operator aborts a running item's card and the API stops it +- **THEN** the status line says it is stopped and the card stands under + Backlog once the next read sees it + +#### Scenario: A failed card is retried + +- **WHEN** the operator retries a failed item's card, confirms, and the API + accepts +- **THEN** the status line says it is back in the backlog and the card + stands under Backlog once the next read sees it + +#### Scenario: A retry asks first + +- **WHEN** the operator asks to retry a failed card +- **THEN** the board asks, and nothing is sent until yes is given + +#### Scenario: A declined retry sends nothing + +- **WHEN** the operator declines or abandons the retry question +- **THEN** nothing is sent, the board says so, and the card remains under + Failed + +#### Scenario: A removal asks first + +- **WHEN** the operator asks to remove a backlog card +- **THEN** the board asks, and nothing is sent until yes is given + +#### Scenario: A declined removal sends nothing + +- **WHEN** the operator declines or abandons the removal question +- **THEN** nothing is sent, the board says so, and the card remains + +#### Scenario: A refusal keeps the board + +- **WHEN** an action the API refuses is attempted — a running item + removed, a backlog item aborted, a done item retried — or the API cannot be reached for it +- **THEN** the status line carries the refusal worded as the API states it, + or the fault, and the board keeps drawing + diff --git a/openspec/changes/archive/2026-10-05-add-work-retry/specs/work-commands/spec.md b/openspec/changes/archive/2026-10-05-add-work-retry/specs/work-commands/spec.md new file mode 100644 index 00000000..bfef6944 --- /dev/null +++ b/openspec/changes/archive/2026-10-05-add-work-retry/specs/work-commands/spec.md @@ -0,0 +1,67 @@ +## MODIFIED Requirements + +### Requirement: The work commands as work list clients + +`spinloop work` SHALL be a top-level command group with the subcommands +add, list, abort, retry, remove, logs and board, each a client of the +orchestrator's work list API. Every subcommand SHALL take a `--url` flag +naming the API's base address, and SHALL present the API's token as a +bearer on every request it makes — resolved from `--api-token`, else +`--api-token-file`, else the `SPINLOOP_API_TOKEN` environment variable, +two of the flags given at once being a refusal naming both. A subcommand +that names no `--url` SHALL fail before it calls the API, naming the flag. +The commands SHALL be clients of the API alone: they SHALL NOT read or +write the items file, the state, or the logs directly, and SHALL NOT take +any lock beside them. + +#### Scenario: The API's address is named + +- **WHEN** the operator runs a work command naming the API's address with + `--url` +- **THEN** it calls that API, presenting the token as a bearer + +#### Scenario: No address is named + +- **WHEN** the operator runs a work command with no `--url` +- **THEN** it fails, naming the `--url` flag, and calls no API + +#### Scenario: Two token flags at once + +- **WHEN** the operator gives both `--api-token` and `--api-token-file` +- **THEN** the command fails, naming both flags, and calls no API + +#### Scenario: The token comes from the environment + +- **WHEN** the operator names the API's address, sets no token flag, and the + `SPINLOOP_API_TOKEN` environment is set +- **THEN** the command presents that value as the bearer + +## ADDED Requirements + +### Requirement: Retrying a failed item through the work list API + +`spinloop work retry ` SHALL call the API's `POST /v1/items/{id}/retry` +path to put a failed item back in the backlog, and report the API's answer: +where the API accepts, the command SHALL say the item is back in the +backlog; where it refuses — an item that is not failed, naming its state, or +an id the run does not carry — the command SHALL fail naming the refusal the +way the API states it. The command SHALL take exactly one id, and SHALL NOT +itself check the item's state: the API is the one that holds it. + +#### Scenario: A failed item is retried + +- **WHEN** the operator runs `work retry` naming a failed item and the API + accepts +- **THEN** the command says the item is back in the backlog + +#### Scenario: An item that has not failed is refused + +- **WHEN** the operator runs `work retry` naming an item that is running, + done, or backlog +- **THEN** the API refuses, and the command fails naming the item and its + state + +#### Scenario: An id the run does not carry is refused + +- **WHEN** the operator runs `work retry` naming an id the API does not carry +- **THEN** the command fails, naming the id, the way the API states it diff --git a/openspec/changes/archive/2026-10-05-add-work-retry/tasks.md b/openspec/changes/archive/2026-10-05-add-work-retry/tasks.md new file mode 100644 index 00000000..62b10d18 --- /dev/null +++ b/openspec/changes/archive/2026-10-05-add-work-retry/tasks.md @@ -0,0 +1,20 @@ +## 1. Work list API + +- [x] 1.1 Add `WorkList.Retry(id)` in `internal/orchestrator/worklist.go` and verify with unit tests: failed item becomes backlog and the state is saved, running/done/backlog items are refused with a conflict naming the state, an unknown id is a miss, the items file is unchanged +- [x] 1.2 Add `POST /v1/items/{id}/retry` to `ServeHTTP`, `pathsServed` and a `handleRetry` in `internal/orchestrator/api.go`; verify in `api_test.go` for 200, 409, 404, wrong method, and token rules +- [x] 1.3 Verify the loop re-admits a retried item on its next pass with a test in `orchestrator_test.go` + +## 2. Command + +- [x] 2.1 Add `workRetryCmd` to `cmd/spinloop/work.go` and register it; verify in `work_test.go` for the request path, the success message, a refusal, a missing `--url`, and the argument count +- [x] 2.2 Add `retry` to the work subcommands' item id completion and verify with the existing completion tests + +## 3. Work board + +- [x] 3.1 Add the `workRetry` verb, the `t` key and its request and status line wording to `work_board_model.go`; verify in `work_board_test.go` that retrying a failed card sends the request and a refusal reaches the status line +- [x] 3.2 Show `t retry` in `boardKeys` on a failed card only, in `work_board_render.go`; verify with a render test for failed, running, backlog and done cards + +## 4. Docs and finish + +- [x] 4.1 Document the retry path in `docs/commands/orchestrator.md`, the `work retry` command and the board key in the work command docs +- [x] 4.2 Run `gofmt`, `go vet ./...`, `go test ./... -cover` and `openspec validate add-work-retry --strict`, and verify all pass diff --git a/openspec/specs/fleet-orchestrator/spec.md b/openspec/specs/fleet-orchestrator/spec.md index 591785ee..97252cb8 100644 --- a/openspec/specs/fleet-orchestrator/spec.md +++ b/openspec/specs/fleet-orchestrator/spec.md @@ -137,6 +137,7 @@ a caller of the API. already records an item done, failed, or running from a prior run - **THEN** the startup work list shows that item in its recorded state, not backlog + ### Requirement: Serving the work list API The orchestrator SHALL serve a work list API for the life of the run, on @@ -484,3 +485,38 @@ action — the marker goes, and nothing else changes. - **WHEN** a marker stands beside the file for an item that is not running - **THEN** the marker is taken up, and nothing else changes + +### Requirement: Retrying a failed item through the work list API + +The work list API SHALL serve `POST /v1/items/{id}/retry`, which puts a +failed item back in the backlog: the item's record is removed from the run's +view and the state, so the item is backlog, and the run's loop admits it on +a later pass the way it admits any backlog item. Only a failed item SHALL be +retried: an item recorded done, running, or with no record SHALL be refused +with a conflict naming the item and its state, and an id the items file does +not carry SHALL be refused as not found, naming it. The item's fields in the +items file SHALL NOT change. The pass that handles the request SHALL NOT +admit the item before the record is gone, and a retried item SHALL be +subject to the same matching and limits as any backlog item. + +#### Scenario: A failed item returns to the backlog + +- **WHEN** a request retries an item the run records failed +- **THEN** the item shows backlog to a request that reads the list, and the + loop's next pass may admit it + +#### Scenario: An item that has not failed is refused + +- **WHEN** a request retries an item that is running, done, or backlog +- **THEN** the API answers a conflict naming the item and its state, and + the run's view is unchanged + +#### Scenario: An unknown id is refused + +- **WHEN** a request retries an id the items file does not carry +- **THEN** the API answers not found, naming the id + +#### Scenario: The items file is untouched + +- **WHEN** a failed item is retried +- **THEN** the items file is byte-for-byte what it was before diff --git a/openspec/specs/work-board/spec.md b/openspec/specs/work-board/spec.md index 596b1d76..c086abf2 100644 --- a/openspec/specs/work-board/spec.md +++ b/openspec/specs/work-board/spec.md @@ -5,7 +5,9 @@ Watches a running orchestrator's work list as a full-screen kanban board — one column per state, a card per item, kept current as the run works — and lets the operator add, abort, remove and read items, all through the same work list API the one-shot work commands use. + ## Requirements + ### Requirement: The board is a full-screen view of the run's work `spinloop work board` SHALL open an interactive, full-screen view of the @@ -150,18 +152,19 @@ there are only those that do something there. ### Requirement: The board acts through the work list API The board SHALL offer the work list API's actions on the selected -item: a key that aborts a running item and a key that removes one that is -not running. A removal SHALL ask on screen for confirmation before it is -sent, defaulting to not proceeding, a declined or abandoned question +item: a key that aborts a running item, a key that retries a failed one, and a +key that removes one that is not running. A removal and a retry SHALL each ask on screen for confirmation before +they are sent, defaulting to not proceeding, a declined or abandoned question sending nothing and saying so. An accepted action SHALL be shown underway on the status line until the API answers, and the answered action's effect — the card moving, the card leaving — SHALL follow on the next read that -sees it. Where the API refuses — aborting what is not running, removing a +sees it. Where the API refuses — aborting what is not running, retrying what has not failed, removing a running item, an id the list no longer carries — the board SHALL show the refusal on its status line, worded the way the API states it, and keep drawing. While an action is underway the board SHALL keep a spinner moving, and the keys it names on screen SHALL be only those that would do -something to what is selected. +something to what is selected: abort on a running card, retry on a failed +one, and remove on any card that is not running. #### Scenario: A running card is aborted @@ -169,6 +172,24 @@ something to what is selected. - **THEN** the status line says it is stopped and the card stands under Backlog once the next read sees it +#### Scenario: A failed card is retried + +- **WHEN** the operator retries a failed item's card, confirms, and the API + accepts +- **THEN** the status line says it is back in the backlog and the card + stands under Backlog once the next read sees it + +#### Scenario: A retry asks first + +- **WHEN** the operator asks to retry a failed card +- **THEN** the board asks, and nothing is sent until yes is given + +#### Scenario: A declined retry sends nothing + +- **WHEN** the operator declines or abandons the retry question +- **THEN** nothing is sent, the board says so, and the card remains under + Failed + #### Scenario: A removal asks first - **WHEN** the operator asks to remove a backlog card @@ -182,7 +203,7 @@ something to what is selected. #### Scenario: A refusal keeps the board - **WHEN** an action the API refuses is attempted — a running item - removed, a backlog item aborted — or the API cannot be reached for it + removed, a backlog item aborted, a done item retried — or the API cannot be reached for it - **THEN** the status line carries the refusal worded as the API states it, or the fault, and the board keeps drawing @@ -271,4 +292,3 @@ SHALL end the board cleanly, restoring the terminal. - **WHEN** the operator quits the board - **THEN** the command ends cleanly and the terminal is the terminal that was there before - diff --git a/openspec/specs/work-commands/spec.md b/openspec/specs/work-commands/spec.md index 9f756ad4..98394158 100644 --- a/openspec/specs/work-commands/spec.md +++ b/openspec/specs/work-commands/spec.md @@ -4,11 +4,13 @@ Work the items of a running orchestrator from the shell — add an item, read the backlog, stop a running item, remove an item — as a client of the orchestrator's work list API: the commands name the API's address, present its token, and the run's view of the items is the source of truth. + ## Requirements + ### Requirement: The work commands as work list clients `spinloop work` SHALL be a top-level command group with the subcommands -add, list, abort, remove, logs and board, each a client of the +add, list, abort, retry, remove, logs and board, each a client of the orchestrator's work list API. Every subcommand SHALL take a `--url` flag naming the API's base address, and SHALL present the API's token as a bearer on every request it makes — resolved from `--api-token`, else @@ -233,3 +235,30 @@ in: the API's call is the whole ask, and it answers once the item is out. - **WHEN** the operator removes an id the items file does not carry - **THEN** the API refuses, naming the id, and the command fails naming it +### Requirement: Retrying a failed item through the work list API + +`spinloop work retry ` SHALL call the API's `POST /v1/items/{id}/retry` +path to put a failed item back in the backlog, and report the API's answer: +where the API accepts, the command SHALL say the item is back in the +backlog; where it refuses — an item that is not failed, naming its state, or +an id the run does not carry — the command SHALL fail naming the refusal the +way the API states it. The command SHALL take exactly one id, and SHALL NOT +itself check the item's state: the API is the one that holds it. + +#### Scenario: A failed item is retried + +- **WHEN** the operator runs `work retry` naming a failed item and the API + accepts +- **THEN** the command says the item is back in the backlog + +#### Scenario: An item that has not failed is refused + +- **WHEN** the operator runs `work retry` naming an item that is running, + done, or backlog +- **THEN** the API refuses, and the command fails naming the item and its + state + +#### Scenario: An id the run does not carry is refused + +- **WHEN** the operator runs `work retry` naming an id the API does not carry +- **THEN** the command fails, naming the id, the way the API states it