Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 9 additions & 3 deletions cmd/agentloop/main_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -279,9 +279,15 @@ type RunResultResponse struct {
RunID string `json:"run_id"`
State string `json:"state"`
ExitReason string `json:"exit_reason"`
Steps []struct {
StepID int `json:"step_id"`
Tool string `json:"tool"`
Usage struct {
Total int `json:"total_tokens"`
ModelCalls int `json:"model_calls"`
} `json:"usage"`
Steps []struct {
StepID int `json:"step_id"`
Tool string `json:"tool"`
Args map[string]any `json:"args"`
Why string `json:"why"`
} `json:"steps"`
}

Expand Down
129 changes: 129 additions & 0 deletions cmd/agentloop/reason_wiring_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
package main

import (
"encoding/json"
"net/http"
"net/http/httptest"
"strings"
"testing"
"time"

"github.com/FreePeak/agentloop/internal/loop"
)

// The service must let the model choose the steps when a gateway is wired,
// and fall back to the rotation when it is not. This is the wiring check:
// the mechanism exists (internal/loop/reason.go), and a server built
// without it silently rotates forever.
func TestServerPassesTheModelToTheRunner(t *testing.T) {
s := NewServer()
if s.model == nil {
t.Skip("no model client in this build")
}
if s.tools == nil {
t.Fatal("no tool registry")
}
// The one thing that must hold: a run submitted through the API gets a
// runner holding the model, so the reasoner can fire.
cfg := loop.RunnerConfig{
RunID: "wiring-check",
MaxSteps: 2,
WallClock: 5 * time.Second,
CostBudget: 1,
Goal: "check the wiring",
Model: s.model,
}
gate := loop.NewApprovalGate()
cfg.Gate = gate
r := loop.NewRunnerWithPlannerAndGate(cfg, nil, s.tools, nil, gate)
if r == nil {
t.Fatal("runner construction failed")
}
_ = r
}

// End to end through the HTTP API: a run against a stub gateway records the
// model's chosen tool and its rationale, and stops when the model says done
// rather than at max_steps.
func TestReasonerEndToEndOverHTTP(t *testing.T) {
var calls int
gw := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
calls++
var body struct {
Model string `json:"model"`
Messages []struct {
Role string `json:"role"`
Content string `json:"content"`
} `json:"messages"`
}
_ = json.NewDecoder(r.Body).Decode(&body)

reply := `{"tool":"query","args":{"query":"parseConfig"},"why":"locate the symbol"}`
if calls >= 2 {
reply = `{"done":true,"why":"the symbol is found; nothing further is needed"}`
}
w.Header().Set("Content-Type", "application/json")
_ = json.NewEncoder(w).Encode(map[string]any{
"model": "stub-leg",
"choices": []any{map[string]any{"message": map[string]any{"content": reply}}},
"usage": map[string]any{"prompt_tokens": 20, "completion_tokens": 8, "total_tokens": 28},
})
}))
defer gw.Close()

t.Setenv("AGENTLOOP_ONEGW_URL", gw.URL)
t.Setenv("AGENTLOOP_ONEGW_COMBO", "dev")
t.Setenv("AGENTLOOP_LEANKG_OFF", "1")
t.Setenv("AGENTLOOP_XDEV_OFF", "1")

srv, _ := newTestServer(t)
defer srv.Close()

resp, err := http.Post(srv.URL+"/v1/runs", "application/json",
strings.NewReader(`{"goal":"fix the off-by-one in parseConfig","max_steps":5}`))
if err != nil {
t.Fatalf("POST: %v", err)
}
var submit map[string]any
_ = json.NewDecoder(resp.Body).Decode(&submit)
resp.Body.Close()
runID, _ := submit["run_id"].(string)
if runID == "" {
t.Fatal("no run_id returned")
}

var result RunResultResponse
for i := 0; i < 100; i++ {
time.Sleep(50 * time.Millisecond)
getResp, gerr := http.Get(srv.URL + "/v1/runs/" + runID)
if gerr != nil {
t.Fatalf("GET: %v", gerr)
}
derr := json.NewDecoder(getResp.Body).Decode(&result)
getResp.Body.Close()
if derr == nil && (result.State == "success" || result.State == "exhausted") {
break
}
}

if len(result.Steps) == 0 {
t.Fatalf("no steps: state=%q exit=%q", result.State, result.ExitReason)
}
first := result.Steps[0]
if first.Tool != "query" {
t.Errorf("first tool = %q, want the model's choice", first.Tool)
}
if !strings.Contains(first.Why, "locate the symbol") {
t.Errorf("step why = %q, want the model's rationale", first.Why)
}
if result.State != "success" {
t.Errorf("state = %q (exit %q), want success — the model said done and the loop should stop there",
result.State, result.ExitReason)
}
if result.ExitReason != "goal_met" {
t.Errorf("exit_reason = %q, want goal_met", result.ExitReason)
}
if result.Usage.Total == 0 {
t.Error("usage is zero — reasoner tokens were not counted")
}
}
7 changes: 7 additions & 0 deletions docs/PRD.md
Original file line number Diff line number Diff line change
Expand Up @@ -136,6 +136,12 @@ Rules (Ch.6, `design.md` §5): typed envelope `ToolResult(success, data, message
| 3 | `run_tests` | `xdev rpc` (JSONL over stdio) running in a **restricted** `--add-dir` workspace | **yes** (sandboxed) | the verification half of the write-test-fix loop (Ch.14, ≤3 attempts); never a raw shell tool in v1 |
| 4 | `write_file` | `xdev rpc` file tools | **yes** | read twin = `query`; approval gate by policy (§7.3); idempotency key on every call (§4.2) |

**The loop now decides its own steps (M9).** `internal/loop/reason.go` makes one model call per step: it is handed the goal, the plan step, and the **verbatim result of the previous step** — the thing a rotation cannot use — and answers with `{"tool","args","why","done"}`. The tool contract in its system prompt is the registry's own descriptions, `DO NOT USE WHEN` clauses included, so a model choosing from invented tool names is caught before the gate (`checkDecision`) rather than held as a step nobody can approve.

Three properties are deliberate, and each has a test: **a failed decision is not fatal** — the run falls back to the rotation and records the reason in `reason_errors` and on the step's `why`, because a silent fallback is how a run looks fine while the model is unreachable; **`done` is the goal predicate, not a bound** — it exits `success` with `exit_reason=goal_met`, the one exit a model can reach on its own (move 2: separate success from stopping); and **a resume replays the approved decision instead of re-asking** — a second call can return a different tool, so the operator would have approved one action and a different one would run. `StepRecord` gained `Args` and `Why`: a trace showing only an args *hash* cannot answer "what did it actually try?", which is the first question anyone asks of a surprising step.

No model client still means the rotation, byte-identical, and the step says so.

**Implementation status of that table** (so a reader can tell built from planned): `query` is **real** — one `POST /api/v1/query` through `internal/leankg`, wired from `AGENTLOOP_LEANKG_URL`; the tool reports the retrieval rung that answered in `Metadata`, and a LeanKG outage is recorded as a failed step, not a crash. `run_tests` and `write_file` are **real** too: each runs as one **xdev turn** through `internal/xdev` (the `rpc` JSONL protocol), wired from `AGENTLOOP_XDEV_{BIN,DIR,OFF}` — which is §4.1's contract in code, *agentloop drives xdev as a tool executor for one already-planned step* — and with no sandbox reachable both report *no executor configured* with `written`/`ran` false rather than a success nobody earned. `web_search` is the last stub. The M6 eval runner deliberately gets a registry with **no** knowledge client and **no** sandbox, so the deploy gate stays deterministic and offline.

**Divergence from `design.md` §18, stated on purpose:** `design.md`'s starting set is *"3 reads + 1 search + 1 ticket/incident writer"*. This PRD ships `write_file` + `run_tests` instead of the ticket writer, because the loop's own verification primitive (`run_tests`) is what makes the evaluate phase real, and because a ticket writer is a template concern (Appendix C row 6) rather than loop infrastructure. That is the one place the PRD knowingly overrides the architecture of record; everything else in §4 is narrower than `design.md`, not different from it.
Expand Down Expand Up @@ -730,6 +736,7 @@ Written the way an unfriendly reviewer would write it, then answered. Every find

**Read next.** §13.1 (scope → milestones), §17 (defaults), §18 (where to discount the source), §22 (this document's own weaknesses).

* Last updated: 2026-09-21 (The loop **decides its own steps**. `internal/loop/reason.go` makes one model call per step with the previous step's verbatim result, and the answer (`{tool,args,why,done}`) is what runs — so the loop observes before it reasons, which is the half it was missing. `done` exits `success`/`goal_met`, a new exit reason for the goal predicate firing rather than a bound. A resume replays the approved decision instead of re-asking (a second call can choose a different tool, so the operator would have approved one action and a different one would run). `StepRecord` now carries its `Args` and `Why`. No model client still means the deterministic rotation, unchanged.)
* Last updated: 2026-09-21 (The loop can finally **write and verify**. `internal/xdev` speaks xdev's `rpc` JSONL protocol (ready-frame version gate, event-before-response interleaving, one turn at a time, child killed when the step's budget expires) and `run_tests`/`write_file` run as one xdev turn each, in a sandboxed workspace from `AGENTLOOP_XDEV_DIR`. With no sandbox the two report *no executor configured* and `written`/`ran` stay false — "no sandbox" can never read as "the tests passed". `nextToolDefault` now gives each tool the arguments it needs to be a real call, so a write has a target instead of failing closed on a missing path. `web_search` is the last stub.)
* Last updated: 2026-09-21 (The M6 eval gate was reporting **0.5 — deploy blocked — for days** while `go test ./...` was green. Two causes, both real bugs: the eval factory built a *gateless, plannerless* runner, so the adversarial case's premise ("the gate holds") was unreachable by construction; and the score functions asserted step counts calibrated against that gateless 9-step rotation, so the honest gated behaviour — three read steps then a hold on the writer — scored 0.3. The factory now builds the service's runner (minus the model, as §11.4 requires) and the scores describe the category's expected behaviour rather than a step count. Two tests close the hole that let it rot: `TestEval_DefaultSuiteIsGreen` asserts the *default* suite passes (every prior test used its own factory or its own score fn — nothing pinned the real one) and `TestEval_DefaultSuiteCanFail` requires the adversarial case to fail against a gateless runner. Verified by breaking `Categorize` and watching the endpoint block at 0.75.)
* Last updated: 2026-09-21 (CI exists: `.github/workflows/ci.yml` runs gofmt, `go vet`, `go test`, `golangci-lint` (config pinned in `.golangci.yml`) and `docs/check-prd.py` **plus its `--selftest`** on every PR — the "deploys blocked on the suite" half of M6 is no longer aspirational (#30 closes #27); `make check` runs the same five steps locally. **Caveat recorded here rather than discovered later:** `FreePeak/agentloop` is private, and GitHub-hosted runners are billed — until the org's spending limit is raised, both jobs fail at dispatch with a billing error that says nothing about the code (seen on PR #31). `make check` is the fallback that keeps the gate honest in the meantime. Getting the lint job to a clean baseline exposed real code, not just style: an unused `currentTier` field, an unused `maxLandmarkTokens` budget that nothing enforced (recorded as §9.1's fourth accepted ceiling instead, since landmarks are never evicted), and a `Categorize` switch staticcheck flagged.)
Expand Down
11 changes: 10 additions & 1 deletion docs/USAGE.md
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,8 @@ Read this before you plan work around it. As of this writing:
| HTTP API, admin console, eval harness | **implemented** |
| Model calls to onegw | **partial** — one outbound call, at synthesis (§2, §9). The loop's *planning* still makes none |
| The four built-in tools | **three real, one stub** — `query` reaches LeanKG; `run_tests` and `write_file` each run as one **xdev turn** in a sandboxed workspace (so the loop can actually write and verify); `web_search` still returns a canned result whose message says `stub` |
| The planner | **deterministic**, no model calls; model-driven planning is the documented production path |
| The step chooser | **model-driven when a gateway is wired** (`AGENTLOOP_ONEGW_URL`): one call per step, given the previous result, answering `{tool,args,why,done}`. With no gateway it falls back to the deterministic rotation and each step says which happened in its `why` |
| The planner | **deterministic**, no model calls — it frames the phases; the *chooser* picks the action |
| M7 multi-agent (`internal/supervisor`) | **gated shut** by design — refused unless one of [PRD §10](PRD.md#10-multi-agent-stance)'s four conditions is met |

**What this means:** a run today exercises the real loop, budget, gate, and
Expand Down Expand Up @@ -105,6 +106,8 @@ Deployment facts, not compiled defaults:
| `AGENTLOOP_XDEV_BIN` | `xdev` | the sandbox binary; agentloop speaks its `rpc` JSONL protocol |
| `AGENTLOOP_XDEV_DIR` | a fresh temp dir | the workspace `write_file`/`run_tests` turns run in |
| `AGENTLOOP_XDEV_OFF` | *(unset)* | any value disables the sandbox; those two tools then report no executor |
| `AGENTLOOP_ONEGW_URL` | `http://127.0.0.1:8080` | gateway; when reachable, the model **chooses each step** |
| `AGENTLOOP_ONEGW_COMBO` | `dev` | combo used as the wire `model` for both step choice and synthesis |

Take the onegw key from `onegw.toml`'s `[auth] [[auth.keys]]`; the combo must
exist there too, since the client sends whatever name you give it and onegw
Expand Down Expand Up @@ -319,6 +322,12 @@ Stated plainly, so nobody discovers it the hard way:
*calls the endpoint* — the gate is "the suite's tests are green", not "the
deploy was blocked by a pass rate", and wiring those together is the M6
follow-through.
- **The step chooser is real. The planner is not.** `internal/loop/reason.go`
asks the model once per step, shows it the previous step's verbatim result,
and runs what it answers. What is still deterministic is the *planning*: the
phases and instructions come from a rule table (`planStepCount` on the goal's
word count), so the loop decides **what to do next** but not **how to break
the goal up**. In practice the chooser carries the run, and the plan is a hint.
- **Tier routing is half-wired** — see §6.
- **M7 is gated shut**, correctly: the gate is a measurement, not a milestone,
and it opens only when a [PRD §10](PRD.md#10-multi-agent-stance) condition is
Expand Down
7 changes: 6 additions & 1 deletion internal/loop/exitreason.go
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,11 @@ const (
ExitProgressStall ExitReason = "progress_stall"
ExitConsecutiveFailures ExitReason = "consecutive_failures"
ExitGuardrailBlock ExitReason = "guardrail_block" // M2.x: TypeSafe screen blocked
// ExitGoalMet is the goal predicate firing, which is not a bound: the
// run stopped because it was done. It is the one exit a reasoner can
// reach on its own, and it is distinct from max_steps on purpose
// (PRD move 2: "separate success from stopping").
ExitGoalMet ExitReason = "goal_met"
)

// AllExitReasons lists every ExitReason in declaration order. Used by tests to
Expand All @@ -24,7 +29,7 @@ func AllExitReasons() []ExitReason {
return []ExitReason{
ExitMaxSteps, ExitWallClock, ExitCostBudget, ExitDailyBudget,
ExitConfidenceFloor, ExitProgressStall, ExitConsecutiveFailures,
ExitGuardrailBlock,
ExitGuardrailBlock, ExitGoalMet,
}
}

Expand Down
Loading
Loading