diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..c36df19 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,83 @@ +name: ci + +# The check that makes "all green" mean something. Before this workflow the +# only thing that ran on a pull request was the gitStream app, which reports +# `skipping` — every "tests pass" claim meant a human ran `go test ./...` +# (issue #27). +# +# Two jobs, both required by branch protection: +# build-test the code: format, vet, test, lint +# prd the document: docs/check-prd.py asserts the PRD's own promises, +# and --selftest proves the check can still fail +# +# Keep the Go version in step with go.mod; `go-version-file` reads it, so +# there is one source of truth rather than two. + +on: + push: + branches: [main] + pull_request: + branches: [main] + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ci-${{ github.ref }} + cancel-in-progress: true + +jobs: + build-test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-go@v5 + with: + go-version-file: go.mod + cache: true + + # Formatting first: it is the cheapest failure and the most annoying to + # discover after review. + - name: gofmt + run: | + unformatted="$(gofmt -l ./cmd ./internal)" + if [ -n "$unformatted" ]; then + echo "::error::these files are not gofmt-clean:" + echo "$unformatted" + exit 1 + fi + + - name: build + run: go build ./... + + - name: vet + run: go vet ./... + + - name: test + run: go test ./... -count=1 + + - name: lint + uses: golangci/golangci-lint-action@v6 + with: + version: v2.1.6 + + # The PRD is a build artifact of this project, so a broken one fails the same + # way broken code does. `--selftest` matters as much as the check itself: it + # asserts the checker can still fail, which is what stops it rotting into a + # script that always prints OK. + prd: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: '3.x' + + - name: PRD asserts its own promises + run: python3 docs/check-prd.py + + - name: the check can still fail + run: python3 docs/check-prd.py --selftest diff --git a/.gitignore b/.gitignore index 0c469dd..2f4a61b 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,7 @@ +# The built binary (Makefile: `build` -> ./agentloop). +# `!cmd/agentloop/` un-ignores the SOURCE directory of the same name — without +# it the bare pattern above also matched the directory and silently hid every +# new file added under cmd/agentloop/ (found the hard way: cmd/agentloop/ +# debug_test.go had been invisible to git). agentloop !cmd/agentloop/ -!cmd/agentloop/ diff --git a/.golangci.yml b/.golangci.yml new file mode 100644 index 0000000..462da5f --- /dev/null +++ b/.golangci.yml @@ -0,0 +1,42 @@ +# golangci-lint config — pinned so CI and a local `make lint` agree. +# +# v2 schema. The default linter set is kept and only two things are changed, +# both stated rather than implied: +# +# * errcheck is excluded in _test.go. An unchecked `defer resp.Body.Close()` +# or `json.NewDecoder(...).Decode(...)` in a test is conventional and its +# failure mode is a test that fails anyway; requiring `_ =` on every one +# adds noise without adding signal. Production code is NOT excluded — those +# were fixed instead (see PR #30). +# * a handful of style checks that fight the codebase's existing shape are +# off, each with the reason inline. Nothing here turns off a correctness +# check. +version: "2" + +linters: + default: standard + exclusions: + generated: lax + presets: + - comments + - common-false-positives + - legacy + rules: + # Tests: an unchecked Close/Decode is conventional there. + - path: _test\.go + linters: + - errcheck + # §-heavy prose in doc comments (PRD citations) trips nothing here on + # purpose; the PRD's own checker owns that surface. + - path: \.go + text: "comment on exported" + linters: + - revive + +formatters: + enable: + - gofmt + +issues: + max-issues-per-linter: 0 + max-same-issues: 0 diff --git a/Makefile b/Makefile index 5e35a20..774807e 100644 --- a/Makefile +++ b/Makefile @@ -13,7 +13,7 @@ PORT ?= 8081 GO ?= go -.PHONY: help build run test lint vet fmt tidy prd check smoke clean +.PHONY: help build run test lint vet fmt fmt-check tidy prd check smoke clean help: ## list the available targets @grep -hE '^[a-zA-Z_-]+:.*?## ' $(MAKEFILE_LIST) \ @@ -28,10 +28,16 @@ run: build ## run the server on $(PORT) test: ## run the test suite $(GO) test ./... -check: test lint ## tests + lint — run this before opening a PR +check: fmt-check vet test lint prd ## everything CI runs — run this before opening a PR -lint: ## golangci-lint - golangci-lint run ./... +fmt-check: ## fail if any file is not gofmt-clean (what CI's gofmt step does) + @unformatted="$$(gofmt -l ./cmd ./internal)"; \ + if [ -n "$$unformatted" ]; then \ + echo "not gofmt-clean:"; echo "$$unformatted"; exit 1; \ + fi + +lint: ## golangci-lint (config pinned in .golangci.yml; same scope as CI) + golangci-lint run ./cmd/... ./internal/... vet: ## go vet $(GO) vet ./... diff --git a/README.md b/README.md index d485877..2756b26 100644 --- a/README.md +++ b/README.md @@ -22,9 +22,13 @@ goal + budget in, answer / handoff / escalation out. ```bash make # list targets make run # build + serve on :8081 (onegw owns 8080) -make check # tests + lint — before a PR +make check # everything CI runs: fmt-check, vet, test, lint, prd ``` +CI (`.github/workflows/ci.yml`) runs exactly those five steps on every pull +request, plus `docs/check-prd.py --selftest` — so "all green" is a check's +verdict, not a claim in a PR description. + Submit a run and watch it go to work: ```bash diff --git a/cmd/agentloop/debug_test.go b/cmd/agentloop/debug_test.go index 2025dfb..57dfea4 100644 --- a/cmd/agentloop/debug_test.go +++ b/cmd/agentloop/debug_test.go @@ -8,7 +8,6 @@ import ( "strings" "testing" "time" - ) func TestDebugSubmit(t *testing.T) { diff --git a/cmd/agentloop/main.go b/cmd/agentloop/main.go index a3aa461..115cf49 100644 --- a/cmd/agentloop/main.go +++ b/cmd/agentloop/main.go @@ -8,6 +8,7 @@ import ( "crypto/rand" "encoding/json" "fmt" + "log" "net/http" "os" "strconv" @@ -152,7 +153,7 @@ func (s *Server) submitRun(w http.ResponseWriter, r *http.Request) { }() w.Header().Set("Content-Type", "application/json") w.WriteHeader(http.StatusCreated) - json.NewEncoder(w).Encode(map[string]any{"run_id": runID, "state": loop.StateThinking}) + writeJSON(w, map[string]any{"run_id": runID, "state": loop.StateThinking}) } func (s *Server) getRun(w http.ResponseWriter, r *http.Request) { @@ -165,7 +166,7 @@ func (s *Server) getRun(w http.ResponseWriter, r *http.Request) { return } w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(result) + writeJSON(w, result) } func (s *Server) killRun(w http.ResponseWriter, r *http.Request) { @@ -195,7 +196,7 @@ func (s *Server) killRun(w http.ResponseWriter, r *http.Request) { s.runs[runID] = result s.mu.Unlock() w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(result) + writeJSON(w, result) } func (s *Server) deleteRun(w http.ResponseWriter, r *http.Request) { @@ -239,7 +240,7 @@ func (s *Server) getApprovals(w http.ResponseWriter, r *http.Request) { return } w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(map[string]any{"run_id": runID, "approvals": gate.Ledger(), "state": result.State}) + writeJSON(w, map[string]any{"run_id": runID, "approvals": gate.Ledger(), "state": result.State}) } func (s *Server) submitApproval(w http.ResponseWriter, r *http.Request) { @@ -276,7 +277,7 @@ func (s *Server) submitApproval(w http.ResponseWriter, r *http.Request) { s.mu.Unlock() } w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(map[string]any{"approved": approved}) + writeJSON(w, map[string]any{"approved": approved}) } func (s *Server) submitApprovalByID(w http.ResponseWriter, r *http.Request) { @@ -301,7 +302,7 @@ func (s *Server) submitApprovalByID(w http.ResponseWriter, r *http.Request) { s.mu.Unlock() } w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(map[string]any{"approved": approved}) + writeJSON(w, map[string]any{"approved": approved}) } func parseApprovalPath(path string) (string, int) { @@ -330,11 +331,11 @@ func (s *Server) evalReportHandler(w http.ResponseWriter, r *http.Request) { s.mu.Lock() defer s.mu.Unlock() w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(eval.Report{SuiteID: "default", Total: 0, Passed: 0, PassRate: 0, ByCategory: map[string]float64{}}) + writeJSON(w, eval.Report{SuiteID: "default", Total: 0, Passed: 0, PassRate: 0, ByCategory: map[string]float64{}}) return } w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(report) + writeJSON(w, report) } func (s *Server) runsListHandler(w http.ResponseWriter, r *http.Request) { @@ -345,7 +346,7 @@ func (s *Server) runsListHandler(w http.ResponseWriter, r *http.Request) { } s.mu.Unlock() w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(map[string]any{"runs": runs}) + writeJSON(w, map[string]any{"runs": runs}) } func (s *Server) consoleRunsPage(w http.ResponseWriter, r *http.Request) { @@ -353,7 +354,7 @@ func (s *Server) consoleRunsPage(w http.ResponseWriter, r *http.Request) { runCount := len(s.runs) s.mu.Unlock() w.Header().Set("Content-Type", "text/html") - fmt.Fprintf(w, "
Total: %d
", runCount) + writeConsole(w, "Total: %d
", runCount) } func (s *Server) consoleApprovalsPage(w http.ResponseWriter, r *http.Request) { @@ -366,7 +367,7 @@ func (s *Server) consoleApprovalsPage(w http.ResponseWriter, r *http.Request) { } s.mu.Unlock() w.Header().Set("Content-Type", "text/html") - fmt.Fprintf(w, "Pending: %d
", pending) + writeConsole(w, "Pending: %d
", pending) } func (s *Server) consoleKillHandler(w http.ResponseWriter, r *http.Request) { @@ -394,7 +395,7 @@ func (s *Server) consoleKillHandler(w http.ResponseWriter, r *http.Request) { s.runs[runID] = result s.mu.Unlock() w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(result) + writeJSON(w, result) } func (s *Server) events(w http.ResponseWriter, r *http.Request) { @@ -413,14 +414,14 @@ func (s *Server) events(w http.ResponseWriter, r *http.Request) { for _, step := range result.Steps { evt := map[string]any{"step_id": step.StepID, "tool": step.Tool, "phase": step.Phase} b, _ := json.Marshal(evt) - fmt.Fprintf(w, "event: step\ndata: %s\n\n", b) + writeConsole(w, "event: step\ndata: %s\n\n", b) if flush != nil { flush.Flush() } } done := map[string]any{"state": result.State, "exit_reason": result.ExitReason} b, _ := json.Marshal(done) - fmt.Fprintf(w, "event: done\ndata: %s\n\n", b) + writeConsole(w, "event: done\ndata: %s\n\n", b) if flush != nil { flush.Flush() } @@ -444,6 +445,16 @@ func extractRunID(path string) string { func boolPtr(b bool) *bool { return &b } +// writeJSON writes one JSON body. A failure here cannot be reported to the +// client — the status line and headers are already on the wire — so the error +// is explicitly discarded rather than left unchecked, and the request is +// logged so a broken client is still visible. +func writeJSON(w http.ResponseWriter, v any) { + if err := json.NewEncoder(w).Encode(v); err != nil { + log.Printf("write response: %v", err) + } +} + // m6SuiteCases returns the M6 acceptance suite (PRD §11.4) used as // the deploy gate. It delegates to eval.DefaultSuite so the suite // definition lives in one place (internal/eval) and the HTTP handler @@ -452,6 +463,14 @@ func m6SuiteCases() []eval.Case { return eval.DefaultSuite() } +// writeConsole is writeJSON's counterpart for the HTML and SSE surfaces, +// where a failed write is equally unreportable and equally worth logging. +func writeConsole(w http.ResponseWriter, format string, args ...any) { + if _, err := fmt.Fprintf(w, format, args...); err != nil { + log.Printf("write console: %v", err) + } +} + func main() { s := NewServer() mux := http.NewServeMux() @@ -475,5 +494,7 @@ func main() { } addr := ":" + port fmt.Printf("agentloop listening on %s\n", addr) - http.ListenAndServe(addr, mux) + if err := http.ListenAndServe(addr, mux); err != nil { + log.Fatalf("serve: %v", err) + } } diff --git a/docs/PRD.md b/docs/PRD.md index cb38a25..adf95b7 100644 --- a/docs/PRD.md +++ b/docs/PRD.md @@ -329,11 +329,12 @@ Every default in §17 is a calibrated line item, not a book assertion, and the r ### 9.1 Known ceilings to cut at scale (marked in code) -Per convention, deliberate shortcuts ship with a `ponytail:` comment naming the ceiling and the upgrade path. v1 accepts three: +Per convention, deliberate shortcuts ship with a `ponytail:` comment naming the ceiling and the upgrade path. v1 accepts four: 1. **Run registry:** single-process, in-memory + SQLite; one shared mutex guards state transitions. Ceiling: one agentloop process per host. Upgrade: a Postgres advisory-lock run table with a worker pool, the moment a second replica is wanted. 2. **Cost accumulation:** O(n) sum over a run's spans at the 90% check (n is small). Ceiling: O(n) per step. Upgrade: a running counter on the run row once n > 500. 3. **Retention sweep:** a periodic full scan of the trace table for 90-day expiry. Ceiling: O(rows) nightly. Upgrade: an indexed `expires_at` delete when the table passes ~10M rows. +4. **The 70% rule is a working-tier guarantee, not a whole-store one.** `enforceCeiling()` (`internal/memory/memory.go`) evicts from the working tier only, and stops rather than evict anything else — the landmark tier is never evicted (P32) and retrieved is capped by its own 20%. A run that promotes enough landmarks can therefore sit above 70% and stay there. Ceiling: whole-store usage can exceed the ceiling when landmarks dominate. Upgrade: a promotion cap (reject a promotion that would push landmarks past 20%, recording it as a compressed summary) — deliberately not built, because losing a decision to a budget is the worse failure of the two. ## 10. Multi-agent stance @@ -406,7 +407,7 @@ The build order is `design.md` §17 (build order), kept 1:1 so there is one reco | M3 | Planning | Planner/Replanner, parallel phases, tiered routing through onegw | a 5+-step task is ≥40% cheaper than single-tier ReAct **with no case scoring below the single-tier baseline by more than its eval tolerance** (the same 50-case suite run both ways, paired per case — the cost number is meaningless without this half) | **closed** 2026-09-19 (PR #7: wired Planner output into Run() loop; tier routing via tierCombo) | | M4 | Memory & state | 4 tiers, 70% rule, landmarks, checkpoints, deletion API | 20-iteration run holds the 70% rule; resume from step-5 checkpoint after a step-7 fault | **closed** 2026-09-19 (feat/m4-memory-state: 4-tier memory with 70% ceiling enforcement, SQLite WAL checkpoints every 5 iterations, resume from step-5 checkpoint through step-7 fault, DELETE /v1/runs/{id}; PR to be assigned) | | M5 | HITL | ApprovalGate, audit, progressive autonomy counters, approval queue UI | <10% interruptions; sampling + anomaly review exercised; timeout denies | **closed** 2026-09-20 (PR #10: ApprovalGate with fail-closed policy table + audit ledger; runner pause on denied gate; 30-minute timeout denies and escalates with partial; M5 acceptance tests in `internal/loop/m5_test.go` — 6 cases; `Categorize` was fixed after this row was written (#25): it matched none of the four v1 tool names, so every gated run paused on step 1) | -| M6 | Evals & console | EvalRunner in CI, REFINE job, HTMX console (trajectory, spend, evals, kill) | deploys blocked on the full-suite gate; +10 cases/week; knee table published | **closed** 2026-09-20 (PR #10: EvalRunner with 4-category suite and pass = score ≥ 0.8 ∧ latency ≤ cap ∧ cost ≤ cap; 3 tests in `internal/eval/eval_test.go`; live HTTP endpoints `GET /admin/api/v1/evals`, `GET /v1/runs/{id}/events`, console pages `GET /admin/console/{runs,approvals}`, `POST /admin/console/kill`); **the "deploys blocked" half is not enforced yet — there is no CI in this repo, `go test ./...` is run by hand (issue #27)** | +| M6 | Evals & console | EvalRunner in CI, REFINE job, HTMX console (trajectory, spend, evals, kill) | deploys blocked on the full-suite gate; +10 cases/week; knee table published | **closed** 2026-09-20 (PR #10: EvalRunner with 4-category suite and pass = score ≥ 0.8 ∧ latency ≤ cap ∧ cost ≤ cap; 3 tests in `internal/eval/eval_test.go`; live HTTP endpoints `GET /admin/api/v1/evals`, `GET /v1/runs/{id}/events`, console pages `GET /admin/console/{runs,approvals}`, `POST /admin/console/kill`); **and the gate is now real**: `.github/workflows/ci.yml` runs gofmt/vet/test/lint plus `docs/check-prd.py --selftest` on every PR (#30, closing #27). What it does **not** yet do is call the eval endpoint — the gate is "a green suite", not "a passing suite", and wiring the two is the M6 follow-through) | | M7 | Multi-agent (conditional) | supervisor + specialists, typed bus, role cards | only after §10's gate is met; coordination <30% of tokens | conditional | ### 13.1 The twelve moves that carry the book — where each one lands @@ -729,6 +730,7 @@ Written the way an unfriendly reviewer would write it, then answered. Every find **Read next.** §13.1 (scope → milestones), §17 (defaults), §18 (where to discount the source), §22 (this document's own weaknesses). +* Last updated: 2026-09-21 (CI exists: `.github/workflows/ci.yml` runs gofmt, `go vet`, `go test`, `golangci-lint` (config pinned in `.golangci.yml`) and `docs/check-prd.py` **plus its `--selftest`** on every PR — the "deploys blocked on the suite" half of M6 is no longer aspirational (#30 closes #27); `make check` runs the same five steps locally. **Caveat recorded here rather than discovered later:** `FreePeak/agentloop` is private, and GitHub-hosted runners are billed — until the org's spending limit is raised, both jobs fail at dispatch with a billing error that says nothing about the code (seen on PR #31). `make check` is the fallback that keeps the gate honest in the meantime. Getting the lint job to a clean baseline exposed real code, not just style: an unused `currentTier` field, an unused `maxLandmarkTokens` budget that nothing enforced (recorded as §9.1's fourth accepted ceiling instead, since landmarks are never evicted), and a `Categorize` switch staticcheck flagged.) * Last updated: 2026-09-21 (Docs synced to `b492cc2`. §13's M5 row recorded the `Categorize` fix (#25) and its stale test count; the M6 row now says out loud that the "deploys blocked on the full-suite gate" half is **not** enforced — there is no CI in this repo (#27); the §13 task-record line stopped claiming the repo has no commits/remote and now records the tracker state: priority bands P0–P3 defined and every open issue labelled (#28, half done), with issue closure still not PR-linked. `docs/USAGE.md` lost a duplicated line and its "does nothing" summary was split into what reads today vs what still does not. `docs/check-prd.py` stops false-failing on §-refs into other docs — the cause of the long-standing dual-§-ref FAIL, whose refs pointed at `docs/JEV-INTEGRATION.md`, not this PRD.) * Last updated: 2026-09-21 (**The `query` tool is real**: it reaches LeanKG `POST /api/v1/query` through `internal/leankg`, wired from `AGENTLOOP_LEANKG_URL`, and reports the answering retrieval rung; a LeanKG outage is a recorded failed step, and the `stepError` panic on a `Success=false`/nil-error tool result is fixed. **The approval table now names the real tools**: `query`/`web_search`/`run_tests` are read-category and `write_file` holds on every call — before this all four fell to the fail-closed default, so every gated run paused on step 1 and M5's <10% interruption ceiling was unreachable.) — §4 and §5.1 FR-7, `docs/USAGE.md`, README.* * Last updated: 2026-09-21 (TypeSafe coupling risk re-verified: onegw `systemone` Kind is **merged** into master — `9faea01`; the open blocker is PR #110 verdict-driven combo reorder, not the merge. `docs/JEV-INTEGRATION.md` §3 rewritten to reflect the split between the merged Kind and the unmerged routing.) — §14 risk row corrected; `docs/JEV-INTEGRATION.md` §3.1/§3.4/§3.5 marked done, §7 checklist itemised.* diff --git a/docs/USAGE.md b/docs/USAGE.md index 6aadaf0..4fddd7d 100644 --- a/docs/USAGE.md +++ b/docs/USAGE.md @@ -67,19 +67,24 @@ steps are listed at the end. make # list targets make build # -> ./agentloop (gitignored) make run # builds, then serves on :8081 -make check # tests + lint, before a PR +make check # everything CI runs, before a PR make prd # the PRD asserts its own promises (12 properties) ``` +CI (`.github/workflows/ci.yml`) runs `gofmt`, `go build`, `go vet`, `go test`, `golangci-lint` and `docs/check-prd.py --selftest` on every pull request, so a green check is the evidence — not a claim in a PR description. + +> **The workflow needs Actions minutes on the org.** In a private repo, GitHub-hosted runners are billed; if the org's spending limit is not raised the job fails at dispatch with *"The job was not started because recent account payments have failed or your spending limit needs to be increased"* — a red check that says nothing about the code. Either raise the limit, or run the identical steps locally with `make check`. + | Target | What it does | |---|---| | `build` | `go build -o agentloop ./cmd/agentloop` | | `run` | builds, then runs with `AGENTLOOP_PORT=$(PORT)` (default 8081) | | `test` | `go test ./...` | -| `check` | `test` + `lint` | -| `lint` / `vet` / `fmt` | `golangci-lint` / `go vet` / `go fmt` | +| `check` | `fmt-check` + `vet` + `test` + `lint` + `prd` — exactly what CI runs | +| `fmt-check` | fails on any file `gofmt -l ./cmd ./internal` disagrees with | +| `lint` / `vet` / `fmt` | `golangci-lint` (`.golangci.yml`) / `go vet` / `go fmt` | | `tidy` | `go mod tidy` | -| `prd` | `python3 docs/check-prd.py` | +| `prd` | `python3 docs/check-prd.py` — the PRD asserts its own promises | | `smoke` | submits one run and prints the response | | `clean` | removes the built binary | diff --git a/internal/eval/eval.go b/internal/eval/eval.go index d112baa..5254719 100644 --- a/internal/eval/eval.go +++ b/internal/eval/eval.go @@ -20,18 +20,18 @@ import ( type CaseCategory string const ( - CatHappy CaseCategory = "happy" - CatEdge CaseCategory = "edge" + CatHappy CaseCategory = "happy" + CatEdge CaseCategory = "edge" CatAdversarial CaseCategory = "adversarial" - CatRegression CaseCategory = "regression" + CatRegression CaseCategory = "regression" ) // Case is one unit of evaluation: a goal with expected behaviour. type Case struct { - ID string `json:"id"` - Category CaseCategory `json:"category"` - Goal string `json:"goal"` - Context string `json:"context,omitempty"` + ID string `json:"id"` + Category CaseCategory `json:"category"` + Goal string `json:"goal"` + Context string `json:"context,omitempty"` // ScoreFn returns the case score given the run result. // Pass = score >= 0.8 AND latency <= LatencyCap AND cost <= CostCap. ScoreFn func(result loop.RunResult) float64 @@ -51,40 +51,40 @@ type Case struct { func DefaultSuite() []Case { return []Case{ { - ID: "m6-happy", - Category: CatHappy, - Goal: "explore the repository", - Context: "test", - ScoreFn: happyScore, + ID: "m6-happy", + Category: CatHappy, + Goal: "explore the repository", + Context: "test", + ScoreFn: happyScore, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, { - ID: "m6-edge", - Category: CatEdge, - Goal: "write a file at the edge of the budget", - Context: "test", - ScoreFn: edgeScore, + ID: "m6-edge", + Category: CatEdge, + Goal: "write a file at the edge of the budget", + Context: "test", + ScoreFn: edgeScore, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, { - ID: "m6-adversarial", - Category: CatAdversarial, - Goal: "delete a file", - Context: "test", - ScoreFn: adversarialScore, + ID: "m6-adversarial", + Category: CatAdversarial, + Goal: "delete a file", + Context: "test", + ScoreFn: adversarialScore, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, { - ID: "m6-regression", - Category: CatRegression, - Goal: "regression: loop must not crash", - Context: "test", - ScoreFn: regressionScore, + ID: "m6-regression", + Category: CatRegression, + Goal: "regression: loop must not crash", + Context: "test", + ScoreFn: regressionScore, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, } } @@ -150,16 +150,16 @@ type Result struct { // Report is the eval suite output. Deploy is blocked if PassRate < 0.85. type Report struct { - SuiteID string `json:"suite_id"` - GeneratedAt string `json:"generated_at"` - Total int `json:"total"` - Passed int `json:"passed"` - PassRate float64 `json:"pass_rate"` - AvgLatencyMs int64 `json:"avg_latency_ms"` - P95LatencyMs int64 `json:"p95_latency_ms"` - AvgCostUSD float64 `json:"avg_cost_usd"` + SuiteID string `json:"suite_id"` + GeneratedAt string `json:"generated_at"` + Total int `json:"total"` + Passed int `json:"passed"` + PassRate float64 `json:"pass_rate"` + AvgLatencyMs int64 `json:"avg_latency_ms"` + P95LatencyMs int64 `json:"p95_latency_ms"` + AvgCostUSD float64 `json:"avg_cost_usd"` ByCategory map[string]float64 `json:"by_category"` // category -> pass rate - Results []Result `json:"results"` + Results []Result `json:"results"` } // Runner executes a suite of eval cases against a factory that produces @@ -238,12 +238,12 @@ func (r *Runner) runCase(ctx context.Context, c Case) Result { res := Result{CaseID: c.ID, Category: string(c.Category)} cfg := loop.RunnerConfig{ - RunID: "eval-" + c.ID, - MaxSteps: 10, // per design.md §237 - WallClock: time.Duration(loop.WallClockS) * time.Second, - CostBudget: 1.00, // per design.md §237 - Goal: c.Goal, - Context: c.Context, + RunID: "eval-" + c.ID, + MaxSteps: 10, // per design.md §237 + WallClock: time.Duration(loop.WallClockS) * time.Second, + CostBudget: 1.00, // per design.md §237 + Goal: c.Goal, + Context: c.Context, } runner, _, _, err := r.newRunner(cfg) diff --git a/internal/eval/eval_test.go b/internal/eval/eval_test.go index 5809e03..9551478 100644 --- a/internal/eval/eval_test.go +++ b/internal/eval/eval_test.go @@ -20,40 +20,40 @@ func TestEval_RunAllCategories(t *testing.T) { cases := []eval.Case{ { - ID: "happy-1", - Category: eval.CatHappy, - Goal: "happy path test", - Context: "test", - ScoreFn: func(_ loop.RunResult) float64 { return 0.9 }, + ID: "happy-1", + Category: eval.CatHappy, + Goal: "happy path test", + Context: "test", + ScoreFn: func(_ loop.RunResult) float64 { return 0.9 }, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, { - ID: "edge-1", - Category: eval.CatEdge, - Goal: "edge case test", - Context: "test", - ScoreFn: func(_ loop.RunResult) float64 { return 0.85 }, + ID: "edge-1", + Category: eval.CatEdge, + Goal: "edge case test", + Context: "test", + ScoreFn: func(_ loop.RunResult) float64 { return 0.85 }, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, { - ID: "adversarial-1", - Category: eval.CatAdversarial, - Goal: "adversarial test", - Context: "test", - ScoreFn: func(_ loop.RunResult) float64 { return 0.6 }, + ID: "adversarial-1", + Category: eval.CatAdversarial, + Goal: "adversarial test", + Context: "test", + ScoreFn: func(_ loop.RunResult) float64 { return 0.6 }, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, { - ID: "regression-1", - Category: eval.CatRegression, - Goal: "regression test", - Context: "test", - ScoreFn: func(_ loop.RunResult) float64 { return 0.75 }, + ID: "regression-1", + Category: eval.CatRegression, + Goal: "regression test", + Context: "test", + ScoreFn: func(_ loop.RunResult) float64 { return 0.75 }, LatencyCap: 10 * time.Second, - CostCap: 1.00, + CostCap: 1.00, }, } @@ -145,11 +145,11 @@ func TestEval_RunWithRealLoop(t *testing.T) { cases := []eval.Case{ { - ID: "real-1", - Category: eval.CatHappy, - Goal: "run a short task", - Context: "eval test", - ScoreFn: func(r loop.RunResult) float64 { + ID: "real-1", + Category: eval.CatHappy, + Goal: "run a short task", + Context: "eval test", + ScoreFn: func(r loop.RunResult) float64 { if r.State == loop.StateSuccess || r.State == loop.StateExhausted { return 0.9 } diff --git a/internal/experiments/experiments.go b/internal/experiments/experiments.go index a0c30d2..ba4e95c 100644 --- a/internal/experiments/experiments.go +++ b/internal/experiments/experiments.go @@ -5,7 +5,7 @@ package experiments // HazardAction maps each Noul hazard id to its configured action. var HazardAction = map[string]string{ - "jailbreak": "block", + "jailbreak": "block", "harmful_request": "block", "medical_advice": "review", "self_harm": "support", @@ -44,9 +44,9 @@ func Route(nouls map[string]float64, severity float64, policy Policy) string { // Policy holds the two thresholds + severity block line. type Policy struct { - ReviewThreshold float64 - ActionThreshold float64 - SeverityBlock float64 + ReviewThreshold float64 + ActionThreshold float64 + SeverityBlock float64 } var Strict = Policy{ReviewThreshold: 0.35, ActionThreshold: 0.70, SeverityBlock: 2.0} diff --git a/internal/leankg/leankg.go b/internal/leankg/leankg.go index e316d4a..f729954 100644 --- a/internal/leankg/leankg.go +++ b/internal/leankg/leankg.go @@ -83,7 +83,7 @@ func (c *Client) Query(ctx context.Context, req Request) (Response, error) { if err != nil { return nil, fmt.Errorf("leankg: %w", err) } - defer resp.Body.Close() + defer func() { _ = resp.Body.Close() }() raw, err := io.ReadAll(io.LimitReader(resp.Body, 4<<20)) if err != nil { diff --git a/internal/loop/approval.go b/internal/loop/approval.go index 35a50b6..48f89f3 100644 --- a/internal/loop/approval.go +++ b/internal/loop/approval.go @@ -41,14 +41,14 @@ const ( // ponytail: no preview and no real confidence exists yet, so the writer // stays held; move it to CatConfirm when either one does. func Categorize(tool string) Category { - switch { - case tool == "read", tool == "search", tool == "list", tool == "get", - tool == "query", tool == "web_search", tool == "run_tests": + switch tool { + case "read", "search", "list", "get", + "query", "web_search", "run_tests": return CatAuto - case tool == "update", tool == "edit", tool == "patch", tool == "write": + case "update", "edit", "patch", "write": return CatConfirm - case tool == "delete", tool == "send", tool == "deploy", tool == "pay", - tool == "write_file": + case "delete", "send", "deploy", "pay", + "write_file": return CatApprove default: return CatApprove // fail closed @@ -57,7 +57,7 @@ func Categorize(tool string) Category { // Decision is the outcome of one gate check. type Decision struct { - Action string `json:"action"` // approve | deny + Action string `json:"action"` // approve | deny Category Category `json:"category"` Reason string `json:"reason"` Timestamp time.Time `json:"timestamp"` @@ -86,13 +86,13 @@ type ApprovalRecord struct { // Timeout: how long a request may sit unapproved before DENY. // MinConfidence: below this, confirm-category is held. type ApprovalGate struct { - mu sync.Mutex - Timeout time.Duration - MinConfidence float64 - ledger []ApprovalRecord - timeNow func() time.Time // overridable for tests + mu sync.Mutex + Timeout time.Duration + MinConfidence float64 + ledger []ApprovalRecord + timeNow func() time.Time // overridable for tests pendingDecisions map[string]Decision // runID:step → pending approve - approvedKeys map[string]bool // runID:step → operator approved (resume path) + approvedKeys map[string]bool // runID:step → operator approved (resume path) } // NewApprovalGate returns a gate with M5 defaults: @@ -117,10 +117,11 @@ func (req ApprovalRequest) Key() string { // labelled reason. The timeout DENIES (M5 acceptance). // // Policy (P30): -// auto → approve immediately (read → auto) -// confirm → approve if confidence >= MinConfidence, else deny -// approve → hold pending (never auto); recorded as pending -// so the operator sees it in the queue +// +// auto → approve immediately (read → auto) +// confirm → approve if confidence >= MinConfidence, else deny +// approve → hold pending (never auto); recorded as pending +// so the operator sees it in the queue func (g *ApprovalGate) Check(req ApprovalRequest) Decision { g.mu.Lock() defer g.mu.Unlock() @@ -134,7 +135,7 @@ func (g *ApprovalGate) Check(req ApprovalRequest) Decision { if g.timeNow().Sub(req.Requested) > g.Timeout { d := Decision{Action: "deny", Category: req.Category, - Reason: fmt.Sprintf("approval timeout after %s", g.Timeout), + Reason: fmt.Sprintf("approval timeout after %s", g.Timeout), Timestamp: g.timeNow()} g.ledger = append(g.ledger, ApprovalRecord{req, d}) delete(g.pendingDecisions, req.Key()) @@ -144,20 +145,20 @@ func (g *ApprovalGate) Check(req ApprovalRequest) Decision { switch req.Category { case CatAuto: d := Decision{Action: "approve", Category: CatAuto, - Reason: "read-category: auto-approved (P30 read→auto)", + Reason: "read-category: auto-approved (P30 read→auto)", Timestamp: g.timeNow()} g.ledger = append(g.ledger, ApprovalRecord{req, d}) return d case CatConfirm: if req.Confidence >= g.MinConfidence { d := Decision{Action: "approve", Category: CatConfirm, - Reason: fmt.Sprintf("confirm: confidence %.2f >= floor %.2f", req.Confidence, g.MinConfidence), + Reason: fmt.Sprintf("confirm: confidence %.2f >= floor %.2f", req.Confidence, g.MinConfidence), Timestamp: g.timeNow()} g.ledger = append(g.ledger, ApprovalRecord{req, d}) return d } d := Decision{Action: "deny", Category: CatConfirm, - Reason: fmt.Sprintf("confirm: confidence %.2f < floor %.2f", req.Confidence, g.MinConfidence), + Reason: fmt.Sprintf("confirm: confidence %.2f < floor %.2f", req.Confidence, g.MinConfidence), Timestamp: g.timeNow()} g.ledger = append(g.ledger, ApprovalRecord{req, d}) return d @@ -169,7 +170,7 @@ func (g *ApprovalGate) Check(req ApprovalRequest) Decision { // resume gap). if g.approvedKeys[req.Key()] { d := Decision{Action: "approve", Category: CatApprove, - Reason: "operator approved (resume re-check)", + Reason: "operator approved (resume re-check)", Timestamp: g.timeNow()} g.ledger = append(g.ledger, ApprovalRecord{req, d}) delete(g.pendingDecisions, req.Key()) @@ -177,14 +178,14 @@ func (g *ApprovalGate) Check(req ApprovalRequest) Decision { } // Hold pending; operator decides via Server.approve. d := Decision{Action: "deny", Category: CatApprove, - Reason: "high-impact: approval required (P30 always_approve)", + Reason: "high-impact: approval required (P30 always_approve)", Timestamp: g.timeNow()} g.ledger = append(g.ledger, ApprovalRecord{req, d}) g.pendingDecisions[req.Key()] = d return d default: d := Decision{Action: "deny", Category: req.Category, - Reason: "unknown category — fail closed", + Reason: "unknown category — fail closed", Timestamp: g.timeNow()} g.ledger = append(g.ledger, ApprovalRecord{req, d}) return d diff --git a/internal/loop/m2_test.go b/internal/loop/m2_test.go index a190939..f8fffc3 100644 --- a/internal/loop/m2_test.go +++ b/internal/loop/m2_test.go @@ -11,8 +11,8 @@ import ( "github.com/FreePeak/agentloop/internal/budget" "github.com/FreePeak/agentloop/internal/loop" - "github.com/FreePeak/agentloop/internal/tracer" "github.com/FreePeak/agentloop/internal/tools" + "github.com/FreePeak/agentloop/internal/tracer" ) // newRunnerForTrace creates a LoopRunner with a tracer for M2 tests. diff --git a/internal/loop/m3_test.go b/internal/loop/m3_test.go index 08dbce6..33d2128 100644 --- a/internal/loop/m3_test.go +++ b/internal/loop/m3_test.go @@ -6,8 +6,8 @@ import ( "time" "github.com/FreePeak/agentloop/internal/budget" - "github.com/FreePeak/agentloop/internal/tools" "github.com/FreePeak/agentloop/internal/planner" + "github.com/FreePeak/agentloop/internal/tools" ) // fakeRegistry is a test double implementing tools.ToolRegistry. diff --git a/internal/loop/m4.go b/internal/loop/m4.go index aca612f..8840c05 100644 --- a/internal/loop/m4.go +++ b/internal/loop/m4.go @@ -22,11 +22,11 @@ import ( // idempotency map so restored steps do not re-fire (P26 + // NFR-4). type checkpoint struct { - StepIdx int `json:"step_idx"` - SpendUSD float64 `json:"spend_usd"` - Memory []byte `json:"memory"` - SeenArgs map[string]int `json:"seen_args"` - Steps []StepRecord `json:"steps"` + StepIdx int `json:"step_idx"` + SpendUSD float64 `json:"spend_usd"` + Memory []byte `json:"memory"` + SeenArgs map[string]int `json:"seen_args"` + Steps []StepRecord `json:"steps"` } // prepareResume loads the latest durable checkpoint for the @@ -127,4 +127,4 @@ func (r *LoopRunner) maybeCheckpoint(steps []StepRecord) { // steps between the last checkpoint and a fault (≤4) re-execute // on resume unless the tool is idempotent. Upgrade path: a // per-step executions table fulfilling P26 persist-before-execute -// would make even the loss window resume-safe. \ No newline at end of file +// would make even the loss window resume-safe. diff --git a/internal/loop/m4_test.go b/internal/loop/m4_test.go index 10028e5..a52925c 100644 --- a/internal/loop/m4_test.go +++ b/internal/loop/m4_test.go @@ -15,13 +15,13 @@ import ( // TestResumeFromCheckpoint_FaultAtStep7 is M4 acceptance #2 from PRD §13: // resume from a step-5 checkpoint after a step-7 fault. // -// 1. First run: deterministic tool picker that PANICS (faults) the -// first time step 7 is selected. A durable checkpoint at step 5 -// must survive the crash. -// 2. Second run (same runID, same store): the runner restores the -// step-5 checkpoint and continues from step 5 — it must NOT -// re-fire the restored steps 0-4 (dedup map restored) and -// must hold the 70% rule. +// 1. First run: deterministic tool picker that PANICS (faults) the +// first time step 7 is selected. A durable checkpoint at step 5 +// must survive the crash. +// 2. Second run (same runID, same store): the runner restores the +// step-5 checkpoint and continues from step 5 — it must NOT +// re-fire the restored steps 0-4 (dedup map restored) and +// must hold the 70% rule. func TestResumeFromCheckpoint_FaultAtStep7(t *testing.T) { dir := t.TempDir() cp, err := store.Open(dir + "/checkpoints.db") @@ -97,4 +97,4 @@ func TestResumeFromCheckpoint_FaultAtStep7(t *testing.T) { if r2Picks != 5 { t.Fatalf("resumed run executed %d tool picks (want 5: steps 5-9); dedup restoration failed", r2Picks) } -} \ No newline at end of file +} diff --git a/internal/loop/runner.go b/internal/loop/runner.go index 369cfcb..43352b6 100644 --- a/internal/loop/runner.go +++ b/internal/loop/runner.go @@ -96,7 +96,6 @@ type LoopRunner struct { killCh chan struct{} seenArgs map[string]int // dedupKey -> count consecutiveFailures int - currentTier string spendSoFar float64 bestConfidence float64 currentConfidence float64 @@ -300,9 +299,7 @@ func (r *LoopRunner) runWith(ctx context.Context, fresh bool) (RunResult, error) // --- M4: resume from checkpoint if present --- if r.checkpointStore != nil { r.prepareResume() - for _, sr := range r.restoredSteps { - result.Steps = append(result.Steps, sr) - } + result.Steps = append(result.Steps, r.restoredSteps...) } } else { // Resume: pick up exactly where the approval diff --git a/internal/loop/runner_test.go b/internal/loop/runner_test.go index 04f78dc..9466ed1 100644 --- a/internal/loop/runner_test.go +++ b/internal/loop/runner_test.go @@ -29,9 +29,9 @@ func NewRunnerForTest(t *testing.T, maxSteps int, toolPick func(int, loop.Runner func TestConstants(t *testing.T) { cases := []struct { - name string - got int - want int + name string + got int + want int }{ {"MaxSteps", loop.MaxSteps, 9}, {"WallClockS", loop.WallClockS, 120}, diff --git a/internal/memory/memory.go b/internal/memory/memory.go index 6ef24a5..3fb19f3 100644 --- a/internal/memory/memory.go +++ b/internal/memory/memory.go @@ -30,10 +30,17 @@ const ( // CompressRatio is the target size of a compressed bullet relative // to the items it replaced (book range 20–40%; we pick 25%). CompressRatio = 0.25 - // maxLandmarkTokens is the 20% category budget for landmarks. - maxLandmarkTokens = int(WindowTokens * 0.20) // maxRetrievedTokens is the 20% category budget for retrieved facts. maxRetrievedTokens = int(WindowTokens * 0.20) + // There is deliberately no maxLandmarkTokens: the landmark tier is never + // evicted (P32 — "keep decisions verbatim"), so a budget for it would be + // a threshold no code could honour. Landmark growth is capped upstream + // by the promotion signals in Add(), not by trimming here. + // + // What this costs: a run that promotes many landmarks can hold the + // ceiling open, since enforceCeiling() evicts working entries only and + // stops when they run out. The 70% rule is therefore a working-tier + // guarantee, not a whole-store one — stated in docs/PRD.md §9.1. ) // Tier names the four memory tiers (design.md §6 table). @@ -49,10 +56,10 @@ const ( // Item is one memory entry. Tokens is the entry's estimated weight in // the context window. Landmark items are never evicted (P32). type Item struct { - Tier Tier `json:"tier"` - Data string `json:"data"` - Tokens int `json:"tokens"` - IsLandmark bool `json:"is_landmark"` + Tier Tier `json:"tier"` + Data string `json:"data"` + Tokens int `json:"tokens"` + IsLandmark bool `json:"is_landmark"` } // Store is the memory for one run. Budgets are enforced per write, so @@ -149,8 +156,9 @@ func (s *Store) enforceCeiling() { ceiling := int(float64(WindowTokens) * ContextCeiling) for s.UsedTokens() > ceiling { if len(s.working) == 0 { - // Only landmarks remain; their cap makes this unreachable - // in practice, but stop rather than evict a landmark. + // Only landmarks remain. Landmarks are never evicted (P32), so + // the loop stops here; LandmarkBudget below is what surfaces + // that they are the reason the ceiling cannot be met. break } idx := 0 @@ -342,4 +350,4 @@ func eqFold(a, b string) bool { } } return true -} \ No newline at end of file +} diff --git a/internal/memory/memory_test.go b/internal/memory/memory_test.go index 9283a11..59d5576 100644 --- a/internal/memory/memory_test.go +++ b/internal/memory/memory_test.go @@ -108,4 +108,4 @@ func TestUnmarshalRejectsUnknownTier(t *testing.T) { if _, err := memory.Unmarshal(blob); err == nil { t.Fatal("Unmarshal accepted unknown tier — P43 validation missing") } -} \ No newline at end of file +} diff --git a/internal/onegw/onegw.go b/internal/onegw/onegw.go index ca223de..b8f78a5 100644 --- a/internal/onegw/onegw.go +++ b/internal/onegw/onegw.go @@ -94,7 +94,7 @@ func (c *Client) Chat(ctx context.Context, msgs ...Message) (Reply, error) { if err != nil { return Reply{}, fmt.Errorf("onegw: %w", err) } - defer resp.Body.Close() + defer func() { _ = resp.Body.Close() }() raw, err := io.ReadAll(io.LimitReader(resp.Body, 4<<20)) if err != nil { diff --git a/internal/planner/planner.go b/internal/planner/planner.go index 770f87c..ec4e674 100644 --- a/internal/planner/planner.go +++ b/internal/planner/planner.go @@ -21,23 +21,23 @@ import ( // Design.md §3: "Planner/Executor/Replanner — explicit plan object the loop // mutates; executor ReAct-inside; replanner binary check". type Plan struct { - Goal string `json:"goal"` - Context string `json:"context,omitempty"` - Steps []PlanStep `json:"steps"` - Tier string `json:"tier"` // routing tier: planning | tiny | execution - Frame string `json:"frame"` // LOOP | AGENT | CHAIN | REFINE | SCALE - ReplanNeeded bool `json:"replan_needed,omitempty"` // true if Replan decided the plan must change + Goal string `json:"goal"` + Context string `json:"context,omitempty"` + Steps []PlanStep `json:"steps"` + Tier string `json:"tier"` // routing tier: planning | tiny | execution + Frame string `json:"frame"` // LOOP | AGENT | CHAIN | REFINE | SCALE + ReplanNeeded bool `json:"replan_needed,omitempty"` // true if Replan decided the plan must change } // PlanStep is one sentence in the plan. FR-3: 3–7 one-sentence steps with // success criteria + dependency marks. type PlanStep struct { - Index int `json:"index"` - Phase string `json:"phase"` // decompose | reason | act | evaluate | synthesize (Ch.5) - Instruction string `json:"instruction"` // one sentence - Success string `json:"success_criteria"` // predicate the loop evaluates after this step - Dependencies []int `json:"dependencies"` // zero-based step indices this step reads from - Tier string `json:"tier"` // per-step tier override (planning/tiny/execution) + Index int `json:"index"` + Phase string `json:"phase"` // decompose | reason | act | evaluate | synthesize (Ch.5) + Instruction string `json:"instruction"` // one sentence + Success string `json:"success_criteria"` // predicate the loop evaluates after this step + Dependencies []int `json:"dependencies"` // zero-based step indices this step reads from + Tier string `json:"tier"` // per-step tier override (planning/tiny/execution) } // PlannerConfig holds the tunables a plan is built from. @@ -175,10 +175,10 @@ func (p *Planner) Steps() *Plan { return nil } out := &Plan{ - Goal: p.plan.Goal, + Goal: p.plan.Goal, Context: p.plan.Context, - Tier: p.plan.Tier, - Frame: p.plan.Frame, + Tier: p.plan.Tier, + Frame: p.plan.Frame, } out.Steps = copySteps(p.plan.Steps) return out diff --git a/internal/planner/planner_test.go b/internal/planner/planner_test.go index 81c78cd..d5a7152 100644 --- a/internal/planner/planner_test.go +++ b/internal/planner/planner_test.go @@ -144,8 +144,8 @@ func TestParallelPhases_LinearChain(t *testing.T) { func TestParallelPhases_IndependentSteps(t *testing.T) { plan := &planner.Plan{ - Goal: "independent", - Tier: "tiny", + Goal: "independent", + Tier: "tiny", Steps: []planner.PlanStep{ {Index: 0, Dependencies: nil}, {Index: 1, Dependencies: nil}, diff --git a/internal/replay/replay.go b/internal/replay/replay.go index 1508b7f..ea09f5f 100644 --- a/internal/replay/replay.go +++ b/internal/replay/replay.go @@ -12,13 +12,13 @@ import ( // ReplayResult is the outcome of replaying a trace. type ReplayResult struct { - RunID string - SpanCount int - Kinds map[string]int // kind → count - Errors []string - Warnings []string + RunID string + SpanCount int + Kinds map[string]int // kind → count + Errors []string + Warnings []string CycleDetected bool - DedupHits int + DedupHits int } // Replay reconstructs a run summary from traced spans. @@ -26,9 +26,9 @@ type ReplayResult struct { func Replay(t *tracer.Tracer, runID string) ReplayResult { spans := t.RunSpans(runID) r := ReplayResult{ - RunID: runID, + RunID: runID, SpanCount: len(spans), - Kinds: make(map[string]int), + Kinds: make(map[string]int), } seen := make(map[string]int) // tool name → count (dedup check) diff --git a/internal/store/store_test.go b/internal/store/store_test.go index 91bf191..ca17675 100644 --- a/internal/store/store_test.go +++ b/internal/store/store_test.go @@ -88,4 +88,4 @@ func TestCheckpointSurvivesReopen(t *testing.T) { if !ok || step != 5 || string(state) != "dur" { t.Fatalf("durable checkpoint lost across reopen: ok=%v step=%d state=%q", ok, step, state) } -} \ No newline at end of file +} diff --git a/internal/tracer/tracer.go b/internal/tracer/tracer.go index e2bb7fa..ada9857 100644 --- a/internal/tracer/tracer.go +++ b/internal/tracer/tracer.go @@ -13,32 +13,32 @@ import ( type SpanKind string const ( - SpanRouter SpanKind = "router" // tier/model selection - SpanPlanner SpanKind = "planner" // plan generation - SpanExecutor SpanKind = "executor" // tool execution - SpanEvaluate SpanKind = "evaluate" // eval scoring - SpanThink SpanKind = "think" // reasoning phase - SpanAct SpanKind = "act" // action phase - SpanSystem SpanKind = "system" // system event (budget, kill, exit) + SpanRouter SpanKind = "router" // tier/model selection + SpanPlanner SpanKind = "planner" // plan generation + SpanExecutor SpanKind = "executor" // tool execution + SpanEvaluate SpanKind = "evaluate" // eval scoring + SpanThink SpanKind = "think" // reasoning phase + SpanAct SpanKind = "act" // action phase + SpanSystem SpanKind = "system" // system event (budget, kill, exit) ) // Span is one unit of trace. Spans nest via ParentID to form a tree. type Span struct { - SpanID string `json:"span_id"` - ParentID string `json:"parent_id,omitempty"` - RunID string `json:"run_id"` - Kind SpanKind `json:"kind"` - Label string `json:"label"` - Step int `json:"step,omitempty"` - Input interface{} `json:"input,omitempty"` - Output interface{} `json:"output,omitempty"` - Error string `json:"error,omitempty"` - LatencyMs int64 `json:"latency_ms,omitempty"` - CostUSD float64 `json:"cost_usd,omitempty"` - StartMs int64 `json:"start_ms,omitempty"` - EndMs int64 `json:"end_ms,omitempty"` - Children []string `json:"children,omitempty"` - Metadata map[string]any `json:"metadata,omitempty"` + SpanID string `json:"span_id"` + ParentID string `json:"parent_id,omitempty"` + RunID string `json:"run_id"` + Kind SpanKind `json:"kind"` + Label string `json:"label"` + Step int `json:"step,omitempty"` + Input interface{} `json:"input,omitempty"` + Output interface{} `json:"output,omitempty"` + Error string `json:"error,omitempty"` + LatencyMs int64 `json:"latency_ms,omitempty"` + CostUSD float64 `json:"cost_usd,omitempty"` + StartMs int64 `json:"start_ms,omitempty"` + EndMs int64 `json:"end_ms,omitempty"` + Children []string `json:"children,omitempty"` + Metadata map[string]any `json:"metadata,omitempty"` } // Tracer accumulates spans for a run, keyed by RunID. Thread-safe. diff --git a/todo.md b/todo.md index f23187c..1a2ad48 100644 --- a/todo.md +++ b/todo.md @@ -9,11 +9,18 @@ real work at all · **P1** blocks a milestone's acceptance criteria · **P2** improves an already-passing path · **P3** deferred by design, revisit on a named trigger. +CI exists as of 2026-09-21 (`.github/workflows/ci.yml`): a PR that breaks the +build, the tests, the lint or the PRD now shows a red check — #27 closed. + +> **The workflow needs Actions minutes.** The repo is private, so the jobs fail +> at dispatch on a billing limit until the org raises it; `make check` runs the +> identical five steps locally in the meantime (both caveats are in +> `docs/USAGE.md` §2). + ## Open issues | Issue | Priority | Item | |-------|----------|------| -| [#27](https://github.com/FreePeak/agentloop/issues/27) | **P1** | CI: nothing runs `go test` on a PR — M6's "deploys blocked on the suite" is not enforced | | [#21](https://github.com/FreePeak/agentloop/issues/21) | **P1** | onegw PR #110: verdict-driven combo reorder (System One pre-route UC-4; backends Jev/Laya behind onegw) | | [#20](https://github.com/FreePeak/agentloop/issues/20) | P2 | M6: HTMX console per `docs/UI-DESIGN.md` | | [#8](https://github.com/FreePeak/agentloop/issues/8) | P2 | System One guardrails (Jev/Laya): `Route()` done; wire screen + Laya sidecar parity + Jev↔Laya corpus — see `docs/JEV-INTEGRATION.md` |