Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
143 changes: 143 additions & 0 deletions cmd/agentloop/guardrail_wiring_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,143 @@
package main

import (
"encoding/json"
"net/http"
"net/http/httptest"
"strings"
"testing"
"time"
)

// The PRD claims (§7.2, NFR-1b) that every tool result reaching the model is
// screened. Before this wiring, nothing in the service ever set a screen —
// only unit tests did — so the claim was false in production.
//
// This test drives a run through the HTTP API against a stub System One and
// requires the screen to actually fire and its verdict to reach the run.
func TestGuardrailScreenFiresInProduction(t *testing.T) {
var screens int
var sawState string
so := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.URL.Path != "/v1/systemone" {
t.Errorf("path = %q, want /v1/systemone", r.URL.Path)
}
var body struct {
State string `json:"state"`
}
_ = json.NewDecoder(r.Body).Decode(&body)
screens++
// The FIRST screen is the goal; the rest are tool results. Keep the
// goal's text so the assertion is about what was judged, not just
// that something was.
if screens == 1 {
sawState = body.State
}
w.Header().Set("Content-Type", "application/json")
// A jailbreak so the verdict is unambiguous.
_, _ = w.Write([]byte(`{"model":"jev-stub","answers":{
"jailbreak":0.95,"harmful_request":0.9,"medical_advice":0.1,
"self_harm":0.0,"severity":2.4}}`))
}))
defer so.Close()

t.Setenv("AGENTLOOP_GUARDRAIL_URL", so.URL)
t.Setenv("AGENTLOOP_GUARDRAIL_MODEL", "jev-stub")
t.Setenv("AGENTLOOP_LEANKG_OFF", "1")
t.Setenv("AGENTLOOP_XDEV_OFF", "1")

srv, _ := newTestServer(t)
defer srv.Close()

resp, err := http.Post(srv.URL+"/v1/runs", "application/json",
strings.NewReader(`{"goal":"screen me","max_steps":3}`))
if err != nil {
t.Fatalf("POST: %v", err)
}
var submit map[string]any
_ = json.NewDecoder(resp.Body).Decode(&submit)
resp.Body.Close()
runID, _ := submit["run_id"].(string)

var result RunResultResponse
for i := 0; i < 100; i++ {
time.Sleep(50 * time.Millisecond)
getResp, gerr := http.Get(srv.URL + "/v1/runs/" + runID)
if gerr != nil {
t.Fatalf("GET: %v", gerr)
}
derr := json.NewDecoder(getResp.Body).Decode(&result)
getResp.Body.Close()
// Any terminal state, including a goal-screen block, ends the wait.
// Waiting only for exhausted/success made this test hang on the
// very outcome it was asserting.
if derr == nil && result.State != "" && result.State != "thinking" {
break
}
}

if screens == 0 {
t.Fatal("the System One endpoint was never called — the screen is not wired into the service")
}
if sawState == "" {
t.Error("the screen was called with empty state")
}
// Exactly one screen is expected here: the goal. The block must land
// BEFORE any step runs, so no tool result is ever screened — an
// injection must not get a single tool call first. (A run that passes
// the goal screen then screens each tool result too; that path is
// covered by TestScreenSeesTheResultText in internal/loop.)
if screens != 1 {
t.Errorf("screens = %d, want exactly 1 (the goal) — a blocked goal must not run any step", screens)
}
// The goal is judged first, and the block lands there.
if sawState != "screen me" {
t.Errorf("the first screen judged %q, want the submitted goal", sawState)
}
if result.ExitReason != "guardrail_block" {
t.Errorf("exit_reason = %q, want guardrail_block (jailbreak > 0.70)", result.ExitReason)
}
// The block must land on the GOAL screen, before any step ran — an
// injection must not get a single tool call first.
if len(result.Steps) != 1 || result.Steps[0].Tool != "(goal screen)" {
t.Errorf("steps = %+v, want exactly the goal-screen record", result.Steps)
}
}

// With no screening endpoint configured, the run must proceed and SAY it was
// unscreened — not silently look clean.
func TestNoGuardrailEndpointIsRecordedNotAssumed(t *testing.T) {
t.Setenv("AGENTLOOP_GUARDRAIL_URL", "")
t.Setenv("AGENTLOOP_LEANKG_OFF", "1")
t.Setenv("AGENTLOOP_XDEV_OFF", "1")

srv, _ := newTestServer(t)
defer srv.Close()

resp, err := http.Post(srv.URL+"/v1/runs", "application/json",
strings.NewReader(`{"goal":"no screen configured","max_steps":2}`))
if err != nil {
t.Fatalf("POST: %v", err)
}
var submit map[string]any
_ = json.NewDecoder(resp.Body).Decode(&submit)
resp.Body.Close()
runID, _ := submit["run_id"].(string)

var result RunResultResponse
for i := 0; i < 100; i++ {
time.Sleep(50 * time.Millisecond)
getResp, gerr := http.Get(srv.URL + "/v1/runs/" + runID)
if gerr != nil {
t.Fatalf("GET: %v", gerr)
}
derr := json.NewDecoder(getResp.Body).Decode(&result)
getResp.Body.Close()
if derr == nil && result.State != "" && result.State != "thinking" {
break
}
}
if result.ExitReason == "guardrail_block" {
t.Error("a run with no screen configured was blocked by a screen")
}
}
108 changes: 107 additions & 1 deletion cmd/agentloop/main.go
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,8 @@ import (

"github.com/FreePeak/agentloop/internal/budget"
"github.com/FreePeak/agentloop/internal/eval"
"github.com/FreePeak/agentloop/internal/experiments"
"github.com/FreePeak/agentloop/internal/guardrail"
"github.com/FreePeak/agentloop/internal/leankg"
"github.com/FreePeak/agentloop/internal/loop"
"github.com/FreePeak/agentloop/internal/onegw"
Expand All @@ -40,6 +42,10 @@ type Server struct {
// sandboxDir is the workspace sandboxed turns run in; empty when no
// executor is configured.
sandboxDir string
// guardrail screens every message against the System One battery
// (§7.2). Nil means no screen is configured, which the run records as
// errors rather than reporting as a clean verdict.
guardrail *guardrail.Client
}

// NewServer creates a Server with the v1 tool set and empty stores.
Expand Down Expand Up @@ -68,6 +74,7 @@ func NewServer() *Server {
gates: make(map[string]*loop.ApprovalGate),
tools: reg,
sandboxDir: sandboxDir,
guardrail: guardrailFromEnv(),
model: onegw.New(
envOr("AGENTLOOP_ONEGW_URL", "http://127.0.0.1:8080"),
os.Getenv("AGENTLOOP_ONEGW_KEY"),
Expand Down Expand Up @@ -105,6 +112,54 @@ func envOr(key, def string) string {
//
// No LeanKG API key: the service is local and its REST surface is
// unauthenticated by design for a single-tenant deployment (PRD §7.4).
// guardrailFromEnv builds the System One screening client.
//
// AGENTLOOP_GUARDRAIL_URL endpoint root (onegw, or TypeSafe directly);
// unset means NO screen, and runs record that
// AGENTLOOP_GUARDRAIL_KEY bearer key (Jev); a local Laya needs none
// AGENTLOOP_GUARDRAIL_MODEL default jev-latest
//
// There is deliberately no default URL: pointing the screen at a gateway
// that does not serve /v1/systemone would turn every step into a failed
// screen, and a run that is unscreened must say so rather than pretend.
func guardrailFromEnv() *guardrail.Client {
url := os.Getenv("AGENTLOOP_GUARDRAIL_URL")
if url == "" {
return nil
}
return guardrail.New(url, os.Getenv("AGENTLOOP_GUARDRAIL_KEY"), os.Getenv("AGENTLOOP_GUARDRAIL_MODEL"))
}

// screenFunc adapts the client to the loop's screen hook. A nil client
// yields nil, so the runner's "no screen configured" path is what runs —
// not a screen that always passes.
func (s *Server) screenFunc() loop.ScreenFunc {
if s.guardrail == nil {
return nil
}
return func(text string) (map[string]float64, float64, error) {
v, err := s.guardrail.Screen(context.Background(), text)
if err != nil {
return nil, 0, err
}
return v.Nouls, v.Severity, nil
}
}

// screenGoal judges the submitted goal before any step runs. It returns
// the routed action ("pass"/"review"/"block") or an error when the screen
// could not run — which the caller records rather than treating as clean.
func (s *Server) screenGoal(goal string) (string, error) {
if s.guardrail == nil {
return "", nil
}
v, err := s.guardrail.Screen(context.Background(), goal)
if err != nil {
return "", err
}
return experiments.Route(v.Nouls, v.Severity, experiments.Strict), nil
}

// tiersFromEnv maps this loop's three routing tiers onto onegw combos.
//
// The gateway ships whatever combos an operator configured — in this
Expand Down Expand Up @@ -221,7 +276,58 @@ func (s *Server) submitRun(w http.ResponseWriter, r *http.Request) {
// must stay deterministic and offline.
Model: s.model,
}
runner := loop.NewRunnerWithPlannerAndGate(cfg, guard, s.tools, planner.NewPlanner(), gate)
runner := loop.NewRunnerWithPlannerAndGate(cfg, guard, s.tools, planner.NewPlanner(), gate).
WithGuardrailScreen(s.screenFunc())

// Screen the GOAL before the loop takes a step (§7.2: "the model
// context is the injection surface" — the goal is the first thing that
// enters it). A block here is the run never starting, which is the
// correct outcome for an injection; a review holds it for a human.
if verdict, err := s.screenGoal(body.Goal); err != nil {
runResult := loop.RunResult{
RunID: runID,
State: loop.StateExhausted,
ExitReason: loop.ExitGuardrailBlock,
ScreenErrors: []string{err.Error()},
}
runResult.Success = boolPtr(false)
s.mu.Lock()
s.runs[runID] = runResult
s.mu.Unlock()
// The run id must reach the caller even when the goal is blocked:
// a 201 with no body is a client that cannot look up what happened.
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(http.StatusCreated)
writeJSON(w, map[string]any{"run_id": runID, "state": runResult.State, "exit_reason": runResult.ExitReason})
return
} else if verdict != "" && verdict != "pass" {
state := loop.StateExhausted
if verdict == "review" {
state = loop.StatePausedApproval
}
runResult := loop.RunResult{
RunID: runID,
State: state,
ExitReason: loop.ExitGuardrailBlock,
Steps: []loop.StepRecord{{
StepID: 0,
Phase: "evaluate",
Tool: "(goal screen)",
Why: "guardrail: " + verdict,
Screens: []loop.ScreenResult{{
Hazard: "noul_battery", Action: verdict,
}},
}},
}
runResult.Success = boolPtr(false)
s.mu.Lock()
s.runs[runID] = runResult
s.mu.Unlock()
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(http.StatusCreated)
writeJSON(w, map[string]any{"run_id": runID, "state": runResult.State, "exit_reason": runResult.ExitReason})
return
}
s.mu.Lock()
s.runners[runID] = runner
s.gates[runID] = gate
Expand Down
1 change: 1 addition & 0 deletions cmd/agentloop/main_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -289,6 +289,7 @@ type RunResultResponse struct {
Args map[string]any `json:"args"`
Why string `json:"why"`
} `json:"steps"`
ScreenErrors []string `json:"screen_errors,omitempty"`
}

// TestM6_EvalSuite runs 4 eval cases across all categories
Expand Down
7 changes: 7 additions & 0 deletions docs/PRD.md
Original file line number Diff line number Diff line change
Expand Up @@ -285,6 +285,12 @@ Bearer-token auth on every route (mirror LeanKG's role model: admin/contributor/
### 7.2 Retrieved content (untrusted → model context)
Prompt injection is the tool path's default failure mode: sanitize fetched content, strip instruction-like patterns (blocklist + embedding-similarity check), and never let a tool result change the policy table or the budget. Tool output is data; the only thing that may act on it is the loop, under policy.

**TypeSafe scoring layer — now wired (2026-09-21).** This section described the layer as "live, not aspirational" while **nothing in the service ever turned it on**: `guardrailScreen` was set only by unit tests, so §7.2's guarantee ("every tool result that reaches the model — and the model's reply before it reaches the operator — is screened") was false in production. `internal/guardrail` now speaks `POST /v1/systemone` (the wire Jev and Laya both answer, and the one onegw forwards), the service screens the **goal before the first step** and every **tool result** before it can reach the model, and `Route()` decides pass/review/block.

Three honesty properties, each with a test: a screen that **cannot run** is recorded on the step and in `screen_errors` — not passed silently, and not fail-closed either, because blocking every run on a screening outage is its own failure; a blocked **goal** returns its `run_id` with `exit_reason=guardrail_block` rather than an empty 201; and empty text is refused before the round trip, because "clean" is a verdict the battery did not give.

**Still not done, and stated so:** the model's *reply* is not screened on the way out (the tool-result and goal directions are), and the measured ~740 ms/call is not yet counted against `BudgetGuard` — a per-step cost the PRD budgets but nothing meters.

**TypeSafe scoring layer (live, not aspirational).** Because the model context is the injection surface, every tool result that reaches the model — and the model's reply before it reaches the operator — is screened with the Noul/Score battery (§4.3 rule): four Noul questions (jailbreak, harmful_request, medical_advice, self_harm) plus one severity Score, run once per message, routed under the strict policy by default. The cookbook's three actions map onto the loop: `block` → halt the run, return the labelled partial, log the hazard; `review` → hold the step, surface it on the approval queue; `pass` → nothing. Two live experiments on 2026-09-19 (see GitHub issue #1) established:

- **Outputs screened with 5 replies:** 2 benign pass, 3 harmful correctly blocked (dosage advice at sev 2.05, jailbreak-compliance at sev 1.44, harmful lockpick at sev 2.12). Zero harmful output reached the operator.
Expand Down Expand Up @@ -742,6 +748,7 @@ Written the way an unfriendly reviewer would write it, then answered. Every find

**Read next.** §13.1 (scope → milestones), §17 (defaults), §18 (where to discount the source), §22 (this document's own weaknesses).

* Last updated: 2026-09-21 (**The guardrail screen is wired.** §7.2 and NFR-1b claimed every message was screened; nothing in the service ever set a screen — only tests did, so the containment claim was false in production. `internal/guardrail` speaks `POST /v1/systemone`, the goal is judged before the first step and every tool result before it reaches the model, a screen that cannot run is recorded rather than passed or fail-closed, and a blocked goal returns its `run_id` instead of an empty 201. The reply-out direction and the ~740ms/call budget entry are still open.)
* Last updated: 2026-09-21 (Tier routing reaches the wire. `ChatTier(ctx, tier, msgs…)` replaced the tier-less `Chat` on the model interface, `onegw.Client.WithTiers` maps tier→combo, and the three tiers are now planning/execution/synthesis instead of planning/execution/tiny — the old middle names were combos onegw does not ship. `Reply` records the combo asked for *and* the answering leg, so a fallback is visible. Verified against a stub gateway: three calls to `exec-combo`, one to `synth-combo`.)
* 2026-09-21 — (The loop **decides its own steps**. `internal/loop/reason.go` makes one model call per step with the previous step's verbatim result, and the answer (`{tool,args,why,done}`) is what runs — so the loop observes before it reasons, which is the half it was missing. `done` exits `success`/`goal_met`, a new exit reason for the goal predicate firing rather than a bound. A resume replays the approved decision instead of re-asking (a second call can choose a different tool, so the operator would have approved one action and a different one would run). `StepRecord` now carries its `Args` and `Why`. No model client still means the deterministic rotation, unchanged.)
* 2026-09-21 — (The loop can finally **write and verify**. `internal/xdev` speaks xdev's `rpc` JSONL protocol (ready-frame version gate, event-before-response interleaving, one turn at a time, child killed when the step's budget expires) and `run_tests`/`write_file` run as one xdev turn each, in a sandboxed workspace from `AGENTLOOP_XDEV_DIR`. With no sandbox the two report *no executor configured* and `written`/`ran` stay false — "no sandbox" can never read as "the tests passed". `nextToolDefault` now gives each tool the arguments it needs to be a real call, so a write has a target instead of failing closed on a missing path. `web_search` is the last stub.)
Expand Down
9 changes: 9 additions & 0 deletions docs/USAGE.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,7 @@ Read this before you plan work around it. As of this writing:
| The step chooser | **model-driven when a gateway is wired** (`AGENTLOOP_ONEGW_URL`): one call per step, given the previous result, answering `{tool,args,why,done}`. With no gateway it falls back to the deterministic rotation and each step says which happened in its `why` |
| The planner | **deterministic**, no model calls — it frames the phases; the *chooser* picks the action |
| M7 multi-agent (`internal/supervisor`) | **gated shut** by design — refused unless one of [PRD §10](PRD.md#10-multi-agent-stance)'s four conditions is met |
| Guardrail screening (`internal/guardrail`) | **live when configured** — the goal is judged before the first step and each tool result before it reaches the model. Unset URL means no screen, recorded in `screen_errors` rather than assumed clean |

**What this means:** a run today exercises the real loop, budget, gate, and
observability machinery end to end, and it **reads**: `query` returns real hits
Expand Down Expand Up @@ -106,6 +107,9 @@ Deployment facts, not compiled defaults:
| `AGENTLOOP_XDEV_BIN` | `xdev` | the sandbox binary; agentloop speaks its `rpc` JSONL protocol |
| `AGENTLOOP_XDEV_DIR` | a fresh temp dir | the workspace `write_file`/`run_tests` turns run in |
| `AGENTLOOP_XDEV_OFF` | *(unset)* | any value disables the sandbox; those two tools then report no executor |
| `AGENTLOOP_GUARDRAIL_URL` | *(unset — **no screen**) | System One endpoint (onegw or TypeSafe). Unset means unscreened, and runs say so in `screen_errors` |
| `AGENTLOOP_GUARDRAIL_KEY` | *(empty)* | bearer key for Jev; a local Laya needs none |
| `AGENTLOOP_GUARDRAIL_MODEL` | `jev-latest` | backend alias |
| `AGENTLOOP_ONEGW_URL` | `http://127.0.0.1:8080` | gateway; when reachable, the model **chooses each step** |
| `AGENTLOOP_ONEGW_COMBO` | `dev` | default combo — the wire `model` when no tier-specific one is set |
| `AGENTLOOP_ONEGW_COMBO_PLANNING` | *(falls back to `COMBO`)* | combo for plan/replan steps |
Expand Down Expand Up @@ -331,6 +335,11 @@ Stated plainly, so nobody discovers it the hard way:
phases and instructions come from a rule table (`planStepCount` on the goal's
word count), so the loop decides **what to do next** but not **how to break
the goal up**. In practice the chooser carries the run, and the plan is a hint.
- **The guardrail screens in two of three directions.** The goal and each
tool result are judged (§7.2). The model's **reply on the way out** is not,
and the ~740 ms/call the PRD budgets is not yet metered by `BudgetGuard`. With
`AGENTLOOP_GUARDRAIL_URL` unset there is no screen at all, and the run records
that in `screen_errors` rather than looking clean.
- **Tier routing reaches the wire, but the 40–70% number is unproven.** The
loop sends a tier per call (`planning`/`execution`/`synthesis`) and each maps
to a combo; unmapped tiers fall through to `AGENTLOOP_ONEGW_COMBO`, so a
Expand Down
Loading
Loading