diff --git a/src/ui/format.ts b/src/ui/format.ts index 5c0430a..fac3ab8 100644 --- a/src/ui/format.ts +++ b/src/ui/format.ts @@ -11,6 +11,7 @@ const COL_AGENT = 25; const COL_STATUS = 10; const COL_DURATION = 10; const COL_SCORE = 7; +const COL_CATEGORY_LABEL = 20; const SEP_SUMMARY = 72; const SEP_SCORED = 102; const SEP_REPORT = 100; @@ -146,27 +147,122 @@ export function buildScoreInsight(score: ScoreResult): string | null { return parts.join(" | "); } +// --- Score breakdown --- +// Compact terminal summary of the HTML report's "Score breakdown" section: only +// the worst few interactions, each with its single weakest dimension. The full +// per-interaction table lives in the HTML report (renderCategoryBreakdown() in +// src/report-ui/src/scripts/render.ts), which decides what counts as imperfect. + +/** Total line width, including the 4-space indent. Fits an 80-column terminal / Actions log. */ +const BREAKDOWN_WIDTH = 80; +/** Interactions listed per category; the rest are summarized in a count. */ +const BREAKDOWN_MAX_ROWS = 3; +/** Width of the id column ("#12", "Necessity") and metric column ("Relevance 40", "12 unnecessary"). */ +const COL_BREAKDOWN_ID = 11; +const COL_BREAKDOWN_METRIC = 16; + +/** Scale a 0-1 audit dimension onto the 0-100 scale used everywhere else. Mirrors fmt01() in the HTML report. */ +function fmt01(value: number): string { + return (value * 100).toFixed(0); +} + +/** Relevance and necessity only contribute to the Agent category's score. */ +function categoryShowsRelevance(label: string): boolean { + return label === "Agent"; +} + +/** The lowest dimension of one audit that feeds this category's score. */ +function weakestDimension(audit: InteractionAudit, showRelevance: boolean): { label: string; value: number } { + const dims = [ + { label: "Success", value: audit.success }, + { label: "Speed", value: audit.speed }, + ...(showRelevance ? [{ label: "Relevance", value: audit.contextRelevance }] : []), + ]; + return dims.reduce((min, d) => (d.value < min.value ? d : min)); +} + +/** Collapse whitespace and truncate a judge rationale to fit the remaining line width. */ +function truncateRationale(rationale: string, max: number): string { + const text = rationale.replace(/\s+/g, " ").trim(); + return text.length > max ? text.slice(0, max - 1) + "…" : text; +} + /** - * Find the non-default audit with the lowest composite score. - * Returns a truncated rationale string, or null if no non-default audits exist. + * Render the worst interactions for one category as a short, borderless list. + * Returns an empty array when the category has nothing to explain (no audits + * with real rationales, or all of them perfect and no unnecessary calls). */ -function findWeakestAuditRationale(audits: InteractionAudit[]): string | null { - let weakest: InteractionAudit | null = null; - let weakestComposite = Infinity; - - for (const audit of audits) { - if (audit.rationale === "default") continue; - const composite = (audit.success + audit.speed + audit.weight + audit.contextRelevance) / 4; - if (composite < weakestComposite) { - weakestComposite = composite; - weakest = audit; - } +function renderBreakdownTable(label: string, cat: CategoryScore): string[] { + const nonDefaultAudits = cat.audits.filter((a) => a.rationale !== "default"); + if (nonDefaultAudits.length === 0) return []; + + const showRelevance = categoryShowsRelevance(label); + const imperfect = nonDefaultAudits + .map((audit) => ({ audit, weakest: weakestDimension(audit, showRelevance) })) + .filter((r) => r.weakest.value < 1) + .sort((a, b) => a.weakest.value - b.weakest.value); + const passingCount = nonDefaultAudits.length - imperfect.length; + + const necessity = + showRelevance && cat.necessity.rationale !== "default" && cat.necessity.unnecessaryIds.length > 0 + ? cat.necessity + : null; + if (imperfect.length === 0 && !necessity) return []; + + const rationaleWidth = BREAKDOWN_WIDTH - 4 - COL_BREAKDOWN_ID - COL_BREAKDOWN_METRIC; + const renderRow = (id: string, metric: string, rationale: string) => + ` ${id.padEnd(COL_BREAKDOWN_ID)}${metric.padEnd(COL_BREAKDOWN_METRIC)}${truncateRationale(rationale, rationaleWidth)}`; + + const lines: string[] = []; + + for (const { audit, weakest } of imperfect.slice(0, BREAKDOWN_MAX_ROWS)) { + lines.push(renderRow(`#${audit.id}`, `${weakest.label} ${fmt01(weakest.value)}`, audit.rationale)); + } + if (necessity) { + lines.push(renderRow("Necessity", `${necessity.unnecessaryIds.length} unnecessary`, necessity.rationale)); } - if (!weakest) return null; + const hiddenCount = imperfect.length - Math.min(imperfect.length, BREAKDOWN_MAX_ROWS); + const footer = [ + ...(hiddenCount > 0 ? [`${hiddenCount} more flagged`] : []), + ...(passingCount > 0 ? [`${passingCount} passing`] : []), + ]; + if (footer.length > 0) lines.push(` ${footer.join(", ")} not shown`); + + return lines; +} + +/** Width of the scorecard and section rules, including the 2-space indent. */ +const SCORECARD_WIDTH = 80; +/** Cells in a score bar; each cell is 5 points. */ +const SCORE_BAR_CELLS = 20; + +/** A 0-100 score as a fixed-width bar, e.g. "████████████░░░░░░░░". */ +function scoreBar(score: number): string { + const filled = Math.round((Math.max(0, Math.min(100, score)) / 100) * SCORE_BAR_CELLS); + return "█".repeat(filled) + "░".repeat(SCORE_BAR_CELLS - filled); +} + +/** One scorecard row: label, "NN / 100", bar, and an optional trailing note. */ +function scorecardRow(label: string, score: number, note?: string): string { + const row = ` ${label.padEnd(COL_CATEGORY_LABEL)}${String(score).padStart(3)} / 100 ${scoreBar(score)}`; + return note ? `${row} ${note}` : row; +} + +/** A section heading followed by a rule to the scorecard width, e.g. " Agent ─────". */ +function sectionHeading(title: string): string { + return ` ${title} ${"─".repeat(Math.max(0, SCORECARD_WIDTH - title.length - 3))}`; +} - const rationale = weakest.rationale.length > 100 ? weakest.rationale.slice(0, 97) + "..." : weakest.rationale; - return `#${weakest.id} ${rationale}`; +/** Verbose detail for one process-quality category: dimension roll-up plus the worst interactions. */ +function renderCategoryDetail(label: string, cat: CategoryScore): string[] { + const d = cat.dimensions; + return [ + sectionHeading(label), + ` Success ${d.success} · Speed ${d.speed} · Weight ${d.weight} · ` + + `Relevance ${d.relevance} · Necessity ${d.necessity}`, + ...renderBreakdownTable(label, cat), + ]; } export function renderFinalOutput(output: RunOutput, verbose: boolean, agentCount?: number): string { @@ -289,78 +385,47 @@ function renderScoredResult(result: ScoredRunResult, verbose: boolean): string { const { score } = result; const sep = "─".repeat(SEP_DETAIL); const lines: string[] = []; + const categories: Array<[label: string, cat: CategoryScore]> = [ + ["Environment", score.environment], + ["Service", score.service], + ["Agent", score.agent], + ]; lines.push(""); lines.push(` AXIS Report: ${result.scenarioName}`); lines.push(` ${sep}`); lines.push(""); - lines.push(` AXIS Result ${score.axisScore} / 100`); - if (score.judging) { - const judgeLabel = score.judging.model ? `${score.judging.agent}|${score.judging.model}` : score.judging.agent; - lines.push(` Agent used for judging: ${judgeLabel}`); - } - lines.push(""); - // Goal Achievement - lines.push(` Goal Achievement ${score.goalAchievement.score} / 100`); - for (const c of score.goalAchievement.criteria) { - const icon = c.score >= CRITERION_HIGH ? "\u2714" : c.score >= CRITERION_MEDIUM ? "\u25D0" : "\u2717"; - const label = c.check.length > 38 ? c.check.slice(0, 35) + "..." : c.check; - lines.push(` ${icon} ${label.padEnd(40)} (${c.score}/10)`); - } + // Scorecard: every score up front, aligned, so they read at a glance. + lines.push(scorecardRow("AXIS Result", score.axisScore)); lines.push(""); - - // Environment - lines.push(` Environment ${score.environment.score} / 100`); - lines.push( - ` ${score.environment.interactionCount} interactions | ` + `${score.environment.auditedCount} audited`, - ); - if (verbose) { - const d = score.environment.dimensions; - lines.push( - ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + - `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, - ); - const envRationale = findWeakestAuditRationale(score.environment.audits); - if (envRationale) lines.push(` ${envRationale}`); + lines.push(scorecardRow("Goal Achievement", score.goalAchievement.score)); + for (const [label, cat] of categories) { + lines.push(scorecardRow(label, cat.score, `${cat.interactionCount} interactions · ${cat.auditedCount} audited`)); } lines.push(""); - - // Service - lines.push(` Service ${score.service.score} / 100`); - lines.push(` ${score.service.interactionCount} interactions | ` + `${score.service.auditedCount} audited`); - if (verbose) { - const d = score.service.dimensions; - lines.push( - ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + - `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, - ); - const svcRationale = findWeakestAuditRationale(score.service.audits); - if (svcRationale) lines.push(` ${svcRationale}`); + lines.push(` Agent: ${result.agentName}`); + if (score.judging) { + const judgeLabel = score.judging.model ? `${score.judging.agent}|${score.judging.model}` : score.judging.agent; + lines.push(` Judged by: ${judgeLabel}`); } - lines.push(""); - // Agent - lines.push(` Agent ${score.agent.score} / 100`); - lines.push(` ${score.agent.interactionCount} interactions | ` + `${score.agent.auditedCount} audited`); - if (verbose) { - const d = score.agent.dimensions; - lines.push( - ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + - `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, - ); - const agentRationale = findWeakestAuditRationale(score.agent.audits); - if (agentRationale) lines.push(` ${agentRationale}`); + // Breakdowns underneath the scorecard. + if (score.goalAchievement.criteria.length > 0) { + lines.push(""); + lines.push(sectionHeading("Goal Achievement")); + for (const c of score.goalAchievement.criteria) { + const icon = c.score >= CRITERION_HIGH ? "\u2714" : c.score >= CRITERION_MEDIUM ? "\u25D0" : "\u2717"; + const label = c.check.length > 38 ? c.check.slice(0, 35) + "..." : c.check; + lines.push(` ${icon} ${label.padEnd(40)} (${c.score}/10)`); + if (verbose && c.rationale) lines.push(` ${c.rationale}`); + } } - lines.push(""); - lines.push(` Agent: ${result.agentName}`); - - // Verbose: show rationale per criterion if (verbose) { - lines.push(""); - for (const c of score.goalAchievement.criteria) { - lines.push(` [${c.check}] ${c.rationale}`); + for (const [label, cat] of categories) { + lines.push(""); + lines.push(...renderCategoryDetail(label, cat)); } } diff --git a/test/unit/ui/format.test.ts b/test/unit/ui/format.test.ts index e29c2c0..f4c56df 100644 --- a/test/unit/ui/format.test.ts +++ b/test/unit/ui/format.test.ts @@ -8,7 +8,7 @@ import { renderBaselineList, renderBaselineComparison, } from "../../../src/ui/format.js"; -import type { CategoryScore, ScoreResult } from "../../../src/types/scoring.js"; +import type { CategoryScore, InteractionAudit, ScoreResult } from "../../../src/types/scoring.js"; import type { RunOutput, RunResult } from "../../../src/types/output.js"; import type { AgentOutput } from "../../../src/types/agent.js"; import type { Baseline, BaselineComparison } from "../../../src/types/baseline.js"; @@ -389,3 +389,164 @@ describe("renderBaselineComparison", () => { expect(output).not.toContain("New (not in baseline)"); }); }); + +// --- Score breakdown table (verbose scored output) --- + +function makeAudit(overrides: Partial & { id: number }): InteractionAudit { + return { + categories: ["agent"], + success: 1, + speed: 1, + weight: 1, + contextRelevance: 1, + rationale: "Looked fine.", + ...overrides, + }; +} + +/** A scored result whose Agent category carries the given audits and necessity judgment. */ +function makeScoredResult(agentCat: Partial): RunResult { + const agent: CategoryScore = { + ...makeCategory(60), + auditedCount: 3, + necessity: { category: "agent", score: 1, unnecessaryIds: [], rationale: "default" }, + ...agentCat, + }; + return { + ...makeResult({ exitCode: 0 }), + score: makeScoreResult({ agent }), + } as RunResult; +} + +describe("renderScenarioDetail score breakdown", () => { + it("renders one row per imperfect audit with its weakest dimension", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [ + makeAudit({ id: 3, success: 0.5, speed: 0.8, rationale: "Install failed once." }), + makeAudit({ id: 7, speed: 0.6, rationale: "Slow directory scan." }), + ], + }), + ); + + expect(out).toMatch(/#3\s+Success 50\s+Install failed once\./); + expect(out).toMatch(/#7\s+Speed 60\s+Slow directory scan\./); + expect(out).not.toMatch(/#3.*Speed 80/); + }); + + it("orders rows worst-first and caps them at three", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [ + makeAudit({ id: 1, speed: 0.9 }), + makeAudit({ id: 2, success: 0.2 }), + makeAudit({ id: 3, speed: 0.5 }), + makeAudit({ id: 4, contextRelevance: 0.3 }), + makeAudit({ id: 5 }), + ], + }), + ); + + const ids = [...out.matchAll(/^\s+#(\d+)\s/gm)].map((m) => Number(m[1])); + expect(ids).toEqual([2, 4, 3]); + expect(out).toContain("1 more flagged, 1 passing not shown"); + }); + + it("counts passing audits instead of listing them", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [ + makeAudit({ id: 1, success: 0.5 }), + makeAudit({ id: 2 }), + makeAudit({ id: 3 }), + makeAudit({ id: 4, rationale: "default" }), + ], + }), + ); + expect(out).toContain("2 passing not shown"); + expect(out).not.toMatch(/^\s+#2\s/m); + }); + + it("renders a necessity row with the unnecessary-call count", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [makeAudit({ id: 1, success: 0.5 })], + necessity: { category: "agent", score: 0.7, unnecessaryIds: [4, 12], rationale: "Redundant greps." }, + }), + ); + expect(out).toMatch(/Necessity\s+2 unnecessary\s+Redundant greps\./); + }); + + it("omits the necessity row when nothing was flagged", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [makeAudit({ id: 1, success: 0.5 })], + necessity: { category: "agent", score: 1, unnecessaryIds: [], rationale: "All calls were needed." }, + }), + ); + expect(out).not.toMatch(/^\s+Necessity\s/m); + }); + + it("omits the table when every audit is a default placeholder", () => { + const out = renderScenarioDetail( + makeScoredResult({ audits: [makeAudit({ id: 1, success: 0.5, rationale: "default" })] }), + ); + expect(out).not.toMatch(/^\s+#1\s/m); + }); + + it("omits the table when all audits pass and necessity is default", () => { + const out = renderScenarioDetail(makeScoredResult({ audits: [makeAudit({ id: 1 })] })); + expect(out).not.toMatch(/^\s+#1\s/m); + }); + + it("ignores context relevance outside the Agent category", () => { + const envAudit = makeAudit({ id: 1, categories: ["environment"], contextRelevance: 0.2 }); + const env: CategoryScore = { ...makeCategory(60), auditedCount: 1, audits: [envAudit] }; + const out = renderScenarioDetail({ ...makeResult({ exitCode: 0 }), score: makeScoreResult({ env }) } as RunResult); + + // Relevance does not feed Env's score, so this audit counts as passing. + expect(out).not.toMatch(/#1\s+Relevance 20/); + }); + + it("keeps every breakdown line within 80 columns", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [makeAudit({ id: 1, success: 0.5, rationale: "A ".repeat(200) })], + necessity: { + category: "agent", + score: 0.1, + unnecessaryIds: Array.from({ length: 40 }, (_, i) => i + 100), + rationale: "Long rationale. ".repeat(40), + }, + }), + ); + + const rows = out.split("\n").filter((l) => /^\s+(#\d+|Necessity)\s/.test(l)); + expect(rows).toHaveLength(2); + for (const line of rows) { + expect([...line].length).toBeLessThanOrEqual(80); + } + }); +}); + +describe("renderScenarioDetail scorecard", () => { + const out = renderScenarioDetail({ + ...makeResult({ exitCode: 0 }), + score: makeScoreResult({ env: makeCategory(62), svc: makeCategory(95), agent: makeCategory(0) }), + } as RunResult); + + it("lists every score with a bar before any breakdown", () => { + const scorecard = out.slice(0, out.indexOf("─────\n", out.indexOf("Environment"))); + expect(scorecard).toMatch(/AXIS Result\s+80 \/ 100 {2}█{16}░{4}/); + expect(scorecard).toMatch(/Goal Achievement\s+80 \/ 100/); + expect(scorecard).toMatch(/Environment\s+62 \/ 100 {2}█{12}░{8}/); + expect(scorecard).toMatch(/Service\s+95 \/ 100 {2}█{19}░/); + expect(scorecard).toMatch(/Agent\s+0 \/ 100 {2}░{20}/); + }); + + it("puts the category breakdowns underneath the scorecard", () => { + const scorecardAt = out.indexOf("Agent 0 / 100"); + expect(out.indexOf(" Environment ──")).toBeGreaterThan(scorecardAt); + expect(out.indexOf(" Agent ──")).toBeGreaterThan(out.indexOf(" Service ──")); + }); +});