From 12f54b9cbca3acde17c59283d2249a7c419590ca Mon Sep 17 00:00:00 2001 From: Kunle Oshiyoye Date: Wed, 2 Sep 2026 14:52:28 -0400 Subject: [PATCH 1/2] feat: Add results breakdown to Actions output --- src/ui/format.ts | 194 ++++++++++++++++++++++++++---------- test/unit/ui/format.test.ts | 148 ++++++++++++++++++++++++++- 2 files changed, 287 insertions(+), 55 deletions(-) diff --git a/src/ui/format.ts b/src/ui/format.ts index 5c0430a..b228e81 100644 --- a/src/ui/format.ts +++ b/src/ui/format.ts @@ -11,6 +11,7 @@ const COL_AGENT = 25; const COL_STATUS = 10; const COL_DURATION = 10; const COL_SCORE = 7; +const COL_CATEGORY_LABEL = 20; const SEP_SUMMARY = 72; const SEP_SCORED = 102; const SEP_REPORT = 100; @@ -146,27 +147,147 @@ export function buildScoreInsight(score: ScoreResult): string | null { return parts.join(" | "); } +// --- Score breakdown table --- +// Terminal counterpart of the HTML report's "Score breakdown" section. The row +// selection here mirrors renderCategoryBreakdown() in +// src/report-ui/src/scripts/render.ts — keep the two in sync. + +/** Total width of the breakdown table, excluding its 4-space indent. Matches SEP_SCORED's 102-column footprint. */ +const SEP_BREAKDOWN = 98; +/** Gutter between the id/dimensions columns and the next column. */ +const BREAKDOWN_GUTTER = 2; +/** Never squeeze the rationale below this, even if the dimensions column is wide. */ +const BREAKDOWN_RATIONALE_MIN = 30; /** - * Find the non-default audit with the lowest composite score. - * Returns a truncated rationale string, or null if no non-default audits exist. + * Hard cap on the dimensions column. The widest audit cell is fixed + * ("Success: 99 Speed: 99 Relevance: 99" = 37), but a necessity row's flagged-id + * list is unbounded, so cap the column rather than let one row widen the table. */ -function findWeakestAuditRationale(audits: InteractionAudit[]): string | null { - let weakest: InteractionAudit | null = null; - let weakestComposite = Infinity; - - for (const audit of audits) { - if (audit.rationale === "default") continue; - const composite = (audit.success + audit.speed + audit.weight + audit.contextRelevance) / 4; - if (composite < weakestComposite) { - weakestComposite = composite; - weakest = audit; - } +const COL_BREAKDOWN_DIMS_MAX = 46; +/** Flagged ids listed inline on a necessity row before spilling into "+N more". */ +const BREAKDOWN_MAX_FLAGGED_IDS = 4; + +/** Scale a 0-1 audit dimension onto the 0-100 scale used everywhere else. Mirrors fmt01() in the HTML report. */ +function fmt01(value: number): string { + return (value * 100).toFixed(0); +} + +/** Relevance and necessity only contribute to the Agent category's score. */ +function categoryShowsRelevance(label: string): boolean { + return label === "Agent"; +} + +/** An audit is "imperfect" when a dimension that feeds this category's score is below 1. */ +function isImperfectAudit(audit: InteractionAudit, showRelevance: boolean): boolean { + return audit.success < 1 || audit.speed < 1 || (showRelevance && audit.contextRelevance < 1); +} + +/** The sub-perfect dimensions of one audit, formatted as "Success: 50 Speed: 80". */ +function formatAuditDims(audit: InteractionAudit, showRelevance: boolean): string { + const dims: Array<{ label: string; value: number }> = [ + { label: "Success", value: audit.success }, + { label: "Speed", value: audit.speed }, + ...(showRelevance ? [{ label: "Relevance", value: audit.contextRelevance }] : []), + ]; + return dims + .filter((d) => d.value < 1) + .map((d) => `${d.label}: ${fmt01(d.value)}`) + .join(" "); +} + +/** Truncate a pre-formatted cell to fit, preserving its internal spacing. */ +function truncateCell(text: string, max: number): string { + return text.length > max ? text.slice(0, max - 1) + "…" : text; +} + +/** Collapse whitespace and truncate a judge rationale to fit one table cell. */ +function truncateRationale(rationale: string, max: number): string { + return truncateCell(rationale.replace(/\s+/g, " ").trim(), max); +} + +/** The necessity row's flagged interactions, e.g. "Unnecessary: #4, #12" or "… +9 more". */ +function formatFlaggedIds(ids: number[]): string { + if (ids.length === 0) return ""; + const shown = ids.slice(0, BREAKDOWN_MAX_FLAGGED_IDS).map((id) => `#${id}`); + const overflow = ids.length - shown.length; + if (overflow > 0) shown.push(`+${overflow} more`); + return `Unnecessary: ${shown.join(", ")}`; +} + +/** + * Render the per-interaction score breakdown for one category as a table. + * Returns an empty array when the category has nothing to explain (no audits + * with real rationales), matching the HTML report's behaviour. + */ +function renderBreakdownTable(label: string, cat: CategoryScore): string[] { + const nonDefaultAudits = cat.audits.filter((a) => a.rationale !== "default"); + if (nonDefaultAudits.length === 0) return []; + + const showRelevance = categoryShowsRelevance(label); + const imperfect = nonDefaultAudits.filter((a) => isImperfectAudit(a, showRelevance)); + const passingCount = nonDefaultAudits.length - imperfect.length; + + const necessity = showRelevance && cat.necessity.rationale !== "default" ? cat.necessity : null; + if (imperfect.length === 0 && !necessity) return []; + + // Build the cells first so the columns can be sized to the content: a + // dimensions cell ranges from "Speed: 60" to "Success: 99 Speed: 99 Relevance: 99". + const rows: Array<[id: string, dims: string, rationale: string]> = []; + if (necessity) { + rows.push(["Necessity", formatFlaggedIds(necessity.unnecessaryIds), necessity.rationale]); } + for (const audit of imperfect) { + rows.push([`#${audit.id}`, formatAuditDims(audit, showRelevance), audit.rationale]); + } + + const header: [string, string, string] = ["Interaction", "Dimensions", "Rationale"]; + const widthOf = (i: 0 | 1) => Math.max(header[i].length, ...rows.map((r) => r[i].length)) + BREAKDOWN_GUTTER; + const idWidth = widthOf(0); + const dimsWidth = Math.min(widthOf(1), COL_BREAKDOWN_DIMS_MAX + BREAKDOWN_GUTTER); + const rationaleWidth = Math.max(BREAKDOWN_RATIONALE_MIN, SEP_BREAKDOWN - idWidth - dimsWidth); + + const renderRow = ([id, dims, rationale]: [string, string, string]) => + ` ${id.padEnd(idWidth)}${truncateCell(dims, dimsWidth - BREAKDOWN_GUTTER).padEnd(dimsWidth)}` + + truncateRationale(rationale, rationaleWidth); + + const sep = "─".repeat(idWidth + dimsWidth + rationaleWidth); + const lines: string[] = []; + + lines.push(""); + lines.push(" Score breakdown"); + lines.push(` ${sep}`); + lines.push(renderRow(header)); + lines.push(` ${sep}`); + lines.push(...rows.map(renderRow)); + lines.push(` ${sep}`); + + if (passingCount > 0) { + lines.push(` ${passingCount} other passing interaction${passingCount !== 1 ? "s" : ""} not shown`); + } + + return lines; +} + +/** + * Render one process-quality category: score header, interaction counts, and — + * in verbose mode — the dimension roll-up plus the per-interaction breakdown table. + */ +function renderCategorySection(label: string, cat: CategoryScore, verbose: boolean): string[] { + const lines: string[] = []; - if (!weakest) return null; + lines.push(` ${label.padEnd(COL_CATEGORY_LABEL)}${cat.score} / 100`); + lines.push(` ${cat.interactionCount} interactions | ${cat.auditedCount} audited`); - const rationale = weakest.rationale.length > 100 ? weakest.rationale.slice(0, 97) + "..." : weakest.rationale; - return `#${weakest.id} ${rationale}`; + if (verbose) { + const d = cat.dimensions; + lines.push( + ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + + `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, + ); + lines.push(...renderBreakdownTable(label, cat)); + } + + return lines; } export function renderFinalOutput(output: RunOutput, verbose: boolean, agentCount?: number): string { @@ -310,48 +431,13 @@ function renderScoredResult(result: ScoredRunResult, verbose: boolean): string { } lines.push(""); - // Environment - lines.push(` Environment ${score.environment.score} / 100`); - lines.push( - ` ${score.environment.interactionCount} interactions | ` + `${score.environment.auditedCount} audited`, - ); - if (verbose) { - const d = score.environment.dimensions; - lines.push( - ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + - `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, - ); - const envRationale = findWeakestAuditRationale(score.environment.audits); - if (envRationale) lines.push(` ${envRationale}`); - } + lines.push(...renderCategorySection("Environment", score.environment, verbose)); lines.push(""); - // Service - lines.push(` Service ${score.service.score} / 100`); - lines.push(` ${score.service.interactionCount} interactions | ` + `${score.service.auditedCount} audited`); - if (verbose) { - const d = score.service.dimensions; - lines.push( - ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + - `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, - ); - const svcRationale = findWeakestAuditRationale(score.service.audits); - if (svcRationale) lines.push(` ${svcRationale}`); - } + lines.push(...renderCategorySection("Service", score.service, verbose)); lines.push(""); - // Agent - lines.push(` Agent ${score.agent.score} / 100`); - lines.push(` ${score.agent.interactionCount} interactions | ` + `${score.agent.auditedCount} audited`); - if (verbose) { - const d = score.agent.dimensions; - lines.push( - ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + - `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, - ); - const agentRationale = findWeakestAuditRationale(score.agent.audits); - if (agentRationale) lines.push(` ${agentRationale}`); - } + lines.push(...renderCategorySection("Agent", score.agent, verbose)); lines.push(""); lines.push(` Agent: ${result.agentName}`); diff --git a/test/unit/ui/format.test.ts b/test/unit/ui/format.test.ts index e29c2c0..83ec31a 100644 --- a/test/unit/ui/format.test.ts +++ b/test/unit/ui/format.test.ts @@ -8,7 +8,7 @@ import { renderBaselineList, renderBaselineComparison, } from "../../../src/ui/format.js"; -import type { CategoryScore, ScoreResult } from "../../../src/types/scoring.js"; +import type { CategoryScore, InteractionAudit, ScoreResult } from "../../../src/types/scoring.js"; import type { RunOutput, RunResult } from "../../../src/types/output.js"; import type { AgentOutput } from "../../../src/types/agent.js"; import type { Baseline, BaselineComparison } from "../../../src/types/baseline.js"; @@ -389,3 +389,149 @@ describe("renderBaselineComparison", () => { expect(output).not.toContain("New (not in baseline)"); }); }); + +// --- Score breakdown table (verbose scored output) --- + +function makeAudit(overrides: Partial & { id: number }): InteractionAudit { + return { + categories: ["agent"], + success: 1, + speed: 1, + weight: 1, + contextRelevance: 1, + rationale: "Looked fine.", + ...overrides, + }; +} + +/** A scored result whose Agent category carries the given audits and necessity judgment. */ +function makeScoredResult(agentCat: Partial): RunResult { + const agent: CategoryScore = { + ...makeCategory(60), + auditedCount: 3, + necessity: { category: "agent", score: 1, unnecessaryIds: [], rationale: "default" }, + ...agentCat, + }; + return { + ...makeResult({ exitCode: 0 }), + score: makeScoreResult({ agent }), + } as RunResult; +} + +describe("renderScenarioDetail score breakdown", () => { + it("renders a table row per imperfect audit with its sub-perfect dimensions", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [ + makeAudit({ id: 3, success: 0.5, speed: 0.8, rationale: "Install failed once." }), + makeAudit({ id: 7, speed: 0.6, rationale: "Slow directory scan." }), + ], + }), + ); + + expect(out).toContain("Score breakdown"); + expect(out).toContain("Interaction"); + expect(out).toMatch(/#3\s+Success: 50\s+Speed: 80\s+Install failed once\./); + expect(out).toMatch(/#7\s+Speed: 60\s+Slow directory scan\./); + }); + + it("omits dimensions that are already perfect", () => { + const out = renderScenarioDetail(makeScoredResult({ audits: [makeAudit({ id: 1, speed: 0.4 })] })); + expect(out).toMatch(/#1\s+Speed: 40/); + expect(out).not.toContain("Success: 100"); + }); + + it("counts passing audits instead of listing them", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [ + makeAudit({ id: 1, success: 0.5 }), + makeAudit({ id: 2 }), + makeAudit({ id: 3 }), + makeAudit({ id: 4, rationale: "default" }), + ], + }), + ); + expect(out).toContain("2 other passing interactions not shown"); + expect(out).not.toMatch(/^\s+#2\s/m); + }); + + it("singularizes the passing-interaction count", () => { + const out = renderScenarioDetail( + makeScoredResult({ audits: [makeAudit({ id: 1, success: 0.5 }), makeAudit({ id: 2 })] }), + ); + expect(out).toContain("1 other passing interaction not shown"); + }); + + it("renders a necessity row with the flagged interaction ids", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [makeAudit({ id: 1, success: 0.5 })], + necessity: { category: "agent", score: 0.7, unnecessaryIds: [4, 12], rationale: "Redundant greps." }, + }), + ); + expect(out).toMatch(/Necessity\s+Unnecessary: #4, #12\s+Redundant greps\./); + }); + + it("spills a long flagged-id list into a +N more suffix", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [makeAudit({ id: 1, success: 0.5 })], + necessity: { + category: "agent", + score: 0.1, + unnecessaryIds: [1, 2, 3, 4, 5, 6, 7], + rationale: "Many redundant calls.", + }, + }), + ); + expect(out).toContain("Unnecessary: #1, #2, #3, #4, +3 more"); + }); + + it("omits the table when every audit is a default placeholder", () => { + const out = renderScenarioDetail( + makeScoredResult({ audits: [makeAudit({ id: 1, success: 0.5, rationale: "default" })] }), + ); + expect(out).not.toContain("Score breakdown"); + }); + + it("omits the table when all audits pass and necessity is default", () => { + const out = renderScenarioDetail(makeScoredResult({ audits: [makeAudit({ id: 1 })] })); + expect(out).not.toContain("Score breakdown"); + }); + + it("ignores context relevance outside the Agent category", () => { + const envAudit = makeAudit({ id: 1, categories: ["environment"], contextRelevance: 0.2 }); + const env: CategoryScore = { ...makeCategory(60), auditedCount: 1, audits: [envAudit] }; + const out = renderScenarioDetail({ ...makeResult({ exitCode: 0 }), score: makeScoreResult({ env }) } as RunResult); + + // Relevance does not feed Env's score, so this audit counts as passing. + expect(out).not.toMatch(/#1\s+Relevance: 20/); + }); + + it("keeps every line within the 102-column table footprint", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [ + makeAudit({ + id: 1, + success: 0.99, + speed: 0.99, + contextRelevance: 0.99, + rationale: "A ".repeat(200), + }), + ], + necessity: { + category: "agent", + score: 0.1, + unnecessaryIds: Array.from({ length: 40 }, (_, i) => i + 100), + rationale: "Long rationale. ".repeat(40), + }, + }), + ); + + for (const line of out.split("\n")) { + expect([...line].length).toBeLessThanOrEqual(102); + } + }); +}); From 3b13b94912e84928124478c55ddd583ee94dc7bd Mon Sep 17 00:00:00 2001 From: Kunle Oshiyoye Date: Thu, 24 Sep 2026 15:00:20 -0400 Subject: [PATCH 2/2] feat: Clean up output markdown/text from an eval run --- src/ui/format.ts | 225 ++++++++++++++++-------------------- test/unit/ui/format.test.ts | 99 +++++++++------- 2 files changed, 159 insertions(+), 165 deletions(-) diff --git a/src/ui/format.ts b/src/ui/format.ts index b228e81..fac3ab8 100644 --- a/src/ui/format.ts +++ b/src/ui/format.ts @@ -147,25 +147,19 @@ export function buildScoreInsight(score: ScoreResult): string | null { return parts.join(" | "); } -// --- Score breakdown table --- -// Terminal counterpart of the HTML report's "Score breakdown" section. The row -// selection here mirrors renderCategoryBreakdown() in -// src/report-ui/src/scripts/render.ts — keep the two in sync. - -/** Total width of the breakdown table, excluding its 4-space indent. Matches SEP_SCORED's 102-column footprint. */ -const SEP_BREAKDOWN = 98; -/** Gutter between the id/dimensions columns and the next column. */ -const BREAKDOWN_GUTTER = 2; -/** Never squeeze the rationale below this, even if the dimensions column is wide. */ -const BREAKDOWN_RATIONALE_MIN = 30; -/** - * Hard cap on the dimensions column. The widest audit cell is fixed - * ("Success: 99 Speed: 99 Relevance: 99" = 37), but a necessity row's flagged-id - * list is unbounded, so cap the column rather than let one row widen the table. - */ -const COL_BREAKDOWN_DIMS_MAX = 46; -/** Flagged ids listed inline on a necessity row before spilling into "+N more". */ -const BREAKDOWN_MAX_FLAGGED_IDS = 4; +// --- Score breakdown --- +// Compact terminal summary of the HTML report's "Score breakdown" section: only +// the worst few interactions, each with its single weakest dimension. The full +// per-interaction table lives in the HTML report (renderCategoryBreakdown() in +// src/report-ui/src/scripts/render.ts), which decides what counts as imperfect. + +/** Total line width, including the 4-space indent. Fits an 80-column terminal / Actions log. */ +const BREAKDOWN_WIDTH = 80; +/** Interactions listed per category; the rest are summarized in a count. */ +const BREAKDOWN_MAX_ROWS = 3; +/** Width of the id column ("#12", "Necessity") and metric column ("Relevance 40", "12 unnecessary"). */ +const COL_BREAKDOWN_ID = 11; +const COL_BREAKDOWN_METRIC = 16; /** Scale a 0-1 audit dimension onto the 0-100 scale used everywhere else. Mirrors fmt01() in the HTML report. */ function fmt01(value: number): string { @@ -177,117 +171,98 @@ function categoryShowsRelevance(label: string): boolean { return label === "Agent"; } -/** An audit is "imperfect" when a dimension that feeds this category's score is below 1. */ -function isImperfectAudit(audit: InteractionAudit, showRelevance: boolean): boolean { - return audit.success < 1 || audit.speed < 1 || (showRelevance && audit.contextRelevance < 1); -} - -/** The sub-perfect dimensions of one audit, formatted as "Success: 50 Speed: 80". */ -function formatAuditDims(audit: InteractionAudit, showRelevance: boolean): string { - const dims: Array<{ label: string; value: number }> = [ +/** The lowest dimension of one audit that feeds this category's score. */ +function weakestDimension(audit: InteractionAudit, showRelevance: boolean): { label: string; value: number } { + const dims = [ { label: "Success", value: audit.success }, { label: "Speed", value: audit.speed }, ...(showRelevance ? [{ label: "Relevance", value: audit.contextRelevance }] : []), ]; - return dims - .filter((d) => d.value < 1) - .map((d) => `${d.label}: ${fmt01(d.value)}`) - .join(" "); -} - -/** Truncate a pre-formatted cell to fit, preserving its internal spacing. */ -function truncateCell(text: string, max: number): string { - return text.length > max ? text.slice(0, max - 1) + "…" : text; + return dims.reduce((min, d) => (d.value < min.value ? d : min)); } -/** Collapse whitespace and truncate a judge rationale to fit one table cell. */ +/** Collapse whitespace and truncate a judge rationale to fit the remaining line width. */ function truncateRationale(rationale: string, max: number): string { - return truncateCell(rationale.replace(/\s+/g, " ").trim(), max); -} - -/** The necessity row's flagged interactions, e.g. "Unnecessary: #4, #12" or "… +9 more". */ -function formatFlaggedIds(ids: number[]): string { - if (ids.length === 0) return ""; - const shown = ids.slice(0, BREAKDOWN_MAX_FLAGGED_IDS).map((id) => `#${id}`); - const overflow = ids.length - shown.length; - if (overflow > 0) shown.push(`+${overflow} more`); - return `Unnecessary: ${shown.join(", ")}`; + const text = rationale.replace(/\s+/g, " ").trim(); + return text.length > max ? text.slice(0, max - 1) + "…" : text; } /** - * Render the per-interaction score breakdown for one category as a table. + * Render the worst interactions for one category as a short, borderless list. * Returns an empty array when the category has nothing to explain (no audits - * with real rationales), matching the HTML report's behaviour. + * with real rationales, or all of them perfect and no unnecessary calls). */ function renderBreakdownTable(label: string, cat: CategoryScore): string[] { const nonDefaultAudits = cat.audits.filter((a) => a.rationale !== "default"); if (nonDefaultAudits.length === 0) return []; const showRelevance = categoryShowsRelevance(label); - const imperfect = nonDefaultAudits.filter((a) => isImperfectAudit(a, showRelevance)); + const imperfect = nonDefaultAudits + .map((audit) => ({ audit, weakest: weakestDimension(audit, showRelevance) })) + .filter((r) => r.weakest.value < 1) + .sort((a, b) => a.weakest.value - b.weakest.value); const passingCount = nonDefaultAudits.length - imperfect.length; - const necessity = showRelevance && cat.necessity.rationale !== "default" ? cat.necessity : null; + const necessity = + showRelevance && cat.necessity.rationale !== "default" && cat.necessity.unnecessaryIds.length > 0 + ? cat.necessity + : null; if (imperfect.length === 0 && !necessity) return []; - // Build the cells first so the columns can be sized to the content: a - // dimensions cell ranges from "Speed: 60" to "Success: 99 Speed: 99 Relevance: 99". - const rows: Array<[id: string, dims: string, rationale: string]> = []; - if (necessity) { - rows.push(["Necessity", formatFlaggedIds(necessity.unnecessaryIds), necessity.rationale]); - } - for (const audit of imperfect) { - rows.push([`#${audit.id}`, formatAuditDims(audit, showRelevance), audit.rationale]); - } - - const header: [string, string, string] = ["Interaction", "Dimensions", "Rationale"]; - const widthOf = (i: 0 | 1) => Math.max(header[i].length, ...rows.map((r) => r[i].length)) + BREAKDOWN_GUTTER; - const idWidth = widthOf(0); - const dimsWidth = Math.min(widthOf(1), COL_BREAKDOWN_DIMS_MAX + BREAKDOWN_GUTTER); - const rationaleWidth = Math.max(BREAKDOWN_RATIONALE_MIN, SEP_BREAKDOWN - idWidth - dimsWidth); + const rationaleWidth = BREAKDOWN_WIDTH - 4 - COL_BREAKDOWN_ID - COL_BREAKDOWN_METRIC; + const renderRow = (id: string, metric: string, rationale: string) => + ` ${id.padEnd(COL_BREAKDOWN_ID)}${metric.padEnd(COL_BREAKDOWN_METRIC)}${truncateRationale(rationale, rationaleWidth)}`; - const renderRow = ([id, dims, rationale]: [string, string, string]) => - ` ${id.padEnd(idWidth)}${truncateCell(dims, dimsWidth - BREAKDOWN_GUTTER).padEnd(dimsWidth)}` + - truncateRationale(rationale, rationaleWidth); - - const sep = "─".repeat(idWidth + dimsWidth + rationaleWidth); const lines: string[] = []; - lines.push(""); - lines.push(" Score breakdown"); - lines.push(` ${sep}`); - lines.push(renderRow(header)); - lines.push(` ${sep}`); - lines.push(...rows.map(renderRow)); - lines.push(` ${sep}`); - - if (passingCount > 0) { - lines.push(` ${passingCount} other passing interaction${passingCount !== 1 ? "s" : ""} not shown`); + for (const { audit, weakest } of imperfect.slice(0, BREAKDOWN_MAX_ROWS)) { + lines.push(renderRow(`#${audit.id}`, `${weakest.label} ${fmt01(weakest.value)}`, audit.rationale)); + } + if (necessity) { + lines.push(renderRow("Necessity", `${necessity.unnecessaryIds.length} unnecessary`, necessity.rationale)); } + const hiddenCount = imperfect.length - Math.min(imperfect.length, BREAKDOWN_MAX_ROWS); + const footer = [ + ...(hiddenCount > 0 ? [`${hiddenCount} more flagged`] : []), + ...(passingCount > 0 ? [`${passingCount} passing`] : []), + ]; + if (footer.length > 0) lines.push(` ${footer.join(", ")} not shown`); + return lines; } -/** - * Render one process-quality category: score header, interaction counts, and — - * in verbose mode — the dimension roll-up plus the per-interaction breakdown table. - */ -function renderCategorySection(label: string, cat: CategoryScore, verbose: boolean): string[] { - const lines: string[] = []; +/** Width of the scorecard and section rules, including the 2-space indent. */ +const SCORECARD_WIDTH = 80; +/** Cells in a score bar; each cell is 5 points. */ +const SCORE_BAR_CELLS = 20; - lines.push(` ${label.padEnd(COL_CATEGORY_LABEL)}${cat.score} / 100`); - lines.push(` ${cat.interactionCount} interactions | ${cat.auditedCount} audited`); +/** A 0-100 score as a fixed-width bar, e.g. "████████████░░░░░░░░". */ +function scoreBar(score: number): string { + const filled = Math.round((Math.max(0, Math.min(100, score)) / 100) * SCORE_BAR_CELLS); + return "█".repeat(filled) + "░".repeat(SCORE_BAR_CELLS - filled); +} - if (verbose) { - const d = cat.dimensions; - lines.push( - ` Success: ${d.success} | Speed: ${d.speed} | Weight: ${d.weight} | ` + - `Relevance: ${d.relevance} | Necessity: ${d.necessity}`, - ); - lines.push(...renderBreakdownTable(label, cat)); - } +/** One scorecard row: label, "NN / 100", bar, and an optional trailing note. */ +function scorecardRow(label: string, score: number, note?: string): string { + const row = ` ${label.padEnd(COL_CATEGORY_LABEL)}${String(score).padStart(3)} / 100 ${scoreBar(score)}`; + return note ? `${row} ${note}` : row; +} - return lines; +/** A section heading followed by a rule to the scorecard width, e.g. " Agent ─────". */ +function sectionHeading(title: string): string { + return ` ${title} ${"─".repeat(Math.max(0, SCORECARD_WIDTH - title.length - 3))}`; +} + +/** Verbose detail for one process-quality category: dimension roll-up plus the worst interactions. */ +function renderCategoryDetail(label: string, cat: CategoryScore): string[] { + const d = cat.dimensions; + return [ + sectionHeading(label), + ` Success ${d.success} · Speed ${d.speed} · Weight ${d.weight} · ` + + `Relevance ${d.relevance} · Necessity ${d.necessity}`, + ...renderBreakdownTable(label, cat), + ]; } export function renderFinalOutput(output: RunOutput, verbose: boolean, agentCount?: number): string { @@ -410,43 +385,47 @@ function renderScoredResult(result: ScoredRunResult, verbose: boolean): string { const { score } = result; const sep = "─".repeat(SEP_DETAIL); const lines: string[] = []; + const categories: Array<[label: string, cat: CategoryScore]> = [ + ["Environment", score.environment], + ["Service", score.service], + ["Agent", score.agent], + ]; lines.push(""); lines.push(` AXIS Report: ${result.scenarioName}`); lines.push(` ${sep}`); lines.push(""); - lines.push(` AXIS Result ${score.axisScore} / 100`); - if (score.judging) { - const judgeLabel = score.judging.model ? `${score.judging.agent}|${score.judging.model}` : score.judging.agent; - lines.push(` Agent used for judging: ${judgeLabel}`); - } - lines.push(""); - - // Goal Achievement - lines.push(` Goal Achievement ${score.goalAchievement.score} / 100`); - for (const c of score.goalAchievement.criteria) { - const icon = c.score >= CRITERION_HIGH ? "\u2714" : c.score >= CRITERION_MEDIUM ? "\u25D0" : "\u2717"; - const label = c.check.length > 38 ? c.check.slice(0, 35) + "..." : c.check; - lines.push(` ${icon} ${label.padEnd(40)} (${c.score}/10)`); - } - lines.push(""); - lines.push(...renderCategorySection("Environment", score.environment, verbose)); + // Scorecard: every score up front, aligned, so they read at a glance. + lines.push(scorecardRow("AXIS Result", score.axisScore)); lines.push(""); - - lines.push(...renderCategorySection("Service", score.service, verbose)); - lines.push(""); - - lines.push(...renderCategorySection("Agent", score.agent, verbose)); + lines.push(scorecardRow("Goal Achievement", score.goalAchievement.score)); + for (const [label, cat] of categories) { + lines.push(scorecardRow(label, cat.score, `${cat.interactionCount} interactions · ${cat.auditedCount} audited`)); + } lines.push(""); - lines.push(` Agent: ${result.agentName}`); + if (score.judging) { + const judgeLabel = score.judging.model ? `${score.judging.agent}|${score.judging.model}` : score.judging.agent; + lines.push(` Judged by: ${judgeLabel}`); + } - // Verbose: show rationale per criterion - if (verbose) { + // Breakdowns underneath the scorecard. + if (score.goalAchievement.criteria.length > 0) { lines.push(""); + lines.push(sectionHeading("Goal Achievement")); for (const c of score.goalAchievement.criteria) { - lines.push(` [${c.check}] ${c.rationale}`); + const icon = c.score >= CRITERION_HIGH ? "\u2714" : c.score >= CRITERION_MEDIUM ? "\u25D0" : "\u2717"; + const label = c.check.length > 38 ? c.check.slice(0, 35) + "..." : c.check; + lines.push(` ${icon} ${label.padEnd(40)} (${c.score}/10)`); + if (verbose && c.rationale) lines.push(` ${c.rationale}`); + } + } + + if (verbose) { + for (const [label, cat] of categories) { + lines.push(""); + lines.push(...renderCategoryDetail(label, cat)); } } diff --git a/test/unit/ui/format.test.ts b/test/unit/ui/format.test.ts index 83ec31a..f4c56df 100644 --- a/test/unit/ui/format.test.ts +++ b/test/unit/ui/format.test.ts @@ -419,7 +419,7 @@ function makeScoredResult(agentCat: Partial): RunResult { } describe("renderScenarioDetail score breakdown", () => { - it("renders a table row per imperfect audit with its sub-perfect dimensions", () => { + it("renders one row per imperfect audit with its weakest dimension", () => { const out = renderScenarioDetail( makeScoredResult({ audits: [ @@ -429,16 +429,27 @@ describe("renderScenarioDetail score breakdown", () => { }), ); - expect(out).toContain("Score breakdown"); - expect(out).toContain("Interaction"); - expect(out).toMatch(/#3\s+Success: 50\s+Speed: 80\s+Install failed once\./); - expect(out).toMatch(/#7\s+Speed: 60\s+Slow directory scan\./); + expect(out).toMatch(/#3\s+Success 50\s+Install failed once\./); + expect(out).toMatch(/#7\s+Speed 60\s+Slow directory scan\./); + expect(out).not.toMatch(/#3.*Speed 80/); }); - it("omits dimensions that are already perfect", () => { - const out = renderScenarioDetail(makeScoredResult({ audits: [makeAudit({ id: 1, speed: 0.4 })] })); - expect(out).toMatch(/#1\s+Speed: 40/); - expect(out).not.toContain("Success: 100"); + it("orders rows worst-first and caps them at three", () => { + const out = renderScenarioDetail( + makeScoredResult({ + audits: [ + makeAudit({ id: 1, speed: 0.9 }), + makeAudit({ id: 2, success: 0.2 }), + makeAudit({ id: 3, speed: 0.5 }), + makeAudit({ id: 4, contextRelevance: 0.3 }), + makeAudit({ id: 5 }), + ], + }), + ); + + const ids = [...out.matchAll(/^\s+#(\d+)\s/gm)].map((m) => Number(m[1])); + expect(ids).toEqual([2, 4, 3]); + expect(out).toContain("1 more flagged, 1 passing not shown"); }); it("counts passing audits instead of listing them", () => { @@ -452,52 +463,40 @@ describe("renderScenarioDetail score breakdown", () => { ], }), ); - expect(out).toContain("2 other passing interactions not shown"); + expect(out).toContain("2 passing not shown"); expect(out).not.toMatch(/^\s+#2\s/m); }); - it("singularizes the passing-interaction count", () => { - const out = renderScenarioDetail( - makeScoredResult({ audits: [makeAudit({ id: 1, success: 0.5 }), makeAudit({ id: 2 })] }), - ); - expect(out).toContain("1 other passing interaction not shown"); - }); - - it("renders a necessity row with the flagged interaction ids", () => { + it("renders a necessity row with the unnecessary-call count", () => { const out = renderScenarioDetail( makeScoredResult({ audits: [makeAudit({ id: 1, success: 0.5 })], necessity: { category: "agent", score: 0.7, unnecessaryIds: [4, 12], rationale: "Redundant greps." }, }), ); - expect(out).toMatch(/Necessity\s+Unnecessary: #4, #12\s+Redundant greps\./); + expect(out).toMatch(/Necessity\s+2 unnecessary\s+Redundant greps\./); }); - it("spills a long flagged-id list into a +N more suffix", () => { + it("omits the necessity row when nothing was flagged", () => { const out = renderScenarioDetail( makeScoredResult({ audits: [makeAudit({ id: 1, success: 0.5 })], - necessity: { - category: "agent", - score: 0.1, - unnecessaryIds: [1, 2, 3, 4, 5, 6, 7], - rationale: "Many redundant calls.", - }, + necessity: { category: "agent", score: 1, unnecessaryIds: [], rationale: "All calls were needed." }, }), ); - expect(out).toContain("Unnecessary: #1, #2, #3, #4, +3 more"); + expect(out).not.toMatch(/^\s+Necessity\s/m); }); it("omits the table when every audit is a default placeholder", () => { const out = renderScenarioDetail( makeScoredResult({ audits: [makeAudit({ id: 1, success: 0.5, rationale: "default" })] }), ); - expect(out).not.toContain("Score breakdown"); + expect(out).not.toMatch(/^\s+#1\s/m); }); it("omits the table when all audits pass and necessity is default", () => { const out = renderScenarioDetail(makeScoredResult({ audits: [makeAudit({ id: 1 })] })); - expect(out).not.toContain("Score breakdown"); + expect(out).not.toMatch(/^\s+#1\s/m); }); it("ignores context relevance outside the Agent category", () => { @@ -506,21 +505,13 @@ describe("renderScenarioDetail score breakdown", () => { const out = renderScenarioDetail({ ...makeResult({ exitCode: 0 }), score: makeScoreResult({ env }) } as RunResult); // Relevance does not feed Env's score, so this audit counts as passing. - expect(out).not.toMatch(/#1\s+Relevance: 20/); + expect(out).not.toMatch(/#1\s+Relevance 20/); }); - it("keeps every line within the 102-column table footprint", () => { + it("keeps every breakdown line within 80 columns", () => { const out = renderScenarioDetail( makeScoredResult({ - audits: [ - makeAudit({ - id: 1, - success: 0.99, - speed: 0.99, - contextRelevance: 0.99, - rationale: "A ".repeat(200), - }), - ], + audits: [makeAudit({ id: 1, success: 0.5, rationale: "A ".repeat(200) })], necessity: { category: "agent", score: 0.1, @@ -530,8 +521,32 @@ describe("renderScenarioDetail score breakdown", () => { }), ); - for (const line of out.split("\n")) { - expect([...line].length).toBeLessThanOrEqual(102); + const rows = out.split("\n").filter((l) => /^\s+(#\d+|Necessity)\s/.test(l)); + expect(rows).toHaveLength(2); + for (const line of rows) { + expect([...line].length).toBeLessThanOrEqual(80); } }); }); + +describe("renderScenarioDetail scorecard", () => { + const out = renderScenarioDetail({ + ...makeResult({ exitCode: 0 }), + score: makeScoreResult({ env: makeCategory(62), svc: makeCategory(95), agent: makeCategory(0) }), + } as RunResult); + + it("lists every score with a bar before any breakdown", () => { + const scorecard = out.slice(0, out.indexOf("─────\n", out.indexOf("Environment"))); + expect(scorecard).toMatch(/AXIS Result\s+80 \/ 100 {2}█{16}░{4}/); + expect(scorecard).toMatch(/Goal Achievement\s+80 \/ 100/); + expect(scorecard).toMatch(/Environment\s+62 \/ 100 {2}█{12}░{8}/); + expect(scorecard).toMatch(/Service\s+95 \/ 100 {2}█{19}░/); + expect(scorecard).toMatch(/Agent\s+0 \/ 100 {2}░{20}/); + }); + + it("puts the category breakdowns underneath the scorecard", () => { + const scorecardAt = out.indexOf("Agent 0 / 100"); + expect(out.indexOf(" Environment ──")).toBeGreaterThan(scorecardAt); + expect(out.indexOf(" Agent ──")).toBeGreaterThan(out.indexOf(" Service ──")); + }); +});