diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4ad1f9823..e93cbba3f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -94,6 +94,7 @@ jobs: lint: ${{ steps.filter.outputs.lint || steps.all.outputs.forced }} tpn: ${{ steps.filter.outputs.tpn || steps.all.outputs.forced }} docs: ${{ steps.filter.outputs.docs || steps.all.outputs.forced }} + skills: ${{ steps.filter.outputs.skills || steps.all.outputs.forced }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -175,6 +176,13 @@ jobs: - '.readthedocs.yaml' - '.github/workflows/**' + # Published Agent Skills, graded by skillscope. The dataset and the + # skill's prose are one unit — a prompt can stop matching because the + # description changed — so any edit under skills/ re-runs the check. + skills: + - 'skills/**' + - '.github/workflows/**' + - name: Force full run off pull requests id: all if: github.event_name != 'pull_request' @@ -1093,6 +1101,81 @@ jobs: if: needs.changes.outputs.lint == 'true' run: cargo xtask powershell-lint --shell powershell + # Grades the published Agent Skills under skills/ with AMD's skillscope + # harness. Only the `structural` command runs: it needs no agent, no API key + # and no network beyond the install, so a red run is never wrong for a + # reason of its own. It is advisory today, not blocking: `Skill checks + # (skillscope)` (this job's `name:`, not the `skill-evals` job id) is not a + # required status check in branch protection, so a failing run does not + # stop a merge. An admin must add that exact context to make it blocking. + # It checks the frontmatter an agent runtime + # parses (a `name` that disagrees with the folder makes the skill unloadable), + # the dataset's coverage bar, and every internal markdown reference — a link + # into a renamed file is not cosmetic here, because the agent follows it, + # finds nothing, and improvises. + # + # `routing` and `behavioral` are deliberately NOT run: both need an + # authenticated `claude` CLI and an ANTHROPIC_API_KEY, and this repo has no + # such secret. Turning them on is a caller change to skillscope's + # `skill-evals.yml` reusable workflow once one exists — they are the half of + # the question that asks whether the skill actually fires, so this job is not + # a substitute for them. + # + # skills/rocm-cli-assistant/ is deliberately out of scope. Despite its path it + # is not a published skill: apps/rocm/src/main.rs embeds it verbatim into the + # chat assistant's system prompt with include_str!, so giving it the YAML + # frontmatter this check requires would inject that frontmatter into a live + # prompt. + skill-evals: + name: Skill checks (skillscope) + runs-on: ubuntu-latest + timeout-minutes: 10 + needs: changes + # A manual E2E dispatch only exercises the E2E jobs; skip the rest. + if: github.event_name != 'workflow_dispatch' && needs.changes.outputs.skills == 'true' + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + + # The `skillscope` step below names skills/rocm-doctor explicitly, so a + # third skills// folder added later would pass this job green + # while never being graded -- the same silent-non-grading failure this + # job exists to catch, just one level deeper. Fail loudly instead: any + # directory under skills/ that isn't graded above or excluded by design + # (rocm-cli-assistant -- see AGENTS.md #7) forces a deliberate edit here. + - name: Guard against an ungraded skills/ directory + run: | + set -euo pipefail + known="rocm-doctor rocm-cli-assistant" + for dir in skills/*/; do + name="${dir%/}" + name="${name#skills/}" + case " $known " in + *" $name "*) ;; + *) + echo "::error::skills/$name is neither graded by skill-evals nor excluded by design ($known). Add it to the skillscope step's 'skills:' input, or add it to this guard's exclusion list with the reason, before merging." >&2 + exit 1 + ;; + esac + done + + # The action is only a launcher: it sets up Python and uv, resolves which + # build of the harness to run, and execs it. It imports nothing from + # skillscope, so pinning it to a release SHA cannot break on a payload + # version it predates. What the pin does NOT cover: the composite's own + # steps (setup-python, setup-uv, setup-node) are on their own moving + # tags, and the harness itself is resolved at runtime via + # `uvx --from git+.../skillscope@` with no lockfile. Unlike this + # repo's usual SHA pin (AGENTS.md #6), this one does not reach the code + # that actually runs. install-claude stays off; structural needs no + # agent. + - name: Check skill structure, dataset, and references + uses: amd/skillscope@6dde8e8a34ad5456d3c2ff3418bc81c2cd9d5d69 # v0.1.0 + with: + command: structural + skills: skills/rocm-doctor + args: --skill-files skill-card.md --skill-sections Description,Owner,License + install-claude: "false" + license-headers: name: License header check (hawkeye) runs-on: ubuntu-latest diff --git a/.gitignore b/.gitignore index fdec19493..f94b9a657 100644 --- a/.gitignore +++ b/.gitignore @@ -13,6 +13,8 @@ active-implementation-log.md current-user-asks.md __pycache__/ /.hawkeye-bin/ +# skillscope writes its JSON run reports here. +/.skillscope/ /.claude /plans/ /workspace/ diff --git a/AGENTS.md b/AGENTS.md index 46d0141bd..a91499cbb 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -229,6 +229,33 @@ frontmatter for the skill loader, and skills published this way carry no license headers of their own. The licence is stated in `skill-card.md` instead. +Two checks gate it, and they cover different things: + +- **`skill-evals` (skillscope, advisory today)** — the frontmatter an agent + runtime parses, the `evals/evals.json` coverage bar (at least 3 prompts that + should trigger the skill and 2 near misses that should not), the + `skill-card.md` sections, and every internal markdown link. It is not yet a + required status check in branch protection: an admin must add its exact + context, `Skill checks (skillscope)` (the job's `name:`, not the + `skill-evals` job id), before a red run actually blocks a merge. Reproduce + it locally with + `uv tool install git+https://github.com/amd/skillscope@v0.1.0`, then + `skillscope structural --skills-dir 'skills/rocm-doctor' --skill-files + skill-card.md --skill-sections Description,Owner,License`. Add `--external` + to check the outbound URLs too; CI does not, because a rate-limited host is + not a broken link. +- **`rocm_doctor_skill.feature` (e2e, blocking)** — whether the prose still + describes the binary, as above. + +Neither grades whether the skill actually *fires*. That is skillscope's +`routing` and `behavioral`, which need an authenticated `claude` CLI and an +`ANTHROPIC_API_KEY` this repo does not have. The dataset is written and checked +so those can be switched on without further work. + +`skills/rocm-cli-assistant/` is **not** in scope for skillscope: it is embedded +verbatim into the chat system prompt with `include_str!`, so the YAML +frontmatter a published skill needs would end up inside that prompt. + ## 8) Verification Matrix For This Repo Minimum quality gate before upstream-ready status: diff --git a/skills/rocm-doctor/evals/evals.json b/skills/rocm-doctor/evals/evals.json new file mode 100644 index 000000000..62fe6da27 --- /dev/null +++ b/skills/rocm-doctor/evals/evals.json @@ -0,0 +1,80 @@ +{ + "evaluations": [ + { + "id": "rocm-hip-no-binary-for-gpu", + "skill_should_trigger": true, + "prompt": "torch.cuda.is_available() returns False on my AMD GPU and I get 'hipErrorNoBinaryForGpu' when I run my script. What's wrong?", + "expected_behavior": [ + "Drive the `rocm` CLI to diagnose -- check `rocm --version`, offer to install it with the user's consent, then `rocm diagnose` -- instead of applying a ROCm fix invented from general knowledge", + "Hand the user the CLI install path rather than guessing, if the CLI cannot be installed here" + ], + "unexpected_behavior": [ + "Execute a mutating or sudo command without first getting the user's explicit consent" + ], + "logs_contain": [ + "rocm --version" + ] + }, + { + "id": "rocm-permission-denied-kfd", + "skill_should_trigger": true, + "prompt": "I get 'permission denied' opening /dev/kfd and rocminfo can't see my AMD GPU. How do I fix it?", + "note": "The symptom is Linux-only and the runner has no GPU, so this grades the Phase 0 probe and the consent rule, not a successful diagnosis.", + "unexpected_behavior": [ + "Execute a mutating or sudo command (usermod, modprobe, and the like) without first getting the user's explicit consent" + ], + "logs_contain": [ + "rocm --version" + ] + }, + { + "id": "rocm-lemonade-routes-upstream", + "skill_should_trigger": true, + "prompt": "The Lemonade app (from lemonade-sdk) fails to load a model on my Radeon GPU -- its bundled ROCm runtime throws an error. Where should I report this?", + "note": "Lemonade ships its own ROCm runtime. The host probe cannot detect it, so `route_when_no_match` never names it -- the skill hands over this tracker from its own routing table. The URL is the proof that it did.", + "logs_contain": [ + "lemonade-sdk/lemonade/issues" + ] + }, + { + "id": "rocm-nothing-established-routes-upstream", + "skill_should_trigger": true, + "prompt": "My AMD GPU setup feels wrong on Linux -- something about ROCm is off but I can't paste an exact error. Can you work out what's broken and fix it?", + "note": "A vague report the catalog cannot explain. `matched` can still come back non-empty with sub-threshold entries, so an agent gating on emptiness would propose a fix for a healthy host. The rule is to read `has_match` and route upstream when it is false.", + "expected_behavior": [ + "Treat `has_match` being false as 'no cause established' and route the user upstream, rather than proposing a fix from a sub-threshold entry in `matched`", + "Ask the user for the exact error text instead of inventing a diagnosis" + ], + "unexpected_behavior": [ + "Propose or apply a remediation when the CLI established no cause", + "Invent a failure mode that is not in the CLI's closed catalog" + ] + }, + { + "id": "cuda-broken-on-nvidia", + "skill_should_trigger": false, + "note": "Same symptom sentence this skill exists for, on the wrong vendor's GPU. The scope gate makes a non-AMD GPU a hard decline, so nothing should fire.", + "prompt": "torch.cuda.is_available() is False on my NVIDIA RTX 4090 and CUDA seems broken. Help me fix it." + }, + { + "id": "rocm-under-wsl2", + "skill_should_trigger": true, + "note": "ROCm vocabulary and an AMD GPU, but WSL2 is a separate platform the scope gate declines rather than troubleshoots -- the skill still has to load to make that call, so this grades the decline itself, not a diagnosis.", + "prompt": "I'm running ROCm inside WSL2 on Windows and torch.cuda.is_available() is False for my AMD GPU. How do I fix it?", + "expected_behavior": [ + "State plainly that WSL2 is out of scope for this skill and why, and point at AMD's ROCm-on-WSL guide", + "Give no troubleshooting for it: no commands to run, no driver advice, no diagnostic checklist" + ], + "unexpected_behavior": [ + "Run `rocm examine`, `rocm diagnose`, or `rocm fix`", + "Offer any WSL2 troubleshooting steps or generic GPU advice" + ] + }, + { + "id": "nvidia-container-toolkit-unrelated", + "skill_should_trigger": false, + "note": "NVIDIA GPU and a container runtime error, with no ROCm/AMD vocabulary at all -- doesn't match this skill's frontmatter description, so nothing should load. Restores the two-negative floor `rocm-under-wsl2` moved off of.", + "prompt": "My NVIDIA Docker container can't see the GPU -- 'nvidia-container-cli: initialization error: nvml error: driver/library version mismatch'. How do I get the NVIDIA Container Toolkit working?" + } + ] +}