diff --git a/CHANGELOG.md b/CHANGELOG.md index 6b57ca2119..2b6c0f7afc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,18 @@ > completion state and remaining P0 gates. No version bump or release claim is > made here while that status holds. +## [1.64.17.0] - 2026-07-24 + +**The broad fix: claimed limitations now require evidence, everywhere.** + +Nine live-release failures in two days shared one root: the agent asserting folklore as fact — "Apple requires an app-specific password," "the API can't do this," "screenshots need a key" — instead of running the ten-second check that would have disproven it. The shared judgment contract that binds all six skills gains clause 13: a claimed limitation or requirement is a material claim, stated only with the verbatim error, the documented statement, or a live probe in hand. Pattern-matching a failure to a familiar story is not evidence. When a cheap probe settles the question, run it before asking the user anything or declaring a gate. The Apple adapter's specific rules remain as regression pins; this clause is the umbrella that covers the cases nobody has hit yet. + +### Itemized changes + +### Changed + +- `references/SHARED-JUDGMENT.md` (all six skill trees): clause 13 — evidence-before-claimed-limitations, probe-before-gate. + ## [1.64.16.0] - 2026-07-24 **A metadata error is not a login problem.** diff --git a/VERSION b/VERSION index cf52465fdf..052f562808 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -1.64.16.0 +1.64.17.0 diff --git a/evals/parity/transcripts/policy-units.json b/evals/parity/transcripts/policy-units.json index 1a82a79326..870a6263c7 100644 --- a/evals/parity/transcripts/policy-units.json +++ b/evals/parity/transcripts/policy-units.json @@ -43,7 +43,7 @@ "prompt_sha256": "a0fcff9cad68f9305da861fa5a58990e1ef51d9d84298fe2df7ba8a2f56b1dba", "semantic_attempt_sha256": "7724c3d954f421753c597364c383bd76bc01ba9803f99fb14488287bb925aeb6" }, - "policy_sha256": "a797f3dc1190e0e6533dc4e11c3f1814ab448d2d80a3c1ceb784b5dd53dcf076", + "policy_sha256": "130c30b4cbc2fd06ff90a9374e3d09ed0745c9cd92398984cfccf1c983843e05", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -88,7 +88,7 @@ "prompt_sha256": "04d82a21358f188cb823dcceeecaebea0b4ed8b9c1f8d00220ef3b499c48b1c8", "semantic_attempt_sha256": "9639406995955284517acb30f56b37dedc42a6c08142163dfa8fe0295105cbbf" }, - "policy_sha256": "5f92342d96becb5d8311ef45e0206cf268807af30b0c2d2b3492ffa9c8a42b43", + "policy_sha256": "b56db8dbcf9aab7783509626e0f81cc6d49db4c4de1023ac07cc19db0f84bb95", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -138,7 +138,7 @@ "prompt_sha256": "8ad639f708a2ccd5c7c438b1dedf6afacdfa4eac3883a7a046ea23f29314e1c4", "semantic_attempt_sha256": "4c450039e4c622fbeaa34b7f7dfc810d3b7369aaf7e5f4d240cd6cc18a31331a" }, - "policy_sha256": "dd5be25637f658f4b8e830b1f7d9d2ef02c75bf2595b14c863bab355f04c0460", + "policy_sha256": "3dfa58861d1b8f3e646b761c0d07ddaee9612ca08e1ecf093ce725ecced62b87", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -188,7 +188,7 @@ "prompt_sha256": "d2aa9aee06a116bf04cb62772e8abdf8dbd606811b6d38565389c88e86c79ede", "semantic_attempt_sha256": "64885a66bdc6688b880cbb9a59babf314834c068eb040657501bf287bfcb0d2b" }, - "policy_sha256": "a797f3dc1190e0e6533dc4e11c3f1814ab448d2d80a3c1ceb784b5dd53dcf076", + "policy_sha256": "130c30b4cbc2fd06ff90a9374e3d09ed0745c9cd92398984cfccf1c983843e05", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -235,7 +235,7 @@ "prompt_sha256": "a01bb977de84e53d8ce3dfa427bcc93d73c6e449cd4e9c1ff63b436fd41fb0d1", "semantic_attempt_sha256": "ce5c64c62c3ff9b60949f34717dffbc908f62ce30f8a4a48ec182eba4363a206" }, - "policy_sha256": "a95b0d10de018117bf779a2b691054e6e0d8d9177dd4cae8dca056dd6b311402", + "policy_sha256": "37246df302d82acf05bc8ad70ff9939b157e259fca26f1e20b9ca3d8f48f13f1", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -286,7 +286,7 @@ "prompt_sha256": "95e97e26268ad7e509527e0f54c943ed6c4105d919f250648a2ad59ce77cbdb9", "semantic_attempt_sha256": "dacd11a78e32aeb6c0065496bedd4a9770f7bbac1036e003d3768fac41eb84f0" }, - "policy_sha256": "a797f3dc1190e0e6533dc4e11c3f1814ab448d2d80a3c1ceb784b5dd53dcf076", + "policy_sha256": "130c30b4cbc2fd06ff90a9374e3d09ed0745c9cd92398984cfccf1c983843e05", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -337,7 +337,7 @@ "prompt_sha256": "bdd7e8adfa7c15cf8531f84c3adaaacc725075f1a78f7487225300750547b82d", "semantic_attempt_sha256": "32c11b63412c5d873ff29dcc87c45bef1b50daa218a5c6c1075864aa24d7c3b6" }, - "policy_sha256": "a797f3dc1190e0e6533dc4e11c3f1814ab448d2d80a3c1ceb784b5dd53dcf076", + "policy_sha256": "130c30b4cbc2fd06ff90a9374e3d09ed0745c9cd92398984cfccf1c983843e05", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" @@ -382,7 +382,7 @@ "prompt_sha256": "0a05f6c487992bbd9de7133a7eb546f8fd039eca08d876fb04e8c7d535919d8e", "semantic_attempt_sha256": "06f77b27c9bd360562655b23ca00a0dd021bdb14ebaaf3d836845e4e0cb43d68" }, - "policy_sha256": "fc4029b1efbd8b3e038a3612964bd7d0a6d588a28e02d0badc51ebbbdaf7d2a6", + "policy_sha256": "96231044a64f223854817fbda1273dd392958cdf98fcc72f25189037e24ef335", "policy_present": true, "prompt_is_not_authority_input": true, "verdict": "PASS" diff --git a/package.json b/package.json index 1b6b2ce34f..5e3e5fc490 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "gstack", - "version": "1.64.16.0", + "version": "1.64.17.0", "description": "GStack 2 \u2014 six portable Agent Skills with an optional host-neutral runtime.", "license": "MIT", "type": "module", diff --git a/scripts/gstack2/generate-skill-tree.ts b/scripts/gstack2/generate-skill-tree.ts index f9c499b5f9..ad33f0b8ae 100644 --- a/scripts/gstack2/generate-skill-tree.ts +++ b/scripts/gstack2/generate-skill-tree.ts @@ -481,6 +481,7 @@ function sharedJudgmentContract(): string { '10. Ask only what cannot be inferred. Question rounds are for decisions that are consequential and still open after the prompt, the repository, and platform convention are consulted; infer the rest, state each inferred default in one line, and batch what remains. Never spend a round confirming what a handoff already names or offering optional extras.', '11. A user-stated time constraint binds every phase and every chained skill. Skip or compress optional phases that do not fit it, noting each skip in one line.', '12. The user makes the final decision.', + '13. A claimed limitation or requirement is a material claim. Never state that a tool, API, or platform cannot do something — or that a credential, key, account, or manual step is required — without evidence in hand: the verbatim error, the documented statement, or a live probe. Pattern-matching a failure to a familiar story is not evidence; diagnose from the actual output. When a cheap probe settles the question (run the command, list installed capabilities, attempt the operation), run it before asking the user for anything or declaring a gate.', '', ].join('\n'); } diff --git a/skills/debug/references/SHARED-JUDGMENT.md b/skills/debug/references/SHARED-JUDGMENT.md index 0db5d4da05..40d9fe9136 100644 --- a/skills/debug/references/SHARED-JUDGMENT.md +++ b/skills/debug/references/SHARED-JUDGMENT.md @@ -15,3 +15,4 @@ This contract constrains every specialist without replacing specialist judgment. 10. Ask only what cannot be inferred. Question rounds are for decisions that are consequential and still open after the prompt, the repository, and platform convention are consulted; infer the rest, state each inferred default in one line, and batch what remains. Never spend a round confirming what a handoff already names or offering optional extras. 11. A user-stated time constraint binds every phase and every chained skill. Skip or compress optional phases that do not fit it, noting each skip in one line. 12. The user makes the final decision. +13. A claimed limitation or requirement is a material claim. Never state that a tool, API, or platform cannot do something — or that a credential, key, account, or manual step is required — without evidence in hand: the verbatim error, the documented statement, or a live probe. Pattern-matching a failure to a familiar story is not evidence; diagnose from the actual output. When a cheap probe settles the question (run the command, list installed capabilities, attempt the operation), run it before asking the user for anything or declaring a gate. diff --git a/skills/plan/references/SHARED-JUDGMENT.md b/skills/plan/references/SHARED-JUDGMENT.md index 0db5d4da05..40d9fe9136 100644 --- a/skills/plan/references/SHARED-JUDGMENT.md +++ b/skills/plan/references/SHARED-JUDGMENT.md @@ -15,3 +15,4 @@ This contract constrains every specialist without replacing specialist judgment. 10. Ask only what cannot be inferred. Question rounds are for decisions that are consequential and still open after the prompt, the repository, and platform convention are consulted; infer the rest, state each inferred default in one line, and batch what remains. Never spend a round confirming what a handoff already names or offering optional extras. 11. A user-stated time constraint binds every phase and every chained skill. Skip or compress optional phases that do not fit it, noting each skip in one line. 12. The user makes the final decision. +13. A claimed limitation or requirement is a material claim. Never state that a tool, API, or platform cannot do something — or that a credential, key, account, or manual step is required — without evidence in hand: the verbatim error, the documented statement, or a live probe. Pattern-matching a failure to a familiar story is not evidence; diagnose from the actual output. When a cheap probe settles the question (run the command, list installed capabilities, attempt the operation), run it before asking the user for anything or declaring a gate. diff --git a/skills/qa/references/SHARED-JUDGMENT.md b/skills/qa/references/SHARED-JUDGMENT.md index 0db5d4da05..40d9fe9136 100644 --- a/skills/qa/references/SHARED-JUDGMENT.md +++ b/skills/qa/references/SHARED-JUDGMENT.md @@ -15,3 +15,4 @@ This contract constrains every specialist without replacing specialist judgment. 10. Ask only what cannot be inferred. Question rounds are for decisions that are consequential and still open after the prompt, the repository, and platform convention are consulted; infer the rest, state each inferred default in one line, and batch what remains. Never spend a round confirming what a handoff already names or offering optional extras. 11. A user-stated time constraint binds every phase and every chained skill. Skip or compress optional phases that do not fit it, noting each skip in one line. 12. The user makes the final decision. +13. A claimed limitation or requirement is a material claim. Never state that a tool, API, or platform cannot do something — or that a credential, key, account, or manual step is required — without evidence in hand: the verbatim error, the documented statement, or a live probe. Pattern-matching a failure to a familiar story is not evidence; diagnose from the actual output. When a cheap probe settles the question (run the command, list installed capabilities, attempt the operation), run it before asking the user for anything or declaring a gate. diff --git a/skills/review/references/SHARED-JUDGMENT.md b/skills/review/references/SHARED-JUDGMENT.md index 0db5d4da05..40d9fe9136 100644 --- a/skills/review/references/SHARED-JUDGMENT.md +++ b/skills/review/references/SHARED-JUDGMENT.md @@ -15,3 +15,4 @@ This contract constrains every specialist without replacing specialist judgment. 10. Ask only what cannot be inferred. Question rounds are for decisions that are consequential and still open after the prompt, the repository, and platform convention are consulted; infer the rest, state each inferred default in one line, and batch what remains. Never spend a round confirming what a handoff already names or offering optional extras. 11. A user-stated time constraint binds every phase and every chained skill. Skip or compress optional phases that do not fit it, noting each skip in one line. 12. The user makes the final decision. +13. A claimed limitation or requirement is a material claim. Never state that a tool, API, or platform cannot do something — or that a credential, key, account, or manual step is required — without evidence in hand: the verbatim error, the documented statement, or a live probe. Pattern-matching a failure to a familiar story is not evidence; diagnose from the actual output. When a cheap probe settles the question (run the command, list installed capabilities, attempt the operation), run it before asking the user for anything or declaring a gate. diff --git a/skills/ship/references/SHARED-JUDGMENT.md b/skills/ship/references/SHARED-JUDGMENT.md index 0db5d4da05..40d9fe9136 100644 --- a/skills/ship/references/SHARED-JUDGMENT.md +++ b/skills/ship/references/SHARED-JUDGMENT.md @@ -15,3 +15,4 @@ This contract constrains every specialist without replacing specialist judgment. 10. Ask only what cannot be inferred. Question rounds are for decisions that are consequential and still open after the prompt, the repository, and platform convention are consulted; infer the rest, state each inferred default in one line, and batch what remains. Never spend a round confirming what a handoff already names or offering optional extras. 11. A user-stated time constraint binds every phase and every chained skill. Skip or compress optional phases that do not fit it, noting each skip in one line. 12. The user makes the final decision. +13. A claimed limitation or requirement is a material claim. Never state that a tool, API, or platform cannot do something — or that a credential, key, account, or manual step is required — without evidence in hand: the verbatim error, the documented statement, or a live probe. Pattern-matching a failure to a familiar story is not evidence; diagnose from the actual output. When a cheap probe settles the question (run the command, list installed capabilities, attempt the operation), run it before asking the user for anything or declaring a gate.