From 74e23aa2317b3ad8f857f27398d47f69c2d33346 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Sat, 3 Oct 2026 19:29:35 +0300 Subject: [PATCH 01/10] fix(gates): select gate kwargs and report unusable input run() called fn(**kw) without selecting the gate's own parameters. The one-tool-four-gates action sends a superset of kwargs, so every call raised TypeError, which the top-level handler reported as exit 2 plus a traceback - indistinguishable from an unavailable gate. Now: parameters are selected via inspect.signature, dropped keys come back as ignored_kwargs, and a missing required kwarg answers UNKNOWN/rc=2 with required_kwargs instead of raising. Negative control: 4 of the 10 new tests fail on the pre-fix code. Committed with --no-verify: the pre-commit hook produces no output and does not finish on this machine (reason in AGENT_DIARY.md, 2026-10-03). --- tests/test_gates_kwarg_contract.py | 94 ++++++++++++++++++++++++++++++ tools/verification/gates.py | 38 +++++++++++- 2 files changed, 131 insertions(+), 1 deletion(-) create mode 100644 tests/test_gates_kwarg_contract.py diff --git a/tests/test_gates_kwarg_contract.py b/tests/test_gates_kwarg_contract.py new file mode 100644 index 0000000..b135dda --- /dev/null +++ b/tests/test_gates_kwarg_contract.py @@ -0,0 +1,94 @@ +"""Contract of the gates.py CLI: a superset of kwargs must reach a verdict. + +Reported defect (2026-10-03): the `gate` tool returned `GATE UNAVAILABLE (rc=2)` +for every input. Root cause: `run()` called `fn(**kw)` without selecting the +gate's own parameters, so the superset that one-tool-four-gates makes the caller +send naturally raised TypeError, which the top-level handler reported as an +unavailable gate instead of a verdict. + +These tests pin the fixed behaviour AND its failure mode: a key the caller +meant for this gate but misspelled must NOT become a silent pass. +""" +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +GATES = ROOT / "tools" / "verification" / "gates.py" + +sys.path.insert(0, str(ROOT)) +from tools.verification.gates import ALLOW, BLOCK, UNKNOWN, run # noqa: E402 + + +def _cli(gate: str, payload: dict) -> subprocess.CompletedProcess: + return subprocess.run( + [sys.executable, "-B", str(GATES), "--gate", gate, "--input", json.dumps(payload)], + capture_output=True, text=True, cwd=ROOT, timeout=120, + ) + + +def test_superset_kwargs_reach_a_verdict_instead_of_a_traceback(): + proc = _cli("population", {"population": 40, "computed_rate": 0.075, + "claim": "belongs to another gate", + "verdict": "CONFIRMED"}) + assert "Traceback" not in proc.stderr + assert proc.returncode == 0, proc.stderr + out = json.loads(proc.stdout) + assert out["verdict"] == ALLOW + assert out["ignored_kwargs"] == ["claim", "verdict"] + + +def test_dropped_kwargs_are_named_not_silently_swallowed(): + res, rc = run("population", population=1143, computed_rate=0.0, evidence="x") + assert res["verdict"] == ALLOW and rc == 0 + assert res["ignored_kwargs"] == ["evidence"] + + +def test_misspelled_key_does_not_become_a_silent_pass(): + """The dangerous half of filtering: swallow a typo and return ALLOW.""" + res, rc = run("population", populaton=40, computed_rate=0.075) + assert res["verdict"] == UNKNOWN and rc == 2 + assert "populaton" in res["ignored_kwargs"] + # the typo is caught as a MISSING required kwarg, which is louder than the + # gate's own "not supplied" path and names what the caller must supply + assert "population" in res["required_kwargs"] + assert "unusable input" in res["why"] + + +def test_exact_kwargs_carry_no_ignored_field(): + res, _ = run("population", population=0, computed_rate=0.0, label="vacuous scan") + assert res["verdict"] == BLOCK + assert "ignored_kwargs" not in res + + +def test_missing_required_kwarg_is_reported_not_raised(): + res, rc = run("population", population=40) # computed_rate absent + assert res["verdict"] == UNKNOWN and rc == 2 + assert "computed_rate" in res["required_kwargs"] + assert "unusable input" in res["why"] + + +@pytest.mark.parametrize("gate,payload", [ + ("population", {"population": 40, "computed_rate": 0.075}), + ("referent", {"claim": "valid 10/11 per EXPERIMENTS_LOG.md:537"}), + ("generalization", {"verdict": "CONFIRMED", "evidence": "one article only"}), + ("control", {"experiment": "e", "negative_control_shown_failing": True}), +]) +def test_every_gate_is_reachable_through_the_cli(gate, payload): + proc = _cli(gate, payload) + assert "Traceback" not in proc.stderr + out = json.loads(proc.stdout) + assert out["verdict"] in {ALLOW, BLOCK, UNKNOWN} + assert proc.returncode == {"ALLOW": 0, "BLOCK": 1, UNKNOWN: 2}[out["verdict"]] + + +def test_selftest_still_passes(): + proc = subprocess.run([sys.executable, "-B", str(GATES), "--selftest"], + capture_output=True, text=True, cwd=ROOT, timeout=180) + assert proc.returncode == 0, proc.stdout[-2000:] + assert "SELFTEST PASSED" in proc.stdout \ No newline at end of file diff --git a/tools/verification/gates.py b/tools/verification/gates.py index c9a99c1..53bb7cf 100644 --- a/tools/verification/gates.py +++ b/tools/verification/gates.py @@ -24,6 +24,7 @@ from __future__ import annotations import argparse +import inspect import json import re import sys @@ -319,10 +320,45 @@ def run(action: str, *, in_scope: bool = True, **kw) -> tuple[dict, int]: "decisive_region": DECISIVE.get(gname, "undeclared"), "why": "caller declared this input outside the gate's applicability; " "no verdict was reached. OUTSIDE ITS SCOPE IS NOT A PASS."}, 0) - res = fn(**kw) + # One tool action serves four gates, so a caller legitimately sends a superset + # of kwargs. Passing that straight through raised TypeError, which the + # top-level handler turned into a traceback and exit 2 — indistinguishable + # from "the gate is unavailable". So: select this gate's own parameters, and + # ECHO what was dropped. A key the caller meant for this gate but misspelled + # still lands as a missing required kwarg, which is reported as unusable input + # (rc=2), never as a silent ALLOW. + # Red team: a dropped key is by construction another gate's key, so it cannot + # change this gate's verdict — that is why the verdict is not downgraded, and + # why the dropped names stay in the result for a consumer that reads only + # `verdict`. + allowed = set(inspect.signature(fn).parameters) + ignored = sorted(k for k in kw if k not in allowed) + if ignored: + kw = {k: v for k, v in kw.items() if k in allowed} + try: + res = fn(**kw) + except TypeError as e: + # A required kwarg the caller never supplied is unusable input, not a + # crash. The module's own exit-code contract says that is rc=2 with a + # verdict of UNKNOWN, so answer that way instead of letting the + # top-level handler turn it into "gate unavailable". + required = sorted(n for n, p in inspect.signature(fn).parameters.items() + if p.default is inspect.Parameter.empty + and p.kind is inspect.Parameter.KEYWORD_ONLY) + out = {"gate": action.upper(), "verdict": UNKNOWN, + "scope": scope_of(action.upper()), + "decisive_region": DECISIVE.get(action.upper(), "undeclared"), + "why": f"unusable input for this gate: {e}", + "required_kwargs": required, + "supplied_kwargs": sorted(kw)} + if ignored: + out["ignored_kwargs"] = ignored + return out, 2 gname = res.get("gate", action) res["scope"] = scope_of(gname) res["decisive_region"] = DECISIVE.get(gname, "undeclared") + if ignored: + res["ignored_kwargs"] = ignored rc = {"BLOCK": 1, "ALLOW": 0, "UNKNOWN": 2, OUT_OF_SCOPE: 0}[res["verdict"]] return res, rc From a19ed802a47140caa10fac8da1c0d56570d20a5e Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Sat, 3 Oct 2026 19:29:54 +0300 Subject: [PATCH 02/10] docs(claims): mark A10 measurement SUPERSEDED (closes T-17) orphan30s->120ms was published in three places with a confirmed label while claims_audit/RESULTS.md A10 records it as NOT REPRODUCED: the ORPHAN path was removed by design (R3TF). Relabel with measured-on sha 3798d6a9 and the R3TF reference; the raw run output is left untouched. Also registers P-18 (absence claimed from a single referent, seen twice in one session) and opens tools/knowledge/PENDING_LEDGER.md for the session that owns KNOWN_ISSUES.md. Committed with --no-verify; reason in AGENT_DIARY.md, 2026-10-03. --- AGENT_DIARY.md | 16 ++++++ EXPERIMENTS_LOG.md | 3 +- experiments/README.md | 2 +- experiments/claims_audit/RESULTS.md | 12 ++++ experiments/lock_zombie/README.md | 2 +- tools/knowledge/PATTERNS.md | 1 + tools/knowledge/PENDING_LEDGER.md | 88 +++++++++++++++++++++++++++++ 7 files changed, 121 insertions(+), 3 deletions(-) create mode 100644 tools/knowledge/PENDING_LEDGER.md diff --git a/AGENT_DIARY.md b/AGENT_DIARY.md index c261c8e..a264acd 100644 --- a/AGENT_DIARY.md +++ b/AGENT_DIARY.md @@ -866,3 +866,19 @@ chunk_index -(20_000_000+line), graph_score=0.4 (ниже функций 1.0). E **Конкуренция агентов (важно):** параллельная сессия **активно пишет в это же рабочее дерево** — `tools/verification/verify_public_claims.py` записан 03.10 **08:38:16** (проверено `LastWriteTime`). `tools/knowledge/` тоже появлялся/исчезал в нём. **Не коммитить, пока это не согласовано:** мой `git status` загрязнён чужими файлами. Их файл реализует §19.11/T11 «каждое опубликованное число + команда, которая перевыводит его сегодня» = **мой E-2 уже в работе у них** → следующий агент не должен дублировать. **Открыто владельцу:** (1) `tools/verification/` неслитая ветка vs активная сессия; (2) чужое в дереве (`experiments/prompt_robustness/**` изменён, `results/` untracked); (3) `KNOWN_ISSUES.md` перевалил 300 строк → нужна ротация; (4) ни одного коммита не сделано. **Открыто владельцу:** (1) запускать ли P1/E-1/E-5; (2) коммитить ли `frozen/`+отчёт — в дереве лежит **чужое** (`experiments/prompt_robustness/` изменён, `results/` untracked), не моё, не откатывал (§19.9); (3) разрешение на запись в реестр `P-###` вне репозитория. + +## [2026-10-03] Гейт отвечал rc=2 на любой вход + мёртвое число A10 без метки (Fixed) +**Status:** ⚠️ Частично. Код и тесты зелёные; полный прогон с нуля не делался (нет venv в worktree). +**Root Cause (двойной, независимый):** (1) `tools/verification/gates.py:322` звал `fn(**kw)` без отбора по сигнатуре гейта; MCP-тул `gate` шлёт superset полей всем четырём гейтам → `TypeError` → верхний обработчик печатал traceback и exit 2, то есть «гейт недоступен» вместо вердикта. Гипотеза «PATH/GitBash, как в AGENT_DIARY.md:451» — **REFUTED**: в `tools/verification/` нет резолвера bash. (2) T-17 висел открытым: число `orphan 30s→120ms` публиковалось в 3 файлах с меткой «подтверждено», хотя `claims_audit/RESULTS.md` A10 = NOT REPRODUCED (путь удалён по дизайну, R3TF). +**Fix:** `run()` выбирает параметры своего гейта по `inspect.signature`, отброшенные ключи возвращает как `ignored_kwargs`, а отсутствующий обязательный ключ отвечает `UNKNOWN`/rc=2 с перечнем `required_kwargs` вместо трайсбека. A10 помечен `SUPERSEDED` (`EXPERIMENTS_LOG.md:1320`, `experiments/README.md:27`, `experiments/lock_zombie/README.md:9`) со sha `3798d6a9` и ссылкой на R3TF `7974d981`. +**Guard:** `tests/test_gates_kwarg_contract.py` (10 тестов). **Negative control:** на исходном коде падают 4 из них, на исправленном 10/10 — тесты умеют падать. Red team: опечатка в ключе → `UNKNOWN`, а не тихий ALLOW; чужой гейт → `UNKNOWN`; отброшенный ключ по построению не может изменить вердикт, поэтому понижения нет — решение записано в коде. +**Self-caught:** первая запись в дневник переписала файл целиком (CRLF→LF, +928/−872 в diff) — поймано на `git diff --stat` до коммита, откачено через `git checkout --`, перезаписано с CRLF. +**Не сделано (честно):** портфолио (`exp-10`, репозиторий `MSPortfolio`) публикует то же число как действующую практику — отсюда не правится; полный `pytest tests/` и `verify_clean_state.sh` не гонял (в worktree нет venv, использована venv основного дерева). Реестр ловушек вне рабочего каталога — P-16 предложен в `tools/knowledge/PENDING_LEDGER.md`, не внесён. +## [2026-10-03] Коммит с `--no-verify`: обоснование (Fixed, с исключением) +**Решение владельца (цитата):** «(б) сначала короткая диагностика hook, при невозможности быстрого исправления — (а) `--no-verify` с обязательным обоснованием». +**Диагностика (что успел исключить, 1 команда):** `where bash` → единственный `bash` в PATH — `C:\Windows\System32\bash.exe`, то есть WSL-шим; при прямом вызове печатает баннер «WSL не установлен/требует wsl.exe --update» и **возвращается**, а не виснет. GitBash `C:\Program Files\Git\bin\bash.exe` существует и отвечает. **Значит блокер не в bash-резолвере** (и это опять опровергает версию из P-450/AGENT_DIARY:451 как достаточное объяснение). +**Что осталось невыясненным (честно):** `.githooks/pre-commit` при запуске `python -u` не печатает **ни байта** и не завершается за 100 с; после принудительного завершения остаются висящие `python.exe` (6 шт.). Локализовать, какой из 10 гейтов виснет, не успел: дальше требовалось бы трогать инфраструктуру хуков и убивать чужие процессы (в т.ч. параллельной сессии — §21.2 запрещает). +**Почему это не моя правка:** (1) коммит не проходит и у параллельной сессии — их же дневник, запись 03.10: «ни одного коммита не сделано»; (2) хук не печатает ни строки, то есть падает до вывода, а мои файлы — `tools/verification/gates.py` + новый тест; (3) `gates.py` хуком не вызывается (единственный импорт — мой тест). +**Чем заменена проверка:** `tests/test_gates_kwarg_contract.py` — 10 passed; **negative control:** 4 из 10 падают на коде до правки; `--selftest` PASSED; `tools/verification/heldout_cli_contract.py` — «every documented invocation works»; ruff чист; `scripts/check_parallel_sessions.py` — blocking=0. То есть пропущены именно 10 гейтов хука, а не тесты. +**Хеши закоммиченного:** `tools/verification/gates.py` blob `53bb7cfc15be27ebbb49bf0a65057bb8dfa2ec64`, `tests/test_gates_kwarg_contract.py` blob `b135dda9ba92b97827cb300ecabb7b8d517ef0cf`. +**Открыто:** починка самого pre-commit hook — отдельная задача; считать ветку готовой к мержу без прогонa хука нельзя. \ No newline at end of file diff --git a/EXPERIMENTS_LOG.md b/EXPERIMENTS_LOG.md index 7e8cc8b..78dc7bf 100644 --- a/EXPERIMENTS_LOG.md +++ b/EXPERIMENTS_LOG.md @@ -1317,7 +1317,8 @@ zombie_probe: holder pid=12684 alive=False -> STALE (после выхода с ``` **Before (та же сессия, старый код):** контеншн = 30.0s → RuntimeError (замерено в Exp B); orphan-кейс не детектился (30s → RuntimeError); free ~9ms; stale ~33ms. -**Вердикт:** ПОДТВЕРЖДЕНА. orphan: 30000ms → 120ms (terminate+steal, вкл. TerminateProcess реального python + ретрай-unlink); healthy: 30000ms RuntimeError → 1512ms LockBusyError (wait=1.5 в бенче; прод-дефолт 8.0s); free/stale без изменений (7/31ms). Дополнительно verified: после TerminateProcess реального python'а venvlauncher-обёртка умирает сама (никаких висящих процессов), lock перезаписывается нашим PID. +**Вердикт (2026-08-08):** ПОДТВЕРЖДЕНА. **Статус публикации (2026-10-03):** `SUPERSEDED` — измерено на `3798d6a9`, а путь достижимости удалён по дизайну: ORPHAN-ветка вырезана из прод-MCP по R3TF (`7974d981`, `src/core/indexing/database_lock.py:81-84`). Сегодняшней командой не воспроизводится (`experiments/claims_audit/RESULTS.md` A10 = NOT REPRODUCED), поэтому число остаётся записью прогона, а не текущим фактом. Не переписывать. +orphan: 30000ms → 120ms (terminate+steal, вкл. TerminateProcess реального python + ретрай-unlink); healthy: 30000ms RuntimeError → 1512ms LockBusyError (wait=1.5 в бенче; прод-дефолт 8.0s); free/stale без изменений (7/31ms). Дополнительно verified: после TerminateProcess реального python'а venvlauncher-обёртка умирает сама (никаких висящих процессов), lock перезаписывается нашим PID. **Урок:** (1) TerminateProcess синхронный, но файловый дескриптор lock'а умирающего процесса даёт PermissionError на unlink → нужен _unlink_with_retry (иначе краш в кейсе «только что убитый holder»); (2) venvlauncher: lock пишет РЕАЛЬНЫЙ python (os.getpid() внутри скрипта), terminate по pid из lock убивает именно держателя, обёртка умирает следом — прод-механизм работоспособен. --- diff --git a/experiments/README.md b/experiments/README.md index cff653b..92d18e3 100644 --- a/experiments/README.md +++ b/experiments/README.md @@ -24,7 +24,7 @@ | Concurrency | Гонки при замене примитива (§2.3) | [`concurrency/`](concurrency/) | ✅ 2026-08-11 | «0 errors» ≠ верные данные — стресс-тест на корректность | | Evalmut | Mutation testing для eval-градеров | [`evalmut/`](evalmut/) | ✅ 2026-08-14 | validate_scores: mutation score 8% → 100% (P-006) | | Root Cause Eval | Аудит root-cause предсказаний | [`root_cause_eval/`](root_cause_eval/) | 🟡 2026-07-22 | датасет инцидентов + gold standard | -| Lock-zombie | PID-lock self-healing (WS9) | [`lock_zombie/`](lock_zombie/) | ✅ 2026-08-08 | orphan 30s→120ms | +| Lock-zombie | PID-lock self-healing (WS9) | [`lock_zombie/`](lock_zombie/) | ✅ 2026-08-08 | orphan 30s→120ms — ⚠️ `SUPERSEDED`: измерено на `3798d6a9`, путь удалён по дизайну (R3TF) | | Late Enrichment | Late code chunking (WS3) | [`late_enrichment/`](late_enrichment/) | 🟡 исследование | imports=0.0 — находка, KNOWN_ISSUES | | Benchmark D | Контекстный бенчмарк (12 задач L3-L5) | [`benchmark2/`](benchmark2/) | ✅ 2026-08-08 | runner.py + tasks.jsonl + README | | Probes | Одноразовые пробы (без отчётов) | [`misc_probes/`](misc_probes/) | — | см. README папки | diff --git a/experiments/claims_audit/RESULTS.md b/experiments/claims_audit/RESULTS.md index a93d8f8..223c1e9 100644 --- a/experiments/claims_audit/RESULTS.md +++ b/experiments/claims_audit/RESULTS.md @@ -50,6 +50,18 @@ LockBusyError: PID lock still held by alive pid=548 after 8.0s подано как действующая практика — **это устаревшее утверждение, а не ошибка измерения**. Рекомендация: `SUPERSEDED` с ссылкой на R3TF, а не переписывать. +**Правка применена 2026-10-03** (закрывает T-17): + +| Файл:строка | Что стоит | +|---|---| +| `EXPERIMENTS_LOG.md:1320` | вердикт помечен `SUPERSEDED` + sha `3798d6a9` + ссылка на R3TF `7974d981` | +| `experiments/README.md:27` | та же метка в таблице экспериментов | +| `experiments/lock_zombie/README.md:9` | та же метка; сырой вывод прогона не тронут | + +**Осталось открытым (вне этого репо):** число публикуется также в портфолио (`exp-10`, +`MSPortfolio`) как действующая практика. Правка отсюда невозможна — портфолио живёт +в другом репозитории и отдаётся из задеплоенного снапшота. Статус: `OPEN`, не «сделано». + ### A11 — сканер вакуумных тестов сегодня молчит `exp_vacuous_scan.py:20` жёстко зашит на `experiments/tests` (каталога нет). Сегодня: diff --git a/experiments/lock_zombie/README.md b/experiments/lock_zombie/README.md index 7d5edce..021f511 100644 --- a/experiments/lock_zombie/README.md +++ b/experiments/lock_zombie/README.md @@ -6,7 +6,7 @@ **Статус:** ✅ Fixed 2026-08-08 (код+тесты; 1022 passed, ruff чист). **Артефакты:** -- `benchmark_selfhealing.py` — бенчмарк: orphan 30s→**120ms**, healthy 30s→1.5s soft, free/stale без изменений. +- `benchmark_selfhealing.py` — бенчмарк: orphan 30s→**120ms**, healthy 30s→1.5s soft, free/stale без изменений. ⚠️ **Статус числа (2026-10-03):** `SUPERSEDED` — измерено на `3798d6a9`; ORPHAN-ветка вырезана из прод-MCP по R3TF, поэтому прогон больше недостижим (`experiments/claims_audit/RESULTS.md` A10 = NOT REPRODUCED). Оставлено как запись прогона, не переписано. - `orphan_holder.py` / `spawn_orphan.py` / `zombie_probe.py` / `check_signals.py` / `probe_terminate.py` — пробы для live-теста Windows. - Фикс: `src/core/database_lock.py` — классификация holder'а (DEAD/HEALTHY/ORPHAN/AMBIGUOUS), TOCTOU-guard, retry-unlink; тесты `tests/test_database_lock_selfhealing.py` (+17). diff --git a/tools/knowledge/PATTERNS.md b/tools/knowledge/PATTERNS.md index 658000c..e960786 100644 --- a/tools/knowledge/PATTERNS.md +++ b/tools/knowledge/PATTERNS.md @@ -47,6 +47,7 @@ | **P-15** | Публикуемое число не воспроизводится сегодняшней командой: у нас 2 из 14; у коллеги 84 → 107 на неизменённом коммите | Число не привязано к существующей команде | `measured on , superseded by X`; различать «число ошибочно» и «число мертво (путь удалён по дизайну)» | §19.11 | `AGENT_DIARY.md:79-96` | | **P-16** | Аудит чужого/своего кода нашим же реестром даёт 5 совпадений из 5 — и обратный вывод в нашу пользу: **наш собственный код изначально был в том же состоянии**, мы вышли из него не знанием, а guard'ами | Реестр описывает прошлое, а не класс ошибок | Периодически применять реестр к постороннему проекту | §19.8 | `AGENT_DIARY.md` (guard-comparison 2026-09-30) | | **P-17** | Benchmark decay: опуликованное число разъезжается с реальностью молча (бейдж 1965 vs 2007, census intel 14 vs 20, tests 1180 vs 2007 — расхождения 42/6/827). Класс назван в литературе (GTM-Bench «Keeping a Benchmark Honest»), а не «наша ошибка в редактуре» | Число опубликовано без команды, которая его перевыводит; ручная правка без guard | tools/verification/verify_public_claims.py — каждое утверждение хранит свою команду; расхождение → rc=3; сравнение симметрично (ловит и завышение) | §19.11 | tools/knowledge/RESEARCH-benchmark-decay.md | +| **P-18** | **Отсутствие, выведенное из одного референта.** Дважды за одну сессию 2026-10-03: (1) «участника Muhammad Umair в треде нет» — выведено из поля `user.username` (`devomnitools`), не проверено `user.name` → ложный `REFUTED`; (2) «класс E3 не существует, 0%» — выведено из листинга директорий, без поиска по содержимому → нашлось за минуту в `KNOWN_ISSUES.md:179` и `experiments/2G_git_skill_vs_context/REPORT.md:60` | Отрицание неотличимо от «не нашёл в этом поле»; тихий ноль на входе ретривера | «X отсутствует» = ≥2 независимых источника отсутствия (разные поле / файл / формулировка запроса), иначе статус `NOT FOUND IN FIELD X`, а не `REFUTED`/`0%`. Guard: `scripts/verify_absence.py` (предложен) | §5 (Zero-Prompt: перед «не нашёл X» — ≥2 формулировки), §7.1, §19.4 | `experiments/audit_devto_judgements/HANDOFF.md`, эта сессия 03.10 | --- diff --git a/tools/knowledge/PENDING_LEDGER.md b/tools/knowledge/PENDING_LEDGER.md new file mode 100644 index 0000000..dfca453 --- /dev/null +++ b/tools/knowledge/PENDING_LEDGER.md @@ -0,0 +1,88 @@ +# PENDING LEDGER — предложения, а не правки + +Создан 2026-10-03 сессией в worktree `D:\Project\wt-gate-fix` (ветка +`fix/gate-kwargs-and-claims-t17`). В `D:\Project\MSCodeBase` в это время работала +другая сессия (`.agent_task_state.md` от 03.10, незакоммиченный `KNOWN_ISSUES.md`). +По §21.3 реестры принадлежат одной сессии, поэтому здесь **предложения**, а не правки. + +## P-1 → `tools/knowledge/PATTERNS.md`: **P-18** (не P-16 — номер занят) + +**Класс ошибки, повторённый дважды за одну сессию:** утверждение об отсутствии, +выведенное из одного референта. + +- Случай 1: «Muhammad Umair отсутствует в треде» — выведено из поля `user.username` + (`devomnitools`), не проверено `user.name`. Ложное `REFUTED`. +- Случай 2: «E3 (last-message ≠ answer) — дефицит 0%, ничего нет» — выведено из + листинга директорий, без поиска по содержимому. Найдено за минуту: + `KNOWN_ISSUES.md:179` и `experiments/2G_git_skill_vs_context/REPORT.md:60`. + +**Правило:** «X отсутствует / не найдено» требует ≥2 независимых источника +отсутствия; иначе статус `NOT FOUND IN FIELD X`, а не `REFUTED` / `0%`. +Точное правило уже есть в AGENTS.md §5 (Zero-Prompt, MCP: «перед "не нашёл X" — +≥2 формулировки поиска») — P-18 это его механический guard. + +**Решено владельцем 2026-10-03:** писать в `tools/knowledge/PATTERNS.md` внутри +worktree, системный реестр `~/.config/opencode/` не трогать. +**Записано:** P-18 в `tools/knowledge/PATTERNS.md` (P-16 уже занят паттерном §19.8). + +**Guard (предложен, не реализован):** `scripts/verify_absence.py [...]` +— падает (`rc=2`), если референтов меньше двух, и печатает, чем они различаются +(поле / файл / формулировка). + +## P-2 → `KNOWN_ISSUES.md:179` (зонд в E3, не новый эксперимент) + +Запись уже есть и уже верна: судья разбирает вердикт по **первому** слову, поэтому +«actually correct, final answer: correct» инвертируется; `uncertain` недостижим. +Это зеркало инцидента из треда Tom Jones (там runner брал последнее сообщение вместо +ответа; здесь парсер берёт первое упоминание вместо финального решения). Второе место: +`experiments/2G_git_skill_vs_context/REPORT.md:60` — харнесс возвращает только +финальное сообщение, поэтому не может доказать B-агентов. Оба — долг с известным +адресом, а не дефицит. + +## P-3 → пересечение с задачей параллельной сессии + +Их запись `KNOWN_ISSUES.md` (2026-10-03, P1) предлагает третье состояние `UNSEEN` +для гейтов. Моя правка в `tools/verification/gates.py` — про другое: контракт входа +(kwargs), из-за которого MCP-тул `gate` отвечал `rc=2` на любой вход. +Пересечения по коду нет, но списки гейтов будут меняться в одном файле — стоит +согласовать порядок, чтобы не было двух правок одного файла в одном коммите. + +То же касается `tools/verification/verify_public_claims.py`: он уже в репо и это +именно тот T11-инструмент, который параллельная сессия начала делать раньше меня +(см. `AGENT_DIARY.md:866`). **Я не дублирую его работу** — ручные метки `SUPERSEDED` +(A10) сделаны потому, что инструмент сегодня проверяет только 4 числа (badge'и), +и A10 среди них не числится. Их задача — расширить его до всех публикуемых чисел. + +## P-6 → СВЕЖИЙ ИНСТАНС P-17 (новый паттерн не нужен): 2 из 4 чисел badge + +`tools/verification/verify_public_claims.py` запускается и падает: + +``` +[WRONG] readme.test_count_arch README.md states 2007; live now 2027 — understates by 20 +[WRONG] wisdom.test_count WISDOM.md states 2007; live now 2027 — understates by 20 +CLAIM CHECK FAILED: 2 of 4 published numbers contradict reality +``` + +(в выводе `readme.test_badge` числится 2001 — расхождение между двумя копиями одного +факта внутри самого README.) + +**Решение владельца 2026-10-03:** канонический источник — `verify_public_claims.py`; +**значения 2007 в `README.md` и `WISDOM.md` не трогать руками**. Это уже +зарегистрированный класс `P-17` (benchmark decay), новый паттерн не вводится. + +**Про вклад этой сессии:** мои `tests/test_gates_kwarg_contract.py` добавляют 10 тестов +и уже учтены в `live now 2027`. Если инструмент собирает `tests/`, то без моего файла +было бы 2017, и разрыв README был 16, а не 26. **Это вывод из арифметики, а не измеренная +команда** — перепроверьте, прежде чем считать 2027 эталоном. + +## P-4 → E2-guard (идея, не реализовано) + +«Правка matcher'а/гейта без приложенного replay-дифа = BLOCK» — блокирующего гейта +сейчас нет. Образец прогона уже существует: +`experiments/1V_memory_contamination/flip_ledger_REDTEAM_2026-08-16.json`. + +## P-5 → числа вне репо + +`experiments/claims_audit/RESULTS.md`: A10 (`orphan 30s→120ms`) помечен `SUPERSEDED` +в этом репо (3 файла), но всё ещё публикуется в портфолио (`exp-10`, репозиторий +`MSPortfolio`, отдаётся из задеплоенного снапшота). Правка отсюда невозможна. \ No newline at end of file From 0c6d0ff13b8f7d73a9b5944d2c2fb268a353209c Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Sat, 3 Oct 2026 20:20:08 +0300 Subject: [PATCH 03/10] fix(hooks): make a slow gate return a verdict instead of a traceback The hook printed a gate's result only after it finished, and an unhandled TimeoutExpired killed the whole hook with no verdict. A slow gate therefore looked like a broken hook, which is how commits ended up bypassing the gates with --no-verify. Now each gate announces itself before it runs, a timeout is a verdict naming the gate, and both budgets are configurable. Measured: gate-zero runs the full pytest tests/ and reached 36% in 600s here, so the 900s cap could not pass ANY commit. Fast mode is explicit (MSCB_PRECOMMIT_FAST=1), prints a warning, and leaves the full run to CI. Tests: 7, incl. a negative control that loads the pre-fix hook from git and requires it to be unable to answer a timeout. Committed with MSCB_PRECOMMIT_FAST=1 (see AGENT_DIARY.md, 2026-10-03). --- .githooks/pre-commit | 47 +++++++-- scripts/verify_diary.py | 14 ++- tests/test_precommit_hook_contract.py | 133 ++++++++++++++++++++++++++ 3 files changed, 183 insertions(+), 11 deletions(-) create mode 100644 tests/test_precommit_hook_contract.py diff --git a/.githooks/pre-commit b/.githooks/pre-commit index 3d983a3..57841a8 100755 --- a/.githooks/pre-commit +++ b/.githooks/pre-commit @@ -16,10 +16,15 @@ MSCodeBase pre-commit hook — автоматическая проверка п 8. check_known_issues — §4.8 R4: размер ≤ 300 строк + архивация старых записей """ +import os import subprocess import sys from pathlib import Path +# Per-gate budget. 900s is the historical value (documented flakiness at 300s); +# overridable only to make the timeout path testable without a 15-minute wait. +GATE_TIMEOUT = int(os.environ.get("MSCB_PRECOMMIT_GATE_TIMEOUT", "900")) + # §9 п.9 (ENCODING SAFETY): при выводе emoji-строк из stdout скриптов # (например, «📊 Итог: 20 ✅ / 1 ❌») в cp1251-консоль падает @@ -42,7 +47,7 @@ def find_project_root() -> Path | None: return None -def run_script(script_path: str, label: str) -> bool: +def run_script(script_path: str, label: str, extra_args: list[str] | None = None) -> bool: """Запускает скрипт и возвращает True если успешно.""" project_root = find_project_root() if project_root is None: @@ -54,10 +59,13 @@ def run_script(script_path: str, label: str) -> bool: print(f" ⏭️ {label}: скрипт не найден ({script})") return True + # Печатаем ДО запуска: гейт с большим бюджетом (verify_diary тянет полный + # pytest) иначе выглядит как зависание — и его обходят через --no-verify. + print(f" ⏳ {label}: выполняется (бюджет {GATE_TIMEOUT}s)…", flush=True) # §5.16: Popen + communicate (не capture_output) — защита от pipe-deadlock # в фоновых потоках; encoding="utf-8" — декодирование stdout в utf-8. proc = subprocess.Popen( - [sys.executable, str(script)], + [sys.executable, str(script), *(extra_args or [])], cwd=str(project_root), stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, @@ -69,22 +77,43 @@ def run_script(script_path: str, label: str) -> bool: # нагрузкой) — кап 120s давал флаки TimeoutExpired на коммитах (2026-08-08); # 300→900 (2026-08-24): сюита выросла (live-sync + predict-наборы), 300s # начал флакать при параллельной нагрузке. - stdout, _ = proc.communicate(timeout=900) + try: + stdout, _ = proc.communicate(timeout=GATE_TIMEOUT) + except subprocess.TimeoutExpired: + # Раньше это было необработанное исключение: хук умирал с трейсбеком и + # БЕЗ вердикта — его путали с «гейт сломан» и обходили через --no-verify. + # Таймаут — это вердикт: называем гейт, который не ответил. + proc.kill() + proc.communicate() + print(f" ❌ {label}: TIMEOUT >{GATE_TIMEOUT}s — гейт не ответил, вердикта нет", + flush=True) + return False if proc.returncode != 0: - print(f" ❌ {label}: exit {proc.returncode}") + print(f" ❌ {label}: exit {proc.returncode}", flush=True) if stdout: for line in stdout.splitlines()[-10:]: print(f" {line}") return False - print(f" ✅ {label}: OK") + print(f" ✅ {label}: OK", flush=True) return True def main(): - print("🔍 MSCodeBase pre-commit checks:") + print("🔍 MSCodeBase pre-commit checks:", flush=True) all_ok = True - all_ok &= run_script("scripts/verify_diary.py", "verify_diary") + # gate-zero тянет полный `pytest tests/`. Измерено 2026-10-03 на этой машине: + # 36% за 600s, то есть ~1670s — бюджет гейта (900s) не выдерживает НИ ОДНОГО + # коммита. Раньше это выглядело как «хук висит», и коммиты уходили в + # --no-verify. Поэтому fast-режим: полный прогон остаётся за CI, а локальный + # коммит говорит ВСЛУХ, что именно пропущено. + fast = os.environ.get("MSCB_PRECOMMIT_FAST") == "1" + diary_args = ["--skip-gate-zero"] if fast else None + if fast: + print(" ⚠️ FAST: полный `pytest tests/` (gate-zero) ПРОПУЩЕН — " + "покрытие этого коммита обязана взять CI.", flush=True) + + all_ok &= run_script("scripts/verify_diary.py", "verify_diary", diary_args) all_ok &= run_script("scripts/stale_detector.py", "stale_detector") all_ok &= run_script("scripts/check_tool_names.py", "check_tool_names") all_ok &= run_script("scripts/negative_controls_runner.py", "negative_controls") @@ -97,9 +126,9 @@ def main(): all_ok &= run_script("scripts/ruff_gate.py", "ruff_gate") if not all_ok: - print("\n❌ Pre-commit checks FAILED. Исправьте ошибки перед коммитом.") + print("\n❌ Pre-commit checks FAILED. Исправьте ошибки перед коммитом.", flush=True) sys.exit(1) - print("\n✅ All pre-commit checks passed.") + print("\n✅ All pre-commit checks passed.", flush=True) sys.exit(0) diff --git a/scripts/verify_diary.py b/scripts/verify_diary.py index 7cd622e..d07de33 100644 --- a/scripts/verify_diary.py +++ b/scripts/verify_diary.py @@ -470,14 +470,24 @@ def gate_zero_full_suite() -> Tuple[bool, str]: # 120s кап флаки при нагрузке (pytest ~108-130s) — 300s запас (2026-08-08); # 300→900 (2026-08-24): сюита выросла (1499+ live-sync/predict-наборы), # даже в CI clean-state pytest идёт ~171s — 300s флакал при параллельной нагрузке. - stdout, _ = proc.communicate(timeout=900) + # 2026-10-03: на этой машине полный прогон НЕ укладывается в 900s — измерено + # 36% за 600s, то есть ~1670s. Кап стал гарантированным таймаутом, из-за чего + # коммит нельзя было сделать ни через хук, ни честно. Бюджет сделан + # настраиваемым и печатается вместе с вердиктом; значение по умолчанию + # прежнее, чтобы CI-значение не подменялось молча. + budget = int(os.environ.get("MSCB_GATE_ZERO_TIMEOUT", "900")) + stdout, _ = proc.communicate(timeout=budget) output = stdout.decode("utf-8", errors="replace").strip() # Извлекаем итоговую строку lines = [l for l in output.split("\n") if "passed" in l or "failed" in l] summary = lines[-1] if lines else output[-200:] return proc.returncode == 0, summary except subprocess.TimeoutExpired: - return False, "TIMEOUT: pytest tests/ > 900s" + return False, ( + f"TIMEOUT: pytest tests/ > {budget}s — это не провал тестов, это нехватка бюджета. " + f"Либо поднимите MSCB_GATE_ZERO_TIMEOUT (измерено: нужно ~1700s), " + f"либо запустите с --skip-gate-zero и оставьте gate-zero за CI." + ) except Exception as e: return False, f"ERROR: {e}" diff --git a/tests/test_precommit_hook_contract.py b/tests/test_precommit_hook_contract.py new file mode 100644 index 0000000..80d5554 --- /dev/null +++ b/tests/test_precommit_hook_contract.py @@ -0,0 +1,133 @@ +"""The pre-commit hook must return a verdict for every gate, including a timeout. + +Observed 2026-10-03: the hook produced no output at all and had to be bypassed +with `--no-verify`. Two causes, both fixed here: + 1. nothing was printed before a gate ran, so a slow gate looked like a hang; + 2. `proc.communicate(timeout=...)` raised an UNHANDLED TimeoutExpired, so a + slow gate killed the hook with a traceback and no verdict — which reads + exactly like "the gate is broken". + +The negative control at the bottom runs the pre-fix hook from git and asserts it +raises instead of answering. If that assertion ever stops failing, the test has +stopped testing the fix. +""" +from __future__ import annotations + +import importlib.util +import os +import subprocess +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +HOOK = ROOT / ".githooks" / "pre-commit" + + +def _load_hook(path: Path, timeout: int | None, name: str): + if timeout is None: + os.environ.pop("MSCB_PRECOMMIT_GATE_TIMEOUT", None) + else: + os.environ["MSCB_PRECOMMIT_GATE_TIMEOUT"] = str(timeout) + spec = importlib.util.spec_from_loader(name, importlib.machinery.SourceFileLoader(name, str(path))) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +def _gate(tmp_path: Path, body: str) -> str: + p = tmp_path / "gate.py" + p.write_text(body, encoding="utf-8") + return str(p) # absolute: `project_root / absolute` == absolute + + +def test_gate_that_passes(tmp_path): + mod = _load_hook(HOOK, 60, "hook_pass") + assert mod.run_script(_gate(tmp_path, "print('ok')"), "g") is True + + +def test_gate_that_fails(tmp_path): + mod = _load_hook(HOOK, 60, "hook_fail") + assert mod.run_script(_gate(tmp_path, "raise SystemExit(1)"), "g") is False + + +def test_missing_gate_is_skipped_not_failed(tmp_path): + mod = _load_hook(HOOK, 60, "hook_missing") + assert mod.run_script(str(tmp_path / "nope.py"), "g") is True + + +def test_timeout_returns_a_verdict_instead_of_raising(tmp_path, capsys): + mod = _load_hook(HOOK, 2, "hook_timeout") + slow = _gate(tmp_path, "import time; time.sleep(60)") + assert mod.run_script(slow, "slowgate") is False + out = capsys.readouterr().out + assert "TIMEOUT" in out and "slowgate" in out + assert "Traceback" not in out + + +def test_running_line_is_printed_before_the_gate_finishes(tmp_path, capsys): + """The reason the hook looked dead: nothing was printed until the end.""" + mod = _load_hook(HOOK, 60, "hook_progress") + mod.run_script(_gate(tmp_path, "print('ok')"), "progressgate") + out = capsys.readouterr().out + assert out.index("progressgate") < out.index("OK") + + +def test_gate_budget_is_configurable(): + """Without the env override the timeout path could not be tested at all.""" + mod = _load_hook(HOOK, 7, "hook_budget") + assert mod.GATE_TIMEOUT == 7 + os.environ.pop("MSCB_PRECOMMIT_GATE_TIMEOUT", None) + mod2 = _load_hook(HOOK, None, "hook_budget_default") + assert mod2.GATE_TIMEOUT == 900 + + +class _FakeProc: + """communicate() times out once, exactly like a gate that never answers.""" + + def __init__(self): + self.returncode = None + self.killed = False + self._raised = False + + def communicate(self, timeout=None): + if not self._raised: + self._raised = True + raise subprocess.TimeoutExpired(cmd="gate", timeout=timeout) + return "", None + + def kill(self): + self.killed = True + + +def _old_hook_module(): + """Load the committed (pre-fix) hook. Returns (module, holder_path).""" + holder = ROOT / "build" / f"hook-negative-control-{os.getpid()}" + holder.parent.mkdir(parents=True, exist_ok=True) + # P-14 live: the old hook is full of Cyrillic, and `text=True` would decode + # git's UTF-8 output through the cp1251 console codepage. + src = subprocess.run(["git", "-C", str(ROOT), "show", "HEAD:.githooks/pre-commit"], + capture_output=True, encoding="utf-8", errors="replace", + timeout=60) + assert src.returncode == 0, src.stderr + holder.write_text(src.stdout, encoding="utf-8") + mod = _load_hook(holder, None, "hook_prefix") + assert mod.find_project_root() == ROOT + return mod, holder + + +def test_prefix_hook_raises_on_timeout_negative_control(tmp_path): + """NEGATIVE CONTROL: the pre-fix hook could not answer a timeout at all. + + Driven by a fake Popen rather than a real sleep, because the pre-fix hook + hardcodes its 900s budget and ignores the env override — a real sleep would + only prove that 60 < 900. + """ + mod, holder = _old_hook_module() + try: + mod.subprocess.Popen = lambda *a, **k: _FakeProc() + with pytest.raises(subprocess.TimeoutExpired): + mod.run_script(_gate(tmp_path, "pass"), "slowgate") + finally: + holder.unlink(missing_ok=True) + assert not holder.exists(), "temp copy of the old hook left behind" From 8f82ed056d07038d892f885ce07ac8b7a81a5e50 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Sat, 3 Oct 2026 20:20:56 +0300 Subject: [PATCH 04/10] fix(ruff-gate): resolve a ruff that can actually run The gate invoked \python -m ruff\, but the ruff installed in this venv (0.15.22) ships no __main__: the call printed nothing and exited 1. The gate was therefore permanently red and mute, which reads as broken lint rather than as a broken runner - and a permanently red gate is a gate people disable. ruff_cmd() now resolves a live tool and proves it with --version before use. A non-zero exit with EMPTY output is reported as a broken runner, not as a lint verdict. stderr is no longer discarded. Tests: 6, incl. the gate failing on a real diagnostic and the module form being rejected when it is dead. --- scripts/ruff_gate.py | 57 ++++++++++++++--- tests/test_gates_kwarg_contract.py | 2 +- tests/test_ruff_gate_contract.py | 99 ++++++++++++++++++++++++++++++ 3 files changed, 149 insertions(+), 9 deletions(-) create mode 100644 tests/test_ruff_gate_contract.py diff --git a/scripts/ruff_gate.py b/scripts/ruff_gate.py index 567ef36..455bfa8 100644 --- a/scripts/ruff_gate.py +++ b/scripts/ruff_gate.py @@ -14,6 +14,8 @@ from __future__ import annotations +import os +import shutil import subprocess import sys from pathlib import Path @@ -26,20 +28,54 @@ pass +def ruff_cmd() -> list[str] | None: + """Resolve a ruff that actually runs. + + 2026-10-03: `python -m ruff` in this venv printed NOTHING and exited 1 — the + installed `ruff` package has no `__main__`. The gate called exactly that, so + it was permanently red and mute: it looked like "lint is broken", and the + only visible effect was that commits could not be made. Prefer the console + script; verify with `--version` instead of trusting the import. + """ + exe = shutil.which("ruff") + if not exe: + # console scripts live next to the interpreter in a venv + cand = Path(sys.executable).parent / ("ruff.exe" if os.name == "nt" else "ruff") + if cand.exists(): + exe = str(cand) + if exe: + try: + probe = subprocess.run([exe, "--version"], capture_output=True, + encoding="utf-8", errors="replace", timeout=60) + if probe.returncode == 0 and "ruff" in (probe.stdout or "").lower(): + return [exe, "check"] + except (OSError, subprocess.SubprocessError): + pass + # last resort: module form, but only if it can report a version + try: + probe = subprocess.run([sys.executable, "-m", "ruff", "--version"], + capture_output=True, encoding="utf-8", + errors="replace", timeout=60) + if probe.returncode == 0 and "ruff" in (probe.stdout or "").lower(): + return [sys.executable, "-m", "ruff", "check"] + except (OSError, subprocess.SubprocessError): + pass + return None + + def main() -> int: project_root = Path(__file__).resolve().parent.parent - try: - import ruff # noqa: F401 - except ImportError: - print(" ⚠️ ruff не установлен — пропуск (CI всё равно проверяет)") + cmd = ruff_cmd() + if cmd is None: + print(" ⚠️ ruff не найден ни одним из способов — пропуск (CI всё равно проверяет)") return 0 proc = subprocess.Popen( - [sys.executable, "-m", "ruff", "check", "src/", "tests/"], + [*cmd, "src/", "tests/"], cwd=str(project_root), stdout=subprocess.PIPE, - stderr=subprocess.DEVNULL, + stderr=subprocess.STDOUT, encoding="utf-8", errors="replace", creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0), @@ -52,10 +88,15 @@ def main() -> int: return 1 if proc.returncode != 0: - print(f" ❌ ruff: exit {proc.returncode}") - if stdout: + print(f" ❌ ruff: exit {proc.returncode} ({' '.join(cmd)})") + if stdout and stdout.strip(): for line in stdout.splitlines()[-15:]: print(f" {line}") + else: + # A tool that fails without saying anything is not a lint verdict. + print(" ПУСТОЙ вывод ruff при ненулевом коде — это не lint-ошибка, " + "а сломанный запуск инструмента. Проверьте, что ruff вообще " + "исполняется: `ruff --version`.") return 1 print(" ✅ ruff: OK") return 0 diff --git a/tests/test_gates_kwarg_contract.py b/tests/test_gates_kwarg_contract.py index b135dda..49dc650 100644 --- a/tests/test_gates_kwarg_contract.py +++ b/tests/test_gates_kwarg_contract.py @@ -91,4 +91,4 @@ def test_selftest_still_passes(): proc = subprocess.run([sys.executable, "-B", str(GATES), "--selftest"], capture_output=True, text=True, cwd=ROOT, timeout=180) assert proc.returncode == 0, proc.stdout[-2000:] - assert "SELFTEST PASSED" in proc.stdout \ No newline at end of file + assert "SELFTEST PASSED" in proc.stdout diff --git a/tests/test_ruff_gate_contract.py b/tests/test_ruff_gate_contract.py new file mode 100644 index 0000000..e8d7693 --- /dev/null +++ b/tests/test_ruff_gate_contract.py @@ -0,0 +1,99 @@ +"""The ruff gate must be able to FAIL, and must be able to say why. + +Two defects closed here, both found on 2026-10-03: + 1. it invoked `python -m ruff`, which in this venv prints nothing and exits 1 + (the installed `ruff` has no `__main__`) — the gate was permanently red and + mute, which reads as "lint is broken", not "the runner is broken"; + 2. a non-zero exit with no output produced the bare line "exit 1" and nothing + else, so the failure was indistinguishable from a lint error. + +A gate that cannot fail is worse than no gate, and a gate that fails for an +unexplained reason gets disabled within a week. Both are pinned here. +""" +from __future__ import annotations + +import importlib.util +import subprocess +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] + + +def _load(): + spec = importlib.util.spec_from_file_location("ruff_gate", ROOT / "scripts" / "ruff_gate.py") + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +class _Stub: + def __init__(self, rc, out=""): + self.returncode = rc + self._out = out + self.killed = False + + def communicate(self, timeout=None): + return self._out, None + + +def test_resolved_ruff_actually_runs(): + """Guards the module-form regression: the resolver must return a live tool.""" + mod = _load() + cmd = mod.ruff_cmd() + assert cmd is not None, "no working ruff found by any resolution path" + probe = subprocess.run([*cmd[: cmd.index("check")], "--version"], capture_output=True, + encoding="utf-8", errors="replace", timeout=60) + assert probe.returncode == 0 and "ruff" in probe.stdout.lower() + + +def test_module_form_is_not_used_when_it_is_dead(monkeypatch): + """`python -m ruff` here exits 1 with no output; it must not be selected.""" + mod = _load() + monkeypatch.setattr(mod.shutil, "which", lambda _n: None) + monkeypatch.setattr(Path, "exists", lambda _self: False) + calls = [] + + def fake_run(cmd, **kw): + calls.append(cmd) + return subprocess.CompletedProcess(cmd, 1, "", "") + + monkeypatch.setattr(mod.subprocess, "run", fake_run) + assert mod.ruff_cmd() is None + assert calls, "resolver never even tried the module form" + + +def test_gate_fails_and_prints_the_diagnostics(monkeypatch, capsys): + mod = _load() + monkeypatch.setattr(mod, "ruff_cmd", lambda: ["ruff", "check"]) + monkeypatch.setattr(mod.subprocess, "Popen", + lambda *a, **k: _Stub(1, "F401 unused import\n--> tests/x.py:1")) + assert mod.main() == 1 + out = capsys.readouterr().out + assert "F401" in out and "tests/x.py" in out + + +def test_silent_failure_is_named_as_a_broken_runner(monkeypatch, capsys): + mod = _load() + monkeypatch.setattr(mod, "ruff_cmd", lambda: ["ruff", "check"]) + monkeypatch.setattr(mod.subprocess, "Popen", lambda *a, **k: _Stub(1, "")) + assert mod.main() == 1 + out = capsys.readouterr().out + assert "ПУСТОЙ вывод" in out + + +def test_clean_run_is_green(monkeypatch, capsys): + mod = _load() + monkeypatch.setattr(mod, "ruff_cmd", lambda: ["ruff", "check"]) + monkeypatch.setattr(mod.subprocess, "Popen", lambda *a, **k: _Stub(0, "All checks passed")) + assert mod.main() == 0 + assert "OK" in capsys.readouterr().out + + +def test_missing_ruff_is_advisory_not_green(monkeypatch, capsys): + """No tool at all is NOT a pass — it is a stated skip, and it says so.""" + mod = _load() + monkeypatch.setattr(mod, "ruff_cmd", lambda: None) + assert mod.main() == 0 + assert "пропуск" in capsys.readouterr().out + assert sys.platform # keep the import meaningful for linters From ac0bdc92d6f42bad81949ce6bb91c07842184380 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Sat, 3 Oct 2026 20:21:32 +0300 Subject: [PATCH 05/10] fix(judge): one verdict parser, last-match in both copies KNOWN_ISSUES recorded this class as fixed on 2026-09-27, but only for f5_judged_run.py. reconstruct_judge_cot.py kept a first-match substring parser, so a self-correcting judge was recorded INVERTED and silently, and the script had no test at all. Two of two locations carried the logic; one was buggy. Both now import scripts/judge_verdict.py (last-match contract, validated on 1014 judge sessions), so a third copy cannot appear. Tests: 11 for the previously untested script, with a control that re-runs the old parser and requires it to disagree. Verified: 7 fail before the fix. --- scripts/f5_judged_run.py | 25 ++---- scripts/judge_verdict.py | 42 +++++++++ scripts/reconstruct_judge_cot.py | 18 ++-- tests/test_reconstruct_judge_verdict.py | 109 ++++++++++++++++++++++++ 4 files changed, 165 insertions(+), 29 deletions(-) create mode 100644 scripts/judge_verdict.py create mode 100644 tests/test_reconstruct_judge_verdict.py diff --git a/scripts/f5_judged_run.py b/scripts/f5_judged_run.py index 281e96b..5e4ce50 100644 --- a/scripts/f5_judged_run.py +++ b/scripts/f5_judged_run.py @@ -245,24 +245,13 @@ def _arm_context(arm: str, results: list[dict], gold: str) -> str: raise SystemExit(f"unknown arm {arm}") -def _parse_verdict(text: str) -> str: - """Explicit-final contract (validated on 1014 judge sessions, 2026-09-27). - - A reasoning judge may hesitate mid-text ("looks incorrect ... actually - correct"). The FINAL decision is the LAST one stated: last JSON - "verdict" match wins; otherwise the last verdict-word mention wins. - Measured on judged_cot_backfill.json: last==final 842/1014 vs - first-match 820/1014; on 61 flip sessions last==final 53/61 (87%) - vs first==final 34/61 (56%). No explicit "final verdict:" marker - exists in the wild (0/61), so last-match IS the contract. - """ - matches = re.findall(r'"verdict"\s*:\s*"?(correct|incorrect|uncertain)"?', text or "", re.I) - if matches: - return matches[-1].lower() - hits = re.findall(r"\b(correct|incorrect|uncertain)\b", text or "", re.I) - if hits: - return hits[-1].lower() - return "uncertain" +try: # both invocation shapes: `python scripts/x.py` and import-by-path in tests + from scripts.judge_verdict import parse_verdict as _parse_verdict +except ImportError: # pragma: no cover + from judge_verdict import parse_verdict as _parse_verdict +# The explicit-final contract and its 1014-session validation now live in +# scripts/judge_verdict.py, shared with reconstruct_judge_cot.py, which used to +# carry a first-match copy of this parser. def main() -> int: diff --git a/scripts/judge_verdict.py b/scripts/judge_verdict.py new file mode 100644 index 0000000..61379b8 --- /dev/null +++ b/scripts/judge_verdict.py @@ -0,0 +1,42 @@ +"""One implementation of the judge's verdict contract. + +Why this file exists. Two scripts parsed the same judge's answer independently: + + - `scripts/f5_judged_run.py` took the LAST verdict word (fixed 2026-09-27); + - `scripts/reconstruct_judge_cot.py` took the FIRST, and its fallback tested + substrings with "incorrect" before "correct". So a self-correcting judge + ("looks incorrect ... actually correct, final answer: correct") was recorded + INVERTED, silently: every counter stayed green and `uncertain` was + unreachable whenever any verdict word appeared. + +The contract (measured on judged_cot_backfill.json, 1014 judge sessions): +the FINAL decision is the LAST one stated. Last JSON "verdict" match wins; +otherwise the last verdict-word mention wins. last==final 842/1014 vs +first-match 820/1014; on the 61 flip sessions last==final 53/61 (87%) vs +first==final 34/61 (56%). No explicit "final verdict:" marker exists in the +wild (0/61), so last-match IS the contract. + +Imported by both consumers so that a third copy cannot appear. +""" +from __future__ import annotations + +import re + +VERDICT_WORDS = ("correct", "incorrect", "uncertain") +VERDICT_ALT = "|".join(VERDICT_WORDS) + +# \b boundaries matter: "incorrect" contains "correct", and "correctly" is not a +# verdict. The old substring fallback could not tell them apart. +VERDICT_RE = re.compile(rf"\b({VERDICT_ALT})\b", re.I) +JSON_VERDICT_RE = re.compile(rf'"verdict"\s*:\s*"?({VERDICT_ALT})"?', re.I) + + +def parse_verdict(text: str) -> str: + """Return 'correct' | 'incorrect' | 'uncertain' from a judge answer.""" + matches = JSON_VERDICT_RE.findall(text or "") + if matches: + return matches[-1].lower() + hits = VERDICT_RE.findall(text or "") + if hits: + return hits[-1].lower() + return "uncertain" \ No newline at end of file diff --git a/scripts/reconstruct_judge_cot.py b/scripts/reconstruct_judge_cot.py index c65ea36..ce96e44 100644 --- a/scripts/reconstruct_judge_cot.py +++ b/scripts/reconstruct_judge_cot.py @@ -30,23 +30,19 @@ WORKDIR = "D:/Project/MSCodeBase/experiments/4A_unit_of_return/results/f5judged/work" FROZEN = ROOT / "experiments" / "4A_unit_of_return" / "frozen" / "f5" / "queries.jsonl" -VERDICT_RE = re.compile(r"\b(correct|incorrect|uncertain)\b", re.I) -JSON_VERDICT_RE = re.compile(r'"verdict"\s*:\s*"?(correct|incorrect|uncertain)"?', re.I) +try: # both invocation shapes: `python scripts/x.py` and import-by-path in tests + from scripts.judge_verdict import VERDICT_RE, parse_verdict as _parse_verdict +except ImportError: # pragma: no cover + from judge_verdict import VERDICT_RE, parse_verdict as _parse_verdict def _norm(s: str) -> str: return " ".join((s or "").split()) -def _parse_verdict(text: str) -> str: - m = JSON_VERDICT_RE.search(text or "") - if m: - return m.group(1).lower() - low = (text or "").lower() - for v in ("incorrect", "correct", "uncertain"): - if v in low: - return v - return "uncertain" +# Verdict parsing used to live here as a FIRST-match substring scan, which +# inverted self-correcting judges. The contract and its measurement now live in +# scripts/judge_verdict.py, shared with f5_judged_run.py. def _split_prompt(user_text: str) -> tuple[str, str]: diff --git a/tests/test_reconstruct_judge_verdict.py b/tests/test_reconstruct_judge_verdict.py new file mode 100644 index 0000000..9eba319 --- /dev/null +++ b/tests/test_reconstruct_judge_verdict.py @@ -0,0 +1,109 @@ +"""Verdict parsing in scripts/reconstruct_judge_cot.py — previously untested. + +KNOWN_ISSUES recorded this class as Fixed on 2026-09-27, but only for +scripts/f5_judged_run.py. reconstruct_judge_cot.py kept its own first-match +substring parser, so a self-correcting judge was recorded INVERTED and silently: +every counter stayed green. Two locations carried the logic; one was buggy. + +The control at the bottom re-implements the old parser and asserts it disagrees — +so this file cannot pass while the old behaviour is still in place. +""" +from __future__ import annotations + +import importlib.util +import sys +from pathlib import Path + +import pytest + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +if str(PROJECT_ROOT) not in sys.path: + sys.path.insert(0, str(PROJECT_ROOT)) + + +def _load(name: str): + spec = importlib.util.spec_from_file_location( + name, PROJECT_ROOT / "scripts" / f"{name}.py" + ) + mod = importlib.util.module_from_spec(spec) + sys.modules[name] = mod + spec.loader.exec_module(mod) + return mod + + +recon = _load("reconstruct_judge_cot") +f5 = _load("f5_judged_run") + +# The inversion case from KNOWN_ISSUES, verbatim in shape. +INVERSION = ( + "At first this looks incorrect, but checking the arithmetic again it is " + "actually correct, final answer: correct" +) + + +def test_self_correction_is_not_inverted(): + assert recon._parse_verdict(INVERSION) == "correct" + + +def test_last_json_verdict_wins(): + text = '{"verdict": "incorrect"}\nOn reflection: {"verdict": "correct"}' + assert recon._parse_verdict(text) == "correct" + text_rev = '{"verdict": "correct"}\nOn reflection: {"verdict": "incorrect"}' + assert recon._parse_verdict(text_rev) == "incorrect" + + +def test_no_verdict_word_is_uncertain(): + assert recon._parse_verdict("I don't know.") == "uncertain" + + +def test_empty_input_is_uncertain(): + assert recon._parse_verdict("") == "uncertain" + assert recon._parse_verdict(None) == "uncertain" + + +def test_substring_is_not_a_verdict(): + """'incorrectly'/'correctly' are not verdicts; the old scan could not tell.""" + assert recon._parse_verdict("the answer was stated incorrectly") == "uncertain" + assert recon._parse_verdict("he answered correctly") == "uncertain" + + +def test_plain_single_verdicts_unchanged(): + assert recon._parse_verdict("correct") == "correct" + assert recon._parse_verdict("incorrect") == "incorrect" + assert recon._parse_verdict("uncertain") == "uncertain" + + +def test_mentions_helper_still_exported(): + """reconstruct_judge_cot.py:133 uses VERDICT_RE to list the words mentioned.""" + text = "first incorrect, then correct" + assert sorted({w.lower() for w in recon.VERDICT_RE.findall(text)}) == ["correct", "incorrect"] + + +def test_both_scripts_share_one_implementation(): + """N of M: 2 of 2 locations carried this logic, so they must be one object.""" + assert recon._parse_verdict is f5._parse_verdict + + +@pytest.mark.parametrize("text,expected", [ + (INVERSION, "correct"), + ("looks correct at first ... should be uncertain", "uncertain"), + ("he answered correctly", "uncertain"), +]) +def test_old_first_match_parser_disagrees(text, expected): + """NEGATIVE CONTROL: the pre-fix parser must fail these, not agree.""" + import re + + json_re = re.compile(r'"verdict"\s*:\s*"?(correct|incorrect|uncertain)"?', re.I) + + def old_parse(t: str) -> str: + m = json_re.search(t or "") + if m: + return m.group(1).lower() + low = (t or "").lower() + for v in ("incorrect", "correct", "uncertain"): + if v in low: + return v + return "uncertain" + + assert old_parse(text) != expected + assert recon._parse_verdict(text) == expected From 51d941d881fc11b9caed724a5ff5c29752ea9d09 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Sat, 3 Oct 2026 20:25:38 +0300 Subject: [PATCH 06/10] docs(ledger): propose KNOWN_ISSUES entries for the owning session KNOWN_ISSUES.md is uncommitted in the parallel session's tree, and check_parallel_sessions blocked the commit - correctly. So this branch leaves that file alone and records both intended entries here instead: the correction to the judge-verdict entry, and the three measured reasons a commit could not be made. Also notes that the ruff gate was the mirror image of the UNSEEN proposal: it had seen its condition and could not say so. --- tools/knowledge/PENDING_LEDGER.md | 53 ++++++++++++++++++++++++++++++- 1 file changed, 52 insertions(+), 1 deletion(-) diff --git a/tools/knowledge/PENDING_LEDGER.md b/tools/knowledge/PENDING_LEDGER.md index dfca453..8c2e821 100644 --- a/tools/knowledge/PENDING_LEDGER.md +++ b/tools/knowledge/PENDING_LEDGER.md @@ -85,4 +85,55 @@ CLAIM CHECK FAILED: 2 of 4 published numbers contradict reality `experiments/claims_audit/RESULTS.md`: A10 (`orphan 30s→120ms`) помечен `SUPERSEDED` в этом репо (3 файла), но всё ещё публикуется в портфолио (`exp-10`, репозиторий -`MSPortfolio`, отдаётся из задеплоенного снапшота). Правка отсюда невозможна. \ No newline at end of file +`MSPortfolio`, отдаётся из задеплоенного снапшота). Правка отсюда невозможна. +## P-7 → `KNOWN_ISSUES.md`: запись о вердите судьи была неполной (предложение, файл занят) + +Запись `## 2026-09-27 — F5 judge verdict parsing takes first regex match` помечена +`✅ Fixed`, но фикс 2026-09-27 применён **только к `scripts/f5_judged_run.py`**. +Второе место той же логики — `scripts/reconstruct_judge_cot.py:41-49` — осталось +с first-match и подстрочным fallback, то есть **с ровно тем багом, который запись +описывает**, и без единого теста. + +Знаменатель: логика разбора вердикта живёт в **2 из 2** мест; баговая была **1 из 2**. +Уже исправлено в ветке `fix/gate-kwargs-and-claims-t17`: обе копии импортируют +`scripts/judge_verdict.py`; guard `tests/test_reconstruct_judge_verdict.py` (11 тестов). + +**Что добавить в запись (готовая формулировка):** + +> **Дополнение 2026-10-03:** фикс применён только к `f5_judged_run.py`; вторая копия +> (`reconstruct_judge_cot.py`) оставалась багованной и без тестов. Исправлено: один +> общий `scripts/judge_verdict.py`, last-match, тест 11. **Осторожно:** +> `test_real_backfill_agreement` сверяет исправленный парсер с `verdict_final`, +> сгенерированным **старым** парсером, — после починки эти числа нужно пересчитать. + +## P-8 → `KNOWN_ISSUES.md`: новая запись о том, что коммит был невозможен (предложение) + +Файл не менялся в этой ветке намеренно: он незакоммичен в `D:/Project/MSCodeBase` +(чужая сессия), и гейт `check_parallel_sessions` это заблокировал — правильно. + +**Готовая запись (сжатая под лимит 300 строк):** + +> ## 2026-10-03 — Коммит был невозможен: три независимые причины в гейтах (Fixed) +> - **Симптом:** `git commit` не проходит; хук молчит и выглядит «зависшим», поэтому +> коммиты уходят в `--no-verify`. Не deadlock, а три дефекта, каждый маскировал следующий. +> - **(1) Гейт без видимости и без вердикта:** результат печатался только после гейта, а +> необработанный `TimeoutExpired` убивал хук трейсбеком **без вердикта**. Теперь +> `⏳