diff --git a/.claude/skills/codebase-index/references/response-contract.md b/.claude/skills/codebase-index/references/response-contract.md index 059d87a..798f152 100644 --- a/.claude/skills/codebase-index/references/response-contract.md +++ b/.claude/skills/codebase-index/references/response-contract.md @@ -18,7 +18,10 @@ Each result can contain: - `elided_lines` `recommended_reads` is the read plan. Start with its first one to three entries -and use exact line ranges. +and use exact line ranges. An entry with `truncated: true` was capped at the +definition head (`max_read_lines`, default 120); `line_end_full` gives the real +extent. Read the capped range first and continue only when the head is not +enough. `pagination.has_more` and `pagination.next_offset` indicate additional results. Prefer a more specific command or a larger token budget before paging. diff --git a/.codex/skills/codebase-index/references/response-contract.md b/.codex/skills/codebase-index/references/response-contract.md index 059d87a..798f152 100644 --- a/.codex/skills/codebase-index/references/response-contract.md +++ b/.codex/skills/codebase-index/references/response-contract.md @@ -18,7 +18,10 @@ Each result can contain: - `elided_lines` `recommended_reads` is the read plan. Start with its first one to three entries -and use exact line ranges. +and use exact line ranges. An entry with `truncated: true` was capped at the +definition head (`max_read_lines`, default 120); `line_end_full` gives the real +extent. Read the capped range first and continue only when the head is not +enough. `pagination.has_more` and `pagination.next_offset` indicate additional results. Prefer a more specific command or a larger token budget before paging. diff --git a/.github/FUNDING.yml b/.github/FUNDING.yml deleted file mode 100644 index 365d548..0000000 --- a/.github/FUNDING.yml +++ /dev/null @@ -1,8 +0,0 @@ -github: - patreon: # placeholder - ko_fi: # placeholder - tidelift: # placeholder - community_bridge: # placeholder - liberapay: # placeholder - issuehunt: # placeholder - custom: # placeholder diff --git a/.github/ISSUE_TEMPLATE/benchmark_report.yml b/.github/ISSUE_TEMPLATE/benchmark_report.yml new file mode 100644 index 0000000..f8d2669 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/benchmark_report.yml @@ -0,0 +1,62 @@ +name: Benchmark report +description: Share a reproduced, failed, or new benchmark run (good or bad numbers both welcome) +title: "[Benchmark]: " +labels: ["benchmark"] +body: + - type: markdown + attributes: + value: | + Independent runs are how this project keeps its numbers honest. Unflattering results are as useful as flattering ones. + + - type: dropdown + id: kind + attributes: + label: What did you run? + options: + - tests/eval/run_eval.py (retrieval eval, own corpus) + - tests/eval/run_eval.py (shipped query sets) + - tests/benchmark_public.py (synthetic suite) + - tests/benchmark_honest.py (index vs rg+window) + - Something else (describe below) + validations: + required: true + + - type: input + id: version + attributes: + label: codebase-index version / commit + placeholder: "1.9.0 or git SHA" + validations: + required: true + + - type: input + id: corpus + attributes: + label: Corpus + description: Public repository URL + commit SHA, or language / size if private. + placeholder: "https://github.com/pallets/flask @ d318b68 — Python, ~18k LOC" + validations: + required: true + + - type: textarea + id: command + attributes: + label: Exact commands + render: shell + validations: + required: true + + - type: textarea + id: results + attributes: + label: Results table / raw output + description: Paste the pooled table and significance rows as printed. Attach the JSON if large. + render: markdown + validations: + required: true + + - type: textarea + id: notes + attributes: + label: Observations + description: Anything surprising, queries that failed, or where the baseline beat the index. diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index 4eacf48..e7f3348 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -1,74 +1,74 @@ -name: Bug Report -description: Report a bug or unexpected behavior +name: Bug report +description: Something is wrong, crashes, or returns the wrong result title: "[Bug]: " labels: ["bug", "triage"] body: - type: markdown attributes: value: | - Thanks for taking the time to report a bug. Please fill in as much of the form below as possible. + Thanks. A reproducible report gets fixed fastest. Do not paste real secrets — the tool redacts snippets, your logs may not. - type: input id: version attributes: label: codebase-index version - description: What version of codebase-index are you running? - placeholder: "0.1.0" + description: "`pip show codebase-index` (there is no --version flag)" + placeholder: "1.9.0" validations: required: true - type: input - id: python-version + id: env attributes: - label: Python version - description: What Python version are you using? - placeholder: "3.12.3" + label: Python and OS + placeholder: "Python 3.12.3 · macOS 14.4 / Ubuntu 24.04 / Windows 11" validations: required: true - - type: input - id: os + - type: dropdown + id: surface attributes: - label: Operating system - description: What OS and version are you using? - placeholder: "macOS 14.4, Ubuntu 24.04, Windows 11" + label: Surface + options: + - CLI + - Claude Code skill / plugin + - Codex CLI + - OpenCode + - MCP server + - Installer scripts + - Other validations: required: true - type: textarea id: what-happened attributes: - label: What happened? - description: Describe the bug and what you expected to happen. + label: What happened, and what did you expect? placeholder: | - 1. Ran `codebase-index index` on my project - 2. Expected: index completes successfully - 3. Actual: got error "..." + Ran `codebase-index refs open_session --json` on . Expected the call in ctx.py; got an empty list with coverage.partial=false. validations: required: true - type: textarea - id: reproduction + id: repro attributes: label: Steps to reproduce - description: Provide minimal steps to reproduce the issue. + description: Ideally against a public repository or a tiny snippet we can index. placeholder: | - 1. `codebase-index init` - 2. `codebase-index index` - 3. `codebase-index search "query"` - 4. See error + 1. git clone https://github.com/... && cd ... + 2. codebase-index index + 3. codebase-index search "..." --json validations: required: true - type: textarea - id: logs + id: doctor attributes: - label: Relevant log output - description: Paste any relevant terminal output or error messages. + label: "`codebase-index doctor` output" render: shell - type: textarea - id: additional + id: logs attributes: - label: Additional context - description: Add any other context, screenshots, or configuration details. + label: Relevant output (`--json` where applicable) or traceback + render: shell diff --git a/.github/ISSUE_TEMPLATE/config.yml b/.github/ISSUE_TEMPLATE/config.yml new file mode 100644 index 0000000..00e12b5 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/config.yml @@ -0,0 +1,8 @@ +blank_issues_enabled: false +contact_links: + - name: Questions and ideas + url: https://github.com/denfry/codebase-index/discussions + about: Ask how to use the tool, discuss design, or share what you built with it. + - name: Security vulnerability + url: https://github.com/denfry/codebase-index/security/advisories/new + about: Report privately. Please do not open a public issue (see SECURITY.md). diff --git a/.github/ISSUE_TEMPLATE/feature_request.yml b/.github/ISSUE_TEMPLATE/feature_request.yml index b3f039c..ab8c587 100644 --- a/.github/ISSUE_TEMPLATE/feature_request.yml +++ b/.github/ISSUE_TEMPLATE/feature_request.yml @@ -1,56 +1,50 @@ -name: Feature Request -description: Suggest an idea or improvement for codebase-index +name: Feature request +description: Propose an improvement or a new capability title: "[Feature]: " labels: ["enhancement", "triage"] body: - - type: markdown - attributes: - value: | - Thanks for suggesting a feature. Please describe your idea and its use case. - - type: textarea id: problem attributes: - label: Problem statement - description: Is your feature request related to a problem? Describe it. - placeholder: "I'm frustrated when..." + label: Problem + description: What are you trying to do with your agent that the tool makes hard today? + placeholder: "When I ask Claude Code about ..., codebase-index returns ..., so the agent ends up reading ..." validations: required: true - type: textarea - id: solution + id: proposal attributes: - label: Proposed solution - description: Describe what you want to happen. - placeholder: "It would be great if codebase-index could..." + label: Proposal + description: What should happen instead? A CLI example or a sketch of the JSON is ideal. validations: required: true - - type: textarea - id: alternatives - attributes: - label: Alternatives considered - description: Describe any alternative solutions or workarounds you've tried. - - type: dropdown - id: scope + id: area attributes: - label: Scope - description: Which area does this feature primarily affect? + label: Area options: - - Indexing (discovery, parsing, storage) - - Retrieval (search, ranking, fusion) - - CLI (commands, output format) - - Skill (SKILL.md, allowed-tools, workflow) - - Configuration (config.json, hooks) + - Retrieval quality (ranking, fusion, budget) + - Graph extraction (edges, impact, paths) + - Language support + - Indexing (discovery, parsing, freshness) + - CLI / output format + - MCP server + - Claude Code / Codex / OpenCode integration + - Benchmarks / evaluation - Documentation - - Testing / CI - Other validations: required: true - type: textarea - id: additional + id: evidence + attributes: + label: How would we know it worked? + description: Retrieval changes need a measurable effect (see tests/eval). A query and the file you expected is enough to start. + + - type: textarea + id: alternatives attributes: - label: Additional context - description: Add any other context, mockups, or examples. + label: Alternatives or workarounds you tried diff --git a/.github/ISSUE_TEMPLATE/language_support.yml b/.github/ISSUE_TEMPLATE/language_support.yml new file mode 100644 index 0000000..4b8485b --- /dev/null +++ b/.github/ISSUE_TEMPLATE/language_support.yml @@ -0,0 +1,48 @@ +name: Language support +description: Request or offer symbol/graph extraction for a language +title: "[Language]: " +labels: ["language support"] +body: + - type: input + id: language + attributes: + label: Language and file extensions + placeholder: "Swift — .swift" + validations: + required: true + + - type: dropdown + id: current + attributes: + label: Current status (see docs/LANGUAGES.md) + options: + - Not detected at all + - Tier C (line chunks + FTS only) + - Tier B (generic Tree-sitter, no graph edges) + - Tier A but extraction is wrong or incomplete + validations: + required: true + + - type: input + id: grammar + attributes: + label: tree-sitter-language-pack grammar name + description: "`tree_sitter_language_pack.get_language('')` must succeed." + placeholder: "swift" + + - type: textarea + id: expectations + attributes: + label: What should be extracted? + description: Definition kinds (function/class/...), calls, imports, inheritance — with a 10–20 line sample file. + render: text + validations: + required: true + + - type: checkboxes + id: offer + attributes: + label: Contribution + options: + - label: I can open a PR adding the `LangSpec` and fixture (see docs/DEVELOPMENT.md → "Add a language") + - label: I can provide a public repository in this language for the retrieval eval diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index b7d25b8..8b54ad6 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -1,38 +1,21 @@ -## Description +## What and why -Describe your changes and their impact. + -Fixes #(issue) +## How I verified it -## Type of change +```bash +# commands you ran, e.g. +pytest tests/test_xxx.py -q --no-cov +python tests/eval/run_eval.py --ablate # required for any ranking change; paste the table below +``` -- [ ] Bug fix (non-breaking change that fixes an issue) -- [ ] New feature (non-breaking change that adds functionality) -- [ ] Breaking change (fix or feature that would cause existing functionality to not work as expected) -- [ ] Documentation update -- [ ] Refactoring (no functional change) -- [ ] CI / build / tooling change + ## Checklist -- [ ] I have read the [CONTRIBUTING.md](../CONTRIBUTING.md) guide -- [ ] My code follows the project's style conventions (`ruff check`, `ruff format`) -- [ ] I have added tests that prove my fix is effective or that my feature works -- [ ] All tests pass (`pytest`) -- [ ] Type checking passes (`mypy`, if applicable) -- [ ] I have updated the documentation accordingly -- [ ] I have updated `CHANGELOG.md` under `[Unreleased]` -- [ ] My commits follow [Conventional Commits](https://www.conventionalcommits.org/) - -## Testing - -Describe how you tested your changes: - -```bash -# Example: -pytest tests/test_my_new_feature.py -codebase-index index --root tests/fixtures/sample_repo -codebase-index search "test query" -``` - -## Screenshots (if applicable) +- [ ] `pytest`, `ruff check src tests`, `mypy src/codebase_index` pass +- [ ] `python scripts/sync_skill_copies.py --check` passes (if `skill_template/` changed) +- [ ] Goldens regenerated intentionally and the diff explained (if `--json` / MCP output changed) +- [ ] `CHANGELOG.md` updated under `[Unreleased]` (user-visible changes only; no version bump) +- [ ] No secrets, generated indexes, or local config committed diff --git a/.github/SUPPORT.md b/.github/SUPPORT.md new file mode 100644 index 0000000..769cfd3 --- /dev/null +++ b/.github/SUPPORT.md @@ -0,0 +1,25 @@ +# Support + +**Something broken?** Run `codebase-index doctor` first; it catches most +install, config and stale-index problems. Then open a +[bug report](https://github.com/denfry/codebase-index/issues/new?template=bug_report.yml) +with the doctor output. + +**How do I…?** Check the [FAQ](../docs/FAQ.md), the +[quick start](../docs/QUICKSTART.md) and the +[installation guide](../docs/INSTALLATION.md). For anything else, start a +[discussion](https://github.com/denfry/codebase-index/discussions). + +**Agent integration** (Claude Code, Codex CLI, OpenCode, MCP clients): see +[docs/MCP.md](../docs/MCP.md) and [docs/INSTALLATION.md](../docs/INSTALLATION.md). +Ready-to-copy client configs are templates; if one does not work with the +current version of your client, that is a bug report we want. + +**Benchmarks**: reproduce a published run or add your own corpus with +[tests/eval](../tests/eval/README.md) and post the numbers with the +[benchmark report](https://github.com/denfry/codebase-index/issues/new?template=benchmark_report.yml) +template. + +**Security**: never in a public issue. See [SECURITY.md](../SECURITY.md). + +There is no paid support and no SLA; this is a volunteer-maintained project. diff --git a/.github/labels.yml b/.github/labels.yml new file mode 100644 index 0000000..c9da59a --- /dev/null +++ b/.github/labels.yml @@ -0,0 +1,63 @@ +# Repository labels. Apply with: scripts/apply_labels.sh (requires `gh` auth). +# Format: name, color (hex, no #), description. + +- name: bug + color: d73a4a + description: Something is broken or returns wrong results +- name: enhancement + color: a2eeef + description: New capability or improvement +- name: documentation + color: 0075ca + description: Docs, examples, demo, FAQ +- name: good first issue + color: 7057ff + description: Small, well-scoped, no prior context needed +- name: help wanted + color: 008672 + description: Maintainers would welcome a PR here +- name: benchmark + color: fbca04 + description: Corpora, metrics, baselines, reproductions of published runs +- name: language support + color: 1d76db + description: Tree-sitter LangSpecs, tiers, per-language extraction +- name: graph extraction + color: 5319e7 + description: Edge kinds, resolution, impact/path/describe +- name: retrieval quality + color: 0e8a16 + description: Ranking, fusion, priors, token budgeting +- name: integrations + color: c5def5 + description: Skills, hooks, installers, agent wiring +- name: mcp + color: bfd4f2 + description: MCP server, tools, envelopes, client configs +- name: claude-code + color: e99695 + description: Claude Code skill / plugin specifics +- name: codex + color: f9d0c4 + description: Codex CLI integration specifics +- name: opencode + color: fef2c0 + description: OpenCode integration specifics +- name: release + color: 006b75 + description: Versioning, changelog, packaging, publishing +- name: security + color: b60205 + description: Exclusion gates, redaction, network policy (report vulns privately) +- name: triage + color: ededed + description: Needs a maintainer look before it is actionable +- name: question + color: d876e3 + description: Usage question; may be moved to Discussions +- name: duplicate + color: cfd3d7 + description: Already tracked elsewhere +- name: wontfix + color: ffffff + description: Out of scope for this project diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e0d031a..c313134 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -23,6 +23,27 @@ jobs: run: mypy src/codebase_index - name: Skill copies in sync run: python scripts/sync_skill_copies.py --check + - name: Version mirrors agree + run: python scripts/check_versions.py + - name: Markdown links resolve + run: python scripts/check_links.py + + package: + needs: lint + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5 + with: + python-version: "3.12" + - run: python -m pip install --upgrade pip + - run: pip install -e ".[dev,build]" + - name: Build sdist + wheel + run: python -m build + - name: Twine check + run: twine check dist/* + - name: Clean-venv install smoke (init -> index -> search) + run: python scripts/release_smoke.py test: needs: lint diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 5756979..9ebc840 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -16,6 +16,13 @@ jobs: with: python-version: "3.12" - run: pip install -e ".[dev,build]" + - name: Version mirrors agree with the tag + run: | + python scripts/check_versions.py + if [ "${GITHUB_REF_TYPE}" = "tag" ]; then + ver=$(python -c "import re,pathlib;print(re.search(r'__version__ = \"([^\"]+)\"', pathlib.Path('src/codebase_index/__init__.py').read_text()).group(1))") + [ "${GITHUB_REF_NAME}" = "v${ver}" ] || { echo "tag ${GITHUB_REF_NAME} != package version v${ver}"; exit 1; } + fi - name: Test gate run: pytest - name: Build @@ -43,10 +50,17 @@ jobs: with: name: dist path: dist + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5 + with: + python-version: "3.12" + - name: Release notes from CHANGELOG.md + run: python scripts/release_notes.py --out release_notes.md - name: Create GitHub release uses: softprops/action-gh-release@3bb12739c298aeb8a4eeaf626c5b8d85266b0e65 # v2 with: files: dist/* + body_path: release_notes.md + # Appends the auto-generated "What's Changed" PR list after the changelog body. generate_release_notes: true pypi-publish: diff --git a/.opencode/skills/codebase-index/references/response-contract.md b/.opencode/skills/codebase-index/references/response-contract.md index 059d87a..798f152 100644 --- a/.opencode/skills/codebase-index/references/response-contract.md +++ b/.opencode/skills/codebase-index/references/response-contract.md @@ -18,7 +18,10 @@ Each result can contain: - `elided_lines` `recommended_reads` is the read plan. Start with its first one to three entries -and use exact line ranges. +and use exact line ranges. An entry with `truncated: true` was capped at the +definition head (`max_read_lines`, default 120); `line_end_full` gives the real +extent. Read the capped range first and continue only when the head is not +enough. `pagination.has_more` and `pagination.next_offset` indicate additional results. Prefer a more specific command or a larger token budget before paging. diff --git a/CHANGELOG.md b/CHANGELOG.md index 64f7fa8..f65060c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,69 @@ All notable changes to this project are documented here. The format is based on ## [Unreleased] +### Added + +- **Public baseline benchmark.** `tests/eval/run_baselines.py` compares the index with + a disciplined `rg` + 80-line-window agent and with repo-map-style context on + Flask, Gson and Fastify at pinned commits, using git-derived ground truth, one + tokenizer on every side, and paired bootstrap / permutation significance. The + logged run is committed under `tests/eval/results/`: pooled over 450 queries, + hit@3 0.547 vs 0.304 and MRR 0.456 vs 0.263 (p < 0.001) at 3.8k vs 3.5k context + tokens per query. `scripts/gen_benchmark_chart.py` renders the chart from the log. +- **Reproducible demo.** `examples/demo/run_demo.sh|.ps1` run Find / Trace / Predict + on `pallets/flask` at a pinned commit; `EXPECTED_OUTPUT.md` is the captured run and + `assets/demo-terminal.svg` is rendered from it (`scripts/gen_terminal_svg.py`). + `docs/DEMO.md` documents GIF/video recording. +- **Release and CI gates.** `scripts/check_versions.py` (package, plugin manifest, + lock tag, skill stamps and changelog must agree), `scripts/check_links.py` + (relative Markdown links must resolve), `scripts/release_notes.py` (GitHub release + body comes from the changelog section and an empty section fails the release), and + a `package` CI job that builds, `twine check`s and runs the clean-venv install smoke + on every pull request. +- **Contributor onboarding.** `docs/DEVELOPMENT.md`, rewritten `CONTRIBUTING.md`, + benchmark-report and language-support issue templates, `SUPPORT.md`, a labels + manifest, and `docs/COMMUNITY_AUDIT.md` / `docs/COMMUNITY_LAUNCH.md`. + +### Changed + +- **Read plan is bounded.** `recommended_reads` entries are capped at + `retrieval.max_read_lines` (default 120) and carry `truncated: true` plus + `line_end_full` when capped. A symbol-aligned chunk can be a whole class; on the + public baseline benchmark the uncapped read plan cost 6.8k tokens per query against + 3.5k for grep, the capped one 3.8k, with identical ranking quality. Set the option to + 0 for the previous behaviour. +- **README, docs and examples show real output only.** Fabricated tables (an + `AuthService.ts` example with a `Score` column, an invented `doctor` transcript, + the mock-up `assets/demo.png`) are replaced by output captured on Flask. + Duplicate pages (`DATABASE_SCHEMA`, `RETRIEVAL_PIPELINE`, `docs/SECURITY`) are + redirect stubs; `SCHEMA.md` now matches `storage/schema.sql`. +- **Benchmark headline.** The "13× fewer tokens on a 55k LOC Java repo" figure is + withdrawn: the repository is private and the accounting was asymmetric. The + defensible claim is the public baseline run above. +- `SECURITY.md` points to GitHub private vulnerability reporting and states the + supported line as 1.9.x. + +### Fixed + +- **MCP server did not start on mcp 2.x.** The SDK renamed `FastMCP` to + `MCPServer` and removed the old import path, so `codebase-index mcp` reported + "needs the optional extra" even with the extra installed, and the MCP tests + skipped silently in CI because the skip guard wrapped our own module's import. + The server now imports `MCPServer` first and falls back to `FastMCP` on 1.x + (verified against mcp 1.29 and 2.1); the tests skip only when the SDK itself is + missing. +- **`codebase-index mcp --root ` was rejected.** Every client template + used that form, but `--root` was only a global option (`codebase-index --root + mcp`). The subcommand now accepts it as well. +- **Plugin wrappers refused four documented commands.** `bin/cbx` and `bin/cbx.ps1` + whitelisted ten subcommands while the skill allowed fourteen; `architecture`, + `diff-impact`, `path` and `describe` now work from the plugin. A parity test pins + the two lists together. +- **Benchmark leakage.** `gen_queries` documented that changelog-style files were + excluded from the evaluation corpus, but the harness never applied the list, and + Flask-style `CHANGES.rst` was not covered. Both are fixed; `CHANGES*`, `HISTORY*`, + `NEWS*` and `RELEASE_NOTES*` are refused as answers and excluded from the corpus. + ## [1.9.0] - 2026-09-02 ### Added diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 16837e0..5b10c9d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,162 +1,120 @@ # Contributing -Thank you for your interest in contributing to `codebase-index`. This project is a local-first Claude Code Skill for codebase indexing, and we welcome contributions of all kinds. +Thanks for helping make `codebase-index` better. It is a local-first retrieval +and code-graph layer for AI coding agents, and it lives or dies by three +invariants: -## Development Setup +1. **Retrieval quality is measured, not asserted.** +2. **The default path stays local and fails closed at security boundaries.** +3. **Machine-readable contracts (CLI `--json`, MCP envelopes) stay stable.** -### Prerequisites +The hands-on setup, test, benchmark and recipe guide is +[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md). This page covers *what* to work on +and *how* to get it merged. -- Python 3.11 or later -- `pipx` or `uv` for package management (optional but recommended) -- Git - -### Clone and Install +## Quick start ```bash git clone https://github.com/denfry/codebase-index.git cd codebase-index - -# Using uv (recommended) -uv sync --all-extras - -# Or using pip -pip install -e ".[dev,embeddings-local,watch]" -``` - -### Run Tests - -```bash -pytest -pytest --cov=src/codebase_index --cov-report=term-missing -``` - -### Run Linting - -```bash -ruff check src/ tests/ -ruff format --check src/ tests/ -``` - -### Run Type Checking - -```bash +python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate +pip install -e ".[dev,mcp]" +pytest # coverage gate: 80% +ruff check src tests mypy src/codebase_index +python scripts/sync_skill_copies.py --check ``` -## Commit Style - -We follow [Conventional Commits](https://www.conventionalcommits.org/en/v1.0.0/): - -``` -(): - -[optional body] - -[optional footer(s)] -``` - -Types: `feat`, `fix`, `docs`, `style`, `refactor`, `test`, `chore`, `ci` - -Examples: -``` -feat(retrieval): add RRF fusion for hybrid search -fix(storage): handle FTS5 trigger recreation on schema change -docs(readme): add comparison table and FAQ -test(parsers): add tree-sitter symbol extraction tests for Go -``` - -## Branch Naming - -Use the pattern `/`: - -- `feat/hybrid-search` -- `fix/fts5-trigger-recreate` -- `docs/readme-comparison` -- `test/treesitter-go` - -## Fork and Pull Request Workflow - -1. Fork the repository and clone your fork as `origin`. -2. Add the canonical repository as `upstream`: +Those four commands are exactly what CI runs. `ruff format` is not enforced. +Set `CBX_NO_SKILL_AUTO_UPDATE=1` before running the CLI inside the checkout. + +## Where help is wanted + +Issues are labelled by area so you can pick something that matches your +interests. `good first issue` and `help wanted` are the entry points. + +| Label | What it covers | Start in | +|---|---|---| +| `benchmark` | New corpora, metrics, baselines, reproductions of published runs | `tests/eval/`, `tests/benchmark_*.py` | +| `language support` | New Tier-A `LangSpec`s, better queries for existing grammars | `src/codebase_index/parsers/languages.py` | +| `graph extraction` | New edge kinds, better resolution, edge confidence | `src/codebase_index/parsers/treesitter.py`, `src/codebase_index/graph/` | +| `retrieval quality` | Ranking signals, fusion, priors, budgeting | `src/codebase_index/retrieval/` | +| `integrations` | Claude Code, Codex CLI, OpenCode, hooks, installers | `src/codebase_index/scaffold.py`, `adapters/`, `skill_template/` | +| `mcp` | MCP tools, envelopes, client configs | `src/codebase_index/mcp/server.py`, `docs/MCP.md` | +| `documentation` | Docs, examples, demo, FAQ | `docs/`, `examples/` | + +Reproducing a benchmark on a repository we have not seen is one of the most +useful contributions there is. Use the *Benchmark report* issue template even +when the result is unflattering. + +## What makes a good PR + +- **One concern per PR**, on a branch named `/` + (`feat/kotlin-imports`, `fix/fts-trigger-recreate`). +- **Tests.** New behaviour has tests; bug fixes have a regression test. The + coverage gate is 80% and CI runs on Linux, macOS and Windows with Python + 3.11–3.13. +- **Contracts.** If `--json` or an MCP payload changes, regenerate the goldens + intentionally (`UPDATE_GOLDEN=1 pytest tests/test_cli_golden.py + tests/test_mcp_golden.py`) and explain the diff. Removing a field or changing + a type bumps `MCP_SCHEMA_VERSION`. +- **Changelog.** Add user-visible changes under `[Unreleased]` in + `CHANGELOG.md`. Do not bump the version; maintainers do that at release time. +- **Skill copies.** After touching `src/codebase_index/skill_template/`, run + `python scripts/sync_skill_copies.py` and commit the regenerated copies. +- **Conventional Commits** for messages: `feat(retrieval): ...`, + `fix(storage): ...`, `docs(readme): ...`. + +### Retrieval-quality PRs specifically + +A ranking change is accepted on evidence, not on a plausible story. + +1. Put the new signal behind a boolean flag on `RetrievalTuning` + (`src/codebase_index/retrieval/tuning.py`) and add it to `ABLATABLE` in + `tests/eval/run_eval.py`. `RetrievalTuning.baseline()` must stay unchanged. +2. Run `python tests/eval/run_eval.py --ablate` on the shipped query sets *and* + at least one external corpus (`tests/eval/gen_queries.py` builds one from any + git repository in two commands). +3. Paste the pooled table and the paired significance row for your flag into + the PR description. Ship on `p < 0.05` for the pooled set; otherwise leave the + signal off by default or drop it. +4. Report mean tokens and latency alongside quality. Winning MRR while doubling + the token bill is not a win. + +## Fork and pull request workflow + +1. Fork, clone your fork as `origin`, add the canonical repo as `upstream`: ```bash git remote add upstream https://github.com/denfry/codebase-index.git git fetch upstream - ``` - -3. Create a focused branch from the latest `upstream/main`: - - ```bash - git fetch upstream git switch -c / upstream/main ``` -4. Keep unrelated changes in separate branches and pull requests. Do not commit - generated files, build artifacts, local configuration, credentials, or - editor-specific files. -5. Rebase onto the latest `upstream/main`, run the required checks, and push the - branch to your fork before opening the pull request. -6. Open the pull request against `denfry/codebase-index:main`. Explain the - problem, solution, verification performed, and any compatibility or migration - impact. - -Do not commit directly to `main`, merge `main` into a feature branch, or change -the project version unless a maintainer explicitly requests it. Add user-visible -changes to `CHANGELOG.md` under `[Unreleased]`; maintainers choose the release -version according to the project's versioning policy. - -The version lives in one place: `src/codebase_index/__init__.py` (`__version__`). -`pyproject.toml` reads it via hatch dynamic versioning. After changing the -version or anything under `src/codebase_index/skill_template/`, run -`python scripts/sync_skill_copies.py` to regenerate the committed skill copies -and version stamps; CI rejects the PR if they drift -(`python scripts/sync_skill_copies.py --check`). - -## Test Requirements - -- All new features must include tests. -- All bug fixes must include a regression test. -- Aim for >80% line coverage on new code. -- Tests must pass on Python 3.11+. -- Use the fixture repository under `tests/fixtures/sample_repo/` for integration tests. - -## Documentation Requirements - -- New commands or flags must be documented in the README and relevant docs files. -- New configuration options must be documented in `docs/INSTALLATION.md` and `examples/config.example.json`. -- API changes must be noted in `CHANGELOG.md` under `[Unreleased]`. - -## Pull Request Checklist - -Before submitting a PR, ensure: - -- [ ] Tests pass: `pytest` -- [ ] Linting passes: `ruff check src/ tests/` -- [ ] Formatting is correct: `ruff format src/ tests/` -- [ ] Type checking passes: `mypy src/codebase_index` (if applicable) -- [ ] CHANGELOG.md is updated under `[Unreleased]` -- [ ] The project version is unchanged unless a maintainer requested a bump -- [ ] Documentation is updated (README, docs/, examples/) -- [ ] Commit messages follow Conventional Commits -- [ ] No secrets or credentials are committed -- [ ] The PR description explains the change and links to any related issues - -## Code Review - -- All PRs require at least one approving review before merging. -- Reviewers will check for correctness, test coverage, documentation, and adherence to project conventions. -- Be respectful and constructive in code review comments. - -## Reporting Issues - -- Use the [bug report template](.github/ISSUE_TEMPLATE/bug_report.yml) for bugs. -- Use the [feature request template](.github/ISSUE_TEMPLATE/feature_request.yml) for new features. -- Use the [skill listing request template](.github/ISSUE_TEMPLATE/skill_listing_request.yml) to request skill directory inclusion. - -## Code of Conduct - -This project follows the [Contributor Covenant Code of Conduct](CODE_OF_CONDUCT.md). Please read it before participating. - -## License - -By contributing, you agree that your contributions will be licensed under the [MIT License](LICENSE). +2. Keep unrelated changes out. Do not commit generated indexes, build + artifacts, local configuration, credentials, or editor files. +3. Rebase onto the latest `upstream/main` before opening or updating the PR + (do not merge `main` into your branch). +4. Run the CI commands above, push to your fork, and open the PR against + `denfry/codebase-index:main`. Describe the problem, the solution, how you + verified it, and any compatibility impact. +5. Never force-push once review has started unless the rewrite is necessary and + announced in the PR. + +Maintainers cut releases from `main` following +[docs/RELEASE_CHECKLIST.md](docs/RELEASE_CHECKLIST.md); the version is +single-sourced in `src/codebase_index/__init__.py`. + +## Reporting issues + +- Bugs: [bug report template](.github/ISSUE_TEMPLATE/bug_report.yml), with + `codebase-index doctor` output. +- Features: [feature request template](.github/ISSUE_TEMPLATE/feature_request.yml). +- Benchmark runs: [benchmark report template](.github/ISSUE_TEMPLATE/benchmark_report.yml). +- Languages: [language support template](.github/ISSUE_TEMPLATE/language_support.yml). +- Security: **not** an issue. See [SECURITY.md](SECURITY.md). + +## Code of conduct and license + +This project follows the [Contributor Covenant](CODE_OF_CONDUCT.md). +Contributions are licensed under the [MIT License](LICENSE). diff --git a/README.md b/README.md index e6ad992..96340f7 100644 --- a/README.md +++ b/README.md @@ -1,330 +1,263 @@

- codebase-index logo + codebase-index logo

codebase-index

- Give AI coding agents a precise map of your codebase — locally, - privately, and with evidence. + A local code map for AI coding agents: find the right file, trace how it connects, predict what a change breaks.

- Find implementations. Trace behavior. Predict change impact. -

- -

- Quick start · - Install · - Benchmarks · - Security · - MCP + Works with Claude Code, Codex CLI, OpenCode and any MCP client. Runs on your machine. Sends nothing anywhere.

PyPI version CI Python 3.11+ - MCP ready - No network by default + MCP stdio server + No network by default MIT license

- codebase-index showing a Find, Trace, Predict workflow with precise file and line evidence + Try it · + Why · + Find / Trace / Predict · + Agents · + Evidence · + Contribute

-## The short version - -`codebase-index` is a local retrieval and code-graph layer for Claude Code, -Codex CLI, OpenCode, and MCP clients. It indexes a repository into SQLite, -extracts symbols and relationships with Tree-sitter, and gives agents ranked -`file:line` evidence instead of making them scan broad sets of files. - -```text -Question → ranked retrieval → dependency evidence → precise answer -``` - -It is not an IDE and not another coding agent. Your existing agent remains the -interface; `codebase-index` gives it better aim. - -## Find. Trace. Predict. - -| Job | Question | Command | -|---|---|---| -| **Find** | Where is authentication implemented? | `codebase-index search "authentication"` | -| **Trace** | How does checkout reach the database? | `codebase-index explain "checkout flow"` | -| **Trace** | How are two components connected? | `codebase-index path ApiController Database` | -| **Predict** | What breaks if `User` changes? | `codebase-index impact User` | -| **Predict** | What does my current diff affect? | `codebase-index diff-impact` | - -Every retrieval packet carries: - -- ranked matches and the reason each match scored; -- exact line ranges to read next; -- index freshness; -- answer confidence and targeted fallbacks; -- graph coverage and edge confidence where relevant. +

+ Real codebase-index output on Flask: a ranked search with line ranges, references with confidence, and an impact table +

-That evidence contract lets an agent distinguish “nothing references this” from -“the graph is partial, so verify with a targeted search.” +

Real output on pallets/flask. Reproduce it with examples/demo.

-## Install in five minutes +## Try it in 60 seconds ```bash pip install codebase-index cd your-project -codebase-index init -codebase-index index -codebase-index search "where is authentication implemented?" +codebase-index index # a few seconds for ~250 files +codebase-index search "where is the session cookie signed" +codebase-index impact SessionInterface --direction up # what depends on it +codebase-index init # wire it into Claude Code / Codex / OpenCode ``` -`init` can install resources for Claude Code, Codex CLI, OpenCode, and detected -MCP clients: +That is the whole setup. The index is a SQLite file under `.claude/cache/`, the +skill tells your agent to query it before reading files, and `codebase-index doctor` +tells you if anything is off. -```bash -codebase-index init --target auto -codebase-index init --target codex -codebase-index init --target claude -codebase-index init --target opencode -``` +## Why this exists -`pipx install codebase-index` is supported as an isolated alternative. See the -[installation guide](docs/INSTALLATION.md) for pinned releases, editable -installs, hooks, Windows details, and troubleshooting. +Claude Code and Codex can already read files. The problem is deciding *which* files. +On a repository with a few hundred files an agent that greps for keywords opens the +wrong file more often than the right one, then reads whole files to compensate. That +costs tokens, time, and, worst of all, wrong answers delivered with confidence. -### Claude Code plugin - -```text -/plugin marketplace add denfry/codebase-index -/plugin install codebase-index@codebase-index -``` - -The plugin provisions a private environment on first use. Later sessions run -offline, and the first codebase question builds the index automatically. - -## What the agent receives - -```json -{ - "query": "where is authentication implemented?", - "confidence": "high", - "results": [ - { - "path": "src/auth/AuthService.ts", - "line_start": 12, - "line_end": 148, - "score": 0.92, - "reason": "exact symbol match, 4 callers" - } - ], - "recommended_reads": [ - { - "path": "src/auth/AuthService.ts", - "line_start": 12, - "line_end": 148 - } - ], - "index": { - "exists": true, - "stale": false - } -} -``` +`codebase-index` gives the agent a ranked, evidence-bearing answer to three questions +before it opens anything: -Snippets are skeletonized when that preserves evidence while saving tokens. -Unrelated bodies collapse, but imports, signatures, matched lines, and exact -read ranges remain. +| | Question | What comes back | +|---|---|---| +| **Find** | Where is X implemented? | Ranked `file:line` ranges, a reason for each rank, and a bounded read plan | +| **Trace** | What calls this? How does A reach B? | Definitions vs references, shortest dependency paths, each edge tagged `extracted` / `inferred` / `ambiguous` | +| **Predict** | What breaks if I change this? What does my diff touch? | Upstream and downstream dependents by distance, with graph-coverage honesty | -## Core commands +Every answer carries its own uncertainty: index freshness, result confidence, and +whether the graph is partial for a language. The agent can tell "nothing references +this" from "the graph doesn't know", which grep never can. -```bash -# Find -codebase-index search "auth token refresh" -codebase-index symbol AuthService - -# Trace -codebase-index explain "authentication flow" -codebase-index refs send_email -codebase-index path ApiController Database -codebase-index describe Database -codebase-index architecture - -# Predict -codebase-index impact User --direction up --depth 2 -codebase-index diff-impact --base HEAD --direction up --depth 2 - -# Inspect and visualize -codebase-index graph User --direction both --depth 2 --output graph.html -codebase-index stats -codebase-index doctor -``` +**How it differs from what you already have** -Add `--json` for agents and automation. Search supports `hybrid`, `fts`, -`symbol`, and opt-in `vector` modes. +- **grep / ripgrep**: exact text, no ranking, no notion of definition vs call, no + dependency graph. Use it when you know the string. Use this when you know the + question. +- **Repo-map style context** (Aider and friends): a signature listing pushed into + every prompt. Good orientation, query-agnostic, and it must fit in the budget. + This is a queryable index instead: per-question ranking, line ranges, graph + traversal, and a stable JSON/MCP contract any agent can call. +- **Cloud code search / IDE indexing**: powerful, but your code leaves the machine + or you adopt an IDE. This is a small Python package on SQLite and Tree-sitter: + no server, no account, no telemetry. -## Why not just grep? +The [comparison guide](docs/COMPARISON.md) says when each of those is the better choice. -Grep is excellent when you know the exact text. Repository questions often -need more: +## Find / Trace / Predict -| Capability | `rg` / grep | codebase-index | -|---|---:|---:| -| Exact text matching | Yes | Yes | -| Ranked results | No | Yes | -| Symbol definitions vs calls | No | Yes | -| Dependency and impact graph | No | Yes | -| Token-budgeted read plan | No | Yes | -| Freshness and coverage signals | No | Yes | -| Local and scriptable | Yes | Yes | +All examples below are real output on Flask at commit `d318b683`; see +[examples/demo/EXPECTED_OUTPUT.md](examples/demo/EXPECTED_OUTPUT.md) for the full transcript. -Use grep for one known string. Use `codebase-index` when the agent must locate, -understand, or assess a change across a repository. +**Find** the implementation, not a keyword hit: -The [comparison guide](docs/COMPARISON.md) also covers Cursor, Aider repo-map, -Sourcegraph, Continue, Amp, and Codebase-Memory MCP—including when those tools -are the better choice. +```text +$ codebase-index search "where is the session cookie signed and saved" --limit 3 +| # | Path | Lines | Reason | +| 1 | src/flask/sessions.py | 24-54 | in src/flask/ · 2 callers · source prior +0.08 | +| 2 | src/flask/sessions.py | 284-385 | in src/flask/ · 2 callers · source prior +0.08 | +| 3 | src/flask/sessions.py | 57-80 | in src/flask/ · 1 callers · source prior +0.08 | +``` -## Measured results +**Trace** references with confidence, and paths between symbols: -On the published 55k LOC Java benchmark: +```text +$ codebase-index refs open_session +| kind | path | line | confidence | +| call | src/flask/ctx.py | 388 | ? ambiguous | +| definition | src/flask/sessions.py | 249 | exact | +| definition | src/flask/sessions.py | 323 | exact | + +$ codebase-index path wsgi_app dispatch_request +wsgi_app (src/flask/app.py) → full_dispatch_request → dispatch_request · 2 hop(s) +``` -- Recall@3: **70%** for `codebase-index` versus **40%** for the `rg` baseline; -- answer-context tokens: approximately **13× fewer**; -- raw results and methodology are checked into the repository. +**Predict** the blast radius of a symbol, or of the diff you have right now: -These results are evidence for that benchmark, not a claim of universal -superiority. Large public-repository and framework-graph evaluations remain on -the [roadmap](docs/ROADMAP.md). Read the complete methodology and limitations in -[BENCHMARKS.md](docs/BENCHMARKS.md). +```text +$ codebase-index impact SecureCookieSessionInterface --direction up --depth 2 +| dist | via | node | location | +| 1 | call | Flask | src/flask/app.py:110 | +| 1 | extends | PathAwareSessionInterface | tests/test_reqctx.py:204 | +| 2 | call | CustomFlask | tests/test_reqctx.py:211 | + +$ codebase-index diff-impact # after editing src/flask/sessions.py +affected files (8): src/flask/app.py, src/flask/ctx.py, src/flask/globals.py, ... (edge kind + confidence per row) +``` -## Local by default +More: `explain "how are blueprints registered"`, `describe dispatch_request`, +`architecture` (modules, god nodes, surprising links), `graph User --output graph.html`. +Add `--json` to any command for the machine-readable packet. -The base install: +## Agent integrations -- makes no network requests; -- sends no telemetry; -- stores the derived index inside the project cache; -- excludes dependency, build, binary, oversized, generated, and secret-like - files before indexing; -- redacts secret patterns again at output time; -- exposes `doctor --strict` for CI and security checks. +| Agent | Setup | What it gets | +|---|---|---| +| **Claude Code** | `codebase-index init --target claude`, or the plugin: `/plugin marketplace add denfry/codebase-index` then `/plugin install codebase-index@codebase-index` | A skill that routes repository questions to the index and reads only `recommended_reads` ranges; optional PostToolUse hook keeps the index fresh | +| **Codex CLI** | `codebase-index init --target codex` | A managed block in `AGENTS.md` plus the skill resources | +| **OpenCode** | `codebase-index init --target opencode` | `/codebase-index` command, agent file, skill resources | +| **Any MCP client** (Claude Desktop, Cursor, VS Code, Zed, Windsurf, ...) | `pip install "codebase-index[mcp]"` then `codebase-index mcp --root /path/to/repo` | 11 tools (`search_code`, `find_refs`, `impact_of`, `impact_of_diff`, `path_between`, ...) with a versioned JSON envelope | +| **Anything with a shell** | `codebase-index --json ...` | The same payloads as plain JSON, plus a local SQLite database you can query yourself | -Embeddings are optional. Local embeddings stay on the machine; external -embeddings require explicit configuration, an API key, and an endpoint -acknowledgement. +`init --target auto` detects which of these are present. See +[INSTALLATION.md](docs/INSTALLATION.md) and [MCP.md](docs/MCP.md). -See the [security model](docs/SECURITY_MODEL.md) for trust boundaries, gates, -failure modes, and residual risks. +## Evidence -## How it works +Numbers below come from runs whose raw logs are in this repository. Nothing here is a +claim about tools we have not benchmarked. -```text -Repository - │ - ├─ discovery + ignore and secret gates - ├─ Tree-sitter symbols and relationships - ├─ line and symbol-aligned chunks - └─ optional embeddings - │ - ▼ - local SQLite - FTS5 + symbols + graph - │ - ▼ - intent routing → hybrid retrieval → rerank → token budget - │ - ▼ - ranked file:line evidence for CLI, Skill, and MCP -``` - -The three product surfaces share one service layer, so retrieval behavior does -not drift between the CLI, installed agent skills, and MCP tools. +**Index vs a disciplined grep agent, on public repositories.** Three repos at pinned +commits (Flask, Gson, Fastify), 450 questions mined from git history (commit subject +→ files that commit changed, so neither side wrote the answer key), the same tokenizer +charging both sides for what enters context: -Detailed internals: +| pooled, n = 450 | hit@3 | MRR | context tokens / question | +|---|---:|---:|---:| +| codebase-index | **0.547** | **0.456** | 3,835 | +| `rg` + 80-line windows | 0.304 | 0.263 | 3,495 | -- [Architecture](docs/ARCHITECTURE.md) -- [Retrieval pipeline](docs/RETRIEVAL_PIPELINE.md) -- [Database schema](docs/DATABASE_SCHEMA.md) -- [Language and graph coverage](docs/LANGUAGES.md) +Every delta is significant at p < 0.001 (paired bootstrap CI in the log). Read it as: +**1.8× more likely to put the answer in the top three, at about 10% more context**. +Reproduce with `python tests/eval/run_baselines.py --clone`; details, caveats and the +repo-map-style comparison are in [BENCHMARKS.md](docs/BENCHMARKS.md). -## Supported surfaces +

Bar chart of hit@3, MRR and context tokens for codebase-index versus ripgrep with windows on Flask, Gson and Fastify

-| Surface | Integration | -|---|---| -| Claude Code | Skill, plugin, optional hooks | -| Codex CLI | `AGENTS.md` plus project skill | -| OpenCode | Command, agent, and skill resources | -| MCP clients | stdio server with versioned JSON envelopes | -| Shell and automation | CLI, `--json`, and local SQLite | +**Every ranking signal is ablated.** A change to the ranker ships only if it is +significant on a pooled multi-language query set +([tests/eval](tests/eval/README.md)). 1.9.0 removed two signals that could not +show a benefit and rejected five plausible ones. -Run the MCP server with: +**What is not measured yet**: whether an *agent* completes tasks better with the +index. That needs model calls and a rubric and is the top item in +[BENCHMARKS.md](docs/BENCHMARKS.md#future-work-in-priority-order). Please do not +quote task-success numbers for this project; there are none. -```bash -codebase-index mcp --root /path/to/repository -``` +## How it works -Available MCP tools include search, explain, symbols, references, impact, -diff impact, architecture, shortest path, node description, health, and index statistics. -See [MCP.md](docs/MCP.md) for client configuration. +

Architecture: discovery and secret gates, Tree-sitter symbols and edges, local SQLite with FTS5, one service layer behind CLI, skill and MCP; the query path runs intent detection, retrievers, RRF fusion, rerank and a token budget

+ +Indexing walks the repository through ignore and secret gates, extracts symbols and +import/call/reference/inheritance edges with Tree-sitter for 12 languages, and stores +chunks, symbols and edges in one SQLite file with FTS5. A query goes through intent +detection, path/symbol/FTS retrievers (vector is opt-in), reciprocal-rank fusion that +rewards cross-retriever agreement on a file, a rerank with calibrated source priors, +and a token budget that emits ranked ranges plus a bounded read plan. The CLI, the +installed skills and the MCP server all call the same service layer, so behaviour +cannot drift between surfaces. + +Deep dives: [Architecture](docs/ARCHITECTURE.md) · +[Retrieval](docs/RETRIEVAL.md) · [Schema](docs/SCHEMA.md) · +[Languages](docs/LANGUAGES.md) · [Skill design](docs/SKILL_DESIGN.md) + +## Privacy and security + +- **No network by default.** The base install has no network dependency and makes + no requests. There is no telemetry, no crash reporting, no usage counter. +- **Secrets never get in.** `.env*`, keys, certificates, credential files, + binaries, dependency and build directories, generated and oversized files are + excluded before parsing; `.gitignore`, `.codeindexignore`, `.claudeignore` and + `.cursorignore` are honoured. +- **Secrets never get out.** Snippets are redacted again at output time (AWS keys, + private-key blocks, JWTs, connection strings, Slack tokens, high-entropy values). +- **One opt-in exit, triple-gated.** External embeddings require an explicit config + flag, an API key in the environment, and a printed endpoint warning, or they are + refused. Local embeddings stay local. +- **Verify it yourself.** `codebase-index doctor --strict` audits the gates and exits + non-zero in CI if one is off. + +Details and residual risks: [SECURITY_MODEL.md](docs/SECURITY_MODEL.md). +Report vulnerabilities privately: [SECURITY.md](SECURITY.md). + +## Supported languages + +Symbol and graph extraction (Tier A): Python, JavaScript, TypeScript, Java, Go, Rust, +C, C++, C#, Ruby, PHP, Kotlin. Generic Tree-sitter definitions (Tier B): Lua. +Full-text search for everything else that is text, including Markdown, YAML, JSON, +TOML, SQL, Dockerfiles, Terraform and CI configs. `refs` and `impact` report +`coverage.partial` when a language has no graph edges, so an empty result is never +silently presented as "no callers". Tiers and how to add a language: +[LANGUAGES.md](docs/LANGUAGES.md). ## Project status -The latest released line is **1.9.0**. It includes: - -- hybrid and optional vector retrieval; -- Tree-sitter symbol extraction across the documented language tiers; -- import, call, reference, and inheritance graphs; -- architecture communities, central nodes, and surprising cross-module links; -- shortest dependency paths and node descriptions; -- token-budgeted and skeletonized retrieval packets; -- benchmark-calibrated lexical expansion, fuzzy identifier matching, and source-aware ranking; -- rank fusion that scores cross-retriever agreement at file level, not just at a locator; -- bounded, intent-directed graph discovery with optional diversity and duplicate suppression; -- CLI, Skill, plugin, and MCP delivery; -- incremental updates, watch hooks, diagnostics, skill rollback, and diff-aware - impact analysis; -- a multi-repository retrieval evaluation with leak-free git-derived ground truth, - one-signal ablations, and paired significance tests - ([tests/eval](tests/eval/README.md)). - -Every shipped ranking signal has to survive that evaluation: 1.9.0 removed the -cost of two signals that could not demonstrate a benefit and rejected several -plausible ones outright (IDF-weighted coverage, stemming, graph propagation, MMR, -a file-length prior). Planned work is deliberately separated from shipped -capability. The next product priorities are typed framework edges and an even more -direct task-context workflow. See the [roadmap](docs/ROADMAP.md). - -## Documentation - -| Start here | Deep dives | Project trust | -|---|---|---| -| [Quick start](docs/QUICKSTART.md) | [Retrieval](docs/RETRIEVAL.md) | [Benchmarks](docs/BENCHMARKS.md) | -| [Installation](docs/INSTALLATION.md) | [Architecture](docs/ARCHITECTURE.md) | [Security](docs/SECURITY.md) | -| [FAQ](docs/FAQ.md) | [MCP](docs/MCP.md) | [Release checklist](docs/RELEASE_CHECKLIST.md) | -| [Skill design](docs/SKILL_DESIGN.md) | [Schema](docs/SCHEMA.md) | [Changelog](CHANGELOG.md) | +Current line: **1.9.x**, on PyPI, MIT. CI runs Linux, macOS and Windows on Python +3.11–3.13 with an 80% coverage gate, golden snapshots for every CLI and MCP payload, +a packaging smoke test on every PR, and a skill-copy drift check. -## Contributing +What works today: hybrid retrieval with optional local vectors; Tree-sitter symbols +and edges; `search`, `explain`, `symbol`, `refs`, `impact`, `diff-impact`, `path`, +`describe`, `architecture`, `graph`; token-budgeted, skeletonized packets; incremental +`update`, `watch`, and hooks; CLI, Claude Code plugin/skill, Codex, OpenCode and +MCP delivery; a reproducible multi-repository benchmark. -Contributions should preserve three invariants: +What does not exist yet: framework-aware typed edges (routes, DI, migrations), +multi-repository workspaces, paged MCP results, an agent task-level benchmark, +signed release artifacts. See the [roadmap](docs/ROADMAP.md). -1. retrieval quality is measured, not asserted; -2. the default path remains local and fails closed at security boundaries; -3. machine-readable contracts stay stable across CLI and MCP. +## Contributing -Before opening a pull request: +Fifteen minutes from clone to a first PR: [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md). +Where help is most useful: benchmark reproductions on your own repositories, new +Tier-A languages, graph edge kinds, and MCP client verification. Labels and the +retrieval-quality PR rules are in [CONTRIBUTING.md](CONTRIBUTING.md). ```bash -pytest -ruff check . -mypy src -python scripts/sync_skill_copies.py --check +pip install -e ".[dev,mcp]" +pytest && ruff check src tests && mypy src/codebase_index +python tests/eval/run_eval.py --ablate # required for any ranking change ``` -Add user-visible changes under `[Unreleased]` in [CHANGELOG.md](CHANGELOG.md). -See [CONTRIBUTING.md](CONTRIBUTING.md) if present and the repository -instructions for branch and review policy. +## Roadmap + +Next: verified MCP client configs and paged results; an agent task-level evaluation; +a large-monorepo benchmark run; then framework-aware typed edges behind a hand-labelled +graph benchmark. Full list with the "not a claim until it ships" rule: +[docs/ROADMAP.md](docs/ROADMAP.md). History: [CHANGELOG.md](CHANGELOG.md). ## License diff --git a/ROADMAP.md b/ROADMAP.md index d046f83..254ce1e 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -1,112 +1,34 @@ # Roadmap -`codebase-index` is developed in milestone-driven slices. Each milestone delivers a runnable, testable feature. - -## Milestones - -### M0 — Repository Packaging ✅ -- Polished README, documentation, badges, issue templates, CI workflows. -- MIT license, changelog, code of conduct, contributing guide. -- Claude Code Skill directory structure. -- **Exit:** Repository is ready for public listing and awesome-list submissions. - -### M1 — SQLite + FTS5 Index ✅ -- SQLite database with FTS5 virtual table for full-text search. -- File discovery with layered ignore rules (`.gitignore`, `.codeindexignore`, built-in denylist). -- Incremental indexing with file hash tracking. -- **Exit:** `codebase-index index` populates the database; secrets and binaries are excluded. - -### M2 — Tree-sitter Symbol Extraction ✅ -- AST-based symbol extraction, now expanded beyond the original Python/JavaScript/TypeScript - set to the Tier-A languages listed in `docs/LANGUAGES.md`. -- Symbol-aligned chunking with gap windows. -- `symbol` and `refs` commands for intra-file symbol lookup. -- **Exit:** `codebase-index symbol "AuthService"` returns symbol definitions and references. - -### M3 — Hybrid Retrieval ✅ -- Combined search: path match + symbol match + FTS5 + optional vector search. -- Reciprocal Rank Fusion (RRF) for result merging. -- Confidence scoring and fallback suggestions. -- **Exit:** `codebase-index search "query"` returns ranked results from multiple retrievers. - -### M4 — Graph Expansion ✅ -- Dependency, import, call, and inheritance edge extraction. -- Graph-based result expansion (related files, callers, callees). -- `impact` command for blast radius analysis. -- **Exit:** `codebase-index impact "src/auth/AuthService.ts"` shows affected files and symbols. - -### M5 — Token-Budgeted Retrieval Packets ✅ -- Ranked retrieval packets with file paths, line ranges, snippets, and "next files to read". -- Token budget enforcement (configurable max output size). -- Compact Markdown and JSON output formats. -- **Exit:** Claude reads only the recommended line ranges, not entire files. - -### M6 — Optional Local Embeddings ✅ -- `sqlite-vec` integration for vector similarity search. -- Local embedding models (sentence-transformers) as default. -- External embedding APIs behind explicit opt-in with warnings. -- **Exit:** Semantic queries improve recall when embeddings are enabled. - -### M7 — Multi-CLI skill packaging ✅ -- `init --target claude|codex|opencode|auto|all`. -- Claude Code skill, Codex instructions/resources, OpenCode command/agent files. -- **Exit:** AI CLI setup can be generated from the package without hand-copying files. - -### M8 — Optional Hooks + Watch Mode ✅ -- Post-tool-use hook for automatic index updates. -- `--with-hooks` flag and hook configuration. -- `doctor` reports enabled hooks and their status. -- **Exit:** Index stays fresh automatically after file edits. - -### M9 — Public Release ✅ -- Comprehensive test suite with coverage targets. -- Performance benchmarks on medium-sized repositories. -- GitHub release with tagged version. -- Release pipeline readiness for PyPI trusted publishing. -- Awesome-list submissions (Claude Code skills, AI coding tools). -- **Exit:** tagged GitHub install works on a clean machine. - -### M10 — Distribution hardening -- Publish `codebase-index` to PyPI after trusted publishing is configured and verified. -- Verify `pipx install codebase-index`, `uvx codebase-index init`, and `uv tool install codebase-index`. -- Add Homebrew tap formula: `brew install denfry/tap/codebase-index`. -- Publish signed release checksums and SBOMs for release artifacts. -- **Exit:** users do not need GitHub URL installs for the standard path, and artifacts have a clear supply-chain story. - -### M11 — First-class MCP server ✅ -- Model Context Protocol wrapper for external tool integration. -- MCP server exposing `healthcheck`, `search_code`, `find_symbol`, `find_refs`, - `impact_of`, `explain_code`, and `index_stats`. -- Versioned JSON schema and golden tests for every tool output. -- **Exit:** `codebase-index mcp --root ` can be used by any MCP-compatible client. - -### M11.5 — MCP hardening -- Ready-to-copy configs verified against Claude Desktop, Claude Code, Cursor, VS Code, Zed, - and Windsurf current versions. -- Golden snapshots for every tool output. -- Paging or progressive result support for large repositories. - -### M12 — Public benchmark suite ✅ -- Shipped reproducible public benchmark script: `tests/benchmark_public.py`. -- Retrieval quality: Recall@1/3/5, MRR, nDCG. -- Agent usefulness: answer-correctness proxy on the public fixture. -- Token economy versus grep-window baseline. -- Language-specific results, freshness latency after edits, graph tasks, and scale counters. - -### M12.5 — Real-repo benchmark expansion -- 10k, 100k, 1M LOC repository targets. -- Real-world Python, TypeScript, Java, Go, Rust, C#, PHP repos. -- Human-reviewed answer correctness on real codebase questions. -- Token economy versus repo-map style context and vanilla agent exploration. -- Framework graph tasks: route -> handler -> service -> DB, migrations, config consumers, CI/infra. - -### M13 — Code intelligence graph -- Framework-aware typed edges beyond import/call/reference. -- Routes, tests/fixtures, config consumers, migrations, event flows, DI wiring, - frontend component flows, and error/log-message traces. -- Edge confidence and resolver provenance so agents can distinguish precise - edges from heuristics. - ---- - -See [CHANGELOG.md](CHANGELOG.md) for released versions and their changes. +`codebase-index` was built in milestone-driven slices, each delivering a runnable, tested +feature. The milestones below have shipped; forward work is tracked in the product roadmap, +[docs/ROADMAP.md](docs/ROADMAP.md), which separates *now / next / then* and does not treat an +item as a product claim until it appears in [CHANGELOG.md](CHANGELOG.md). + +## Shipped milestones + +| Milestone | Delivered | Since | +|---|---|---| +| M0 Repository packaging | README, docs, CI, issue templates, MIT license, changelog | 1.0.0 | +| M1 SQLite + FTS5 index | Layered ignore rules, secret/binary/generated exclusion, incremental hashing | 1.0.0 | +| M2 Tree-sitter symbols | `LangSpec` extraction for the Tier-A languages, symbol-aligned chunks, `symbol` / `refs` | 1.0.0, widened in 1.3.0 | +| M3 Hybrid retrieval | Path + symbol + FTS5 (+ optional vector) fused with RRF, confidence and fallbacks | 1.0.0 | +| M4 Graph expansion | Import / call / reference / inheritance edges, `impact` | 1.0.0 | +| M5 Token-budgeted packets | Ranked `file:line` results, `recommended_reads`, Markdown and JSON | 1.0.0 | +| M6 Optional local embeddings | `sqlite-vec`, sentence-transformers, triple-gated external backend | 1.0.0 | +| M7 Multi-CLI packaging | `init --target claude\|codex\|opencode\|auto\|all` (1.0.2); Claude Code plugin one-command install (1.3.0) | 1.0.2 | +| M8 Hooks + watch mode | PostToolUse auto-update, `watch`, `doctor` hook reporting | 1.0.0 | +| M9 Public release | Coverage gate, tagged GitHub releases, clean-machine smoke | 1.0.0 | +| M10 Distribution (PyPI part) | `pip install codebase-index`, `pipx`, Trusted Publishing | 1.6.0 | +| M11 MCP server | Stdio server (1.1.0); versioned envelopes and golden tests for every tool (1.4.0) | 1.1.0 | +| M12 Public benchmark suite | `tests/benchmark_public.py` with Recall/MRR/nDCG/token/freshness/graph metrics (1.1.0); logged run (1.4.0) | 1.1.0 | +| — Retrieval evaluation | Leak-free git-derived ground truth, multi-corpus pooling, significance tests | 1.8.0 / 1.9.0 | + +## Still open (tracked in docs/ROADMAP.md) + +- M10 remainder: `uvx` verification on every CI OS, Homebrew tap, signed checksums / SBOM. +- M11.5: MCP client configs verified against current client releases; paged results. +- M12.5: larger public-repository runs (a 1M-LOC / monorepo target) and an LLM-agent + task-level evaluation. A multi-repository public baseline benchmark now exists at + `tests/eval/run_baselines.py` with a logged run under `tests/eval/results/`. +- M13: typed framework-aware edges (routes, tests, config consumers, migrations, DI). diff --git a/SECURITY.md b/SECURITY.md index 1602659..5d1e94f 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,51 +1,73 @@ # Security Policy -## Supported Versions - -| Version | Supported | -| ------- | ------------------ | -| 1.2.x | :white_check_mark: | -| < 1.2 | :x: | - -Only the latest minor version receives security updates. - -## Reporting a Vulnerability - -We take security seriously. If you discover a vulnerability, please report it responsibly: - -1. **Do not** open a public issue. -2. Email the maintainers with a description of the vulnerability and steps to reproduce. -3. We will acknowledge receipt within 48 hours and provide a timeline for a fix. -4. Once resolved, we will publish a security advisory and credit the reporter (if desired). - -## No Telemetry Promise - -`codebase-index` does **not** collect, transmit, or store any telemetry, usage data, or analytics. All indexing, search, and storage operations occur entirely on your local machine. There are no phone-home mechanisms, crash reporters, or usage counters. - -## Secret Handling - -- **Never indexed**: `.env` files, private keys (`.pem`, `.key`), certificates, tokens, credential files, and binary artifacts are excluded before parsing. -- **Redacted in output**: Any snippets that may contain secret-like patterns (AWS keys, JWTs, bearer tokens, connection strings) are masked before being returned to Claude or printed to the terminal. -- **Respects ignore files**: `.gitignore`, `.claudeignore`, `.codeindexignore`, and `.cursorignore` are all honored during discovery. - -## External Embeddings Opt-In - -The default configuration disables embeddings entirely (`backend = "noop"`). External embedding APIs (which would send code text to a remote service) require: - -1. Explicit `embeddings.allow_external = true` in configuration. -2. A user-provided API key via environment variable. -3. Warnings printed by both `doctor` and `index` commands. - -Without all three conditions, external embeddings are refused. - -## Threat Model - -- **Indexed content**: Treat indexing an untrusted repository the same as opening it in a text editor. Parsers operate over file content but do not execute code. -- **Cache location**: The SQLite index is stored in `.claude/cache/codebase-index/`. Ensure this directory is not committed to version control (it is in the default `.gitignore`). -- **World-writable directories**: `doctor` warns if the cache directory has insecure permissions. - -## Unsafe Patterns to Avoid - -- Do not commit the SQLite index file to a shared repository. -- Do not enable external embeddings on repositories containing proprietary or regulated code without reviewing your organization's data handling policies. -- Do not run `codebase-index index` on repositories you do not trust without reviewing the `doctor` output first. +`codebase-index` reads entire repositories, so its security posture matters +more than for a typical CLI. The full trust model, exclusion pipeline and +redaction rules are documented in [docs/SECURITY_MODEL.md](docs/SECURITY_MODEL.md); +this page covers reporting and the guarantees in short form. + +## Supported versions + +Only the latest minor release line receives security fixes. + +| Version | Supported | +|---|---| +| 1.9.x (latest) | Yes | +| < 1.9 | No — upgrade with `pip install -U codebase-index` | + +## Reporting a vulnerability + +Please do **not** open a public issue for security problems. + +1. Use GitHub private vulnerability reporting: + + (the *Report a vulnerability* button under the repository's **Security** tab). +2. Include the version (`pip show codebase-index`), + platform, a description, and reproduction steps. Redact any real secrets. +3. You will get an acknowledgement within a few days. Fixes are released as a + patch on the supported line and published as a GitHub security advisory, + crediting the reporter unless they prefer otherwise. + +If private reporting is unavailable for any reason, open a minimal issue that +says only "security report, please contact me" without details, and a +maintainer will reach out. + +## What the tool guarantees + +- **No telemetry.** No usage data, analytics, crash reports or phone-home of + any kind. +- **No network by default.** The base install makes no network requests. + The only code path that can send repository text off the machine is the + *external* embeddings backend, which is refused unless all three hold: + `embeddings.allow_external = true` in config, an API key provided via + environment variable, and the endpoint warning printed by `doctor` / `index`. +- **Secrets are never indexed.** `.env*`, private keys and certificates + (`*.pem`, `*.key`, `*.p12`, `*.pfx`, `id_rsa*`, `*.crt`, `*.keystore`), + `credentials*`, `secrets*`, binaries, dependency and build directories, + generated files and oversized files are excluded before parsing. +- **Secrets are redacted on output.** Snippets pass through + `output/redact.py` before reaching the agent or the terminal (AWS keys, + private-key blocks, JWTs and bearer tokens, connection strings with + credentials, Slack tokens, high-entropy values assigned to key-like names). +- **Ignore files are honoured.** `.gitignore`, `.claudeignore`, + `.codeindexignore` and `.cursorignore`. +- **Read-only agent surface.** The generated skill's `allowed-tools` and the + `cbx` wrappers whitelist read-only subcommands; `clean`, `init` and `watch` + are not callable from the skill. +- **Self-audit.** `codebase-index doctor --strict` exits non-zero if any gate + is misconfigured; use it in CI. + +## Threat model in brief + +- Indexing a repository is like opening it in an editor: parsers read file + content, nothing is executed. +- The derived index lives in `.claude/cache/codebase-index/` and is + gitignored by `init`. Do not commit it or share it as if it were sanitised — + redaction happens at output time, and the index stores indexed text. +- Treat `doctor` warnings about world-writable cache directories as real. + +## Release integrity + +Releases are built in GitHub Actions and published to PyPI with Trusted +Publishing (OIDC, no stored tokens). Signed checksums, SBOMs and build +attestations are on the roadmap and are **not** yet provided; do not assume +them. diff --git a/assets/architecture.svg b/assets/architecture.svg new file mode 100644 index 0000000..e1d8664 --- /dev/null +++ b/assets/architecture.svg @@ -0,0 +1,96 @@ + + codebase-index architecture: repository to ranked file:line evidence + + + + + + + + + + Repository + discovery walk + .gitignore / .codeindexignore + secret · binary · generated gates + size cap (1 MB) + + + + Tree-sitter extraction + symbols (12 Tier-A languages) + import · call · reference · + extends / implements edges + symbol-aligned chunks + + + + Local SQLite + files · chunks · symbols + edges (with confidence) + FTS5 · optional sqlite-vec + .claude/cache/codebase-index/ + + + + One service layer + CLI (codebase-index / cbx) + Skill (Claude Code, Codex, OpenCode) + MCP (stdio, schema_version 1) + --json for anything else + + + + + + + Query path (per question) + + + intent detection + locate · trace · impact + + + retrievers + path · symbol · FTS5 + (+ vector, opt-in) + + + RRF fusion + file-level agreement + + + rerank · priors · dedup + every signal ablated + + + token-budgeted packet + ranked file:line · reads · confidence + + + + + + + + + Graph path (refs · path · impact · diff-impact · architecture) + + + bounded expansion over the edge table + direction · depth · node cap + + + edge confidence carried through + extracted · inferred · ambiguous + + + coverage.partial when a language lacks edges + so "no callers" is never silently wrong + + + + + + Network: off by default. Telemetry: none. External embeddings: opt-in, triple-gated. Source: docs/ARCHITECTURE.md + diff --git a/assets/benchmark.svg b/assets/benchmark.svg new file mode 100644 index 0000000..b1650dd --- /dev/null +++ b/assets/benchmark.svg @@ -0,0 +1,98 @@ + +codebase-index vs rg+window on public repositories + +Index vs disciplined grep on public repositories +flask, gson, fastify at pinned commits · 450 git-derived queries · codebase-index 1.9.0 · 2026-09-04 · same tokenizer on both sides +codebase-index +rg + 80-line windows +hit@3 — answer file among the top 3 + +0.00 + +0.25 + +0.50 + +0.75 + +1.00 +flask + +0.49 + +0.23 +gson + +0.57 + +0.35 +fastify + +0.58 + +0.33 +pooled + +0.55 + +0.30 +MRR — how high the first correct file ranks + +0.00 + +0.25 + +0.50 + +0.75 + +1.00 +flask + +0.38 + +0.23 +gson + +0.50 + +0.29 +fastify + +0.48 + +0.27 +pooled + +0.46 + +0.26 +context tokens per query (pooled mean) + +0 + +1,100 + +2,200 + +3,300 + +4,400 +packet + +2,402 + +1,351 ++ reads + +3,835 + +3,495 +Paired significance, index − rg (pooled) +hit@3: +0.242 95% CI [+0.187, +0.298] p < 0.001 +recall@5: +0.231 95% CI [+0.183, +0.278] p < 0.001 +MRR: +0.194 95% CI [+0.151, +0.235] p < 0.001 +tokens: 340 95% CI [238, 440] p < 0.001 +Ground truth: commit subject → files that commit changed. Raw run: tests/eval/results/. +Latency is not charted: in-process index vs a separate ripgrep binary is not a fair pair. + diff --git a/assets/demo-terminal.svg b/assets/demo-terminal.svg new file mode 100644 index 0000000..837ae95 --- /dev/null +++ b/assets/demo-terminal.svg @@ -0,0 +1,49 @@ + + codebase-index 1.9 · pallets/flask @ d318b683 · real output, see examples/demo + + + + codebase-index 1.9 · pallets/flask @ d318b683 · real output, see examples/demo + $ codebase-index search "where is the session cookie signed and saved" --limit 3 +Query: where is the session cookie signed and saved +Intent: `locate_impl` · Confidence: medium + +| # | Path | Lines | Reason | +|---|------|-------|--------| +| 1 | src/flask/sessions.py | 24-54 | in src/flask/ · 2 callers · source prior +0.08 | +| 2 | src/flask/sessions.py | 284-385 | in src/flask/ · 2 callers · source prior +0.08 | +| 3 | src/flask/sessions.py | 57-80 | in src/flask/ · 1 callers · source prior +0.08 | + +src/flask/sessions.py:24-54 +class SessionMixin(MutableMapping[str, t.Any]): +src/flask/sessions.py:284-385 +class SecureCookieSessionInterface(SessionInterface): +src/flask/sessions.py:57-80 +class SecureCookieSession(CallbackDict[str, t.Any], SessionMixin): + +Recommended reads: +- src/flask/sessions.py:24-54 +- src/flask/sessions.py:284-385 +- src/flask/sessions.py:57-80 + +$ codebase-index refs open_session +query: open_session | index: fresh + +| kind | path | line | confidence | +|------|------|------|------------| +| call | src/flask/ctx.py | 388 | ? ambiguous | +| definition | src/flask/sessions.py | 249 | exact | +| definition | src/flask/sessions.py | 323 | exact | +| call | src/flask/testing.py | 165 | ? ambiguous | +| definition | tests/test_reqctx.py | 182 | exact | +| definition | tests/test_session_interface.py | 16 | exact | + +$ codebase-index impact SecureCookieSessionInterface --direction up --depth 2 +impact: `SecureCookieSessionInterface` · direction: up · depth: 2 · affected files: + +| dist | via | kind | node | location | +|------|-----|------|------|----------| +| 1 | call | symbol | Flask | src/flask/app.py:110 | +| 1 | extends | symbol | PathAwareSessionInterface | tests/test_reqctx.py:204 | +| 2 | call | symbol | CustomFlask | tests/test_reqctx.py:211 | + diff --git a/assets/demo.png b/assets/demo.png deleted file mode 100644 index a8077fd..0000000 Binary files a/assets/demo.png and /dev/null differ diff --git a/bin/cbx b/bin/cbx index cfe60b7..d4fb7f6 100644 --- a/bin/cbx +++ b/bin/cbx @@ -3,7 +3,7 @@ # from the venv provisioned by scripts/bootstrap.sh (located via the .venv-path pointer). set -euo pipefail -ALLOWED="search explain symbol refs impact graph stats doctor update index" +ALLOWED="search explain architecture symbol refs impact diff-impact path describe graph stats doctor update index" sub="${1:-}" case " $ALLOWED " in *" ${sub} "*) : ;; diff --git a/bin/cbx.ps1 b/bin/cbx.ps1 index 85face7..d841080 100644 --- a/bin/cbx.ps1 +++ b/bin/cbx.ps1 @@ -5,7 +5,7 @@ param( [Parameter(ValueFromRemainingArguments = $true)] [string[]]$Rest ) $ErrorActionPreference = "Stop" -$allowed = @("search", "explain", "symbol", "refs", "impact", "graph", "stats", "doctor", "update", "index") +$allowed = @("search", "explain", "architecture", "symbol", "refs", "impact", "diff-impact", "path", "describe", "graph", "stats", "doctor", "update", "index") if ($allowed -notcontains $Subcommand) { Write-Error "cbx: refusing subcommand '$Subcommand'. Allowed: $($allowed -join ', ')" exit 2 diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 4bb88af..3f6579d 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -2,8 +2,9 @@ ## 1. Overview -`codebase-index` is a **local-first** code intelligence layer for AI coding agents. In `1.9.0` -it has two shipped faces: +`codebase-index` is a **local-first** code intelligence layer for AI coding agents. This page is +the design reference; the hands-on guide (setup, tests, benchmarks, recipes) is +[DEVELOPMENT.md](DEVELOPMENT.md). It has two shipped faces: 1. **A Claude Code Skill** (`.claude/skills/codebase-index/SKILL.md`) that Claude auto-invokes for codebase questions. The skill is thin: it tells Claude *when* to search, *how* to call the CLI, @@ -71,7 +72,8 @@ codebase-index/ ├── adapters/ # per-CLI install logic (claude/codex/opencode, sh + ps1) ├── lib/ # shared shell helpers for the installer ├── bin/ # plugin wrappers (cbx resolves the provisioned venv) -├── scripts/ # bootstrap.sh/.ps1, release_smoke.py, sync_skill_copies.py +├── scripts/ # bootstrap.sh/.ps1, release_smoke.py, sync_skill_copies.py, +│ # check_versions.py, check_links.py, release_notes.py, gen_assets.py ├── hooks/ # plugin hooks.json (SessionStart bootstrap) ├── .claude-plugin/ # plugin manifest + marketplace catalog ├── .github/ # CI (lint, skill-sync gate, OS/Python test matrix), release @@ -81,6 +83,8 @@ codebase-index/ ├── .claude/ .codex/ .opencode/ # committed installed copies (generated — same script) ├── examples/ # sample queries, configs, hooks ├── tests/ # pytest suite + fixtures (sample_repo, multilang) +│ └── eval/ # retrieval evaluation: harness, metrics, gen_queries, run_eval, +│ # baselines + run_baselines (public-repo comparison), results/ └── src/codebase_index/ ├── cli.py # Typer app: all commands (delegates to service.py) ├── service.py # shared CLI/MCP service layer: paths, search sessions, stats @@ -94,8 +98,8 @@ codebase-index/ │ # symbol_chunks.py, base.py ├── indexer/ # pipeline.py (full + incremental build), freshness.py, │ # doc_chunks.py - ├── graph/ # builder.py (edge resolution), expand.py (impact), - │ # export.py (HTML graph) + ├── graph/ # builder.py (edge resolution), expand.py (impact), analysis.py + │ # (communities/god nodes), navigate.py (path/describe), export.py ├── storage/ # db.py (pragmas, schema, version guard), schema.sql, repo.py ├── retrieval/ # intent.py, searchers.py, fusion.py, rerank.py, priors.py, │ # lexical.py, fuzzy.py, diversity.py, skeleton.py, @@ -158,8 +162,9 @@ upward to nearest `.git`/`.claude`), and `--quiet`. Search-family commands accep | `clean` | `--yes`, `--all` | resets index DB (`--all` wipes cache dir) | removed-count | | `watch` | `--debounce ms` | long-running | event log | -The skill only ever calls the **read-only** family (`search`, `symbol`, `refs`, `impact`, `diff-impact`, -`explain`, `stats`) plus `update`. It never calls `clean` or `init`. See SECURITY.md. +The skill only ever calls the **read-only** family (`search`, `explain`, `architecture`, `symbol`, +`refs`, `impact`, `diff-impact`, `path`, `describe`, `graph`, `stats`, `doctor`) plus `update` and +`index`. It never calls `clean`, `init` or `watch`. See [SECURITY_MODEL.md](SECURITY_MODEL.md). ### Freshness contract @@ -192,9 +197,9 @@ If `exists=false` → skill runs `index`. If `stale=true` and cheap → skill ru ## 8. MCP server -The same `retrieval` + `storage` layers are wrapped in a stdio MCP server exposing tools like -`search_code`, `find_symbol`, `find_refs`, `impact_of`, `impact_of_diff`, `explain_code`, `index_stats`, and -`healthcheck`. +The same `retrieval` + `storage` layers are wrapped in a stdio MCP server exposing +`healthcheck`, `search_code`, `find_symbol`, `find_refs`, `impact_of`, `impact_of_diff`, +`explain_code`, `architecture_overview`, `path_between`, `describe_symbol`, and `index_stats`. Current implementation: diff --git a/docs/BENCHMARKS.md b/docs/BENCHMARKS.md index 5b1d7db..604237f 100644 --- a/docs/BENCHMARKS.md +++ b/docs/BENCHMARKS.md @@ -1,134 +1,168 @@ # Benchmarks -`codebase-index` has four benchmark surfaces. Read them with their status in -mind — the whole point of this page is to keep evidence and aspiration separate. - -| Surface | What it is | Status | Use it as | +The rule of this project is *measure improvements, don't assert them*. This page +lists every benchmark surface, what it can and cannot prove, and the exact +numbers we are willing to put in a README. Anything not on this page is not a +claim. + +## The one-paragraph version + +On three public repositories at pinned commits (Flask, Gson, Fastify: Python, +Java, JavaScript), across 450 questions mined from git history, `codebase-index` +put the right file in its top three **55% of the time versus 30% for a +disciplined ripgrep agent**, at roughly the same context cost (3.8k vs 3.5k +tokens per question, same tokenizer on both sides). Every metric delta is +significant at p < 0.001 with paired bootstrap confidence intervals. Raw run: +[tests/eval/results/2026-09-04-public-baselines.md](../tests/eval/results/2026-09-04-public-baselines.md). +Reproduce it with one command (below). That is the headline; the rest of this +page is the fine print. + +## Surfaces + +| Surface | What it measures | Status | Use it as | |---|---|---|---| -| Retrieval eval (`tests/eval/`) | Ranking quality vs a fixed baseline across multiple real repositories, with significance tests | **Proven (relative)** | The gate for ranking changes; measures *deltas*, not absolute superiority | -| Public suite (`tests/benchmark_public.py`) | Deterministic synthetic multi-language fixture with the full metric framework | **Toy/synthetic** | CI regression gate + metric shape, **not** product-quality evidence | -| Smoke/perf (`test_perf_smoke.py`, `test_benchmark_comparison.py`) | Latency + output-size guards on a tiny fixture | **Toy/smoke** | Regression checks only | -| Honest real-repo (`tests/benchmark_honest.py`) | 55k LOC Java repo, recall@3 vs disciplined `rg` baseline, symmetric token accounting | **Proven (one repo)** | The only headline product-quality number we stand behind today | - -### Claims that should NOT be made yet - -Do not write, imply, or ship any of these until a run with published logs exists: - -- Any 10k / 100k / 1M LOC scale or speed claim (no real run at that size). -- "Beats Cursor / Sourcegraph / Codebase-Memory MCP" — no head-to-head exists. -- Per-language *absolute* quality claims beyond Java. The retrieval eval covers - Python, Java and TypeScript, but it measures this system against its own earlier - versions — it says a change helped, not that the product beats an alternative. -- Generic "Nx faster" / "Nx fewer tokens" without naming the baseline and repo. -- Latency claims against external tools — the honest run explicitly does not - headline latency (Python process start dominates; real `rg` is tens of ms). - Version-over-version latency from the retrieval eval is in-process and is only - comparable to other runs of that harness. - -The defensible headline today is exactly: **on one 55k LOC Java repo, recall@3 was -70% (index) vs 40% (`rg`+window), using ~13× fewer answer tokens.** Everything -else is roadmap. - -## Retrieval evaluation (ranking gate) - -See [tests/eval/README.md](../tests/eval/README.md) for the full protocol. In short: -ground truth comes from hand-written queries verified against the source tree and -from commit-subject → changed-files pairs mined from git history (leak-free: the -query text is not in the indexed corpus). Corpora are pooled across languages, one -index per corpus is shared by every variant, and every non-baseline row carries a -paired bootstrap CI and permutation p-value. - -`1.9.0` was measured over 305 queries across Python, Java and TypeScript corpora -against `1.8.0`: MRR +0.027, MAP +0.028, nDCG@10 +0.024, recall@5 +0.031 (all -p < 0.001), with p50 latency 78.6 ms → 51.2 ms. These are version-over-version -ranking deltas on those corpora, not a universal quality claim. - -## Public benchmark suite +| **Public baselines** (`tests/eval/run_baselines.py`) | Index vs `rg`+window and vs repo-map-style context on public repos, symmetric tokens, significance | **Proven, reproducible** | The headline comparison against *not having an index* | +| **Retrieval eval** (`tests/eval/run_eval.py`) | Ranking quality of the shipped default vs the 1.7.0 baseline and one-signal ablations, pooled across corpora, with significance | **Proven (relative)** | The gate every ranking change must pass; measures deltas between versions, not superiority over other tools | +| Public synthetic suite (`tests/benchmark_public.py`) | Full metric framework on a deterministic toy fixture | Toy | CI regression gate for metric *shape*; not product evidence | +| Smoke/perf (`test_perf_smoke.py`, `test_benchmark_comparison.py`) | Latency and output-size guards on a tiny fixture | Toy | Regression checks only | +| Single-repo honest run (`tests/benchmark_honest.py`) | recall@3 and tokens vs `rg` on one 55k LOC Java repository | **Historical, not reproducible by others** | The repository is private; the script is kept because its read model informed the public baselines. Do not quote its numbers | -Run: +## Public baselines (the headline) ```bash -python tests/benchmark_public.py --workdir .tmp-public-benchmark +pip install -e ".[dev]" tiktoken +python tests/eval/run_baselines.py --clone --workdir .tmp-baselines \ + --out tests/eval/results/$(date +%F)-public-baselines ``` -The public suite builds a deterministic multi-language fixture repository and -reports JSON metrics: - -- Retrieval quality: `recall_at_1`, `recall_at_3`, `recall_at_5`, `mrr`, `ndcg_at_5` -- Agent usefulness: `answer_correctness_at_3` -- Token economy: index context tokens versus a grep-window baseline -- Language breakdown: per-language recall and answer-correctness proxy -- Freshness: stale detection after an edit and incremental update latency -- Graph tasks: callers/dependencies/impact checks -- Scale counters: indexed files, symbols, edges, and bytes - -Example output shape: - -```json -{ - "retrieval_quality": { - "recall_at_1": 0.75, - "recall_at_3": 1.0, - "recall_at_5": 1.0, - "mrr": 0.875, - "ndcg_at_5": 0.9077 - }, - "answer_correctness": { - "answer_correctness_at_3": 1.0 - }, - "token_economy": { - "index_tokens_avg": 25.0, - "grep_window_tokens_avg": 72.375, - "compression_vs_grep": 2.895 - } -} -``` - -The CI gate is `tests/test_public_benchmark.py`. It verifies the suite reports -all required metric families and catches obvious quality, freshness, graph, and -token-accounting regressions. - -## Honest real-repo benchmark +`--clone` fetches the three corpora at the commits pinned in the script. Pass +`--repo /path/to/any/git/repo` to benchmark your own repository the same way; +please post the result with the *benchmark report* issue template, especially +when it is unflattering. + +### Protocol + +- **Ground truth** is mined from git history by `tests/eval/gen_queries.py`: the + query is a human-written commit subject, the answer is the set of files that + commit changed. Neither is produced by the retriever, and the subject text is + not in the indexed corpus, so the set is leak-free by construction. The newest + 150 localised commits per repository are used (no merges, reverts, releases, + formatting commits, or sweeping changes over more than four files). +- **Changelog-style files** (`CHANGELOG*`, `CHANGES*`, `HISTORY*`, `NEWS*`) are + excluded both as answers and from the indexed corpus, because they paraphrase + commit subjects. +- **Index side**: `search()` with the shipped defaults (`limit 10`, + `token_budget 1500`, hybrid mode). Context = the full JSON payload the agent + receives, plus the top-3 `recommended_reads` ranges read in full. +- **rg + window**: stop-words dropped from the question, up to six salient terms + searched with real ripgrep, files ranked by match density, an 80-line window + read around the densest hit in each of the top-3 files. Context = the first + 50 lines of `rg` output plus the three windows. +- **repo-map-style**: file paths plus top-level definition signatures packed + under a 2k or 8k token budget, ordered by graph degree (or by identifier + overlap with the question in the *query-aware* variant). This is an + approximation of the *style* of context Aider builds, not Aider itself. A map + is not a ranking, so it is scored on whether the answer file is present at + all, an upper bound on what an agent could do with it. +- **Tokens** are counted with `tiktoken/cl100k_base` on every side. +- **Significance**: paired bootstrap 95% CI and paired permutation p-value, + seeded, index vs rg. + +### Logged run (2026-09-04, codebase-index 1.9.0) + +| corpus | commit | language | files | queries | +|---|---|---|---:|---:| +| pallets/flask | `d318b683` | Python | 229 | 150 | +| google/gson | `b3f4ca20` | Java | 311 | 150 | +| fastify/fastify | `15ebc8e2` | JavaScript | 389 | 150 | + +Pooled, n = 450: + +| method | hit@3 | recall@5 | MRR | packet / listing tokens | + top-3 reads | +|---|---:|---:|---:|---:|---:| +| codebase-index | **0.547** | **0.525** | **0.456** | 2,402 | 3,835 | +| rg + 80-line windows | 0.304 | 0.293 | 0.263 | 1,351 | 3,495 | + +| index − rg | delta | 95% CI | p | +|---|---:|---:|---:| +| hit@3 | +0.242 | [+0.187, +0.298] | < 0.001 | +| recall@5 | +0.231 | [+0.183, +0.278] | < 0.001 | +| MRR | +0.194 | [+0.151, +0.235] | < 0.001 | +| tokens | +340 | [+238, +440] | < 0.001 | + +Repo-map-style context, pooled: the answer file is present in a 2k-token map +38% of the time (42% query-aware) and in an 8k-token map 99% of the time. Those +repositories are small enough that 8k tokens covers nearly every file, so the +8k row says little; on a large monorepo it would not. + +![chart](../assets/benchmark.svg) + +### How to read it honestly + +- The index is **1.8× more likely** to put the answer in the top three than a + disciplined grep agent, at **about 10% more context tokens**. It is not + "13× fewer tokens"; that earlier number came from charging the index for + signature snippets and grep for code windows, which is not symmetric. The + public run charges both sides for what actually enters context. +- The token parity is *new in this line*. Before the `max_read_lines` cap + (1.9.1), the index's follow-through reads averaged 6.8k tokens because a + symbol-aligned chunk can be a whole 1,500-line class; the benchmark found + that, and the cap fixed it without touching ranking (the "uncapped" row in the + raw log has identical quality). +- Ground truth is one commit's files. A question can have a correct answer the + commit did not touch, so absolute numbers understate every method equally. + Read the deltas, and read the confidence interval before the delta. +- Per-corpus results move together (hit@3 index/rg: 0.49/0.23 Flask, + 0.57/0.35 Gson, 0.58/0.33 Fastify), which is the point of pooling three + languages: the effect is not a Python artefact. +- Latency is deliberately not tabulated. The index runs in-process, ripgrep is + a separate binary; neither number is a fair claim against the other. + +## Retrieval evaluation (the ranking gate) + +See [tests/eval/README.md](../tests/eval/README.md) for the full protocol. This +harness compares the shipped ranker with its own earlier version and with +one-signal-off ablations, pooled across corpora, with the same significance +machinery. It cannot say the index beats an alternative; it says whether a +change to the ranker helped. -Run: - -```bash -python tests/benchmark_honest.py --repo /path/to/real/repo --rebuild -``` - -The current documented run is against a 55k LOC Java repository: -[tests/benchmark_honest_RESULTS.md](../tests/benchmark_honest_RESULTS.md). - -That benchmark compares the index against a disciplined grep-window agent with -objective recall@3 ground truth. - -## Smoke/perf benchmark - -`tests/test_benchmark_comparison.py` and `tests/test_perf_smoke.py` guard basic -latency and output-size behavior. They are useful regression checks, not product -quality evidence. - -## Remaining benchmark work (TODO checklist) +`1.9.0` was measured over 305 queries across Python, Java and TypeScript corpora +against `1.8.0`: MRR +0.027, MAP +0.028, nDCG@10 +0.024, recall@5 +0.031 (all +p < 0.001), with p50 latency 78.6 ms → 51.2 ms. Two of the three corpora used +for that run are private; the generator is the reproducible part, and the public +baselines above use the same generator on public repositories. -The public suite has the metric framework; the next step is real, larger, -documented repositories. Each task must publish raw logs alongside any headline -number (the pattern set by `tests/benchmark_honest_RESULTS.md`). +Every ranking signal that ships has an ablation row. 1.9.0 removed two signals +that could not demonstrate a benefit and rejected several plausible ones +(IDF-weighted coverage, stemming, graph propagation, MMR, a file-length prior). -- [ ] **10k LOC public repo** — Recall@1/3/5, MRR, nDCG, token economy; named repo + commit SHA. -- [ ] **100k LOC public repo** — same metrics, plus full index build time and incremental update latency. -- [ ] **1M LOC target** — feasibility + scale counters (files/symbols/edges/bytes); may be partial. -- [ ] **Multi-language repo** (≥3 Tier-A languages) — per-language recall and answer-correctness breakdown. -- [ ] **vs vanilla agent grep/read** — tokens and recall against an undisciplined agent exploring the same questions. -- [ ] **vs repo-map-style context** — tokens and recall against an Aider-repo-map-style context blob. -- [ ] **Graph task benchmark** — `refs`, `impact`, and route→handler→service paths against hand-labeled ground truth. -- [ ] **Answer grading** — human-reviewed expected answers, not just file-level recall proxies. -- [ ] **Framework graph tasks** — migrations, config consumers, CI/infra wiring once typed edges land. +## Claims that must NOT be made -How to add one without overclaiming: +Do not write, imply, or ship any of these until a run with published logs exists: -1. Pick a public repo; record its URL and commit SHA. -2. Derive ground truth independently of the index (e.g. naming convention), so the - index cannot grade its own homework. -3. Use a symmetric token estimator and read window on both sides. -4. Commit the raw run output next to a short `*_RESULTS.md` summary. -5. Only then update README/COMPARISON headline numbers. +- Any 1M-LOC or monorepo scale or speed claim (no real run at that size). +- "Beats Cursor / Sourcegraph / Aider / Codebase-Memory MCP": no head-to-head + exists. The repo-map baseline here is a style approximation, not Aider. +- Any *token savings* multiplier without naming the read model. Under + symmetric accounting the index costs about the same as disciplined grep. +- Latency comparisons against external tools. +- Any statement about *LLM agent task success*. The baselines here are + retrieval metrics; nobody has measured whether an agent with the index + finishes tasks faster or better. That is the most important open item. + +## Future work, in priority order + +1. **Agent task-level evaluation**: the same questions, an actual coding agent + (Claude Code or Codex CLI) with and without the skill, measuring answer + correctness, files read, tokens, and wall time. Needs model calls and a + grading rubric; not started. +2. **Large repository run**: a 500k–1M LOC monorepo, reporting index build + time, incremental update latency, memory, and the same retrieval metrics. +3. **Graph task benchmark**: hand-labelled `refs`, `impact`, and + route → handler → service paths. +4. **Framework-aware edges** benchmark once typed edges land (roadmap). + +How to add a benchmark without overclaiming: pick a public repository and pin +the commit; derive ground truth independently of the index; use one token +estimator and one read model on both sides; commit the raw run next to a short +summary; only then touch README numbers. diff --git a/docs/COMMUNITY_AUDIT.md b/docs/COMMUNITY_AUDIT.md new file mode 100644 index 0000000..9ca7b76 --- /dev/null +++ b/docs/COMMUNITY_AUDIT.md @@ -0,0 +1,262 @@ +# Community readiness audit + +Date: 2026-09-04. Baseline: `main` at 1.9.0 (`b48ed14`). Every finding lists what +was wrong, why it matters for adoption, the fix, and its status on the +`chore/community-readiness` branch. Severity: + +- **P0** blocks adoption or trust +- **P1** major friction +- **P2** worthwhile improvement +- **P3** polish + +Status legend: **fixed** (in this branch), **partial**, **open** (needs the +maintainer or is future work). + +## Summary + +The engineering under the hood was already unusually careful: a shared service +layer for CLI/skill/MCP, golden snapshots for every payload, a coverage gate, +an ablation harness with significance tests. What kept a serious developer from +trusting it in the first minute was the *presentation of evidence*: the README's +headline number came from a private repository with asymmetric token +accounting, several docs showed invented output, three doc pairs contradicted +each other, and the demo image was a mock-up. Those are fixed. The remaining +gaps are things only a maintainer can do (enable Discussions, upload the social +preview, apply labels, record a GIF) and two real benchmark holes (agent +task-level evaluation, a large-monorepo run). + +Scores are the maintainer-facing view; the definitions are at the end. + +| Dimension | Before | After | +|---|---:|---:| +| Technical quality | 72 | 84 | +| Onboarding | 62 | 82 | +| Documentation | 58 | 82 | +| Credibility | 45 | 78 | +| Discoverability | 65 | 70 | +| Contributor readiness | 50 | 82 | +| Demo quality | 30 | 75 | +| Launch readiness | 35 | 72 | + +## P0 + +### P0-1 Headline benchmark was not reproducible and not symmetric — fixed + +- **Wrong:** README, COMPARISON and BENCHMARKS led with "recall@3 70% vs 40%, + ~13× fewer tokens" from `tests/benchmark_honest.py` on `NewTowny`, a private + repository on the maintainer's disk. The 13× compared index signature snippets + (~27 tokens) with grep's 80-line windows. Nobody outside could run it. +- **Why it matters:** the first thing a Hacker News reader does with a benchmark + claim is try to reproduce it. A private corpus plus asymmetric accounting reads + as marketing, and it discredits the genuinely good ablation harness next to it. +- **Fix:** `tests/eval/run_baselines.py` runs on Flask, Gson and Fastify at pinned + commits with git-derived ground truth, real ripgrep, one tokenizer on both sides, + and significance tests. Logged run committed. Honest result: hit@3 0.547 vs 0.304 + (p < 0.001) at roughly equal tokens. The 13× claim is withdrawn everywhere; the + old script is labelled historical. + +### P0-2 Read plan cost more tokens than grep — fixed + +- **Wrong:** the new benchmark found the index's follow-through reads averaged + 6.8k tokens vs 3.5k for grep, because symbol-aligned chunks can span a whole + 1,500-line class and `recommended_reads` handed that span to the agent verbatim. +- **Why it matters:** the product promise is "read less"; under fair accounting it + was reading more. +- **Fix:** `retrieval.max_read_lines` (default 120) caps read-plan entries at the + definition head, with additive `truncated` / `line_end_full` fields. Tokens + 3.8k, ranking quality unchanged (the "uncapped" row in the log is identical on + every quality metric). + +### P0-3 Fabricated examples in user-facing docs — fixed + +- **Wrong:** QUICKSTART showed an `AuthService.ts` results table with a `Score` + column the renderer does not print; INSTALLATION showed an invented `doctor` + transcript; `assets/demo.png` was a mock-up with made-up paths; README's JSON + example was invented. +- **Why it matters:** a user who runs the quick start and sees different columns + assumes the tool is broken or the docs are stale. Either way they leave. +- **Fix:** every example is now captured output on `pallets/flask@d318b683`; + `examples/demo/EXPECTED_OUTPUT.md` is the transcript; the hero image is an SVG + rendered from it by a committed script. + +### P0-4 Plugin wrappers refused four documented commands — fixed + +- **Wrong:** `bin/cbx` and `bin/cbx.ps1` whitelisted ten subcommands; the skill + advertised fourteen. `architecture`, `diff-impact`, `path`, `describe` failed + from the Claude Code plugin with "refusing subcommand". +- **Why it matters:** the plugin is the lowest-friction install path and it broke + the Trace/Predict half of the pitch. +- **Fix:** whitelists aligned; `tests/test_plugin_wrappers.py` pins all four + wrappers to the same set. + +### P0-5 Security policy had no reporting channel — fixed + +- **Wrong:** SECURITY.md said "email the maintainers" with no address and listed + 1.2.x as the supported version. +- **Why it matters:** a tool that reads whole repositories must have a working + private disclosure path; an unreachable policy is worse than none. +- **Fix:** GitHub private vulnerability reporting URL, supported line 1.9.x, + explicit statement that checksums/SBOM are not yet shipped. **Open for + maintainer:** confirm private vulnerability reporting is enabled in repository + settings (Security → Policy). + +### P0-6 MCP server did not start on the current SDK — fixed + +- **Wrong:** `pip install "codebase-index[mcp]"` resolves `mcp>=1.0` to mcp 2.x, + where `FastMCP` was renamed `MCPServer` and the old import path raises. The + server printed "needs the optional extra" with the extra installed, and the + documented `codebase-index mcp --root ` form was rejected because + `--root` was only a global option. CI never noticed: the MCP test files + wrapped *our own module's* import in the skip guard, so on mcp 2.x all 41 MCP + tests skipped silently. +- **Why it matters:** "MCP-ready" was on the README badge line and the MCP path + is how Cursor/Zed/Windsurf/Claude Desktop users arrive. +- **Fix:** import `MCPServer` first, fall back to `FastMCP`; `--root` accepted on + the subcommand; tests skip only when the SDK itself is absent. Verified against + mcp 1.29.1 and 2.1.1 with a stdio initialize / tools/list / healthcheck + round-trip. + +## P1 + +### P1-1 Benchmark corpus leakage — fixed + +`gen_queries.py` documented that changelog-style files were excluded from the +evaluation corpus; the harness never imported the list, and Flask's `CHANGES.rst` +was not matched at all. Fixed in both places with regression tests. The 1.9.0 +version-over-version numbers were measured with the changelog in the corpus; +re-run on this repository after the fix, the shipped default still beats the +1.7.0 baseline on every metric at p ≤ 0.004, so the conclusion stands. + +### P1-2 Three duplicate doc pairs with contradictions — fixed + +`SCHEMA.md` vs `DATABASE_SCHEMA.md` (the latter described columns and tables that +do not exist), `RETRIEVAL.md` vs `RETRIEVAL_PIPELINE.md`, `docs/SECURITY.md` vs +`SECURITY_MODEL.md` (both claimed `doctor` checks that `doctor.py` does not +implement). One canonical page each; the others are redirect stubs. + +### P1-3 Two roadmaps — fixed + +Root `ROADMAP.md` (milestones, stale: said PyPI not shipped) and `docs/ROADMAP.md` +(product). Root is now a milestone history table; docs owns forward work. + +### P1-4 No contributor path — fixed + +CONTRIBUTING pointed at `uv sync --all-extras` (no lockfile) and `ruff format +--check` (not enforced, fails on 88 files). No architecture walk-through for a +newcomer, no recipe for the contributions the project actually wants. Added +`docs/DEVELOPMENT.md` with verified commands and recipes (language, edge kind, +ranking signal, MCP tool, CLI command), rewritten CONTRIBUTING with label → area → +starting-file table and evidence rules for ranking PRs. + +### P1-5 Release notes were auto-generated PR lists — fixed + +The changelog is well written but GitHub releases showed only "What's Changed". +`scripts/release_notes.py` now provides the body from the changelog section, and +an empty section fails the release job. `check_versions.py` refuses a tag whose +name does not match `__version__`. + +### P1-6 Packaging only tested at tag time — fixed + +`python -m build`, `twine check` and the clean-venv install smoke ran only in +`release.yml`. A `package` job now runs them on every PR. + +### P1-7 Stale version pins and wrong config path across docs — fixed + +`@v1.8.0` install pins, `1.4.0` doctor output, `.codeindex.json` (the config +lives at `.claude/cache/codebase-index/config.json`), FAQ MCP tool list missing +three tools, bug template placeholder `0.1.0`. All corrected; +`check_versions.py` and `check_links.py` run in CI so the next drift is caught. + +### P1-8 Discussions disabled, issue chooser allowed blank issues — partial + +`config.yml` now disables blank issues and links Discussions and the security +form. **Open for maintainer:** enable Discussions (Settings → General → +Features) or the link 404s; run `scripts/apply_labels.sh` once to create the +labels the templates and CONTRIBUTING refer to. + +## P2 + +### P2-1 No demo on a real repository — fixed + +`examples/demo-project/README.md` described a project that does not exist in the +repo. Replaced by `examples/demo/` on Flask with scripts for bash and PowerShell, +expected output, and `docs/DEMO.md` with VHS/asciinema recording steps. +**Open for maintainer:** record the GIF (the tape file is ready); the README +uses the static SVG until then. + +### P2-2 README first screen — fixed + +Twelve badges, a mock image, and a value proposition that needed the second +screen to explain "why not grep". Rebuilt: one-sentence proposition, agent list, +real terminal card, 60-second try-it, "why this exists" with the grep / repo-map +/ cloud comparison, evidence with the chart, then everything else. + +### P2-3 Social preview not uploaded — open + +`assets/social-preview.png` exists; `usesCustomOpenGraphImage` is false. Upload +under Settings → Social preview. Links on X/Slack/Discord render without an +image until then. + +### P2-4 `docs/SEO.md` carried stale launch templates — fixed + +Templates quoted v1.3.0 and `git+https` installs. Folded into +`docs/COMMUNITY_LAUNCH.md` and deleted. + +### P2-5 Installer docs in Russian — fixed + +`docs/installer.md` translated; flags verified against `install.sh`. +**Open:** `install.sh`, `install.ps1`, `lib/`, `adapters/` still print Russian +log/help strings. Low impact (the PyPI path is primary) but confusing for +contributors reading the scripts. + +### P2-6 `explain` lacks `--limit` — open + +`search` accepts `--limit`; `explain` does not, contradicting ARCHITECTURE's +"search-family commands accept `--limit`". Small, additive; left for a follow-up +PR since it changes the CLI contract and goldens. + +### P2-7 MCP client configs unverified — open + +Templates for Cursor / VS Code / Zed / Windsurf are marked unverified. Needs a +person with each client to confirm; the SUPPORT page asks for exactly that. + +### P2-8 `workflow_dispatch` on release.yml can publish untagged builds — open + +A manual run publishes whatever `main` builds to PyPI without a tag check. Low +risk while the maintainer is the only one with the button, but consider +requiring a version input. + +## P3 + +- `pyproject.toml` keywords gained `code-graph`, `impact-analysis`, + `claude-code-plugin`; classifiers unchanged. — fixed +- `assets/demo.png` and its generator removed; `gen_assets.py` still needs + Windows fonts (documented). — fixed / open +- `docs/superpowers/` contains internal planning documents. They are honest + history and cost nothing, but a newcomer may mistake them for current + design. A one-line README in that folder would help. — open +- CRLF/LF: `.gitattributes` covers scripts; Markdown files get CRLF on Windows + checkouts, producing noisy `git diff` warnings for Windows contributors. + Consider `* text=auto eol=lf`. — open + +## What was verified, not just read + +- Full test suite green in a clean venv (`pytest`, coverage 84%), `ruff check`, + `mypy`, `sync_skill_copies.py --check`, `check_versions.py`, `check_links.py`, + `python -m build` + `twine check` + `release_smoke.py`. +- Every README command run on Flask at the pinned commit; output captured. +- Public baseline benchmark run twice (before and after the read cap); both logs + kept in the raw JSON (`index (uncapped reads)` row). +- Self-repository retrieval eval re-run after the leakage fix. + +## Score definitions + +- **Technical quality:** tests, contracts, CI matrix, known bugs in shipped paths. +- **Onboarding:** minutes from landing to first useful result, with no guesswork. +- **Documentation:** accuracy first, then coverage, then duplication. +- **Credibility:** can a sceptic reproduce every number and every example? +- **Discoverability:** description, topics, PyPI page, social card, search terms. +- **Contributor readiness:** can a stranger land a useful PR in an evening? +- **Demo quality:** does the demo show real value on real code, reproducibly? +- **Launch readiness:** are the texts, links, and community channels ready for + the first wave of strangers? diff --git a/docs/COMMUNITY_LAUNCH.md b/docs/COMMUNITY_LAUNCH.md new file mode 100644 index 0000000..63da4db --- /dev/null +++ b/docs/COMMUNITY_LAUNCH.md @@ -0,0 +1,281 @@ +# Community launch kit + +Drafts for the first public wave, plus the feedback loop that decides what +happens after it. Every text below uses only claims backed by logs in this +repository. Do not add numbers that are not in +[BENCHMARKS.md](BENCHMARKS.md), and do not add social proof that does not exist. + +The one benchmark sentence every post may use: + +> On three public repos (Flask, Gson, Fastify; 450 questions mined from git +> history) the index put the right file in its top three 55% of the time vs 30% +> for a disciplined ripgrep agent, at about the same context tokens (p < 0.001, +> raw logs and a one-command rerun in the repo). + +Repository: https://github.com/denfry/codebase-index · PyPI: `pip install codebase-index` + +## Pre-launch checklist (maintainer) + +- [x] Enable **Discussions** (Settings → General → Features); `config.yml`, + `SUPPORT.md` and `DEVELOPMENT.md` link to it. +- [ ] Upload `assets/social-preview.png` (Settings → Social preview). +- [x] Run `scripts/apply_labels.sh` once (needs `gh auth`). +- [x] Confirm private vulnerability reporting is enabled (Security tab). +- [ ] Record the GIF from `docs/demo.tape` (optional; the SVG card works). +- [ ] Tag the release that contains this branch, so the README, PyPI page and + release notes agree. +- [x] Repository description (About): *Local code map for AI coding agents: + find, trace, and predict change impact with file:line evidence. Claude + Code · Codex · OpenCode · MCP. No network by default.* +- [x] Topics (20 max, current set is fine): `ai-agents ai-coding claude-code + cli code-search codebase-indexing codex-cli context-engineering + developer-tools fts5 local-first mcp opencode python rag + semantic-code-search sqlite token-optimization tree-sitter + cursor-alternative`. If a slot frees, swap `cursor-alternative` for + `code-graph`; it describes the product better than a comparison. + +## Hacker News + +**Title** (under 80 chars, no superlatives): + +> Show HN: codebase-index – a local code map so Claude Code/Codex open the right file + +**Post:** + +> I built this because I kept watching Claude Code and Codex grep for a keyword, +> open the wrong file, then read three whole files to recover. On anything past a +> few hundred files that was most of the token bill and a good share of the wrong +> answers. +> +> codebase-index is a Python CLI that indexes a repo into SQLite (FTS5 + Tree-sitter +> symbols + an import/call/reference/inheritance graph) and answers three questions +> with file:line evidence: where is X implemented, what calls this / how does A +> reach B, and what breaks if I change this (including "what does my current diff +> touch"). It installs as a Claude Code skill/plugin, a Codex AGENTS.md block, an +> OpenCode command, or a stdio MCP server; all four call the same service layer. +> +> Things I tried to get right: +> +> - Every answer carries its uncertainty: index freshness, result confidence, +> graph coverage per language, and each edge tagged extracted/inferred/ambiguous, +> so the agent can tell "nothing references this" from "the graph doesn't know". +> - Local only. No network in the base install, no telemetry; secrets excluded +> before parsing and redacted again on output; `doctor --strict` audits that. +> - Measured, not asserted. Ranking signals ship only if they survive an ablation +> with significance tests on a pooled multi-language query set. Against a +> disciplined ripgrep agent on Flask, Gson and Fastify (450 questions mined from +> git history, same tokenizer on both sides) it puts the right file in the top +> three 55% vs 30% at about the same context tokens. Raw logs and a one-command +> rerun are in the repo; I'd genuinely like people to run it on their own repos +> and post the numbers, especially bad ones. +> +> What it is not: an IDE, a cloud search, or an agent. It also does not yet have +> framework-aware edges (routes → handlers → DB) or an agent task-level benchmark; +> both are on the roadmap and I say so in the README rather than implying them. +> +> https://github.com/denfry/codebase-index + +## Reddit + +Post once per community, days apart, with the wording matched to the audience. +Reply to every substantive comment; do not cross-post identical text. + +### r/ClaudeAI (or the Claude Code community) + +> **A local index skill so Claude Code opens the right file instead of grepping around** +> +> I wrote a skill + plugin that gives Claude Code a ranked, line-precise answer to +> "where is X", "who calls this", and "what breaks if I change this" before it +> opens anything. `pip install codebase-index`, `codebase-index init --target +> claude`, done; or `/plugin marketplace add denfry/codebase-index`. +> +> It's local (SQLite + Tree-sitter, no network, no telemetry), and it tells Claude +> when it isn't sure: freshness, confidence, and per-edge extracted/inferred/ +> ambiguous tags. There's also a PostToolUse hook that keeps the index fresh while +> Claude edits. +> +> Numbers, since "it feels better" is worthless: on Flask/Gson/Fastify with 450 +> questions mined from git history, top-3 hit rate 55% vs 30% for a disciplined +> ripgrep agent at about the same context tokens. Raw logs in the repo; the +> benchmark runs on any git repo with one command, and I'd love to see your +> results. +> +> Repo: https://github.com/denfry/codebase-index + +### r/LocalLLaMA / local AI tooling + +> **Local-first code retrieval + graph for coding agents (no embeddings required, MCP server included)** +> +> codebase-index builds a SQLite index of a repository: FTS5 for text, Tree-sitter +> symbols for 12 languages, and an import/call/reference/inheritance graph. It +> serves ranked file:line evidence, reference lookups, dependency paths and +> change-impact analysis over a CLI, a stdio MCP server, and skills for Claude +> Code/Codex/OpenCode. Embeddings are optional (local sentence-transformers via +> sqlite-vec); the default path is purely lexical + structural and makes zero +> network calls. External embedding APIs are refused unless you opt in three +> separate ways. +> +> Benchmarks are the part I care about most: a multi-language ablation harness +> with significance tests, and a public comparison against a ripgrep agent and a +> repo-map-style context blob on Flask/Gson/Fastify (55% vs 30% top-3 hit rate, +> tokens roughly equal). Everything reruns with one command. +> +> https://github.com/denfry/codebase-index + +### r/programming + +> **How I benchmarked a code index against grep without cheating (and what the first run got wrong)** +> +> Short write-up on building a retrieval layer for coding agents and the benchmark +> mistakes I made: a headline number from a private repo, token accounting that +> charged my tool for signatures and grep for whole windows, and a changelog that +> leaked commit subjects into the corpus. The fixed protocol mines ground truth from +> git history (commit subject → files changed), charges both sides with the same +> tokenizer for what actually enters context, and reports paired bootstrap CIs. +> The first honest run showed my read plan cost *more* tokens than grep, which led +> to a real fix. Code, logs, and the harness are all in the repo. +> +> https://github.com/denfry/codebase-index/blob/main/docs/BENCHMARKS.md + +### r/opensource / developer tools + +> **codebase-index: MIT, local, benchmarked code map for AI coding agents; looking for benchmark reproductions** +> +> The most useful contribution right now is running +> `python tests/eval/run_baselines.py --repo /path/to/your/repo` and posting the +> table with the benchmark-report issue template, especially if it's unflattering. +> Second most useful: a Tier-A language spec (Swift, Dart, Scala are the gaps). +> DEVELOPMENT.md gets you from clone to PR in about fifteen minutes. +> +> https://github.com/denfry/codebase-index + +## X / Twitter + +**Short announcement** + +> codebase-index: a local code map for Claude Code, Codex, OpenCode and MCP. +> Find the right file, trace what calls it, predict what a change breaks. SQLite + +> Tree-sitter, no network, no telemetry. `pip install codebase-index` +> https://github.com/denfry/codebase-index + +**Technical announcement** + +> Benchmarked codebase-index against a disciplined ripgrep agent on Flask, Gson +> and Fastify: 450 questions mined from git history, same tokenizer on both sides. +> Top-3 hit rate 0.55 vs 0.30, MRR 0.46 vs 0.26, p<0.001, ~same context tokens. +> Raw logs + one-command rerun: https://github.com/denfry/codebase-index/blob/main/docs/BENCHMARKS.md + +**Thread** + +> 1/ I built codebase-index because Claude Code and Codex kept opening the wrong +> file. Not because they're bad at reading; because grep is bad at deciding. +> Here's what a local code map does differently, with numbers. 🧵 +> +> 2/ Three questions, one SQLite file: FIND (ranked file:line ranges), TRACE +> (callers, shortest dependency path, each edge tagged extracted/inferred/ +> ambiguous), PREDICT (blast radius of a symbol, or of your current diff). +> +> 3/ Every answer carries its uncertainty: index freshness, confidence, graph +> coverage per language. An agent can tell "nothing references this" from "the +> graph doesn't know". grep can't. +> +> 4/ Local only. No network in the base install, no telemetry. Secrets excluded +> before parsing and redacted on output. `codebase-index doctor --strict` audits +> that and fails CI if a gate is off. +> +> 5/ Benchmarks: ground truth mined from git history (commit subject → files +> changed) so neither the tool nor I wrote the answer key. Flask + Gson + Fastify, +> 450 questions, ripgrep with 80-line windows as the baseline, one tokenizer. +> +> 6/ Result: top-3 hit rate 0.55 vs 0.30, MRR 0.46 vs 0.26, p < 0.001, about the +> same context tokens per question. The first run showed my read plan cost MORE +> tokens than grep; the fix (cap reads at the definition head) is in the same PR. +> +> 7/ What it doesn't do yet: framework-aware edges (route → handler → DB), agent +> task-level evaluation, a 1M-LOC run. All on the roadmap, none implied. +> +> 8/ `pip install codebase-index`, `codebase-index init`. Works as a Claude Code +> plugin, Codex AGENTS.md block, OpenCode command, or MCP server. Please run the +> benchmark on your repo and tell me what breaks. +> https://github.com/denfry/codebase-index + +## Discord (developer communities, one message) + +> **codebase-index** — local code map for coding agents (Claude Code / Codex / +> OpenCode / MCP). Ranked file:line search, callers and dependency paths with +> per-edge confidence, and change-impact for a symbol or your current diff. SQLite +> + Tree-sitter, no network, no telemetry. Benchmarked vs a ripgrep agent on +> Flask/Gson/Fastify: top-3 hit rate 55% vs 30% at ~equal tokens, raw logs in the +> repo. `pip install codebase-index` · https://github.com/denfry/codebase-index +> Happy to answer questions here, and benchmark reproductions on your repos are the +> most useful feedback I can get. + +## GitHub release announcement template + +Use `python scripts/release_notes.py` for the body; prepend this header when a +release is worth announcing beyond the changelog. + +> **codebase-index vX.Y.Z** +> +> One paragraph: what changed for a user, in plain words. +> +> **Evidence:** the metric that moved, with the log path +> (`tests/eval/results/-*.md`) or "no ranking change". +> +> **Upgrade:** `pip install -U codebase-index`; run `codebase-index index` if the +> changelog lists a schema bump; installed skills auto-update on the next command +> (`codebase-index skill-update` to force, `skill-rollback` to undo). +> +> _(changelog section follows)_ + +## Where to publish, in order + +Day 0 is the day the release with this branch is tagged. + +| When | Action | Goal | +|---|---|---| +| Day 0 | Tag + PyPI publish; verify `pip install codebase-index==X.Y.Z` on a clean machine; post the GitHub release with the template above | Everything a visitor clicks works | +| Day 0 | Discord message in the Claude Code and one AI-tooling community | Early, technical feedback from people who already use skills/MCP | +| Day 1 | Reddit: Claude Code community post | Install attempts from the primary audience; watch `doctor` bug reports | +| Day 1 | X short announcement + technical announcement | Link in circulation before HN | +| Day 3 | Show HN, morning US Pacific, weekday | Sceptical readers; expect benchmark methodology questions; answer with logs, not adjectives | +| Day 3 | X thread | Reuse HN discussion points | +| Week 1 | Reddit: r/LocalLLaMA and r/programming posts (the benchmark write-up), spaced two days apart | Reach people who care about local and about methodology | +| Week 1 | Submit to awesome-claude-code, awesome-mcp-servers, awesome-code-search lists; PyPI page check | Long-tail discovery | +| Week 2 | Publish a "what people reported" note in Discussions: bugs fixed, benchmark reproductions received, what changed | Show that feedback lands somewhere | +| Week 2 | Patch release with the first-wave fixes | Convert first-week issues into visible responsiveness | +| Month 1 | Pick the highest-signal gap from the loop below and ship it (likely agent task-level benchmark or MCP client verification); post the result where the question was first asked | Retention signal, second wave | + +## Feedback loop + +**Collect** + +- Install failures (`doctor` output) by OS and Python version. +- First-query outcomes: did `search` return a useful top-3 on the user's repo? + Ask for `--json` output with paths redacted if needed. +- Benchmark reproductions via the benchmark-report template, including negative + ones; each becomes a row in a Discussions thread. +- Integration requests by client (Cursor, Zed, Windsurf, VS Code): the + unverified MCP templates need exactly these people. +- Language requests, ranked by how many repos mention them. + +**Metrics that matter**, in order: + +1. Repeat usage: users who run `update`/`search` on day 7 (ask in Discussions; + there is no telemetry and there will not be). +2. Issues from people who are not the maintainer, and how many are closed + within a week. +3. External pull requests merged. +4. Benchmark reproductions posted, with the spread of results. +5. Forks that carry commits. +6. Integration requests and confirmations ("works with Zed 0.x"). +7. PyPI downloads, as a trend only. +8. Stars, as a lagging indicator. + +**Do not optimise for**: stars, follower counts, being first on a list, +"AI tool of the week" placements, or any metric that a bot could move. No +purchased engagement, no automated cross-posting, no fake issues, no +"reminder" comments in other projects' threads, no unsolicited DMs. + +**Respond to feedback with**: a reproduction or a log, a linked commit, or a +roadmap entry that names the gap. Never with a promise that has no issue number. diff --git a/docs/COMPARISON.md b/docs/COMPARISON.md index bed7020..5540628 100644 --- a/docs/COMPARISON.md +++ b/docs/COMPARISON.md @@ -40,12 +40,12 @@ platform. | Agent interface | CLI, Claude Code skill, Codex instructions, OpenCode resources, stdio MCP server | Cursor IDE | Aider CLI | IDE + Sourcegraph platform | MCP clients | Any shell-capable agent | | Retrieval granularity | File, symbol, line range, references, impact graph | IDE-managed code context | Repo map of important files/classes/functions/signatures within token budget | Code search and code graph | Varies by server | File/line text matches | | Offline guarantee | Default local/offline; external embeddings opt-in | Local IDE indexing plus model/provider behavior depends on setup | Local repo map; model calls depend on Aider config | Cloud by default | Varies; often local | Local | -| Benchmarked quality | Public suite with Recall@1/3/5, MRR, nDCG, tokens, freshness, graph tasks; honest 55k LOC Java run | Public methodology not portable to this repo | No direct local benchmark here | Enterprise/product claims; not benchmarked here | Varies | Baseline only | +| Benchmarked quality | Public multi-repo baseline run (Flask, Gson, Fastify) vs `rg`+window and repo-map-style context, with significance; version-over-version ranking eval with ablations | Public methodology not portable to this repo | No direct local benchmark here | Enterprise/product claims; not benchmarked here | Varies | Baseline only | | Multi-repo support | Single repo today | Workspace/project scoped | Current chat repo/worktree | Strong cross-repo support | Varies | Manual | | Language coverage | Tier-A: 12 code languages; Tier-B generic path; configs mostly FTS | IDE proprietary | Tree-sitter repo map support as provided by Aider | Broad enterprise language coverage | Varies | Any text | | Security posture | `.gitignore`/`.codeindexignore`, secret filename gates, redaction, no telemetry, no network by default | Proprietary behavior; depends on settings | Local map, but model provider path depends on Aider config | Requires platform trust and account policy | Varies by server | No built-in redaction | | Update model | Manual `index`/`update`, hooks, optional watcher | IDE-managed | Rebuilt as Aider manages context | Platform-managed | Varies | Always live but manual | -| Extensibility | CLI `--json`; MCP schema v1.0; SQLite local DB | Limited external contract | Aider internals/context | Sourcegraph APIs | MCP by design | Shell pipelines | +| Extensibility | CLI `--json`; MCP envelope `schema_version` 1; SQLite local DB | Limited external contract | Aider internals/context | Sourcegraph APIs | MCP by design | Shell pipelines | ## When to choose what @@ -133,8 +133,8 @@ This is the closest direct alternative, so the comparison is the most careful. OpenCode workflow. - **Transparency:** readable Python, 80% coverage gate, golden CLI snapshots, and a public benchmark suite wired as a CI regression gate. - - **Honest benchmarks:** we publish raw logs (see the 55k LOC Java run) and mark - unproven scale/graph claims as roadmap. + - **Honest benchmarks:** raw logs are committed next to every headline number + (`tests/eval/results/`), and unproven scale/graph claims stay on the roadmap. - **Choose Codebase-Memory MCP when:** you need its broader graph engine, static-binary distribution, or wider language/agent reach today. - **Choose codebase-index when:** you want a simpler, privacy-strict, transparent @@ -157,24 +157,24 @@ The meaningful distinction is different: - Aider's map is good context injection; `codebase-index` aims to be a queryable index with freshness checks, ignore/security gates, and agent-readable packets. -## Competitive benchmark bar +## Benchmark bar -Tiny fixture benchmarks are useful smoke tests, not product evidence. The -competitive bar now includes graph-first systems that report evaluations over -dozens of real repositories, large language coverage, answer-quality scoring, -token savings, and tool-call reductions. +Tiny fixture benchmarks are smoke tests, not product evidence. What this repository +can show today, with raw logs committed: -`codebase-index` now includes a public benchmark suite (`tests/benchmark_public.py`) with: +- **Index vs a disciplined grep agent** on three public repositories (Flask, Gson, + Fastify; 450 git-derived questions): hit@3 0.547 vs 0.304, MRR 0.456 vs 0.263, + p < 0.001, at about 10% more context tokens under symmetric accounting. +- **Index vs repo-map-style context**: a 2k-token signature map contains the answer + file 38–42% of the time; the index puts it in the top three 55% of the time for a + comparable budget. This is an approximation of the repo-map *style*, not a + measurement of Aider. +- **Version over version**: every ranking change passes a pooled, multi-language + ablation with significance testing. -- Retrieval quality: Recall@1/3/5, MRR, nDCG -- Agent usefulness: answer-correctness proxy on the public fixture, plus real-repo recall gate in the honest benchmark -- Token economy: tokens saved versus grep/read, Aider repo-map style context, and vanilla agent exploration -- Languages: separate results for Python, TypeScript, Java, Go, Rust, C#, PHP, and other supported languages -- Freshness: latency from file edit to usable updated result -- Graph tasks: callers, impact, architecture trace, route -> handler -> service -> DB - -The suite is CI-friendly and synthetic today. Real public 10k/100k/1M LOC scale targets remain -the next benchmark milestone. +What it cannot show: head-to-head results against Cursor, Sourcegraph, Continue or +Codebase-Memory MCP (no comparable harness exists), any 1M-LOC scale claim, or +agent task-success rates. See [BENCHMARKS.md](BENCHMARKS.md). ## Positioning diff --git a/docs/DATABASE_SCHEMA.md b/docs/DATABASE_SCHEMA.md index e0d0d7a..6f371e3 100644 --- a/docs/DATABASE_SCHEMA.md +++ b/docs/DATABASE_SCHEMA.md @@ -1,185 +1,6 @@ # Database Schema -SQLite database structure for `codebase-index`. - -## Overview - -The index is stored in a single SQLite file: `.claude/cache/codebase-index/index.sqlite` - -The schema is defined in `src/codebase_index/storage/schema.sql` and applied on first initialization. - -## Tables - -### files - -Stores metadata about indexed files. - -| Column | Type | Description | -|---|---|---| -| `id` | INTEGER PRIMARY KEY | Auto-incrementing file ID | -| `path` | TEXT UNIQUE NOT NULL | Relative path from project root | -| `language` | TEXT | Detected language (python, typescript, etc.) | -| `size_bytes` | INTEGER | File size in bytes | -| `content_hash` | TEXT NOT NULL | SHA-256 hash for change detection | -| `indexed_at` | TEXT | ISO 8601 timestamp of last indexing | -| `is_generated` | INTEGER DEFAULT 0 | Whether file is auto-generated | - -### chunks - -Stores text chunks for FTS5 indexing. - -| Column | Type | Description | -|---|---|---| -| `id` | INTEGER PRIMARY KEY | Auto-incrementing chunk ID | -| `file_id` | INTEGER REFERENCES files(id) | Parent file | -| `chunk_index` | INTEGER | Position within the file | -| `line_start` | INTEGER | Starting line number (1-indexed) | -| `line_end` | INTEGER | Ending line number (1-indexed) | -| `text` | TEXT NOT NULL | Chunk text content | -| `token_estimate` | INTEGER | Approximate token count | - -### symbols - -Stores extracted symbols from AST parsing. - -| Column | Type | Description | -|---|---|---| -| `id` | INTEGER PRIMARY KEY | Auto-incrementing symbol ID | -| `file_id` | INTEGER REFERENCES files(id) | Defining file | -| `name` | TEXT NOT NULL | Symbol name | -| `kind` | TEXT | Type: class, function, method, variable, etc. | -| `line_start` | INTEGER | Definition start line | -| `line_end` | INTEGER | Definition end line | -| `signature` | TEXT | Function/class signature if available | -| `docstring` | TEXT | Extracted docstring if available | - -### edges - -Stores relationships between symbols and files. - -| Column | Type | Description | -|---|---|---| -| `id` | INTEGER PRIMARY KEY | Auto-incrementing edge ID | -| `source_id` | INTEGER REFERENCES symbols(id) | Source symbol (caller, importer) | -| `target_id` | INTEGER REFERENCES symbols(id) | Target symbol (callee, imported) | -| `edge_type` | TEXT | Type: call, import, reference, inheritance | -| `line` | INTEGER | Line where the edge occurs | - -### modules - -Stores module-level information for import resolution. - -| Column | Type | Description | -|---|---|---| -| `id` | INTEGER PRIMARY KEY | Auto-incrementing module ID | -| `file_id` | INTEGER REFERENCES files(id) | File containing the module | -| `module_path` | TEXT | Resolved module path | -| `exports` | TEXT | JSON array of exported symbol names | - -### fts_chunks - -FTS5 virtual table for full-text search (auto-managed by triggers). - -| Column | Type | Description | -|---|---|---| -| `text` | TEXT | Chunk text (indexed by FTS5) | -| `chunk_id` | INTEGER | References chunks(id) | - -### vec_chunks (optional) - -Vector embeddings for semantic search. Created **only** when `embeddings.enabled = true`, via the -`sqlite-vec` extension (a `vec0` virtual table). - -| Column | Type | Description | -|---|---|---| -| `chunk_id` | INTEGER PRIMARY KEY | References chunks(id) | -| `embedding` | FLOAT[dim] | Embedding vector; `dim` is fixed per build by the configured model | - -### vec_meta (optional) - -Records which embedding model/dimension produced the vectors currently in `vec_chunks`. - -| Column | Type | Description | -|---|---|---| -| `model` | TEXT | Embedding model identifier | -| `dim` | INTEGER | Vector dimension | -| `built_at` | TEXT | ISO 8601 timestamp of the embedding pass | - -### vec_cache (optional) - -Content-addressed embedding cache. `chunk_id`s churn on every full rebuild (chunks are deleted and -re-inserted), so this cache is keyed by `(model, content_sha)` instead — letting unchanged content -reuse its vector for free across rebuilds, so only new or changed text hits the backend. - -| Column | Type | Description | -|---|---|---| -| `model` | TEXT NOT NULL | Embedding model identifier | -| `content_sha` | TEXT NOT NULL | SHA-256 of the chunk content | -| `embedding` | BLOB NOT NULL | Pre-serialized float32 vector | - -Primary key: `(model, content_sha)`. - -### summaries - -Stores file-level summaries for quick overview. - -| Column | Type | Description | -|---|---|---| -| `file_id` | INTEGER PRIMARY KEY REFERENCES files(id) | Associated file | -| `summary` | TEXT | Auto-generated or manual summary | -| `top_symbols` | TEXT | JSON array of most important symbols | - -### metadata - -Stores index-level metadata. - -| Column | Type | Description | -|---|---|---| -| `key` | TEXT PRIMARY KEY | Metadata key | -| `value` | TEXT | Metadata value | - -Common keys: -- `schema_version` — current schema version (for migration checks) -- `last_indexed_at` — timestamp of last full index -- `total_files` — number of indexed files -- `total_symbols` — number of extracted symbols -- `total_chunks` — number of text chunks -- `config_hash` — hash of the configuration used for indexing - -## Indexes - -| Index | Table | Columns | Purpose | -|---|---|---|---| -| `idx_files_path` | files | path | Fast path lookup | -| `idx_files_hash` | files | content_hash | Change detection | -| `idx_symbols_name` | symbols | name | Symbol name lookup | -| `idx_symbols_file` | symbols | file_id | Symbols by file | -| `idx_edges_source` | edges | source_id | Outgoing edges | -| `idx_edges_target` | edges | target_id | Incoming edges | -| `idx_chunks_file` | chunks | file_id | Chunks by file | - -## FTS5 Configuration - -The `fts_chunks` virtual table uses: - -- **Tokenizer:** `unicode61` (default SQLite tokenizer) -- **Content:** `chunks` table (external content mode) -- **Triggers:** INSERT, UPDATE, DELETE triggers keep FTS in sync with `chunks` - -### Query Syntax - -FTS5 supports: -- Phrase matching: `"exact phrase"` -- Prefix matching: `auth*` -- Boolean: `auth AND login`, `auth OR login`, `auth NOT test` -- NEAR: `auth NEAR/5 login` - -## Schema Migrations - -The `metadata.schema_version` key tracks the current schema version. On initialization: - -1. If the database doesn't exist, create it with the latest schema. -2. If the database exists, check `schema_version`: - - If equal to current, proceed. - - If less than current, run migrations (not yet implemented). - - If greater than current, refuse to open (future version protection). +This page moved. The canonical, code-verified schema reference is +[SCHEMA.md](SCHEMA.md), which mirrors `src/codebase_index/storage/schema.sql` +(tables `files`, `symbols`, `chunks`, `edges` with a `confidence` column, `modules`, +`meta`, the `fts_chunks` FTS5 table, and the opt-in `vec_*` tables). diff --git a/docs/DEMO.md b/docs/DEMO.md new file mode 100644 index 0000000..726c99b --- /dev/null +++ b/docs/DEMO.md @@ -0,0 +1,103 @@ +# Recording the demo + +The repository ships a reproducible terminal demo, [examples/demo](../examples/demo/), +and a static SVG rendering of its output, `assets/demo-terminal.svg`, which the README +embeds. This page is the recipe for turning the same session into an animated GIF or a +short video without fabricating anything: every frame comes from the real script. + +## What to record + +`examples/demo/run_demo.sh` (or `run_demo.ps1`) against Flask at the pinned commit. +Steps 2, 3, 5 and 6 (search, refs, impact, diff-impact) are the ones worth showing; +the JSON step is for the README text, not for a recording. + +Keep it under 45 seconds. Viewers stop at the first table. + +## Prerequisites + +```bash +pip install codebase-index +git clone https://github.com/pallets/flask.git .tmp-demo/flask +git -C .tmp-demo/flask checkout d318b683471101618febed18996405ad26462110 +``` + +Use a terminal that is 100 columns wide and about 34 rows tall so the Markdown +tables do not wrap. A dark theme with a monospace font at 14-16 px reads best in a +README at 960 px width. + +## Option A: VHS (recommended, fully scripted) + +[VHS](https://github.com/charmbracelet/vhs) renders a `.tape` script to GIF/MP4/WebM +deterministically, so the recording is as reproducible as the demo itself. + +`docs/demo.tape`: + +```tape +Output assets/demo.gif +Set FontSize 15 +Set Width 1200 +Set Height 720 +Set Theme "Catppuccin Mocha" +Set TypingSpeed 40ms + +Type "cd .tmp-demo/flask && codebase-index index" +Enter +Sleep 4s + +Type 'codebase-index search "where is the session cookie signed and saved" --limit 3' +Enter +Sleep 5s + +Type "codebase-index refs open_session" +Enter +Sleep 4s + +Type "codebase-index impact SecureCookieSessionInterface --direction up --depth 2" +Enter +Sleep 5s + +Type "codebase-index diff-impact" +Enter +Sleep 6s +``` + +Make the throwaway edit before recording (`run_demo.sh` shows the `sed` line) so +`diff-impact` has something to report, and revert it afterwards. + +```bash +vhs docs/demo.tape +``` + +## Option B: asciinema + agg + +```bash +asciinema rec --cols 100 --rows 34 demo.cast +bash examples/demo/run_demo.sh .tmp-demo/flask +exit +agg --theme monokai --font-size 15 demo.cast assets/demo.gif +``` + +Trim the `.cast` file to the interesting steps before converting; `agg` supports +`--idle-time-limit 2` to compress pauses. + +## Option C: static SVG (what the README uses today) + +`scripts/gen_terminal_svg.py` renders a captured transcript into an SVG "terminal +card". It has no dependencies and is what produced `assets/demo-terminal.svg`: + +```bash +bash examples/demo/run_demo.sh .tmp-demo/flask > /tmp/demo.txt 2>&1 +python scripts/gen_terminal_svg.py /tmp/demo.txt --out assets/demo-terminal.svg +``` + +The script strips ANSI colour, keeps the `$ command` lines highlighted, and cuts at +the first `--json` step so the card stays short. + +## Checklist before publishing a recording + +- The commit hash in the first frame matches `examples/demo/run_demo.sh`. +- No path from your home directory is visible (use a relative `.tmp-demo`). +- File size: GIF under 3 MB for the README; put the MP4 in the GitHub release, not + in git. +- Update `examples/demo/EXPECTED_OUTPUT.md` if the codebase-index version changed and + the output moved. diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md new file mode 100644 index 0000000..4b50efc --- /dev/null +++ b/docs/DEVELOPMENT.md @@ -0,0 +1,236 @@ +# Development guide + +Everything you need to make a first contribution in about fifteen minutes. For +the full design see [ARCHITECTURE.md](ARCHITECTURE.md); this page is the +hands-on version. + +## 1. Map of the code (two minutes) + +``` +src/codebase_index/ +├── cli.py Typer commands; every command delegates to service.py +├── service.py shared CLI/MCP layer: db resolution, search/impact/stats payloads +├── discovery/ walker.py (file walk), ignore.py (.gitignore & co), classify.py (language, secrets, binary, generated) +├── parsers/ languages.py (LangSpec per language), treesitter.py (symbols + edges), line_chunker.py +├── indexer/ pipeline.py (build_index / update_index), freshness.py +├── graph/ builder.py (edge resolution), expand.py (impact), analysis.py, navigate.py, export.py +├── retrieval/ pipeline.py (search), searchers.py, fusion.py, rerank.py, priors.py, tuning.py, budget.py, skeleton.py +├── storage/ db.py, schema.sql, repo.py (typed SQL accessors) +├── mcp/server.py stdio MCP server over the same service layer +├── output/ markdown.py, json.py, redact.py +└── skill_template/ canonical skill source; copies under .claude/ .codex/ .opencode/ skill/ skills/ are generated +``` + +Query path in one line: `retrieval/pipeline.py::search` → `detect_intent` → +`_run_retrievers` (path, symbol, FTS5, optional vector, optional graph) → `fuse` +(RRF) → `rerank` → dedup / diversify → token budget → payload. + +## 2. Set up (three minutes) + +Python 3.11+ and git are the only prerequisites. + +```bash +git clone https://github.com/denfry/codebase-index.git +cd codebase-index +python -m venv .venv +. .venv/bin/activate # Windows: .venv\Scripts\activate +pip install -e ".[dev,mcp]" +``` + +`dev` brings pytest, ruff, mypy, pyyaml and the `mcp` SDK; `mcp` is listed +separately so the server extra stays installable on its own. Optional extras: +`embeddings`, `embeddings-local`, `watch`, `build`. + +Before running the CLI **inside this checkout**, export +`CBX_NO_SKILL_AUTO_UPDATE=1`. Without it the skill auto-updater may rewrite the +committed `.skill_version` stamps when the installed package metadata is stale. +`tests/conftest.py` sets this for pytest automatically. + +## 3. Run the checks (what CI runs) + +CI (`.github/workflows/ci.yml`) runs exactly these: + +```bash +ruff check src tests +mypy src/codebase_index +python scripts/sync_skill_copies.py --check +pytest # coverage gate: --cov-fail-under=80 (pyproject.toml) +pytest tests/test_perf_smoke.py --runslow --no-cov # Linux + 3.12 only +``` + +Notes: + +- `ruff format` is **not** enforced; running `ruff format --check` on the + current tree reports many files. Format new code if you like, but do not + reformat files you are not otherwise touching. +- Tests marked `slow` are skipped unless you pass `--runslow` + (`tests/conftest.py::pytest_addoption`). +- Golden snapshots live in `tests/golden/*.json`. When an output contract + changes on purpose, regenerate them and review the diff: + + ```bash + UPDATE_GOLDEN=1 pytest tests/test_cli_golden.py tests/test_mcp_golden.py + ``` + + `tests/golden_utils.py` masks timestamps, commit SHAs and the package version + so snapshots are stable across machines; `schema_version` is deliberately not + masked because it is the contract under test. +- The skill copies gate: anything under `src/codebase_index/skill_template/` or + the version in `src/codebase_index/__init__.py` must be propagated with + `python scripts/sync_skill_copies.py` (no flag) before committing. + +Run a single test file quickly without the coverage gate: + +```bash +pytest tests/test_fusion.py -q --no-cov +``` + +## 4. Try the CLI on a fixture + +```bash +export CBX_NO_SKILL_AUTO_UPDATE=1 +codebase-index --root tests/fixtures/sample_repo index +codebase-index --root tests/fixtures/sample_repo search "refresh token" --json +codebase-index --root tests/fixtures/sample_repo impact User +``` + +`tests/fixtures/sample_repo/` deliberately contains files that must never be +indexed (`.env`, `secrets.pem`, `huge.json`, `logo.png`); see +`tests/fixtures/README.md`. + +## 5. Benchmark workflow + +The project rule is *measure improvements, do not assert them*. Three surfaces +exist; know which one you are using. + +| Surface | Command | Use it for | +|---|---|---| +| Retrieval eval (ranking gate) | `python tests/eval/run_eval.py` | Any change that can move ranking | +| Public synthetic suite | `python tests/benchmark_public.py --workdir .tmp-public-benchmark` | Metric-shape regression check (`tests/test_public_benchmark.py` wraps it in CI) | +| Older single-repo script | `python tests/benchmark_honest.py --repo ` | Index vs `rg`+window token/recall comparison against one repository | + +The retrieval eval (`tests/eval/`): + +- `harness.py` builds one index per corpus into a temp dir (`build_corpus_index`) + and reuses it for every variant, then computes recall@5/10, MRR, nDCG@10, + hit@3, P@5, MAP, `useful@budget`, mean tokens, duplicate rate, and latency + percentiles (`evaluate`, `format_table`, `pool`). +- `metrics.py` holds the IR metrics plus `paired_bootstrap_ci` and + `paired_permutation_p` (seeded, reproducible). +- `gen_queries.py` mints leak-free ground truth from git history (commit subject + → files that commit changed): + + ```bash + python tests/eval/gen_queries.py --repo ../some-repo --out /tmp/some-repo.yml + python tests/eval/run_eval.py --corpus ../some-repo:/tmp/some-repo.yml --ablate + ``` + +- `run_eval.py --ablate` runs a one-signal-off sweep over the boolean flags in + `ABLATABLE` and prints a paired significance table for each row. + +Checked-in query sets: `tests/eval/queries/self_repo.yml` (hand-written) and +`self_repo_git.yml` (generated). See [tests/eval/README.md](../tests/eval/README.md). + +## 6. Recipes + +### Add a language (Tier A symbol extraction) + +1. `src/codebase_index/discovery/classify.py`: add the extension to + `_LANG_BY_SUFFIX` and the language id to `_TREE_SITTER_LANGS`. +2. `src/codebase_index/parsers/languages.py`: add a `LangSpec(name, ts_name, + defs_query, calls_query, imports_query)` and register it in `LANGS`. + Definitions are captured as `@def.` with the name node as `@name`; + calls capture `@callee`; imports/inheritance capture `@import.module`, + `@extends.base`, `@implements.iface` (mapped by `_EDGE_PREFIXES` in + `parsers/treesitter.py`). +3. If the grammar's node kinds are not covered by `_definition_kind` / + `_name_node` / `_callee_node` in `parsers/treesitter.py`, extend them. +4. Add a fixture under `tests/fixtures/multilang/` and cases in + `tests/test_languages.py` (query compiles against the grammar) and + `tests/test_multilang_symbols.py` (`test_registry_consistency_every_treesitter_lang_extracts` + is parametrised over the registry, so a language with zero symbols fails loudly). +5. Update the tier table in [LANGUAGES.md](LANGUAGES.md). + +```bash +pytest tests/test_languages.py tests/test_multilang_symbols.py tests/test_graph.py -q --no-cov +``` + +### Add a graph edge kind + +Edges are rows in the `edges` table (`storage/schema.sql`): `edge_type`, +`src_kind`/`src_id`, `dst_kind`/`dst_id`/`dst_name`, `line`, `resolved`, and a +`confidence` of `extracted`, `inferred`, or `ambiguous`. + +1. Emit the edge from `parsers/treesitter.py::_extract_graph_edges` (query + captures) or `_extract_edges` (calls). Today's types are `call`, `reference`, + `import`, `extends`, `implements`. +2. Resolve it in `graph/builder.py::resolve_edges`. Symbol-target types are in + `_SYMBOL_EDGE_TYPES` and resolve only on a repo-unique name; imports resolve + by path suffix. Anything that cannot be pinned to one target is marked + `ambiguous` by `repo.mark_ambiguous_edges`. Never guess. +3. Make sure `graph/expand.py` (impact), `graph/navigate.py` (path/describe) and + `graph/export.py` (HTML/GraphML/DOT/Neo4j) do the right thing with the new + type, and document it in [SCHEMA.md](SCHEMA.md) and [LANGUAGES.md](LANGUAGES.md). +4. Add tests in `tests/test_graph.py` / `tests/test_graph_coverage.py`; goldens + that include edges (`impact_*`, `mcp_impact_of*`, `path_*`) may need + regeneration. + +### Change retrieval ranking + +1. Put every new signal behind a boolean on `RetrievalTuning` + (`src/codebase_index/retrieval/tuning.py`). Defaults are the shipped + configuration; `RetrievalTuning.baseline()` must keep reproducing the 1.7.0 + behaviour, otherwise the ablation table is meaningless. +2. Wire it in `retrieval/pipeline.py::search` (candidate generation, `fuse`, + `rerank`, selection) or in `retrieval/rerank.py` / `retrieval/priors.py` + (`source_role_prior`, capped by `MAX_ABS_PRIOR` so priors stay tiebreakers). +3. Add the flag to `ABLATABLE` in `tests/eval/run_eval.py`. +4. Run `python tests/eval/run_eval.py --ablate` on at least two corpora + (`--corpus`), and paste the pooled table plus the significance row into the + PR. A signal ships only if its row is significant (p < 0.05) on the pooled + set; otherwise it stays off by default or is removed. +5. Unit tests: `tests/test_fusion.py`, `tests/test_hybrid_ranking.py`, + `tests/test_priors.py`, `tests/test_diversity.py`. + +### Add an MCP tool + +1. `src/codebase_index/mcp/server.py`: decorate a function with `@_tool()` and + return `_emit("", payload)`. The `_emit` envelope adds + `schema_version` and `tool`; error paths must use it too + (`_no_index_payload()` is the standard no-index body). +2. Do the real work in `service.py` so the CLI command and the MCP tool share + one implementation. +3. Register the tool in `tests/test_mcp_server.py::test_mcp_server_has_expected_tools` + and in the envelope parametrisation there; add a golden case in + `tests/test_mcp_golden.py` and generate `tests/golden/mcp_.json` with + `UPDATE_GOLDEN=1`. +4. Document it in the tool table of [MCP.md](MCP.md). Bump `MCP_SCHEMA_VERSION` + only for a breaking change (field removal or type change). + +### Add a CLI command + +Add it in `cli.py`, delegate to `service.py`, support `--json`, and if the skill +should be allowed to call it, add it to the `ALLOWED` whitelist in +`src/codebase_index/skill_template/scripts/cbx` and `cbx.ps1`, then run +`python scripts/sync_skill_copies.py` and update `references/commands.md` in the +template. Add a golden in `tests/test_cli_golden.py` for `--json` output. + +## 7. Release process (maintainers) + +The version is single-sourced in `src/codebase_index/__init__.py` +(`__version__`); `pyproject.toml` reads it through hatch dynamic versioning. +Two mirrors are kept in sync by `scripts/sync_skill_copies.py`: +`.claude-plugin/plugin.json` (`version`) and `requirements.lock` (release tag +tarball). Tagging `vX.Y.Z` triggers `.github/workflows/release.yml`: test gate, +build, `twine check`, `scripts/release_smoke.py` (clean venv install → init → +index → search), GitHub release, and PyPI publish via Trusted Publishing. + +Follow [RELEASE_CHECKLIST.md](RELEASE_CHECKLIST.md) top to bottom. Contributors +never bump the version; they add entries under `[Unreleased]` in +`CHANGELOG.md`. + +## 8. Where to ask + +Open a [discussion](https://github.com/denfry/codebase-index/discussions) for +design questions, an issue for bugs and concrete proposals, and read +[CONTRIBUTING.md](../CONTRIBUTING.md) for the PR workflow. diff --git a/docs/FAQ.md b/docs/FAQ.md index 1e25c0d..c381aa2 100644 --- a/docs/FAQ.md +++ b/docs/FAQ.md @@ -13,8 +13,8 @@ or `pipx` (isolated): pip install codebase-index # or: pipx install codebase-index ``` -To pin an exact version or grab an unreleased commit, install from a GitHub tag -instead: `pip install "codebase-index @ git+https://github.com/denfry/codebase-index.git@v1.8.0"`. +To pin an exact version: `pip install codebase-index==1.9.0`. To try an unreleased +commit, install from git: `pip install "codebase-index @ git+https://github.com/denfry/codebase-index.git@main"`. Then run `codebase-index init` inside your project and `codebase-index index` to build the first index. In Claude Code you can instead install the plugin @@ -74,16 +74,12 @@ Yes. Run: codebase-index mcp --root /path/to/repo ``` -The stdio MCP server exposes: +The stdio MCP server exposes eleven tools (`src/codebase_index/mcp/server.py`): -- `healthcheck` -- `search_code` -- `find_symbol` -- `find_refs` -- `impact_of` -- `impact_of_diff` -- `explain_code` -- `index_stats` +- `healthcheck`, `index_stats` +- `search_code`, `explain_code`, `find_symbol`, `find_refs` +- `impact_of`, `impact_of_diff` +- `architecture_overview`, `path_between`, `describe_symbol` See [MCP.md](MCP.md) for schema and client config templates. @@ -158,27 +154,41 @@ Yes. Use any of these methods: 1. **`.codeindexignore`** — Tool-specific ignore file (highest priority) 2. **`.gitignore`** — Standard git ignore file 3. **`.claudeignore`** — Claude-specific ignore file -4. **Configuration** — `extra_ignore` patterns in `.codeindex.json` +4. **Configuration** — `extra_ignore` patterns in + `.claude/cache/codebase-index/config.json` (written by `init`; see + `examples/config.example.json`) ## Is it production-ready? -Yes — `codebase-index` is released as **v1.8.0**. The core indexing and search -functionality is implemented and tested. The current `1.8.0` package includes: - -- Hybrid FTS/path/symbol/vector retrieval with benchmark-calibrated lexical expansion - and bounded fuzzy identifier matching -- Import/call/reference graph expansion, intent-directed graph discovery, and `impact` -- Diff-aware blast-radius analysis for tracked working-tree changes -- Optional local embeddings, with external embeddings gated behind explicit opt-in -- Hooks and watch mode for freshness -- Multi-CLI setup for Claude Code, Codex CLI, and OpenCode - -Known gaps: the public benchmark suite is still small, the MCP server needs -verified client-specific docs and progressive/paged results, and the graph is -closer to an import/call/reference graph than a full framework-aware code -intelligence graph. - -See [ROADMAP.md](../ROADMAP.md) for the full milestone plan. +Yes, with the caveats below. The current line is **1.9.x** (see +[CHANGELOG.md](../CHANGELOG.md)). It ships: + +- Hybrid FTS5 / path / symbol retrieval with optional local embeddings; rank fusion that + scores cross-retriever agreement at file level; evidence-calibrated source priors, + lexical expansion, and fuzzy identifier fallback (1.8.0, 1.9.0). +- Tree-sitter symbols for twelve Tier-A languages; import / call / reference / + inheritance graph with per-edge confidence; `refs`, `path`, `describe`, + `architecture`, `impact`, and diff-aware `diff-impact` (1.5.0, 1.7.0). +- Token-budgeted, skeletonized retrieval packets with bounded `recommended_reads`. +- CLI, Claude Code skill and plugin, Codex CLI and OpenCode resources, and a stdio MCP + server sharing one service layer. +- A reproducible retrieval evaluation with leak-free git-derived ground truth, + multi-corpus pooling, and paired significance tests; every 1.9.0 ranking change had + to pass it. + +Known gaps: the graph is import/call/reference-level rather than framework-aware, MCP +client configs are templates not yet verified against each client release, and there +is no LLM-driven task-level evaluation yet. See [ROADMAP.md](ROADMAP.md). + +## How does it differ from Aider's repo-map, Cursor, or plain grep? + +Grep is exact text matching with no ranking, symbol awareness or read plan. A repo map +is a query-independent context blob fed to one agent. Cursor is an IDE with its own +proprietary index. `codebase-index` is a queryable, local, scriptable retrieval and +graph layer any shell-capable or MCP agent can call. The trade-offs, including when +each alternative is the better choice, are in [COMPARISON.md](COMPARISON.md); the +measured comparison against grep-style and repo-map-style baselines is in +[BENCHMARKS.md](BENCHMARKS.md). ## How do I contribute? diff --git a/docs/INSTALLATION.md b/docs/INSTALLATION.md index 695cf68..ca1e417 100644 --- a/docs/INSTALLATION.md +++ b/docs/INSTALLATION.md @@ -64,10 +64,13 @@ ln -s ~/codebase-index/skill ~/.claude/skills/codebase-index # From PyPI (recommended) pip install codebase-index pipx install codebase-index # isolated environment -uv tool install codebase-index # uv-managed tool +uv tool install codebase-index # uv-managed tool (standard PyPI path; not part of CI yet) -# Pin to a GitHub tag for an exact or unreleased version -pip install "codebase-index @ git+https://github.com/denfry/codebase-index.git@v1.8.0" +# Pin an exact release +pip install codebase-index==1.9.0 + +# Or install an unreleased commit straight from git +pip install "codebase-index @ git+https://github.com/denfry/codebase-index.git@main" # From source (editable mode) git clone https://github.com/denfry/codebase-index.git @@ -90,10 +93,11 @@ pip install -e ".[embeddings-local,watch,dev]" ### uvx / Homebrew status -As of `1.8.0`, **PyPI is shipped** — `pip install codebase-index` and -`pipx install codebase-index` are the verified paths. `uvx codebase-index init`, -Homebrew tap installation, signed checksums, and SBOMs remain distribution -targets for a more complete release story. +As of `1.9.0`, **PyPI is shipped** — `pip install codebase-index` and +`pipx install codebase-index` are the paths exercised by the release smoke test. +`uv tool install` / `uvx` should work because the package is a normal PyPI wheel, but +they are not verified in CI. Homebrew tap installation, signed checksums, and SBOMs +remain distribution targets. Target future commands: @@ -124,21 +128,20 @@ codebase-index --help codebase-index doctor ``` -Expected output: +Real output (1.9.0, in a freshly cloned repository before `init`): ``` -=== codebase-index Doctor === - -[OK] Python 3.12 (requires 3.11+) -[OK] codebase-index package installed (v1.8.0) -[OK] tree-sitter is available -[INFO] Cache directory not yet created: ... -[INFO] Skill not installed in .claude/skills/ -[INFO] No config file (using defaults) - -All checks passed. +!! [high] cache_gitignored: add '.claude/cache/codebase-index/' to .gitignore (run `init`) +-- [info] hooks_enabled: no auto-update hook (run `init --with-hooks`) +OK [medium] index_fresh: index is fresh +OK [medium] symbol_extraction: tree-sitter languages extract symbols +OK [info] graph_coverage: all indexed languages have full dependency-graph support ``` +`!!` marks a failed high-severity check (`doctor --strict` exits non-zero on those), +`--` a failed lower-severity one, `OK` a pass. Running `codebase-index init` clears the +first finding. Package and Python versions: `pip show codebase-index`. + ## Claude Code Setup After installing the Python package, ensure the skill is available to Claude Code: @@ -269,7 +272,7 @@ codebase-index index If `doctor` warns about external embeddings, check your config: ```bash -cat .codeindex.json | grep allow_external +grep allow_external .claude/cache/codebase-index/config.json ``` Set `allow_external` to `false` to disable external API calls. diff --git a/docs/MCP.md b/docs/MCP.md index b89dd1d..9cc6e84 100644 --- a/docs/MCP.md +++ b/docs/MCP.md @@ -7,7 +7,7 @@ pip install "codebase-index[mcp]" codebase-index mcp --root /path/to/repo ``` -The server speaks MCP over stdio through FastMCP. Build the index with +The server speaks MCP over stdio through the official Python SDK (`MCPServer` on mcp 2.x, `FastMCP` on 1.x; both are supported). Build the index with `codebase-index index` before connecting a client. Current shipped interfaces: @@ -126,7 +126,10 @@ Use the client's MCP server configuration UI or JSON file and register: ``` Client-specific config file paths and screenshots should be added only after -they are verified against the current client versions. +they are verified against the current client versions. The templates above are +the standard stdio-server shape every listed client accepts; none has been +re-verified against a specific client release yet (tracked in +[ROADMAP.md](ROADMAP.md)). ## Progressive results diff --git a/docs/QUICKSTART.md b/docs/QUICKSTART.md index 2ae5744..748dc5b 100644 --- a/docs/QUICKSTART.md +++ b/docs/QUICKSTART.md @@ -48,40 +48,47 @@ This creates the cache directory, configuration, and the selected CLI instructio codebase-index index ``` -You should see output like: +The examples below are real output from indexing +[pallets/flask](https://github.com/pallets/flask) at commit `d318b683` with +codebase-index 1.9.0: ``` -Indexing... - Discovered 142 files (excluded 23 sensitive/generated) - Extracted 891 symbols - Built 2,340 chunks - Index built in 3.2s +Indexed 230 files (0 pruned). + parse failures: 0; tree-sitter files with 0 symbols: 17 ``` ## Step 4: Run Your First Search ```bash -codebase-index search "where is authentication implemented?" +codebase-index search "where is the session cookie signed and saved" --limit 5 ``` -Expected output: - ``` -Top matches: -┌──────┬──────────────────────────┬──────────────────┬───────┬────────────────────────────┐ -│ Rank │ Path │ Symbols │ Score │ Reason │ -├──────┼──────────────────────────┼──────────────────┼───────┼────────────────────────────┤ -│ 1 │ src/auth/AuthService.ts │ AuthService │ 0.92 │ exact symbol match │ -│ 2 │ src/routes/auth.ts │ login, logout │ 0.78 │ FTS match · 4 callers │ -│ 3 │ src/middleware/auth.ts │ requireAuth │ 0.65 │ path match · FTS match │ -└──────┴──────────────────────────┴──────────────────┴───────┴────────────────────────────┘ - -Recommended reads: - 1. src/auth/AuthService.ts:12-148 - 2. src/routes/auth.ts:20-91 - 3. src/middleware/auth.ts:5-42 +**Query:** where is the session cookie signed and saved +**Intent:** `locate_impl` · **Confidence:** medium + +| # | Path | Lines | Reason | +|---|------|-------|--------| +| 1 | `src/flask/sessions.py` | 24-54 | in src/flask/ · 2 callers · source prior +0.08 | +| 2 | `src/flask/sessions.py` | 284-385 | in src/flask/ · 2 callers · source prior +0.08 | +| 3 | `src/flask/sessions.py` | 57-80 | in src/flask/ · 1 callers · source prior +0.08 | +| 4 | `tests/test_basic.py` | 542-600 | source prior -0.06 · generated/test demoted | +| 5 | `tests/test_reqctx.py` | 201-249 | source prior -0.06 · generated/test demoted | + +`src/flask/sessions.py:284-385` +``` +class SecureCookieSessionInterface(SessionInterface): +``` +... + +**Recommended reads:** +- `src/flask/sessions.py:24-54` +- `src/flask/sessions.py:284-385` ``` +Rank 2 is the class that signs and saves the cookie. Add `--json` for the +machine-readable packet agents consume. + ## Step 5: Use with Your AI CLI When installed for Claude Code, Codex CLI, or OpenCode, the generated instructions @@ -100,14 +107,18 @@ The agent will: ## Interpreting Results -Each result includes: - -- **Rank** — position in the result list (start with 1-3) -- **Path** — file location -- **Symbols** — extracted symbols in the matched region -- **Score** — relevance score (0.0 to 1.0) -- **Reason** — why this result ranked (e.g., "exact symbol match") -- **Recommended reads** — exact line ranges to open +- **Intent** — how the query was classified (`locate_impl`, `how_it_works`, `impact`, ...); + it selects the retriever mix. +- **Confidence** — `high`: answer from the evidence; `medium`: read the recommended + ranges and confirm the key claim; `low`: follow the fallback suggestions (ripgrep + patterns) instead of trusting the list. +- **#, Path, Lines** — rank and the exact line range that matched. Start with ranks 1–3. +- **Reason** — why it ranked: exact symbol match, callers, path match, and the source + prior (implementation code is preferred over tests and documentation). +- **Snippets** — budgeted, often skeletonized (unrelated body lines folded) and + secret-redacted. +- **Recommended reads** — the read plan: exact ranges to open, capped at 120 lines + each (`truncated: true` in JSON when a longer symbol was cut at its head). ## What Success Looks Like diff --git a/docs/RELEASE_CHECKLIST.md b/docs/RELEASE_CHECKLIST.md index a159688..e5f3886 100644 --- a/docs/RELEASE_CHECKLIST.md +++ b/docs/RELEASE_CHECKLIST.md @@ -9,31 +9,30 @@ recreating a GitHub release (used to publish an already-tagged version). Work top to bottom. Do not tag until every required box is checked. -## 1. Version sync (single source + the two manual mirrors) +## 1. Version sync (single source, mirrors checked by script) The package version is single-sourced from `src/codebase_index/__init__.py` -(hatch dynamic version). Two files mirror it and are **not** auto-synced — bump -them by hand and verify: +(hatch dynamic version). `scripts/sync_skill_copies.py` rewrites every mirror; +`scripts/check_versions.py` proves they agree and runs in CI lint and at the top +of the release build, so a mismatched tag fails before anything is published. - [ ] `src/codebase_index/__init__.py` → `__version__` bumped (canonical). -- [ ] `.claude-plugin/plugin.json` → `"version"` matches. -- [ ] `.claude-plugin/marketplace.json` → version matches (if present). -- [ ] `requirements.lock` → the GitHub tarball tag matches the new tag - (`.../tags/vX.Y.Z.tar.gz`). The plugin bootstrap installs exactly this pin. -- [ ] README / QUICKSTART / INSTALLATION / FAQ / MCP install snippets reference - the new tag (`@vX.Y.Z`). -- [ ] Skill copies + `.skill_version` stamps regenerated and in sync: +- [ ] Mirrors regenerated and verified: ```bash - python scripts/sync_skill_copies.py # regenerate - python scripts/sync_skill_copies.py --check # CI gate: must pass clean + python scripts/sync_skill_copies.py # plugin.json, requirements.lock, skill copies + stamps + python scripts/sync_skill_copies.py --check # CI gate: no drift + python scripts/check_versions.py # CI gate: package == plugin.json == lock tag == stamps, CHANGELOG has the section ``` +- [ ] Docs that quote a pinned tag (`@vX.Y.Z`, `codebase-index==X.Y.Z`) updated; + `python scripts/check_links.py` passes (CI gate for relative links). + ## 2. Tests and lint - [ ] `pytest` green locally (coverage gate `--cov-fail-under=80` enforced). -- [ ] `ruff check src/ tests/` clean. -- [ ] `mypy src/codebase_index` (advisory) reviewed. +- [ ] `ruff check src tests scripts` clean (CI gate). +- [ ] `mypy src/codebase_index` clean (CI gate, not advisory). - [ ] Slow/perf tests considered: `pytest --runslow` for index/search latency. - [ ] CI matrix green (Ubuntu/macOS/Windows × py3.11–3.13) on the release branch. @@ -93,8 +92,15 @@ them by hand and verify: ## 8. Changelog and docs -- [ ] `CHANGELOG.md`: move `[Unreleased]` items under the new `vX.Y.Z` dated +- [ ] `CHANGELOG.md`: move `[Unreleased]` items under the new `## [X.Y.Z] - YYYY-MM-DD` heading; add the version-compare link at the bottom. +- [ ] Preview the GitHub release body — it is generated from that section, and an + empty or missing section fails the release job: + + ```bash + python scripts/release_notes.py # prints the notes for __version__ + ``` + - [ ] ROADMAP / docs reflect anything that shipped or moved. - [ ] `docs/PRODUCT_UPGRADE_PLAN.md` status column updated for shipped items. @@ -102,9 +108,12 @@ them by hand and verify: - [ ] Commit the version bump + changelog on the release branch; open/merge PR. - [ ] Tag: `git tag vX.Y.Z && git push origin vX.Y.Z`. -- [ ] `release.yml` build job green (test gate + `python -m build` + `twine check` - + `release_smoke.py`). -- [ ] GitHub release created with artifacts attached; release notes reviewed. +- [ ] `release.yml` build job green (`check_versions.py` + tag/version match + test + gate + `python -m build` + `twine check` + `release_smoke.py`). The same + build/twine/smoke steps already ran on the PR in CI's `package` job. +- [ ] GitHub release created with artifacts attached; body = CHANGELOG section + (`release_notes.py`) followed by the auto-generated PR list. +- [ ] PyPI shows the new version (`pip index versions codebase-index`). - [ ] Post-publish: re-run `pipx install "...@vX.Y.Z"` once to confirm the tag resolves. diff --git a/docs/RETRIEVAL.md b/docs/RETRIEVAL.md index 6102baa..de233b3 100644 --- a/docs/RETRIEVAL.md +++ b/docs/RETRIEVAL.md @@ -1,5 +1,7 @@ # Retrieval Pipeline +(`docs/RETRIEVAL_PIPELINE.md` was merged into this page.) + The retrieval engine turns a natural-language or symbolic query into a **compact, ranked, token-budgeted** set of file/line ranges for Claude to read. It is hybrid: multiple independent retrievers run, their results are fused and reranked, then trimmed. Graph expansion and MMR are @@ -70,10 +72,12 @@ source)` list so fusion is source-agnostic. `fuzzy_fallback_min` rows. Measured over 305 queries on three repositories it moved no ranking metric while costing ~20% of query latency, so it is kept for typos and abbreviations but no longer runs when the query already spelled its identifier correctly. -- **FTS** — FTS5 `bm25()` over the `fts_chunks` virtual table (chunk text + symbol names + - summaries indexed). Query-time camelCase/snake_case splitting, small down-weighted synonym - expansion, and soft coverage scoring make natural-language questions robust without weakening - exact terms. +- **FTS** — FTS5 `bm25()` over the `fts_chunks` virtual table (chunk text + symbol names + indexed). Query-time camelCase/snake_case splitting, a small down-weighted synonym/inflection + vocabulary, OR-groups for soft matching with bounded term coverage, and ranking by + original-term coverage with BM25 as a tie-break make natural-language questions robust without + weakening exact terms. Every FTS term is quoted so query punctuation cannot inject MATCH + operators. - **Vector** *(opt-in)* — cosine similarity over chunk embeddings via `sqlite-vec`. Only runs if `embeddings.enabled = true`. Adds semantic recall for paraphrased queries. Absent → pipeline degrades gracefully to FTS+symbol. @@ -182,20 +186,27 @@ Results are trimmed to fit `--token-budget` (default per intent, e.g. 1500 token 2. Greedily attach snippets to the highest-ranked results until budget is hit. 3. Snippets are trimmed to the relevant line range (± a few context lines), not whole functions. 4. Lower-ranked results become **`recommended_reads`** (path + range, no snippet) so Claude can - choose to read them itself. + choose to read them itself. A read is capped at `retrieval.max_read_lines` (default 120): a + symbol-aligned chunk can be a whole class, and the read plan should bill the agent for the + definition head, not the body. Capped entries carry `truncated: true` and `line_end_full`. 5. Snippet text passes through secret redaction (see SECURITY.md) before emission. The point: Claude gets enough to decide, and a precise list of what to read next — never a dump. ## 8. Confidence & fallback -A `confidence` score (high/medium/low) is derived from: top RRF score, score gap between #1 and #2, -number of agreeing retrievers, and whether a symbol matched exactly. +A categorical `confidence` (high/medium/low) is derived from exact-symbol evidence, +multi-retriever agreement, score separation between #1 and #2, and result count. An exact symbol +match is `high`, including a single-result response. - **high** → Claude reads `recommended_reads` and answers. - **medium** → Claude reads, but may verify with one Grep. - **low** → skill instructs Claude to **fall back** to `ripgrep`/Grep/Glob with suggested patterns - emitted in `fallback_suggestions` (derived from query terms + detected symbols). + emitted in `fallback_suggestions` (derived from query terms + detected symbols): `rg` patterns, + likely paths, and query-broadening hints. + +Default token budget: 1500 for `search`, 2200 for `explain`, configurable per project in +`.claude/cache/codebase-index/config.json` (`retrieval.token_budget`). ## 9. Output payload (shared by Markdown + JSON) diff --git a/docs/RETRIEVAL_PIPELINE.md b/docs/RETRIEVAL_PIPELINE.md index 5bf0944..b03ddfd 100644 --- a/docs/RETRIEVAL_PIPELINE.md +++ b/docs/RETRIEVAL_PIPELINE.md @@ -1,179 +1,5 @@ # Retrieval Pipeline -How `codebase-index` finds and ranks relevant code for a query. - -## Overview - -The retrieval pipeline combines multiple search strategies into a single ranked result set. - -``` -User query - ↓ -Intent detection (keyword / symbol / impact / general) - ↓ -┌─────────────────────────────────────────┐ -│ Parallel Retrievers │ -├─────────────────────────────────────────┤ -│ 1. Exact symbol match │ -│ 2. Path-based search │ -│ 3. SQLite FTS5 lexical search │ -│ 4. Vector search (optional embeddings) │ -│ 5. Graph expansion (explicit opt-in) │ -└─────────────────────────────────────────┘ - ↓ -Reciprocal Rank Fusion (RRF) - ↓ -Reranking (boosts for symbol match, recency, file type) - ↓ -Token budget enforcement - ↓ -Ranked retrieval packet with confidence score -``` - -## 1. Exact Symbol Match - -**Trigger:** Query matches a known symbol name exactly or with minor variation. - -**Process:** -- Look up the symbol in the `symbols` table -- Return the definition file and line range -- Include all reference locations from the `edges` table - -**Score boost:** Highest priority — exact symbol matches are ranked first. - -## 2. Path-Based Search - -**Trigger:** Query contains file path fragments or recognizable path patterns. - -**Process:** -- Match query terms against file paths in the `files` table -- Use substring matching with path segment awareness - -**Score boost:** Moderate — path matches indicate the user knows where to look. - -## 3. SQLite FTS5 Lexical Search - -**Trigger:** General keyword and natural-language queries. - -**Process:** -- Parse identifiers into camelCase/PascalCase/snake_case subtokens. -- Add a small, explicit synonym/inflection vocabulary at lower weight. -- Use OR groups for soft matching, then require bounded term coverage and rank - by original-term coverage plus BM25. -- Quote every FTS term so query punctuation cannot inject MATCH operators. - -**Score:** Coverage is the primary signal; BM25 is a bounded tie-break. - -## 4. Vector Search (Optional) - -**Trigger:** Enabled when `embeddings.backend` is not "noop". - -**Process:** -- Embed the query using the configured backend -- Search `vec_chunks` for nearest neighbors -- Return chunks with cosine similarity scores - -**Score:** Cosine similarity (0.0 to 1.0). - -> **Indexing note:** chunk embeddings are reused across rebuilds via a content-addressed -> `vec_cache` (keyed by model + content SHA-256), so only new or changed chunks are re-embedded. -> See [DATABASE_SCHEMA.md](DATABASE_SCHEMA.md) and [SCHEMA.md](SCHEMA.md) for details. - -## 5. Graph Expansion - -**Trigger:** `RetrievalTuning(graph_source=True)` and an intent plan with a -graph strategy. It is disabled in the shipped default because the reproducible -self-repository ablation reduced direct-hit MRR. - -**Process:** -- Seed from lexical/symbol candidates already found in SQLite. -- Traverse only indexed, resolved edges with bounded depth and node count. -- Follow `up` (callers/importers), `down` (callees/imports), or `both` - according to the intent plan. -- Apply distance decay and preserve edge confidence in the candidate reason. - -Graph expansion is an opt-in context-discovery signal, not a replacement for -direct lexical or symbol evidence. - -## 6. Diversity and duplicate control - -SimHash suppresses near-duplicate snippets independently of MMR. It does not move -ranking metrics; it earns its place by taking the duplicate rate of returned -snippets from ~1.6% to ~0%. Bounded Maximal Marginal Relevance is available -through `RetrievalTuning(mmr=True)`; the shipped default keeps relevance-only -ordering because MMR moved no metric and roughly doubled p50 latency. - -The over-fetch that feeds these selection stages is `candidate_pool_multiplier` -(default 2), an explicit knob rather than a side effect of enabling dedup. - -## 7. Reciprocal Rank Fusion (RRF) - -Combines ranked lists from the enabled retrievers: - -``` -RRF_score(d) = Σ_r w_r · k / (k + rank_r(d)) -``` - -The implementation multiplies textbook RRF by `k` so fusion and bounded rerank -bonuses share a comparable scale; ordering is unchanged. - -Where: -- `k` is a constant (default 60) -- `rank_r(d)` is the rank of document `d` in retriever `r` -- `w_r` is the intent/tuning weight for retriever `r` - -Candidates are keyed by `(path, line-bucket)`, so co-located hits merge. Because a -symbol definition and a lexical hit in the same file are often *not* co-located, -each candidate additionally receives — at `file_agreement_weight` (0.4) — the RRF -mass of every retriever that found its file at a different locator. Retrievers -already counted at that locator are excluded, so nothing double-counts. Without -this, two retrievers agreeing on a file produced two weak candidates instead of one -strong one, and cross-source agreement never affected the score. - -## 8. Reranking - -After fusion, apply bounded explainable boosts and penalties: - -| Factor | Effect | Rationale | -|---|---:|---| -| Exact symbol match | +0.20 | User named a specific symbol | -| Symbol definition kind | +0.05 | Prefer actionable definitions | -| Path term match | +0.05 | User supplied a location clue | -| Degree / reference evidence | up to +0.08 | Stable structural tiebreaker | -| Implementation source prior | +0.08 | Prefer source over prose/tests | -| Documentation source prior | -0.20 | Prose describing a feature outmatches the code lexically | -| Generated/vendor/build | -0.25 | Suppress low-value derived code | -| Test path on non-test query | -0.06 | Keep tests as supporting evidence | - -## 9. Confidence - -Confidence is categorical (`high`, `medium`, `low`) and is derived from -exact-symbol evidence, multi-retriever agreement, score separation, and result -count. Exact symbol matches are `high`, including a single-result response. - -| Confidence | Action | -|---|---| -| `high` | Read recommended ranges and answer directly | -| `medium` | Read ranges; optionally confirm with one Grep | -| `low` | Use fallback suggestions (ripgrep, Glob) | - -## 10. Token Budget Enforcement - -The output is capped at a configurable token budget: - -1. Results are sorted by final score -2. Snippets are included until the budget is reached -3. Remaining results are listed without snippets -4. The `recommended_reads` field contains only the most critical line ranges - -Default budget: 1500 tokens (configurable in `.codeindex.json`). - -## 11. Fallback Suggestions - -When confidence is low, the pipeline generates fallback strategies: - -- **ripgrep patterns:** Extracted keywords from the query, formatted for `rg` -- **likely paths:** Common directories to search based on query terms -- **broaden query:** Suggestions for rewording the query - -These are included in the response so Claude can fall back gracefully. +This page moved. The canonical description of intent detection, the retrievers, RRF fusion with +file agreement, reranking and source priors, graph expansion, diversity control, token budgeting +and confidence/fallback is [RETRIEVAL.md](RETRIEVAL.md). diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index fa67701..57ae43b 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -46,13 +46,18 @@ release in `CHANGELOG.md`. The current priority is to make existing power obvious, measurable, and easy to invoke. -- [ ] Publish 10k and 100k LOC public-repository benchmark runs with raw logs. +- [~] Publish public-repository benchmark runs with raw logs. **Partially done:** + `tests/eval/run_baselines.py` compares the index with grep-style and + repo-map-style baselines on Flask (~18k LOC Python), Gson (~57k LOC Java) + and Fastify (~78k LOC JavaScript) at pinned commits, with a logged run under + `tests/eval/results/`. Remaining: a 1M-LOC / large-monorepo run. - [ ] Add task-level agent evaluation: success rate, tokens, files read, time, and citation correctness. - [ ] Verify MCP setup against current releases of each documented client. - [ ] Add progressive/paged MCP retrieval for large repositories. - [ ] Complete `uvx` and clean-machine install verification on every CI OS. - [ ] Publish signed checksums and an SBOM with releases. +- [ ] Homebrew tap (`brew install denfry/tap/codebase-index`). ## Next — task-native context diff --git a/docs/SCHEMA.md b/docs/SCHEMA.md index 490dc85..4726a54 100644 --- a/docs/SCHEMA.md +++ b/docs/SCHEMA.md @@ -1,7 +1,8 @@ # Database Schema Single SQLite database at `.claude/cache/codebase-index/index.sqlite`. WAL mode, foreign keys on. -The DDL below is the canonical source mirrored by `src/codebase_index/storage/schema.sql`. +`src/codebase_index/storage/schema.sql` is the applied DDL; this page mirrors it and explains it. +(`docs/DATABASE_SCHEMA.md` was merged into this page.) ## Pragmas @@ -87,6 +88,7 @@ CREATE INDEX idx_edges_src ON edges(src_kind, src_id); CREATE INDEX idx_edges_dst ON edges(dst_kind, dst_id); CREATE INDEX idx_edges_name ON edges(dst_name); CREATE INDEX idx_edges_type ON edges(edge_type); +CREATE INDEX idx_edges_file ON edges(file_id); -- replace_edges deletes per file on every update -- Module / package level summaries for architecture-intent queries. CREATE TABLE modules ( @@ -158,10 +160,21 @@ backend; everything else is copied straight from the cache into `vec_chunks`. Ne are written back to `vec_cache` so subsequent rebuilds reuse them. The reported "embedded" count reflects cache **misses** — i.e. the work actually performed. -## Migrations +## Schema versioning -`meta.schema_version` gates migrations in `storage/db.py`. On version mismatch the CLI either -migrates in place (for additive changes) or, for breaking changes, prompts a `index --rebuild`. +`meta.schema_version` is guarded in `storage/db.py`. There is no in-place migration framework: + +- a **newer** index than the installed CLI supports is refused on open with a message asking for + an updated CLI; +- an **older** index is still readable (queries never touch columns added later); the build + commands (`index`, `update`) detect the mismatch with `peek_schema_version` and rebuild from + scratch, because `schema.sql` is applied with `IF NOT EXISTS` and old tables would persist. + +## FTS5 query syntax + +FTS5 supports phrase matching (`"exact phrase"`), prefix matching (`auth*`), boolean operators +(`auth AND login`, `auth OR login`, `auth NOT test`) and `NEAR/5`. The lexical retriever quotes +every term it derives from the query, so user punctuation cannot inject MATCH operators. ## Incremental indexing keys diff --git a/docs/SECURITY.md b/docs/SECURITY.md index 8e4909a..88123af 100644 --- a/docs/SECURITY.md +++ b/docs/SECURITY.md @@ -1,119 +1,7 @@ -# Security Model +# Security -`codebase-index` is **local-first and offline by default**. Its threat model assumes the indexed -repository may contain secrets and that a skill must not exfiltrate code or run dangerous commands. +This page moved. -> **Trust model in 60 seconds** -> 1. **Offline by default** — the base install has zero network dependencies; nothing leaves your machine (§1, §4). -> 2. **One opt-in exit, triple-gated** — external embeddings require `allow_external` **and** an env API key **and** a printed endpoint warning, or they are refused (§4). -> 3. **Secrets never get in** — `.env`, keys, certs, and credential files are excluded before parsing (§2). -> 4. **Secrets never get out** — every snippet is redacted before it reaches the agent (§3). -> 5. **No telemetry, ever** — no analytics, no phone-home, no usage data. -> 6. **Verify it yourself** — `codebase-index doctor --strict` audits all of the above and gates CI (§6). -> -> The same callout appears in the README so the trust story is identical wherever a reader lands. - -## 1. Principles - -1. **Local-first** — index, query, and storage all happen on the user's machine. -2. **No network by default** — the base install has no network dependency. The only code path that - can leave the machine is an *external embedding API*, which is **opt-in and off by default**. -3. **Never index sensitive material** — secrets, `.env`, keys, certs, build/dependency/binary/ - generated/huge files are excluded before parsing. -4. **Redact secrets in output** — even indexed snippets are scrubbed before being shown to Claude. -5. **Respect ignore files** — `.gitignore`, `.cursorignore`, `.claudeignore`, `.codeindexignore`. -6. **Minimal, safe tool surface** — the skill's `allowed-tools` only permits the read-only CLI - subcommands and read-only fallbacks (ripgrep/Grep/Glob). No `clean`, no arbitrary shell. -7. **Workspace trust** — indexing executes parsers over repo content; treat indexing an untrusted - repo as you would opening it in an editor. `doctor` warns before risky operations. - -## 2. Exclusion pipeline (`discovery/`) - -A file must pass **every** gate to be indexed: - -| Gate | Rule | -|---|---| -| Ignore files | Not matched by `.gitignore` / `.cursorignore` / `.claudeignore` / `.codeindexignore` | -| Built-in denylist | Not in `node_modules`, `.venv`, `dist`, `build`, `target`, `.git`, `vendor`, `__pycache__`, etc. | -| Secret filenames | Not `.env*`, `*.pem`, `*.key`, `*.p12`, `*.pfx`, `id_rsa*`, `*.crt`, `*.keystore`, `credentials*`, `secrets*` | -| Binary | No NUL bytes / not a known binary extension (images, archives, fonts, compiled artifacts) | -| Size | `size_bytes <= max_file_bytes` (default 1 MB) | -| Generated | Not matched by generated-file patterns (`*.min.js`, `*.lock`, `*.pb.go`, `*_pb2.py`, `*.generated.*`) — indexed as summary-only at most | - -`.codeindexignore` is the tool's **own** ignore file (highest specificity) so users can exclude -paths from indexing without affecting git or other tools. - -## 3. Secret redaction (`output/` + parsers) - -Two layers: - -- **At index time** — files that *look* like secret stores are excluded entirely (above). -- **At output time** — every snippet is passed through a redactor before emission. Patterns: - - high-entropy strings assigned to keys named `*key*`, `*secret*`, `*token*`, `*password*`, `*api*` - - common formats: AWS keys (`AKIA...`), private key headers (`-----BEGIN ... PRIVATE KEY-----`), - JWTs, bearer tokens, connection strings with credentials, `xox[baprs]-` Slack tokens. - - Matches are replaced with `«redacted:»`, preserving line numbers. - -Redaction is conservative: it never widens the snippet, only masks within it. - -## 4. Embeddings & network - -- Default: `embeddings.backend = "noop"` (disabled) — pure lexical+symbol+graph search. -- `embeddings.backend = "local"` → on-device model (e.g. sentence-transformers). Still no network - at query time (model downloaded once at setup, which the user initiates explicitly). -- `embeddings.backend = "external"` → sends chunk text to a configured API. This requires: - - explicit `embeddings.allow_external = true` in config, **and** - - an env-provided API key, **and** - - `doctor` and `index` both print a clear warning naming the endpoint. - Without all three, external embedding is refused. - -## 5. Skill tool surface - -`SKILL.md` declares a narrow `allowed-tools`: - -```yaml -allowed-tools: - - Bash(codebase-index search:*) - - Bash(codebase-index explain:*) - - Bash(codebase-index symbol:*) - - Bash(codebase-index refs:*) - - Bash(codebase-index impact:*) - - Bash(codebase-index diff-impact:*) - - Bash(codebase-index architecture:*) - - Bash(codebase-index path:*) - - Bash(codebase-index describe:*) - - Bash(codebase-index graph:*) - - Bash(codebase-index stats:*) - - Bash(codebase-index doctor:*) - - Bash(codebase-index update:*) - - Bash(codebase-index index:*) - - Grep - - Glob -``` - -Explicitly **not** allowed via the skill: `clean`, `init`, `watch`, an unscoped -`codebase-index *`, `python -m codebase_index *`, or unscoped `Bash`. -Destructive or scaffolding actions remain a human/manual decision. The wrapper -scripts (`scripts/cbx`) only forward to the installed `codebase-index` binary -and reject unknown subcommands. - -## 6. `doctor` — safety self-check - -`codebase-index doctor` (and `--strict` for CI) reports: - -- whether the cache is inside `.gitignore` (warns if the index could be committed) -- whether external embeddings are enabled and to which endpoint -- any indexed file that matches a secret pattern (should be none → flags a leak) -- ignore-file coverage and any oversized/binary files that slipped through -- the resolved `allowed-tools` vs. the recommended minimal set -- world-writable cache directory permissions - -`doctor` exits non-zero under `--strict` if any high-severity finding is present, so it can gate CI. - -## 7. What the skill must NOT do - -- Must not run `codebase-index index --rebuild` automatically on huge/untrusted repos without the - freshness check indicating it's needed. -- Must not echo raw `.env`/secret file contents even if a user pastes a path — the CLI refuses to - read excluded files for snippet output. -- Must not enable embeddings or any network path on its own. +- Trust model, exclusion pipeline, redaction, embeddings/network policy, skill tool surface and + what `doctor` checks: [SECURITY_MODEL.md](SECURITY_MODEL.md). +- Reporting a vulnerability and supported versions: the root [SECURITY.md](../SECURITY.md). diff --git a/docs/SECURITY_MODEL.md b/docs/SECURITY_MODEL.md index e383c11..abc96af 100644 --- a/docs/SECURITY_MODEL.md +++ b/docs/SECURITY_MODEL.md @@ -63,8 +63,20 @@ External embeddings require **all three** conditions: Without all three, external embedding is refused. +## Skill tool surface + +The generated `SKILL.md` frontmatter restricts the agent to read-only subcommands +(`search`, `explain`, `architecture`, `symbol`, `refs`, `impact`, `diff-impact`, `path`, +`describe`, `graph`, `stats`, `doctor`, `update`, `index`, and the `cbx` wrapper) plus `Read`, +`Grep`, `Glob`. `clean`, `init`, `watch`, unscoped `Bash`, and `python -m codebase_index` are not +allowed. The `cbx` wrappers (skill `scripts/cbx` and plugin `bin/cbx`) enforce the same whitelist +and refuse other subcommands. + ## Threat Model +(`docs/SECURITY.md` was merged into this page; the reporting policy lives in the root +[SECURITY.md](../SECURITY.md).) + ### Trusted Inputs - The user's own codebase (they control what's in it) - Configuration files they create @@ -83,29 +95,32 @@ Without all three, external embedding is refused. | Malicious file content | Parsers are read-only; no code execution | | Cache leakage | Cache in `.gitignore`; doctor checks permissions | -## Unsafe Patterns - -- **Do not commit the SQLite index** to a shared repository. It contains code snippets. -- **Do not enable external embeddings** on repositories containing proprietary or regulated code without reviewing your organization's data handling policies. -- **Do not run `codebase-index index`** on repositories you do not trust without reviewing the `doctor` output first. +## `doctor` — safety self-check -## Hook Risks +`codebase-index doctor` (`src/codebase_index/doctor.py`) reports exactly these findings today: -Optional hooks (e.g., post-tool-use auto-update) execute CLI commands automatically. Ensure: +| id | severity | what it checks | +|---|---|---| +| `cache_gitignored` | high | the derived index directory is in `.gitignore` (a committed index leaks indexed text) | +| `hooks_enabled` | info | whether the Claude Code PostToolUse auto-update hook is configured | +| `index_fresh` | medium | the index exists and is not stale relative to the tree | +| `symbol_extraction` | medium | Tree-sitter languages actually produce symbols (guards silent parser failure) | +| `graph_coverage` | info | every indexed language has dependency-graph support, or names the partial ones | -- Hook commands are read-only or safe (`codebase-index update --quiet`) -- Hook commands do not contain user-controlled input -- Hook output is not echoed to the user unless necessary +`--strict` exits non-zero when a high-severity finding fails, so it can gate CI. The external- +embedding warning (naming the endpoint) is printed by `index` and `search` when the backend is +resolved, not by `doctor`. Secret-pattern scans of the index, `allowed-tools` diffs and +cache-permission checks are **not** implemented; do not rely on `doctor` for them. -## `doctor` — Safety Self-Check +## Hook risks -`codebase-index doctor` reports: +Optional hooks (the PostToolUse auto-update) execute CLI commands automatically. Keep hook +commands read-only (`codebase-index update --quiet`), free of user-controlled input, and quiet. -- Whether the cache is inside `.gitignore` (warns if the index could be committed) -- Whether external embeddings are enabled and to which endpoint -- Any indexed file that matches a secret pattern (should be none) -- Ignore-file coverage and any oversized/binary files that slipped through -- The resolved `allowed-tools` vs. the recommended minimal set -- World-writable cache directory permissions +## Unsafe patterns -With `--strict` flag, `doctor` exits non-zero if any high-severity finding is present, suitable for CI gating. +- Do not commit the SQLite index to a shared repository: it stores indexed text, and redaction + happens only at output time. +- Do not enable external embeddings on proprietary or regulated code without reviewing your + organisation's data-handling policy. +- Review `doctor` output before indexing a repository you do not trust. diff --git a/docs/SEO.md b/docs/SEO.md deleted file mode 100644 index bd867d3..0000000 --- a/docs/SEO.md +++ /dev/null @@ -1,196 +0,0 @@ -# SEO Plan - -Repository SEO strategy for `codebase-index`. - -## Repository Metadata - -### Repository Name - -``` -codebase-index -``` - -Rationale: Matches the intended package name and primary product keyword. - -### GitHub About Description - -``` -Local-first codebase indexing for Claude Code, Codex CLI, OpenCode & AI coding agents — hybrid FTS5 + Tree-sitter + graph search, fully offline. -``` - -### GitHub Topics - -``` -ai-agents, ai-coding, claude-code, cli, code-search, codebase-indexing, codex-cli, context-engineering, cursor-alternative, developer-tools, fts5, local-first, mcp, opencode, python, rag, semantic-code-search, sqlite, token-optimization, tree-sitter -``` - -GitHub caps topics at 20; this list is the live set (all 20 slots used). `codebase-rag` -is a swap candidate if a slot frees up. - -### Website - -Leave blank initially. Can be set to GitHub Pages docs site later. - -## README SEO Strategy - -### First 150 Words - -The opening paragraph must contain these keywords naturally: - -- AI coding agents -- codebase indexing -- local-first -- Cursor-like indexing -- token-efficient context -- semantic code search -- AST symbol search - -### Searchable Headings - -Use these headings in the README (already implemented): - -1. "Local Codebase Indexing for AI Coding Agents" (hero section) -2. "What Is codebase-index?" (definition) -3. "What Problem Does codebase-index Solve?" (problem) -4. "How Does codebase-index Work?" (method) -5. "Which AI CLIs Does codebase-index Support?" (integration) - -### Keyword Density - -Each primary keyword should appear 1-3 times naturally: - -| Keyword | Target Count | Sections | -|---|---|---| -| AI coding agents | 3-5 | Hero, features, installation | -| codebase indexing | 2-4 | Hero, problem, solution | -| local-first | 3-5 | Hero, features, security | -| Cursor-like | 2-3 | Problem, comparison | -| token-efficient | 2-3 | Solution, features | -| semantic code search | 1-2 | Features, how it works | -| AST symbol search | 1-2 | Features, architecture | -| hybrid code search | 1-2 | Solution, retrieval | -| Tree-sitter | 2-3 | Features, architecture | -| SQLite FTS5 | 1-2 | Architecture, database schema | - -**Do not keyword-stuff.** Write naturally for humans first. - -## Badges - -Include shields.io badges in the README hero section: - -```markdown -![License](https://img.shields.io/badge/license-MIT-blue.svg) -![Python](https://img.shields.io/badge/python-3.11+-blue.svg) -![CI](https://github.com/denfry/codebase-index/actions/workflows/ci.yml/badge.svg) -![Claude Code Skill](https://img.shields.io/badge/Claude%20Code%20Skill-yes-green.svg) -![Codex CLI](https://img.shields.io/badge/Codex%20CLI-supported-green.svg) -![OpenCode](https://img.shields.io/badge/OpenCode-supported-green.svg) -![MCP](https://img.shields.io/badge/MCP-stdio%20server-green.svg) -![Local First](https://img.shields.io/badge/local--first-yes-green.svg) -![No Telemetry](https://img.shields.io/badge/no%20telemetry-yes-green.svg) -![No Network](https://img.shields.io/badge/no%20network%20by%20default-yes-green.svg) -![SQLite](https://img.shields.io/badge/database-SQLite-blue.svg) -![Tree-sitter](https://img.shields.io/badge/parsing-Tree--sitter-orange.svg) -``` - -## Social Preview Image - -Built and committed as `assets/social-preview.png` (1280×640). Regenerate with: - -```bash -python scripts/gen_assets.py -``` - -This also builds `assets/demo.png` (1200×760), the static terminal still embedded -near the top of `README.md`. - -- **Dimensions:** 1280×640 (GitHub recommended) -- **Background:** GitHub dark theme (#0d1117), accent glow -- **Text:** wordmark `codebase-index` + "Local codebase indexing for AI coding agents" -- **Elements:** terminal mock with a ranked search result + capability chips -- **Style:** clean, minimal, professional - -> **Action still required:** the file in the repo is not the social card by itself. -> Upload it in **Settings → General → Social preview** so GitHub serves it as the -> `og:image` on X / Slack / Discord / LinkedIn. (`usesCustomOpenGraphImage` is -> currently `false`.) - -## Launch Checklist - -- [x] Create v1.3.0 release with release notes -- [x] Add all GitHub topics (20/20 slots used; see list above) -- [x] Set repository description in About section -- [ ] Upload social preview image (`assets/social-preview.png` built; must be uploaded in Settings → Social preview) -- [x] Ensure README first 150 words contain target keywords -- [x] Verify all badges render correctly -- [ ] Submit to awesome Claude Code skills lists -- [ ] Submit to awesome AI coding tools lists -- [ ] Post announcement on: - - X/Twitter - - Reddit (r/LocalLLaMA, r/ClaudeAI, r/artificial) - - Hacker News (Show HN) - - Dev.to -- [~] Add demo GIF or terminal recording to README (`assets/demo.png` static still built; animated GIF still pending) -- [x] Ensure comparison page is complete (`docs/COMPARISON.md`) -- [x] Ensure security model page is complete (`docs/SECURITY_MODEL.md`) -- [x] Tag release on GitHub (`v1.3.0`) - -## Backlink Targets - -Submit to these lists for backlinks and discoverability: - -### Awesome Lists -- awesome-claude-code -- awesome-claude -- awesome-ai-coding-tools -- awesome-code-search -- awesome-developer-tools -- awesome-sqlite -- awesome-tree-sitter - -### Directories -- Claude Skill Directory (if exists) -- MCP Server Directory after client config docs are verified -- PyPI package listing after distribution hardening is complete - -### Communities -- Claude Code Discord -- AI Coding Tools communities -- Developer tool forums - -## Release Announcement Template - -``` -codebase-index v1.3.0 - -A local-first codebase index for AI coding agents. - -Instead of scanning your whole repo, Claude Code, Codex CLI, or OpenCode searches -a local hybrid index (FTS5 + symbols + graph) and reads only the relevant line ranges. - -Features: -- Local-first, no network by default -- SQLite + FTS5 full-text search -- Tree-sitter symbol extraction -- Claude Code, Codex CLI, and OpenCode setup -- Token-efficient retrieval packets -- Secret redaction -- Respects .gitignore - -Install: pip install "codebase-index @ git+https://github.com/denfry/codebase-index.git@v1.8.0" -GitHub: https://github.com/denfry/codebase-index -``` - -## Social Post Template - -``` -Gave Claude Code Cursor-like codebase awareness. - -codebase-index builds a local hybrid index so Claude finds the right files without scanning your whole repo. - -- FTS5 + symbols + graph search -- No network by default -- Token-efficient output - -pip install "codebase-index @ git+https://github.com/denfry/codebase-index.git@v1.8.0" -``` diff --git a/docs/demo.tape b/docs/demo.tape new file mode 100644 index 0000000..79f1c59 --- /dev/null +++ b/docs/demo.tape @@ -0,0 +1,29 @@ +# VHS tape for the README demo (see docs/DEMO.md). Run from the repository root +# after cloning Flask into .tmp-demo/flask at the pinned commit: +# vhs docs/demo.tape +Output assets/demo.gif +Set FontSize 15 +Set Width 1200 +Set Height 720 +Set Theme "Catppuccin Mocha" +Set TypingSpeed 40ms + +Type "cd .tmp-demo/flask && codebase-index index" +Enter +Sleep 4s + +Type 'codebase-index search "where is the session cookie signed and saved" --limit 3' +Enter +Sleep 5s + +Type "codebase-index refs open_session" +Enter +Sleep 4s + +Type "codebase-index impact SecureCookieSessionInterface --direction up --depth 2" +Enter +Sleep 5s + +Type "codebase-index diff-impact" +Enter +Sleep 6s diff --git a/docs/installer.md b/docs/installer.md index 10fddd1..fd8bd05 100644 --- a/docs/installer.md +++ b/docs/installer.md @@ -1,29 +1,32 @@ -# Multi-CLI Installer для codebase-index +# Multi-CLI installer for codebase-index -Единый GitHub-hosted installer, который раскладывает skill **codebase-index** -сразу в несколько AI-CLI сред: **Claude Code**, **Codex CLI** и **OpenCode**. +A single GitHub-hosted installer that lays out the **codebase-index** skill into several +AI-CLI environments at once: **Claude Code**, **Codex CLI** and **OpenCode**. -Архитектура: один entrypoint (`install.sh` / `install.ps1`) + отдельные -адаптеры под каждую среду (`adapters/.{sh,ps1}`). Источник правды — -каталог `skill/` в репозитории. Никакой «магии»: все пути логируются и -переопределяются. +Architecture: one entrypoint (`install.sh` / `install.ps1`) plus one adapter per environment +(`adapters/.{sh,ps1}`). The source of truth is the `skill/` directory in the +repository. No magic: every path is logged and can be overridden. + +> Most users should prefer `pip install codebase-index` followed by `codebase-index init` +> (see [INSTALLATION.md](INSTALLATION.md)). The shell installer exists for environments where +> the skill files need to be placed globally without a Python package install first. --- -## Что устанавливается +## What gets installed -| CLI | Что именно | Куда (global по умолчанию) | -|-----|------------|----------------------------| -| **Claude Code** | skill-директория с `SKILL.md` + scripts | `~/.claude/skills/codebase-index/` | -| **Codex CLI** | managed block в `AGENTS.md` + ресурсы (instruction package, **не** Claude Skill) | блок в `~/.codex/AGENTS.md`, ресурсы в `~/.codex/skills/codebase-index/` | -| **OpenCode** | markdown-команда `/codebase-index` + agent-файл + ресурсы | `~/.config/opencode/commands/`, `.../agents/`, `.../skills/codebase-index/` | +| CLI | What | Where (global scope, the default) | +|-----|------|-----------------------------------| +| **Claude Code** | skill directory with `SKILL.md` + scripts | `~/.claude/skills/codebase-index/` | +| **Codex CLI** | managed block in `AGENTS.md` + resources (an instruction package, **not** a Claude Skill) | block in `~/.codex/AGENTS.md`, resources in `~/.codex/skills/codebase-index/` | +| **OpenCode** | markdown command `/codebase-index` + agent file + resources | `~/.config/opencode/commands/`, `.../agents/`, `.../skills/codebase-index/` | -Для project-установки (`--scope project`) пути меняются на -`./.claude/...`, `./AGENTS.md`, `./.opencode/...`. +With a project install (`--scope project`) the paths become `./.claude/...`, `./AGENTS.md`, +`./.opencode/...`. --- -## Быстрая установка +## Quick install **macOS / Linux:** @@ -37,10 +40,9 @@ curl -fsSL https://raw.githubusercontent.com/denfry/codebase-index/main/install. irm https://raw.githubusercontent.com/denfry/codebase-index/main/install.ps1 | iex ``` -### Безопасная установка (с предпросмотром — рекомендуется) +### Safer install (download, read, then run — recommended) -Pipe-to-shell исполняет удалённый код «вслепую». Безопаснее скачать, прочитать -и только потом запустить: +Pipe-to-shell executes remote code blindly. Download, read, then run: ```sh curl -fsSL https://raw.githubusercontent.com/denfry/codebase-index/main/install.sh -o install.sh @@ -56,9 +58,9 @@ pwsh ./install.ps1 --- -## Сценарии использования +## Usage scenarios -**Конкретный target:** +**A specific target:** ```sh sh install.sh --target claude @@ -70,32 +72,32 @@ sh install.sh --target opencode pwsh ./install.ps1 -Target claude ``` -**Все сразу:** +**Everything at once:** ```sh sh install.sh --target all ``` -**Авто-определение (по умолчанию)** — installer сам находит CLI по PATH и -типичным конфиг-директориям (`~/.claude`, `~/.codex`, `~/.config/opencode`): +**Auto-detect (default)** — the installer looks for CLIs on `PATH` and in the usual config +directories (`~/.claude`, `~/.codex`, `~/.config/opencode`): ```sh sh install.sh # = --target auto ``` -**Dry-run** (ничего не меняет, показывает план): +**Dry run** (changes nothing, prints the plan): ```sh sh install.sh --target all --dry-run ``` -**Uninstall** (удаляет только установленное этим installer — по manifest): +**Uninstall** (removes only what this installer wrote, per its manifest): ```sh sh install.sh --target all --uninstall ``` -**Переопределение директории установки:** +**Override the install directory:** ```sh sh install.sh --target claude --install-dir "$HOME/my-skills/codebase-index" @@ -105,53 +107,53 @@ sh install.sh --target claude --install-dir "$HOME/my-skills/codebase-index" pwsh ./install.ps1 -Target claude -InstallDir "D:\skills\codebase-index" ``` -**Pinning по ветке/тегу** (воспроизводимость и безопасность): +**Pin to a branch or tag** (reproducibility and safety): ```sh -sh install.sh --branch v1.8.0 +sh install.sh --branch v1.9.0 ``` --- -## Флаги +## Flags -| install.sh | install.ps1 | Значение | -|------------|-------------|----------| +| install.sh | install.ps1 | Meaning | +|------------|-------------|---------| | `--target` | `-Target` | `claude\|codex\|opencode\|all\|auto` | -| `--install-dir` | `-InstallDir` | переопределить путь установки | -| `--repo-url` | `-RepoUrl` | URL репозитория-источника | -| `--branch` | `-Branch` | ветка/тег для скачивания | +| `--install-dir` | `-InstallDir` | override the install path | +| `--repo-url` | `-RepoUrl` | source repository URL | +| `--branch` | `-Branch` | branch or tag to download | | `--scope` | `-Scope` | `global\|project` | -| `--dry-run` | `-DryRun` | не вносить изменений | -| `--force` | `-Force` | перезаписать существующую установку (с backup) | -| `--no-python-bootstrap` | `-NoPythonBootstrap` | не создавать venv / не запускать bootstrap.py | -| `--verbose` | `-Verbose` | подробный лог | -| `--uninstall` | `-Uninstall` | удалить по manifest | -| `--help` | `-Help` | справка | +| `--dry-run` | `-DryRun` | make no changes | +| `--force` | `-Force` | overwrite an existing install (a backup is taken first) | +| `--no-python-bootstrap` | `-NoPythonBootstrap` | do not create a venv / run `bootstrap.py` | +| `--verbose` | `-Verbose` | verbose log | +| `--uninstall` | `-Uninstall` | remove per manifest | +| `--help` | `-Help` | show help | --- ## Python / runtime -После раскладки файлов installer (если не указан `--no-python-bootstrap`): +After laying out the files, unless `--no-python-bootstrap` was given, the installer: -1. ищет `python3`/`python` (Unix) или `py`/`python` (Windows), **минимум 3.9**; -2. если Python не найден — **не** ставит системный Python сам, а печатает - понятную инструкцию; -3. при наличии Python создаёт `.venv` внутри директории skill; -4. ставит зависимости из `requirements.txt`, если файл есть; -5. запускает `skill/scripts/bootstrap.py` (создаёт `runtime.json`, дополняет manifest). +1. looks for `python3` / `python` (Unix) or `py` / `python` (Windows), **3.9 minimum** for the + bootstrap itself (the `codebase-index` package requires 3.11+); +2. if no Python is found, does **not** install one, but prints a clear instruction; +3. otherwise creates a `.venv` inside the skill directory; +4. installs dependencies from `requirements.txt` if that file exists; +5. runs `skill/scripts/bootstrap.py` (creates `runtime.json`, extends the manifest). --- ## Manifest -После установки в директории skill создаётся `install_manifest.json`: +After installation the skill directory contains `install_manifest.json`: ```json { "skill_name": "codebase-index", - "version": "1.8.0", + "version": "1.9.0", "target": "claude", "os": "linux", "source_repo": "https://github.com/denfry/codebase-index", @@ -162,70 +164,67 @@ sh install.sh --branch v1.8.0 } ``` -Uninstall читает этот файл и удаляет **только** перечисленные в нём файлы. -Для `AGENTS.md` удаляется только managed block, сам файл сохраняется. +Uninstall reads this file and removes **only** the files listed in it. For `AGENTS.md` only the +managed block is removed; the file itself is kept. --- -## Как запускать в OpenCode +## Running in OpenCode -После установки команда доступна как: +After installation the command is available as: ``` -/codebase-index <запрос> +/codebase-index ``` --- ## Troubleshooting -- **«Python 3.9+ не найден»** — установите Python и повторите без - `--no-python-bootstrap`, либо игнорируйте, если venv не нужен. -- **«Уже установлено … (используйте --force)»** — добавьте `--force` - (создаётся backup перед перезаписью). -- **Неверный путь для вашей версии Claude Code** — задайте `--install-dir`. - Дефолтные пути вынесены в функции `*_default_dir` / `Get-*TargetDir`. -- **Нет curl/wget (Unix)** — установите один из них; на Windows используется - `Invoke-WebRequest`. -- **Авто-режим ничего не нашёл (exit 4)** — укажите `--target` явно. +- **"Python 3.9+ not found"** — install Python and re-run without `--no-python-bootstrap`, or + ignore it if you do not need the venv. +- **"Already installed … (use --force)"** — add `--force` (a backup is taken before overwriting). +- **Wrong path for your Claude Code version** — pass `--install-dir`. Default paths live in the + `*_default_dir` / `Get-*TargetDir` functions. +- **No curl/wget (Unix)** — install one of them; Windows uses `Invoke-WebRequest`. +- **Auto mode found nothing (exit 4)** — pass `--target` explicitly. --- ## Security notes -- Удалённый код не исполняется «на лету»: архив сначала скачивается, - проверяется структура (`SKILL.md` + frontmatter) и пути (запрет traversal). -- URL источника всегда печатается перед скачиванием. -- Поддерживается pinning по `--branch`. -- Без `--install-dir` installer не пишет за пределы `HOME`/проекта. -- `sudo` не используется. -- Есть `--dry-run`. +- Remote code is not executed on the fly: the archive is downloaded first, its structure is + checked (`SKILL.md` + frontmatter) and its paths are checked (no traversal). +- The source URL is always printed before downloading. +- Pinning with `--branch` is supported. +- Without `--install-dir` the installer never writes outside `HOME` / the project. +- `sudo` is never used. +- `--dry-run` is available. --- ## Developer notes -### Структура +### Layout ``` -install.sh / install.ps1 entrypoints -lib/common.sh / common.ps1 общие функции (лог, скачивание, manifest, bootstrap) -adapters/.{sh,ps1} логика конкретного CLI -skill/ источник правды (SKILL.md, scripts/bootstrap.py) -tests/installer/smoke.{sh,ps1} smoke-тесты +install.sh / install.ps1 entrypoints +lib/common.sh / common.ps1 shared functions (logging, download, manifest, bootstrap) +adapters/.{sh,ps1} per-CLI logic +skill/ source of truth (SKILL.md, scripts/bootstrap.py) +tests/installer/smoke.{sh,ps1} smoke tests ``` -### Как добавить новый adapter +### Adding an adapter -1. Создайте `adapters/.sh`, определив функции `adapter_install` и - `adapter_uninstall` (используйте функции из `lib/common.sh`). -2. Создайте `adapters/.ps1` с `Invoke-AdapterInstall` / - `Invoke-AdapterUninstall`. -3. Добавьте `` в `--target`/`-Target` и в авто-определение - (`detect_targets` / `Get-AutoTargets`). -4. Допишите строку в smoke-тесты. +1. Create `adapters/.sh` defining `adapter_install` and `adapter_uninstall` (use the + helpers from `lib/common.sh`). +2. Create `adapters/.ps1` with `Invoke-AdapterInstall` / `Invoke-AdapterUninstall`. +3. Add `` to `--target` / `-Target` and to auto-detection (`detect_targets` / + `Get-AutoTargets`). +4. Add a line to the smoke tests. -### Как тестировать локально +### Testing locally ```sh sh tests/installer/smoke.sh @@ -235,5 +234,5 @@ sh tests/installer/smoke.sh pwsh tests/installer/smoke.ps1 ``` -Smoke-тест прогоняет dry-run для всех целей, ставит skill в temp-директорию, -проверяет `SKILL.md` + manifest и выполняет uninstall. +The smoke test dry-runs every target, installs the skill into a temp directory, checks +`SKILL.md` + the manifest, and uninstalls. diff --git a/docs/superpowers/plans/2026-05-29-m9-release.md b/docs/superpowers/plans/2026-05-29-m9-release.md index 363352e..4c2573d 100644 --- a/docs/superpowers/plans/2026-05-29-m9-release.md +++ b/docs/superpowers/plans/2026-05-29-m9-release.md @@ -807,7 +807,7 @@ In `README.md`, replace the `## Status` section body (the "🚧 Blueprint / scaf ✅ **`0.1.0` released.** All milestones M0–M9 are implemented: discovery + storage, FTS5 lexical search, tree-sitter symbols/refs, hybrid ranking, graph impact, optional local embeddings, the packaged skill + freshness contract, hooks/watch, and a tested, `pipx`-installable release. See -[CHANGELOG.md](CHANGELOG.md) and [docs/ROADMAP.md](docs/ROADMAP.md). +[CHANGELOG.md](../../../CHANGELOG.md) and [docs/ROADMAP.md](../../ROADMAP.md). ``` Also add `[CHANGELOG.md](CHANGELOG.md)` to the Documentation list. diff --git a/examples/config.example.json b/examples/config.example.json index f28b046..6d4711c 100644 --- a/examples/config.example.json +++ b/examples/config.example.json @@ -9,7 +9,10 @@ "default_mode": "hybrid", "rrf_k": 60, "token_budget": 1500, - "limit": 10 + "limit": 10, + "compact_snippets": true, + "compact_min_reduction": 0.25, + "max_read_lines": 120 }, "embeddings": { "backend": "noop", diff --git a/examples/demo-project/README.md b/examples/demo-project/README.md deleted file mode 100644 index ee8ed47..0000000 --- a/examples/demo-project/README.md +++ /dev/null @@ -1,55 +0,0 @@ -# Demo Project - -This is a sample project structure for demonstrating `codebase-index` capabilities. - -## Structure - -``` -demo-project/ -├── src/ -│ ├── auth/ -│ │ ├── __init__.py -│ │ ├── service.py # AuthService class -│ │ └── middleware.py # requireAuth middleware -│ ├── models/ -│ │ ├── __init__.py -│ │ └── user.py # User model -│ ├── routes/ -│ │ ├── __init__.py -│ │ └── auth.py # Login/logout routes -│ ├── config.py # Configuration loader -│ └── app.py # Application entry point -├── tests/ -│ ├── test_auth.py -│ └── test_user.py -├── .env # Should be excluded from index -├── .gitignore -└── package.json -``` - -## Try It - -```bash -cd examples/demo-project - -# Initialize and index -codebase-index init -codebase-index index - -# Try some queries -codebase-index search "authentication" -codebase-index symbol "AuthService" -codebase-index refs "login" -codebase-index impact "User" -codebase-index stats -``` - -## Expected Results - -After indexing, you should see: - -- `AuthService` class extracted from `src/auth/service.py` -- `User` model extracted from `src/models/user.py` -- Route handlers extracted from `src/routes/auth.py` -- `.env` file excluded (secret file) -- FTS5 index populated with chunk text from all source files diff --git a/examples/demo/EXPECTED_OUTPUT.md b/examples/demo/EXPECTED_OUTPUT.md new file mode 100644 index 0000000..5f1e4ac --- /dev/null +++ b/examples/demo/EXPECTED_OUTPUT.md @@ -0,0 +1,187 @@ +# Expected output + +Captured by running `examples/demo/run_demo.sh` against +[pallets/flask](https://github.com/pallets/flask) at commit `d318b683` with +codebase-index 1.9.0 on 2026-09-04. Timestamps and ordering of equal-score rows +may differ on your machine; file paths, line ranges and edge kinds should not. + +```text +$ codebase-index index +Indexed 230 files (0 pruned). + parse failures: 0; tree-sitter files with 0 symbols: 17 + +$ codebase-index search where is the session cookie signed and saved --limit 5 +**Query:** where is the session cookie signed and saved +**Intent:** `locate_impl` · **Confidence:** medium + +| # | Path | Lines | Reason | +|---|------|-------|--------| +| 1 | `src/flask/sessions.py` | 24-54 | in src/flask/ · 2 callers · source prior +0.08 | +| 2 | `src/flask/sessions.py` | 284-385 | in src/flask/ · 2 callers · source prior +0.08 | +| 3 | `src/flask/sessions.py` | 57-80 | in src/flask/ · 1 callers · source prior +0.08 | +| 4 | `tests/test_basic.py` | 542-600 | source prior -0.06 · generated/test demoted | +| 5 | `tests/test_reqctx.py` | 201-249 | source prior -0.06 · generated/test demoted | + +`src/flask/sessions.py:24-54` +``` +class SessionMixin(MutableMapping[str, t.Any]): +``` +`src/flask/sessions.py:284-385` +``` +class SecureCookieSessionInterface(SessionInterface): +``` +`src/flask/sessions.py:57-80` +``` +class SecureCookieSession(CallbackDict[str, t.Any], SessionMixin): +``` +`tests/test_basic.py:542-600` +``` +def test_session_vary_cookie(app, client): +``` +`tests/test_reqctx.py:201-249` +``` +def test_session_dynamic_cookie_name(): +``` + +**Recommended reads:** +- `src/flask/sessions.py:24-54` +- `src/flask/sessions.py:284-385` +- `src/flask/sessions.py:57-80` +- `tests/test_basic.py:542-600` +- `tests/test_reqctx.py:201-249` + +$ codebase-index refs open_session +**query:** open_session | **index:** fresh + +| kind | path | line | confidence | +|------|------|------|------------| +| call | `src/flask/ctx.py` | 388 | ? ambiguous | +| definition | `src/flask/sessions.py` | 249 | exact | +| definition | `src/flask/sessions.py` | 323 | exact | +| call | `src/flask/testing.py` | 165 | ? ambiguous | +| definition | `tests/test_reqctx.py` | 182 | exact | +| definition | `tests/test_session_interface.py` | 16 | exact | + + +$ codebase-index path wsgi_app dispatch_request +**path:** `wsgi_app` → `dispatch_request` · **2 hop(s)** + +`wsgi_app` (src/flask/app.py) + → _call_ → +`full_dispatch_request` (src/flask/app.py) + → _call_ → +`dispatch_request` (src/flask/app.py) + + +$ codebase-index impact SecureCookieSessionInterface --direction up --depth 2 +**impact:** `SecureCookieSessionInterface` · **direction:** up · **depth:** 2 · **affected files:** 2 + +| dist | via | kind | node | location | +|------|-----|------|------|----------| +| 1 | call | symbol | `Flask` | `src/flask/app.py:110` | +| 1 | extends | symbol | `PathAwareSessionInterface` | `tests/test_reqctx.py:204` | +| 2 | call | symbol | `CustomFlask` | `tests/test_reqctx.py:211` | + + +$ codebase-index diff-impact +**diff impact:** `HEAD` → working tree +**direction:** `up` · **depth:** 2 + +**changed files (1):** +- `src/flask/sessions.py` + +**affected files (8):** +| distance | path | changed by | edge | confidence | +|---:|---|---|---|---| +| 1 | `src/flask/app.py` | `src/flask/sessions.py` | call | extracted | +| 1 | `src/flask/ctx.py` | `src/flask/sessions.py` | call | extracted | +| 1 | `src/flask/globals.py` | `src/flask/sessions.py` | extends | extracted | +| 1 | `src/flask/testing.py` | `src/flask/sessions.py` | call | extracted | +| 1 | `tests/test_reqctx.py` | `src/flask/sessions.py` | extends | extracted | +| 1 | `tests/test_session_interface.py` | `src/flask/sessions.py` | extends | extracted | +| 2 | `tests/test_signals.py` | `src/flask/sessions.py` | call | extracted | +| 2 | `tests/test_testing.py` | `src/flask/sessions.py` | call | extracted | + + +$ codebase-index --json search where is the session cookie signed and saved --limit 3 +{ + "query": "where is the session cookie signed and saved", + "intent": "locate_impl", + "mode": "hybrid", + "index": { + "exists": true, + "stale": false, + "files_changed_since_build": 0, + "built_at": "2026-09-04T13:13:31Z", + "head_commit": "d318b683471101618febed18996405ad26462110" + }, + "confidence": "medium", + "results": [ + { + "path": "src/flask/sessions.py", + "line_start": 24, + "line_end": 54, + "symbols": [ + "SessionMixin" + ], + "score": 1.9301, + "reason": "in src/flask/ · 2 callers · source prior +0.08", + "token_est": 11, + "rank": 1, + "skeletonized": false, + "elided_lines": 0, + "snippet": "class SessionMixin(MutableMapping[str, t.Any]):" + }, + { + "path": "src/flask/sessions.py", + "line_start": 284, + "line_end": 385, + "symbols": [ + "SecureCookieSessionInterface" + ], + "score": 1.8766, + "reason": "in src/flask/ · 2 callers · source prior +0.08", + "token_est": 13, + "rank": 2, + "skeletonized": false, + "elided_lines": 0, + "snippet": "class SecureCookieSessionInterface(SessionInterface):" + }, + { + "path": "src/flask/sessions.py", + "line_start": 57, + "line_end": 80, + "symbols": [ + "SecureCookieSession" + ], + "score": 1.8679, + "reason": "in src/flask/ · 1 callers · source prior +0.08", + "token_est": 16, + "rank": 3, + "skeletonized": false, + "elided_lines": 0, + "snippet": "class SecureCookieSession(CallbackDict[str, t.Any], SessionMixin):" + } + ], + "recommended_reads": [ + { + "path": "src/flask/sessions.py", + "line_start": 24, + "line_end": 54 + }, + { + "path": "src/flask/sessions.py", + "line_start": 284, + "line_end": 385 + }, + { + "path": "src/flask/sessions.py", + "line_start": 57, + "line_end": 80 + } + ], + "fallback_suggestions": {} +} + +Done. Index lives in C:/Users/dabin/AppData/Local/Temp/claude/D--Projects-codebase-index/7f35eb46-c8c9-49e7-9d02-4c510c6877d0/scratchpad/repos/flask/.claude/cache/codebase-index/ and nothing was sent anywhere. +``` diff --git a/examples/demo/README.md b/examples/demo/README.md new file mode 100644 index 0000000..de4451e --- /dev/null +++ b/examples/demo/README.md @@ -0,0 +1,53 @@ +# Demo: Find, Trace, Predict on a real repository + +This demo runs `codebase-index` against [Flask](https://github.com/pallets/flask) +(about 230 files, 18k lines of Python) at a pinned commit, so what you see is what +[EXPECTED_OUTPUT.md](EXPECTED_OUTPUT.md) shows. It takes under a minute and makes no +network requests after the clone. + +```bash +pip install codebase-index +bash examples/demo/run_demo.sh # or: pwsh examples/demo/run_demo.ps1 +``` + +Pass a path to reuse an existing Flask checkout: `bash examples/demo/run_demo.sh ~/src/flask`. + +## What it shows + +| Step | Question | Command | What to look at | +|---|---|---|---| +| Find | Where is the session cookie signed and saved? | `codebase-index search "where is the session cookie signed and saved" --limit 5` | All three top hits are in `src/flask/sessions.py` with exact line ranges; tests are ranked below implementation ("generated/test demoted"). | +| Trace | Who calls `open_session`? | `codebase-index refs open_session` | Definitions vs calls, and the confidence column: two callers are marked `ambiguous` because several classes define `open_session`. The tool says so instead of guessing. | +| Trace | How does a request reach the view? | `codebase-index path wsgi_app dispatch_request` | A two-hop call chain `wsgi_app → full_dispatch_request → dispatch_request`, all in `app.py`. | +| Predict | What could break if `SecureCookieSessionInterface` changes? | `codebase-index impact SecureCookieSessionInterface --direction up --depth 2` | Upstream dependents at distance 1 and 2, with the edge kind (`call`, `extends`) that connects them. | +| Predict | What does my current diff affect? | edit `sessions.py`, then `codebase-index diff-impact` | Eight affected files ranked by distance from the changed file, each with the edge kind and `extracted` confidence. | +| Agent view | Same search as JSON | `codebase-index --json search ... --limit 3` | The packet an agent receives: freshness (`index`), `confidence`, ranked results with `snippet`, and `recommended_reads` with line ranges. | + +## Things worth noticing + +- **Confidence is part of the answer.** `refs` labels edges `exact` or `ambiguous`; + `impact` and `diff-impact` carry `extracted` / `inferred` / `ambiguous`. An agent can + distinguish "nothing references this" from "the graph is unsure". +- **The read plan is bounded.** `recommended_reads` entries longer than + `retrieval.max_read_lines` (default 120) are capped at the definition head and marked + `truncated: true` with `line_end_full`, so a 1,500-line class does not become a + 1,500-line read. +- **Tests are demoted, not hidden.** Search ranks `tests/test_basic.py` below + `src/flask/sessions.py` for an implementation question, but still lists it. +- **Nothing left the machine.** The index is a SQLite file under + `.claude/cache/codebase-index/`; `codebase-index doctor --strict` audits the + network-off default. + +## Try your own questions + +```bash +cd .tmp-demo/flask +codebase-index explain "how are blueprints registered" +codebase-index describe dispatch_request +codebase-index architecture +codebase-index search "jsonify" --mode symbol +``` + +To run the same demo on your own repository, replace the clone step with `cd your-repo` +and keep the rest. For recording a GIF or video of this session, see +[docs/DEMO.md](../../docs/DEMO.md). diff --git a/examples/demo/run_demo.ps1 b/examples/demo/run_demo.ps1 new file mode 100644 index 0000000..7330f8f --- /dev/null +++ b/examples/demo/run_demo.ps1 @@ -0,0 +1,42 @@ +# Reproducible demo of codebase-index on a real public repository (Flask). +# +# pwsh examples/demo/run_demo.ps1 # clones Flask into .\.tmp-demo +# pwsh examples/demo/run_demo.ps1 C:\src\flask # use an existing checkout +# +# Mirrors run_demo.sh. Nothing leaves the machine. +param([string]$WorkDir = ".tmp-demo\flask") +$ErrorActionPreference = "Stop" +$RepoUrl = "https://github.com/pallets/flask.git" +$RepoSha = "d318b683471101618febed18996405ad26462110" +$env:CBX_NO_SKILL_AUTO_UPDATE = "1" + +function Step { param([Parameter(ValueFromRemainingArguments)] [string[]]$Cmd) + Write-Host "`n$ $($Cmd -join ' ')" -ForegroundColor Blue + & $Cmd[0] @($Cmd[1..($Cmd.Length - 1)]) + if ($LASTEXITCODE -ne 0) { throw "command failed: $($Cmd -join ' ')" } +} + +if (-not (Get-Command codebase-index -ErrorAction SilentlyContinue)) { + Write-Error "codebase-index is not on PATH. Install it first: pip install codebase-index" +} +if (-not (Test-Path (Join-Path $WorkDir ".git"))) { + Write-Host "Cloning Flask into $WorkDir ..." + git clone --quiet $RepoUrl $WorkDir +} +git -C $WorkDir checkout --quiet $RepoSha +Set-Location $WorkDir + +Step codebase-index index +Step codebase-index search "where is the session cookie signed and saved" --limit 5 +Step codebase-index refs open_session +Step codebase-index path wsgi_app dispatch_request +Step codebase-index impact SecureCookieSessionInterface --direction up --depth 2 + +$sessions = "src/flask/sessions.py" +(Get-Content $sessions -Raw) -replace "(?m)^ def open_session\(", " def open_session( # demo edit" | + Set-Content $sessions -NoNewline -Encoding utf8 +Step codebase-index diff-impact +git checkout --quiet -- $sessions + +Step codebase-index --json search "where is the session cookie signed and saved" --limit 3 +Write-Host "`nDone. Index lives in $WorkDir\.claude\cache\codebase-index\ and nothing was sent anywhere." diff --git a/examples/demo/run_demo.sh b/examples/demo/run_demo.sh new file mode 100644 index 0000000..794b635 --- /dev/null +++ b/examples/demo/run_demo.sh @@ -0,0 +1,53 @@ +#!/usr/bin/env bash +# Reproducible demo of codebase-index on a real public repository (Flask). +# +# bash examples/demo/run_demo.sh # clones Flask into ./.tmp-demo +# bash examples/demo/run_demo.sh /path/to/flask # use an existing checkout +# +# Every command below is what an agent (or you) would run; the output is what +# examples/demo/EXPECTED_OUTPUT.md was captured from. Nothing leaves the machine. +set -euo pipefail + +REPO_URL="https://github.com/pallets/flask.git" +REPO_SHA="d318b683471101618febed18996405ad26462110" # pinned so the output is stable +WORKDIR="${1:-.tmp-demo/flask}" +export CBX_NO_SKILL_AUTO_UPDATE=1 + +step() { printf '\n\033[1;34m$ %s\033[0m\n' "$*"; "$@"; } + +if ! command -v codebase-index >/dev/null 2>&1; then + echo "codebase-index is not on PATH. Install it first: pip install codebase-index" >&2 + exit 1 +fi + +if [ ! -d "$WORKDIR/.git" ]; then + echo "Cloning Flask into $WORKDIR ..." + git clone --quiet "$REPO_URL" "$WORKDIR" +fi +git -C "$WORKDIR" checkout --quiet "$REPO_SHA" +cd "$WORKDIR" + +# 1. Build the index (about 230 files; a few seconds). +step codebase-index index + +# 2. FIND — where is the session cookie signed and saved? +step codebase-index search "where is the session cookie signed and saved" --limit 5 + +# 3. TRACE — who calls open_session? (definitions, callers, edge confidence) +step codebase-index refs open_session + +# 4. TRACE — how does a WSGI request reach the view function? +step codebase-index path wsgi_app dispatch_request + +# 5. PREDICT — what could break if SecureCookieSessionInterface changes? +step codebase-index impact SecureCookieSessionInterface --direction up --depth 2 + +# 6. PREDICT — what does my current diff affect? (make a throwaway edit, then revert) +sed -i.bak 's/^ def open_session(/ def open_session( # demo edit/' src/flask/sessions.py +step codebase-index diff-impact +git checkout --quiet -- src/flask/sessions.py && rm -f src/flask/sessions.py.bak + +# 7. The same packet as an agent sees it. +step codebase-index --json search "where is the session cookie signed and saved" --limit 3 + +printf '\nDone. Index lives in %s/.claude/cache/codebase-index/ and nothing was sent anywhere.\n' "$WORKDIR" diff --git a/pyproject.toml b/pyproject.toml index c967628..4761765 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,8 +11,9 @@ requires-python = ">=3.11" license = { text = "MIT" } authors = [{ name = "codebase-index contributors" }] keywords = [ - "claude-code", "codex-cli", "opencode", "mcp", "code-search", - "semantic-code-search", "codebase-indexing", "codebase-rag", "ai-agents", + "claude-code", "claude-code-plugin", "codex-cli", "opencode", "mcp", "mcp-server", + "code-search", "semantic-code-search", "codebase-indexing", "codebase-rag", + "code-graph", "impact-analysis", "ai-agents", "coding-agents", "context-engineering", "local-first", "tree-sitter", "rag", "sqlite", "fts5", "cli", ] classifiers = [ diff --git a/scripts/apply_labels.sh b/scripts/apply_labels.sh new file mode 100644 index 0000000..77f8cdf --- /dev/null +++ b/scripts/apply_labels.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# Create or update the repository labels from .github/labels.yml. +# Requires: gh (authenticated), python3 with PyYAML (installed by the dev extra). +# scripts/apply_labels.sh # denfry/codebase-index +# scripts/apply_labels.sh owner/repo # another repository +set -euo pipefail + +REPO="${1:-denfry/codebase-index}" +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +LABELS="$HERE/../.github/labels.yml" + +python3 - "$LABELS" <<'PY' | while IFS=$'\t' read -r name color desc; do +import sys, yaml +for row in yaml.safe_load(open(sys.argv[1], encoding="utf-8")): + print(f"{row['name']}\t{row['color']}\t{row.get('description', '')}") +PY + echo "label: $name" + gh label create "$name" --repo "$REPO" --color "$color" --description "$desc" --force +done diff --git a/scripts/check_links.py b/scripts/check_links.py new file mode 100644 index 0000000..c06be07 --- /dev/null +++ b/scripts/check_links.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Fail on broken relative links in Markdown files. + + python scripts/check_links.py # whole repository + python scripts/check_links.py docs README.md + +Checks `[text](target)` and `` / `` where the +target is a relative path. `#fragment` suffixes are stripped; http(s):, mailto:, +and fragment-only links are ignored. Prints `path:line: target` for each broken +link and exits 1 when any is found. Stdlib only; runs in CI lint. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +REPO = Path(__file__).resolve().parents[1] +SKIP_DIRS = {".git", ".venv", "venv", "node_modules", "dist", "build", ".tmp-public-benchmark"} + +MD_LINK_RE = re.compile(r"(?]*?\b(?:src|href)=\"([^\"]+)\"", re.I) +FENCE_RE = re.compile(r"^\s*(```|~~~)") +INLINE_CODE_RE = re.compile(r"`[^`]*`") +SCHEME_RE = re.compile(r"^[a-zA-Z][a-zA-Z0-9+.-]*:") + + +def _is_external(target: str) -> bool: + return bool(SCHEME_RE.match(target)) or target.startswith(("#", "//", "${", "<")) + + +def find_broken(md_file: Path, repo: Path = REPO) -> list[tuple[int, str]]: + broken: list[tuple[int, str]] = [] + in_fence = False + for lineno, line in enumerate(md_file.read_text(encoding="utf-8", errors="replace").splitlines(), 1): + if FENCE_RE.match(line): + in_fence = not in_fence + continue + if in_fence: + continue + line = INLINE_CODE_RE.sub("", line) # `[x](y)` inside code spans is prose, not a link + targets = MD_LINK_RE.findall(line) + MD_IMAGE_RE.findall(line) + HTML_SRC_RE.findall(line) + for raw in targets: + target = raw.strip("<>") + if _is_external(target): + continue + target = target.split("#", 1)[0].split("?", 1)[0] + if not target: + continue + base = repo if target.startswith("/") else md_file.parent + resolved = (base / target.lstrip("/")).resolve() + if not resolved.exists(): + broken.append((lineno, raw)) + return broken + + +def iter_markdown(paths: list[Path], repo: Path = REPO): + for root in paths: + if root.is_file(): + yield root + continue + for p in sorted(root.rglob("*.md")): + if any(part in SKIP_DIRS for part in p.relative_to(root).parts): + continue + yield p + + +def main(argv: list[str] | None = None) -> int: + args = argv if argv is not None else sys.argv[1:] + roots = [Path(a).resolve() for a in args] or [REPO] + total_files = 0 + total_broken = 0 + for md in iter_markdown(roots, REPO): + total_files += 1 + for lineno, target in find_broken(md, REPO): + total_broken += 1 + try: + shown = md.relative_to(REPO).as_posix() + except ValueError: + shown = md.as_posix() + print(f"{shown}:{lineno}: {target}") + if total_broken: + print(f"\n{total_broken} broken link(s) in {total_files} markdown file(s)", file=sys.stderr) + return 1 + print(f"links OK: {total_files} markdown file(s), no broken relative links") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/check_versions.py b/scripts/check_versions.py new file mode 100644 index 0000000..8a47e4e --- /dev/null +++ b/scripts/check_versions.py @@ -0,0 +1,101 @@ +#!/usr/bin/env python3 +"""Fail when the version is not the same everywhere it is recorded. + + python scripts/check_versions.py + +Canonical: src/codebase_index/__init__.py::__version__ (hatch dynamic version). +Mirrors that must agree: + .claude-plugin/plugin.json "version" + requirements.lock refs/tags/v.tar.gz + .claude|.codex|.opencode/skills/codebase-index/.skill_version + CHANGELOG.md a `## []` heading, OR the + version is newer than every released + heading and `## [Unreleased]` exists + (main between releases). +Stdlib only; runs in CI lint and before the release build. +""" + +from __future__ import annotations + +import json +import re +import sys +from pathlib import Path + +REPO = Path(__file__).resolve().parents[1] + +VERSION_RE = re.compile(r'^__version__ = "([^"]+)"$', re.M) +LOCK_TAG_RE = re.compile(r"refs/tags/v([0-9][^/]*?)\.tar\.gz") +HEADING_RE = re.compile(r"^## \[([^\]]+)\]", re.M) +STAMPS = ( + Path(".claude/skills/codebase-index/.skill_version"), + Path(".codex/skills/codebase-index/.skill_version"), + Path(".opencode/skills/codebase-index/.skill_version"), +) + + +def _vtuple(version: str) -> tuple[int, ...]: + return tuple(int(p) for p in re.findall(r"\d+", version)) + + +def check(repo: Path = REPO) -> list[str]: + """Return a list of mismatch descriptions; empty means consistent.""" + problems: list[str] = [] + init = (repo / "src/codebase_index/__init__.py").read_text(encoding="utf-8") + match = VERSION_RE.search(init) + if not match: + return ["src/codebase_index/__init__.py: no __version__"] + version = match.group(1) + + plugin = json.loads((repo / ".claude-plugin/plugin.json").read_text(encoding="utf-8")) + if plugin.get("version") != version: + problems.append(f".claude-plugin/plugin.json: {plugin.get('version')!r} != {version!r}") + + lock = (repo / "requirements.lock").read_text(encoding="utf-8") + tag = LOCK_TAG_RE.search(lock) + if not tag: + problems.append("requirements.lock: no refs/tags/vX.Y.Z.tar.gz pin found") + elif tag.group(1) != version: + problems.append(f"requirements.lock: tag v{tag.group(1)} != v{version}") + + for rel in STAMPS: + path = repo / rel + if not path.is_file(): + problems.append(f"{rel.as_posix()}: missing") + continue + stamp = path.read_text(encoding="utf-8").strip() + if stamp != version: + problems.append(f"{rel.as_posix()}: {stamp!r} != {version!r}") + + changelog = (repo / "CHANGELOG.md").read_text(encoding="utf-8") + headings = HEADING_RE.findall(changelog) + released = [h for h in headings if h.lower() != "unreleased"] + if version in released: + pass + elif "Unreleased" in headings and all(_vtuple(version) > _vtuple(h) for h in released): + pass # main between releases: version bumped ahead, notes still under Unreleased + else: + problems.append( + f"CHANGELOG.md: no '## [{version}]' heading and version is not newer than " + f"the latest released heading ({released[0] if released else 'none'})" + ) + return problems + + +def main() -> int: + problems = check() + if problems: + print("version mismatch:", file=sys.stderr) + for p in problems: + print(f" - {p}", file=sys.stderr) + return 1 + version = VERSION_RE.search( + (REPO / "src/codebase_index/__init__.py").read_text(encoding="utf-8") + ).group(1) # type: ignore[union-attr] + print(f"versions consistent: {version} (package, plugin.json, requirements.lock, " + f"{len(STAMPS)} skill stamps, CHANGELOG)") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/gen_assets.py b/scripts/gen_assets.py index 0158659..19aef5f 100644 --- a/scripts/gen_assets.py +++ b/scripts/gen_assets.py @@ -9,7 +9,6 @@ Outputs: assets/mark.png 256x256 -> logo / avatar source assets/social-preview.png 1280x640 -> upload in Settings -> Social preview - assets/demo.png 1200x760 -> README product story """ from __future__ import annotations @@ -247,115 +246,6 @@ def build_social(out: Path) -> None: # --------------------------------------------------------------------------- # # README demo still: 1200 x 760 # --------------------------------------------------------------------------- # -def build_demo(out: Path) -> None: - W, H = 1200, 760 - w, h = W * SS, H * SS - img = gradient_bg(w, h) - add_glow(img, int(w * 0.5), int(h * -0.05), 460 * SS, BLUE, 26) - d = ImageDraw.Draw(img) - - # Header - wm = f_mono_b(34) - d.text((48 * SS, 46 * SS), "codebase", font=wm, fill=FG) - seg = d.textlength("codebase", font=wm) - d.text((48 * SS + seg, 46 * SS), "-index", font=wm, fill=BLUE) - d.text((48 * SS, 94 * SS), "One local map. Three engineering jobs.", - font=f_ui(23), fill=FG2) - - cards = [ - { - "title": "FIND", - "color": BLUE, - "query": '"where is auth implemented?"', - "lines": [ - ("01", "src/auth/AuthService.ts:12", "0.92"), - ("02", "src/routes/auth.ts:20", "0.78"), - ("03", "src/middleware/auth.ts:5", "0.65"), - ], - }, - { - "title": "TRACE", - "color": PURPLE, - "query": '"checkout → database"', - "lines": [ - ("", "CheckoutRoute", "extracted"), - ("→", "CheckoutService", "extracted"), - ("→", "OrderRepository", "inferred"), - ], - }, - { - "title": "PREDICT", - "color": GREEN, - "query": '"what changes with User?"', - "lines": [ - ("1", "direct dependents", "7"), - ("2", "affected tests", "4"), - ("!", "inferred edges", "2"), - ], - }, - ] - - card_w = 352 - card_h = 430 - gap = 24 - x_start = 48 - y0 = 162 - for idx, card in enumerate(cards): - x0 = (x_start + idx * (card_w + gap)) * SS - y = y0 * SS - x1 = x0 + card_w * SS - y1 = y + card_h * SS - d.rounded_rectangle( - [x0, y, x1, y1], radius=18 * SS, fill=PANEL, - outline=BORDER, width=SS + SS // 2, - ) - d.rounded_rectangle( - [x0 + 20 * SS, y + 20 * SS, x0 + 102 * SS, y + 51 * SS], - radius=15 * SS, fill=PANEL_BAR, outline=card["color"], width=SS, - ) - d.text((x0 + 34 * SS, y + 27 * SS), card["title"], - font=f_ui_b(15), fill=card["color"]) - d.text((x0 + 22 * SS, y + 78 * SS), card["query"], - font=f_mono(17), fill=CYAN) - d.line( - [x0 + 22 * SS, y + 122 * SS, x1 - 22 * SS, y + 122 * SS], - fill=BORDER, width=SS, - ) - - row_y = y + 150 * SS - for lead, label, tail in card["lines"]: - d.text((x0 + 24 * SS, row_y), lead, font=f_mono_b(17), - fill=card["color"]) - d.text((x0 + 58 * SS, row_y), label, font=f_mono(16), fill=FG2) - tw = d.textlength(tail, font=f_mono_b(15)) - d.text((x1 - 24 * SS - tw, row_y + 2 * SS), tail, - font=f_mono_b(15), - fill=YELLOW if tail == "inferred" else card["color"]) - row_y += 55 * SS - - d.text((x0 + 24 * SS, y1 - 58 * SS), - ("precise file:line evidence" if idx == 0 else - "auditable dependency chain" if idx == 1 else - "ranked blast radius"), - font=f_ui(15), fill=MUTED) - - # Evidence strip - sy = 626 * SS - d.rounded_rectangle( - [48 * SS, sy, (W - 48) * SS, 702 * SS], - radius=16 * SS, fill=PANEL_BAR, outline=BORDER, width=SS, - ) - d.text((72 * SS, sy + 25 * SS), "EVIDENCE CONTRACT", font=f_ui_b(15), fill=GREEN) - d.text((246 * SS, sy + 23 * SS), - "freshness · confidence · coverage · recommended reads", - font=f_mono(17), fill=FG2) - - d.text((48 * SS, 726 * SS), - "local by default · no telemetry · CLI + Skill + MCP", - font=f_ui(15), fill=MUTED) - - downsave(img, W, H, out) - def main() -> None: root = Path(__file__).resolve().parent.parent @@ -363,7 +253,6 @@ def main() -> None: print("Generating assets:") build_mark(assets / "mark.png") build_social(assets / "social-preview.png") - build_demo(assets / "demo.png") print("Done.") diff --git a/scripts/gen_benchmark_chart.py b/scripts/gen_benchmark_chart.py new file mode 100644 index 0000000..a81a585 --- /dev/null +++ b/scripts/gen_benchmark_chart.py @@ -0,0 +1,119 @@ +#!/usr/bin/env python3 +"""Render the logged public-baseline benchmark as an SVG bar chart. + + python scripts/gen_benchmark_chart.py \ + tests/eval/results/2026-09-04-public-baselines.json --out assets/benchmark.svg + +Reads the raw JSON that `tests/eval/run_baselines.py --out` wrote, so the chart +can never drift from the logged run. Stdlib only. +""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path +from xml.sax.saxutils import escape + +FONT = "-apple-system, BlinkMacSystemFont, 'Segoe UI', Helvetica, Arial, sans-serif" +INK = "#1f2328" +MUTED = "#656d76" +GRID = "#d0d7de" +INDEX = "#0969da" # index +RG = "#8c959f" # grep baseline +CARD = "#ffffff" +BORDER = "#d0d7de" + + +def bar_panel(x: int, y: int, w: int, h: int, title: str, rows: list[tuple[str, float, float]], + *, vmax: float, fmt: str, unit: str = "") -> str: + """Horizontal grouped bars: for each row label, an index bar and an rg bar.""" + out = [f'{escape(title)}'] + label_w = 70 + plot_x = x + label_w + plot_w = w - label_w - 70 + row_h = 40 + top = y + 14 + # grid lines + for i in range(0, 5): + gx = plot_x + plot_w * i / 4 + out.append(f'') + out.append(f'{(vmax * i / 4):{fmt}}{unit}') + for i, (label, a, b) in enumerate(rows): + ry = top + i * row_h + out.append(f'{escape(label)}') + for j, (val, colour) in enumerate(((a, INDEX), (b, RG))): + bw = plot_w * min(val, vmax) / vmax + by = ry + 6 + j * 15 + out.append(f'') + out.append(f'{val:{fmt}}{unit}') + return "\n".join(out) + + +def render(report: dict) -> str: + pooled = report["pooled"]["summary"] + n = report["pooled"]["n_queries"] + corpora = report["corpora"] + sig = report["pooled"]["index_vs_rg"] + + W, H = 960, 470 + parts = [ + f'', + 'codebase-index vs rg+window on public repositories', + f'', + f'Index vs disciplined grep on public repositories', + f'' + f'{escape(", ".join(c["corpus"] for c in corpora))} at pinned commits · {n} git-derived queries · ' + f'codebase-index {escape(report["version"])} · {escape(report["date"])} · same tokenizer on both sides', + # legend + f'codebase-index', + f'rg + 80-line windows', + ] + + # Panel 1: hit@3 per corpus + pooled + rows = [(c["corpus"], c["summary"]["index"]["hit@3"], c["summary"]["rg+window"]["hit@3"]) for c in corpora] + rows.append(("pooled", pooled["index"]["hit@3"], pooled["rg+window"]["hit@3"])) + parts.append(bar_panel(24, 90, 440, 200, "hit@3 — answer file among the top 3", rows, vmax=1.0, fmt=".2f")) + + # Panel 2: MRR per corpus + pooled + rows = [(c["corpus"], c["summary"]["index"]["MRR"], c["summary"]["rg+window"]["MRR"]) for c in corpora] + rows.append(("pooled", pooled["index"]["MRR"], pooled["rg+window"]["MRR"])) + parts.append(bar_panel(500, 90, 440, 200, "MRR — how high the first correct file ranks", rows, vmax=1.0, fmt=".2f")) + + # Panel 3: tokens (packet/listing + top-3 reads), pooled + rows = [ + ("packet", pooled["index"]["packet_tokens_mean"], pooled["rg+window"]["packet_tokens_mean"]), + ("+ reads", pooled["index"]["tokens_mean"], pooled["rg+window"]["tokens_mean"]), + ] + vmax = max(r[1] for r in rows + [("", 0, 0)]) * 1.15 + vmax = max(vmax, max(r[2] for r in rows) * 1.15) + parts.append(bar_panel(24, 320, 440, 110, "context tokens per query (pooled mean)", rows, vmax=round(vmax, -2), fmt=",.0f")) + + # Significance box + y = 320 + parts.append(f'Paired significance, index − rg (pooled)') + for i, (k, s) in enumerate(sig.items()): + fmt = ",.0f" if k == "tokens" else "+.3f" + p_text = "< 0.001" if s["p"] < 0.001 else f"= {s['p']:.3f}" + line = f'{k}: {s["delta"]:{fmt}} 95% CI [{s["ci_lo"]:{fmt}}, {s["ci_hi"]:{fmt}}] p {p_text}' + parts.append(f'{escape(line)}') + parts.append(f'Ground truth: commit subject → files that commit changed. Raw run: tests/eval/results/.') + parts.append(f'Latency is not charted: in-process index vs a separate ripgrep binary is not a fair pair.') + parts.append("") + return "\n".join(parts) + "\n" + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("results", type=Path) + ap.add_argument("--out", type=Path, required=True) + args = ap.parse_args(argv) + report = json.loads(args.results.read_text(encoding="utf-8")) + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(render(report), encoding="utf-8") + print(f"wrote {args.out}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/gen_terminal_svg.py b/scripts/gen_terminal_svg.py new file mode 100644 index 0000000..fc2bace --- /dev/null +++ b/scripts/gen_terminal_svg.py @@ -0,0 +1,149 @@ +#!/usr/bin/env python3 +"""Render a captured terminal transcript as a static SVG "terminal card". + + bash examples/demo/run_demo.sh .tmp-demo/flask > demo.txt 2>&1 + python scripts/gen_terminal_svg.py demo.txt --out assets/demo-terminal.svg + +Why an SVG and not a screenshot: the card is regenerated from a real transcript +(see docs/DEMO.md), diffable in git, crisp at any width, and never shows text the +tool did not actually print. Stdlib only. +""" + +from __future__ import annotations + +import argparse +import re +import sys +from pathlib import Path +from xml.sax.saxutils import escape + +ANSI_RE = re.compile(r"\x1b\[[0-9;]*m") + +# GitHub-dark inspired palette; the card carries its own background so it reads +# identically on light and dark README themes. +BG = "#0d1117" +BAR = "#161b22" +BORDER = "#30363d" +FG = "#e6edf3" +MUTED = "#8b949e" +BLUE = "#58a6ff" +GREEN = "#3fb950" +PURPLE = "#bc8cff" +AMBER = "#d29922" + +CHAR_W = 8.4 # px per monospace character at 14px +LINE_H = 20 +PAD = 18 +FONT = "ui-monospace, SFMono-Regular, Menlo, Consolas, 'Liberation Mono', monospace" + + +def load_transcript(path: Path, *, stop_at: str | None, max_lines: int) -> list[str]: + lines = [] + for raw in path.read_text(encoding="utf-8", errors="replace").splitlines(): + line = ANSI_RE.sub("", raw).rstrip() + if stop_at and line.startswith("$ ") and stop_at in line: + break + if line.startswith("```"): + continue # fence markers carry no information in a terminal card + lines.append(line) + # Trim leading/trailing blank lines and collapse runs of blanks. + out: list[str] = [] + for line in lines: + if line == "" and (not out or out[-1] == ""): + continue + out.append(line) + while out and out[-1] == "": + out.pop() + return out[:max_lines] + + +def _tspan(text: str, fill: str, *, weight: str | None = None) -> str: + attrs = f' fill="{fill}"' + if weight: + attrs += f' font-weight="{weight}"' + return f"{escape(text)}" + + +def render_line(line: str) -> str: + """Colour a line the way a reader would scan it: commands, headers, tables, confidence.""" + if line.startswith("$ "): + return _tspan("$ ", GREEN) + _tspan(line[2:], BLUE, weight="600") + if line.startswith("|"): + cells = line.split("|") + parts = [] + for i, cell in enumerate(cells): + if i: + parts.append(_tspan("|", BORDER)) + stripped = cell.strip() + if re.fullmatch(r"-+:?|:?-+", stripped): + parts.append(_tspan(cell, BORDER)) + elif stripped in ("extracted", "exact"): + parts.append(_tspan(cell, GREEN)) + elif "ambiguous" in stripped or stripped == "inferred": + parts.append(_tspan(cell, AMBER)) + elif stripped.startswith("`") and stripped.endswith("`"): + parts.append(_tspan(cell.replace("`", ""), FG)) + else: + parts.append(_tspan(cell, FG if i in (0, 1, 2, 3) else MUTED)) + return "".join(parts) + if line.startswith("**"): + # "**Query:** text · **Confidence:** medium" + out = [] + for chunk in re.split(r"(\*\*[^*]+\*\*)", line): + if chunk.startswith("**"): + out.append(_tspan(chunk.strip("*"), PURPLE, weight="600")) + else: + out.append(_tspan(chunk, FG)) + return "".join(out) + if line.startswith("- `") or line.startswith(" →") or line.startswith("`"): + return _tspan(line.replace("`", ""), FG) + if line.startswith("```"): + return _tspan("", MUTED) + if line.startswith("Indexed") or line.startswith(" parse"): + return _tspan(line, MUTED) + return _tspan(line, FG) + + +def render_svg(lines: list[str], *, title: str, cols: int) -> str: + width = int(PAD * 2 + cols * CHAR_W) + height = int(PAD * 2 + 36 + len(lines) * LINE_H) + body = [] + y = PAD + 36 + LINE_H - 6 + for line in lines: + body.append( + f'{render_line(line[:cols])}' + ) + y += LINE_H + return f""" + {escape(title)} + + + + {escape(title)} + {chr(10).join(body)} + +""" + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("transcript", type=Path) + ap.add_argument("--out", type=Path, required=True) + ap.add_argument("--title", default="codebase-index · Flask @ d318b683 · examples/demo/run_demo.sh") + ap.add_argument("--cols", type=int, default=100) + ap.add_argument("--max-lines", type=int, default=60) + ap.add_argument("--stop-at", default="--json", help="drop everything from the first command containing this") + args = ap.parse_args(argv) + + lines = load_transcript(args.transcript, stop_at=args.stop_at, max_lines=args.max_lines) + if not lines: + print("empty transcript", file=sys.stderr) + return 1 + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(render_svg(lines, title=args.title, cols=args.cols), encoding="utf-8") + print(f"wrote {args.out} ({len(lines)} lines)") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/release_notes.py b/scripts/release_notes.py new file mode 100644 index 0000000..70aaccc --- /dev/null +++ b/scripts/release_notes.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +"""Extract one version's section from CHANGELOG.md as GitHub release notes. + + python scripts/release_notes.py # current __version__, to stdout + python scripts/release_notes.py 1.9.0 # explicit version + python scripts/release_notes.py --out notes.md # write instead of print + +Exits non-zero when the section is missing or empty, so a tag with an unwritten +changelog fails the release job instead of shipping an empty release page. +Stdlib only; used by .github/workflows/release.yml. +""" + +from __future__ import annotations + +import argparse +import re +import sys +from pathlib import Path + +REPO = Path(__file__).resolve().parents[1] +REPO_URL = "https://github.com/denfry/codebase-index" + +VERSION_RE = re.compile(r'^__version__ = "([^"]+)"$', re.M) +HEADING_RE = re.compile(r"^## \[(?P[^\]]+)\](?:\s*-\s*(?P\S+))?\s*$", re.M) +LINK_DEF_RE = re.compile(r"^\[(?P[^\]]+)\]:\s*(?P\S+)\s*$", re.M) + + +def package_version(repo: Path = REPO) -> str: + text = (repo / "src/codebase_index/__init__.py").read_text(encoding="utf-8") + match = VERSION_RE.search(text) + if not match: + raise SystemExit("could not find __version__ in src/codebase_index/__init__.py") + return match.group(1) + + +def extract_section(changelog: str, version: str) -> tuple[str, str | None, str | None]: + """Return (body, date, previous_version) for `version`. + + `body` is the text between this version's heading and the next `## ` heading, + with trailing link definitions removed. Raises KeyError when the heading is + absent. + """ + headings = list(HEADING_RE.finditer(changelog)) + for idx, match in enumerate(headings): + if match.group("version") != version: + continue + end = headings[idx + 1].start() if idx + 1 < len(headings) else len(changelog) + body = changelog[match.end():end] + # Link definitions live at the bottom of the file; they are not notes. + body = LINK_DEF_RE.sub("", body).strip() + previous = headings[idx + 1].group("version") if idx + 1 < len(headings) else None + return body, match.group("date"), previous + raise KeyError(version) + + +def compare_url(changelog: str, version: str, previous: str | None) -> str | None: + for match in LINK_DEF_RE.finditer(changelog): + if match.group("version") == version: + return match.group("url") + if previous and previous.lower() != "unreleased": + return f"{REPO_URL}/compare/v{previous}...v{version}" + return None + + +def render(changelog: str, version: str) -> str: + body, _date, previous = extract_section(changelog, version) + if not body: + raise SystemExit(f"CHANGELOG.md section for {version} is empty") + footer = [ + "", + "---", + "", + "**Install**", + "", + "```bash", + f"pip install codebase-index=={version}", + f"pipx install codebase-index=={version}", + "```", + ] + url = compare_url(changelog, version, previous) + if url: + footer += ["", f"Full diff: {url}"] + return body + "\n" + "\n".join(footer) + "\n" + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("version", nargs="?", help="default: package __version__") + ap.add_argument("--changelog", default=str(REPO / "CHANGELOG.md")) + ap.add_argument("--out", help="write notes here instead of stdout") + args = ap.parse_args(argv) + + version = args.version or package_version() + changelog = Path(args.changelog).read_text(encoding="utf-8") + try: + notes = render(changelog, version) + except KeyError: + print(f"CHANGELOG.md has no '## [{version}]' section", file=sys.stderr) + return 1 + + if args.out: + Path(args.out).write_text(notes, encoding="utf-8") + print(f"wrote release notes for {version} to {args.out}") + else: + # The changelog uses arrows and dashes; a cp1251 console must not crash. + sys.stdout.buffer.write(notes.encode("utf-8")) + sys.stdout.flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/skill/references/response-contract.md b/skill/references/response-contract.md index 059d87a..798f152 100644 --- a/skill/references/response-contract.md +++ b/skill/references/response-contract.md @@ -18,7 +18,10 @@ Each result can contain: - `elided_lines` `recommended_reads` is the read plan. Start with its first one to three entries -and use exact line ranges. +and use exact line ranges. An entry with `truncated: true` was capped at the +definition head (`max_read_lines`, default 120); `line_end_full` gives the real +extent. Read the capped range first and continue only when the head is not +enough. `pagination.has_more` and `pagination.next_offset` indicate additional results. Prefer a more specific command or a larger token budget before paging. diff --git a/skills/codebase-index/references/response-contract.md b/skills/codebase-index/references/response-contract.md index 059d87a..798f152 100644 --- a/skills/codebase-index/references/response-contract.md +++ b/skills/codebase-index/references/response-contract.md @@ -18,7 +18,10 @@ Each result can contain: - `elided_lines` `recommended_reads` is the read plan. Start with its first one to three entries -and use exact line ranges. +and use exact line ranges. An entry with `truncated: true` was capped at the +definition head (`max_read_lines`, default 120); `line_end_full` gives the real +extent. Read the capped range first and continue only when the head is not +enough. `pagination.has_more` and `pagination.next_offset` indicate additional results. Prefer a more specific command or a larger token budget before paging. diff --git a/src/codebase_index/cli.py b/src/codebase_index/cli.py index ffd216f..82c8f79 100644 --- a/src/codebase_index/cli.py +++ b/src/codebase_index/cli.py @@ -2,8 +2,7 @@ Commands map 1:1 to docs/ARCHITECTURE.md §5 (CLI contract) and delegate to the `indexer`, `retrieval`, and `storage` layers through `service.py` — the same -layer the MCP server uses, so the two surfaces cannot drift. Only `clean` is -still a stub. +layer the MCP server uses, so the two surfaces cannot drift. Conventions: * every command accepts global options via the Typer context: --root, --json, --quiet @@ -742,6 +741,11 @@ def doctor( def mcp( ctx: typer.Context, transport: str = typer.Option("stdio", "--transport", help="Transport: stdio (default)."), + root: Optional[Path] = typer.Option( + None, "--root", + help="Repository root (same as the global --root; accepted here too so client " + "configs can pass `mcp --root `).", + ), ) -> None: """Start the MCP server — exposes codebase-index tools to any MCP client (e.g. Claude Code). @@ -752,8 +756,7 @@ def mcp( "mcpServers": { "codebase-index": { "command": "codebase-index", - "args": ["mcp"], - "cwd": "/path/to/your/project" + "args": ["mcp", "--root", "/path/to/your/project"] } } } @@ -768,12 +771,13 @@ def mcp( ) raise typer.Exit(code=1) - root_opt = ctx.obj.get("root") if ctx.obj else None + # Every doc and client template writes `codebase-index mcp --root `; the + # global `--root` lives on the callback, so the subcommand accepts it as well. + root_opt = root or (ctx.obj.get("root") if ctx.obj else None) if root_opt: - import os - os.environ.setdefault("CBX_ROOT", str(root_opt)) + os.environ["CBX_ROOT"] = str(Path(root_opt).resolve()) - _mcp.run(transport=transport) # type: ignore[arg-type] + _mcp.run(transport=transport) # type: ignore[arg-type,call-overload] @app.command() diff --git a/src/codebase_index/config.py b/src/codebase_index/config.py index 217cad7..eac4cb8 100644 --- a/src/codebase_index/config.py +++ b/src/codebase_index/config.py @@ -28,6 +28,12 @@ class RetrievalConfig(BaseModel): limit: int = 10 compact_snippets: bool = True compact_min_reduction: float = 0.25 + # Longest line span a single `recommended_reads` entry may ask the agent to + # open. A symbol-aligned chunk can be a whole 1,500-line class; the read plan + # points at its head instead and marks the entry `truncated` with the full + # `line_end_full`, so the agent pays for a bounded window and decides itself + # whether the rest is worth it. 0 disables the cap. + max_read_lines: int = 120 class EmbeddingsConfig(BaseModel): diff --git a/src/codebase_index/mcp/server.py b/src/codebase_index/mcp/server.py index 6540297..fae85f5 100644 --- a/src/codebase_index/mcp/server.py +++ b/src/codebase_index/mcp/server.py @@ -29,12 +29,18 @@ if TYPE_CHECKING: from ..config import Config +# mcp 2.x renamed FastMCP to MCPServer and moved it; mcp 1.x only has FastMCP. +# Both expose the same `tool(structured_output=...)` decorator and `run(transport=)` +# entry point this module uses, so one server definition serves both lines. try: - from mcp.server.fastmcp import FastMCP # type: ignore[attr-defined] -except ImportError as exc: # pragma: no cover - raise ImportError( - "MCP server needs the optional extra: pip install codebase-index[mcp]" - ) from exc + from mcp.server.mcpserver import MCPServer as FastMCP # mcp >= 2.0 +except ImportError: # pragma: no cover - exercised on mcp 1.x only + try: + from mcp.server.fastmcp import FastMCP # type: ignore[attr-defined,no-redef] + except ImportError as exc: + raise ImportError( + "MCP server needs the optional extra: pip install codebase-index[mcp]" + ) from exc mcp = FastMCP( "codebase-index", diff --git a/src/codebase_index/retrieval/pipeline.py b/src/codebase_index/retrieval/pipeline.py index 020c956..56c9e70 100644 --- a/src/codebase_index/retrieval/pipeline.py +++ b/src/codebase_index/retrieval/pipeline.py @@ -144,6 +144,26 @@ def _fallback_suggestions(query, ranked) -> dict: return {"ripgrep": rg} +def _bounded_read(entry: dict, max_lines: int) -> dict: + """Cap one read-plan entry to `max_lines`, keeping the original span visible. + + Symbol-aligned chunks can span an entire class. Handing the agent + ``line_start..line_end`` verbatim then bills it for the whole body when the + definition head is usually what it needs first; the capped entry points at + the head and records the full extent so the agent can read on deliberately. + Additive fields only (`truncated`, `line_end_full`): schema unchanged. + """ + span = entry["line_end"] - entry["line_start"] + 1 + if max_lines <= 0 or span <= max_lines: + return entry + return { + **entry, + "line_end": entry["line_start"] + max_lines - 1, + "line_end_full": entry["line_end"], + "truncated": True, + } + + def search( conn: sqlite3.Connection, query: str, @@ -159,6 +179,7 @@ def search( offset: int = 0, compact: bool = True, compact_min_reduction: float = 0.25, + max_read_lines: int = 120, ) -> dict: tuning = tuning or DEFAULT_TUNING plan = detect_intent(query) @@ -215,7 +236,7 @@ def search( paginated = all_results[offset:offset + limit] paginated_keys = {(r["path"], r["line_start"], r["line_end"]) for r in paginated} recommended = [ - r for r in all_recommended + _bounded_read(r, max_read_lines) for r in all_recommended if (r["path"], r["line_start"], r["line_end"]) in paginated_keys ] has_more = len(all_results) > offset + limit diff --git a/src/codebase_index/service.py b/src/codebase_index/service.py index 1857221..7dcfcb5 100644 --- a/src/codebase_index/service.py +++ b/src/codebase_index/service.py @@ -103,6 +103,7 @@ def search_payload( config=cfg, compact=compact, compact_min_reduction=cfg.retrieval.compact_min_reduction, + max_read_lines=cfg.retrieval.max_read_lines, ) diff --git a/src/codebase_index/skill_template/references/response-contract.md b/src/codebase_index/skill_template/references/response-contract.md index 059d87a..798f152 100644 --- a/src/codebase_index/skill_template/references/response-contract.md +++ b/src/codebase_index/skill_template/references/response-contract.md @@ -18,7 +18,10 @@ Each result can contain: - `elided_lines` `recommended_reads` is the read plan. Start with its first one to three entries -and use exact line ranges. +and use exact line ranges. An entry with `truncated: true` was capped at the +definition head (`max_read_lines`, default 120); `line_end_full` gives the real +extent. Read the capped range first and continue only when the head is not +enough. `pagination.has_more` and `pagination.next_offset` indicate additional results. Prefer a more specific command or a larger token budget before paging. diff --git a/tests/benchmark_honest_RESULTS.md b/tests/benchmark_honest_RESULTS.md index 4bba730..46e506f 100644 --- a/tests/benchmark_honest_RESULTS.md +++ b/tests/benchmark_honest_RESULTS.md @@ -1,5 +1,11 @@ # Honest benchmark — codebase-index vs no-skill agent (NewTowny) +> **Historical.** The repository under test is private, so nobody else can rerun +> this, and its token accounting (index charged for signature snippets, grep charged +> for 80-line windows) is not symmetric. The reproducible successor is +> `tests/eval/run_baselines.py` on public repositories; see `docs/BENCHMARKS.md`. +> Do not quote the numbers below as product claims. + Script: `tests/benchmark_honest.py` · Raw run: `tests/benchmark_honest_newtowny.txt` Repo under test: `C:/Users/denfry/IdeaProjects/NewTowny` (303 Java files, ~55k LOC; 574 text files indexed) Token counter: `tiktoken/cl100k_base` (real tokens, identical estimator both sides) diff --git a/tests/benchmark_public_RESULTS.md b/tests/benchmark_public_RESULTS.md index b2a28b6..407abce 100644 --- a/tests/benchmark_public_RESULTS.md +++ b/tests/benchmark_public_RESULTS.md @@ -44,9 +44,10 @@ python tests/benchmark_public.py --workdir .tmp-public-benchmark only; the framework-aware edges that would lift it are designed in `docs/superpowers/specs/2026-06-14-typed-framework-edges-design.md` (M13) and not yet implemented. -- Token economy (~3.4× vs an rg+window baseline) is a synthetic-fixture figure; - the real-repo figure (~13× on a 55k LOC Java repo) lives in - `tests/benchmark_honest_RESULTS.md`. +- Token economy (~3.4× vs an rg+window baseline) is a synthetic-fixture figure and + charges the index only for its snippets. Under symmetric accounting on public + repositories (`tests/eval/results/`), the index costs about the same context as a + disciplined grep agent; the quality gap, not a token multiplier, is the headline. ## Raw output diff --git a/tests/eval/README.md b/tests/eval/README.md index 58903a1..878d28c 100644 --- a/tests/eval/README.md +++ b/tests/eval/README.md @@ -88,6 +88,19 @@ best rank. - `dup%` — fraction of returned results that near-duplicate an earlier result - `p50/p95/p99` latency, in-process, excluding interpreter start-up +## Baselines on public repositories + +`run_eval.py` compares the ranker with itself. `run_baselines.py` compares it with +*not having an index*: a disciplined `rg` agent and a repo-map-style context blob, on +Flask, Gson and Fastify at pinned commits, with symmetric token accounting and the +same significance tests. The logged run lives in `results/`; the read models are in +`baselines.py`. + +```bash +python tests/eval/run_baselines.py --clone --workdir .tmp-baselines --out tests/eval/results/-public-baselines +python tests/eval/run_baselines.py --repo ../your-repo # any local git repository +``` + ## Files | File | Role | @@ -96,4 +109,7 @@ best rank. | `harness.py` | Index build, query execution, aggregation, pooling, tables | | `metrics.py` | IR metrics + paired bootstrap / permutation tests | | `gen_queries.py` | Ground-truth generator from git history | +| `baselines.py` | rg+window and repo-map-style read models, symmetric token accounting | +| `run_baselines.py` | Index vs baselines on public repositories, with significance | +| `results/` | Logged runs (raw JSON + Markdown) | | `queries/` | Checked-in query sets | diff --git a/tests/eval/baselines.py b/tests/eval/baselines.py new file mode 100644 index 0000000..27dd391 --- /dev/null +++ b/tests/eval/baselines.py @@ -0,0 +1,443 @@ +"""Baselines the index is compared against, with symmetric token accounting. + +Two baselines model what an agent does *without* an index: + +* ``rg_window`` — a disciplined grep agent. Drop stop-words from the question, + search the salient terms with ripgrep, rank files by match density, then read + an 80-line window around the densest hit of each of the top-K files. +* ``repo_map`` — a repo-map-style context blob in the spirit of Aider's map: + file paths plus definition signatures, ranked by graph degree (and, in the + query-aware variant, by identifier overlap with the question), packed under a + fixed token budget and handed to the agent up front. + +Both are approximations of a *style* of context construction, not a +re-implementation of any specific product. Neither uses the ranking pipeline; +``repo_map`` reads the symbol table from the index database only because +Tree-sitter signatures are what a repo map is built from. + +Symmetry rules +-------------- +Every side is charged with the same tokenizer for the text that actually enters +the agent's context: + +* index: the JSON payload the agent receives, plus the top-K + ``recommended_reads`` line ranges (a "follow-through read"); +* rg_window: the ripgrep listing (capped at ``LISTING_CAP`` lines — agents + truncate long tool output) plus the top-K windows it reads; +* repo_map: the map itself (the budget it was packed under). + +Latency is reported for context only. The index runs in-process, ripgrep is a +separate binary; neither is a fair wall-clock claim against the other. +""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import sqlite3 +import subprocess +import time +from collections.abc import Iterable, Sequence +from dataclasses import dataclass, field +from pathlib import Path + +from codebase_index.retrieval.pipeline import search +from codebase_index.retrieval.tuning import RetrievalTuning + +from . import metrics +from .harness import EvalQuery + +# --- shared configuration --------------------------------------------------- +WINDOW = 80 +"""Lines read around a grep hit; matches the index's code window.""" +TOP_K = 3 +"""Files an agent opens after a search: it 'starts with ranks 1-3'.""" +LISTING_CAP = 50 +"""Lines of `rg` output an agent actually looks at before deciding what to open.""" +REPO_MAP_BUDGETS = (2000, 8000) +"""Token budgets for the repo-map baseline: a small chat-context map and a large one.""" + +# --- token counting (identical on every side) -------------------------------- +try: # pragma: no cover - exercised only when tiktoken is installed + import tiktoken + + _ENC = tiktoken.get_encoding("cl100k_base") + + def count_tokens(text: str) -> int: + return len(_ENC.encode(text, disallowed_special=())) + + TOKENIZER = "tiktoken/cl100k_base" +except Exception: # pragma: no cover + def count_tokens(text: str) -> int: + return max(0, len(text) // 4) + + TOKENIZER = "chars//4 (tiktoken not installed)" + + +# --- salient-term extraction (what an agent greps for) ------------------------ +STOPWORDS = frozenset( + """ + the a an is are was were be been being how does do did what where which who whom + when why to of in on for and or with from down up it this that these those i would + should could can will show me all work works happen happens other into during if + rename use used uses get got set via across between add adds added fix fixes fixed + remove removes removed update updates updated make makes made support supports + allow allows allowed change changes changed improve improves improved when not + """.split() +) +_TERM_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]+") + + +def salient_terms(query: str, *, max_terms: int = 6) -> list[str]: + """Terms an agent would grep for: identifiers kept whole, stop-words dropped.""" + out: list[str] = [] + seen: set[str] = set() + for term in _TERM_RE.findall(query): + key = term.lower() + if key in STOPWORDS or len(term) < 3 or key in seen: + continue + seen.add(key) + out.append(term) + # Longest terms first: an agent greps the most specific identifier before + # generic words, and ripgrep's OR of all of them is order-independent anyway. + out.sort(key=len, reverse=True) + return out[:max_terms] + + +# --- repository file access ---------------------------------------------------- +TEXT_EXTS = frozenset( + """ + .py .pyi .js .mjs .cjs .jsx .ts .tsx .java .kt .kts .go .rs .rb .php .cs .c .h .cc + .cpp .hpp .swift .scala .lua .sql .sh .ps1 .md .rst .txt .yml .yaml .toml .json + .xml .gradle .cfg .ini + """.split() +) +IGNORE_PARTS = frozenset( + """ + .git node_modules __pycache__ .venv venv dist build target .idea .gradle out + .claude .codex .opencode .tox .mypy_cache .pytest_cache .ruff_cache + """.split() +) + + +def _norm(path: str) -> str: + return path.replace("\\", "/") + + +@dataclass +class Corpus: + root: Path + _lines: dict[str, list[str]] = field(default_factory=dict) + + def lines(self, rel: str) -> list[str]: + rel = _norm(rel) + cached = self._lines.get(rel) + if cached is None: + try: + cached = (self.root / rel).read_text( + encoding="utf-8", errors="ignore" + ).splitlines() + except OSError: + cached = [] + self._lines[rel] = cached + return cached + + def read_range(self, rel: str, start: int, end: int) -> str: + lines = self.lines(rel) + if not lines: + return "" + start = max(1, start) + end = min(len(lines), end) + if end < start: + return "" + return "\n".join(lines[start - 1 : end]) + + def is_text(self, rel: str) -> bool: + parts = _norm(rel).split("/") + if any(p in IGNORE_PARTS for p in parts): + return False + return Path(rel).suffix.lower() in TEXT_EXTS + + +def _merge(ranges: Iterable[tuple[int, int]]) -> list[tuple[int, int]]: + merged: list[list[int]] = [] + for s, e in sorted(ranges): + if merged and s <= merged[-1][1] + 1: + merged[-1][1] = max(merged[-1][1], e) + else: + merged.append([s, e]) + return [(s, e) for s, e in merged] + + +def tokens_for_reads(corpus: Corpus, reads: dict[str, list[tuple[int, int]]]) -> int: + """Charge every (file, line range) once, ranges merged per file.""" + total = 0 + for rel, ranges in reads.items(): + for s, e in _merge(ranges): + total += count_tokens(corpus.read_range(rel, s, e)) + return total + + +# --- outcome ------------------------------------------------------------------- +@dataclass +class BaselineOutcome: + ranked_files: list[str] + tokens: int + """Everything that entered context: listing/packet + top-K reads.""" + packet_tokens: int + """The listing / packet / map alone, before any follow-through read.""" + latency_ms: float + extra: dict = field(default_factory=dict) + + +# --- index side ---------------------------------------------------------------- +def run_index( + conn: sqlite3.Connection, + corpus: Corpus, + q: EvalQuery, + *, + tuning: RetrievalTuning | None = None, + limit: int = 10, + token_budget: int = 1500, + max_read_lines: int = 120, +) -> BaselineOutcome: + start = time.perf_counter() + payload = search( + conn, + q.query, + mode="hybrid", + limit=limit, + token_budget=token_budget, + no_fallback=True, + tuning=tuning or RetrievalTuning(), + max_read_lines=max_read_lines, + ) + latency_ms = (time.perf_counter() - start) * 1000.0 + + ranked: list[str] = [] + for r in payload.get("results", []): + p = _norm(r["path"]) + if p not in ranked: + ranked.append(p) + + # The packet is what an MCP / --json agent literally receives. + packet_tokens = count_tokens(json.dumps(payload, ensure_ascii=False)) + + # Follow-through: the agent opens the top-K recommended ranges. + reads: dict[str, list[tuple[int, int]]] = {} + for rr in payload.get("recommended_reads", [])[:TOP_K]: + s = int(rr.get("line_start") or 1) + e = int(rr.get("line_end") or s) + if e <= s: + e = s + WINDOW - 1 + reads.setdefault(_norm(rr["path"]), []).append((s, e)) + read_tokens = tokens_for_reads(corpus, reads) + + return BaselineOutcome( + ranked_files=ranked, + tokens=packet_tokens + read_tokens, + packet_tokens=packet_tokens, + latency_ms=latency_ms, + extra={"confidence": payload.get("confidence"), "read_tokens": read_tokens}, + ) + + +# --- rg + window --------------------------------------------------------------- +RG = os.environ.get("CBX_RG") or shutil.which("rg") +"""Path to ripgrep; set CBX_RG when `rg` is a shell alias rather than a binary on PATH.""" + + +def rg_version() -> str | None: + """`ripgrep X.Y.Z` for the report, or None when the Python fallback is in use.""" + if not RG: + return None + try: + out = subprocess.run([RG, "--version"], capture_output=True, text=True, timeout=10).stdout + except (OSError, subprocess.SubprocessError): + return None + return out.splitlines()[0].strip() if out else None + + +def _rg_hits(corpus: Corpus, terms: Sequence[str]) -> tuple[dict[str, list[int]], list[str], float]: + """Return ({file: [line numbers]}, listing lines, elapsed ms) using ripgrep. + + Falls back to a pure-Python scan when ripgrep is not installed; the fallback + is labelled in the report because its latency is not comparable. + """ + if not terms: + return {}, [], 0.0 + start = time.perf_counter() + hits: dict[str, list[int]] = {} + listing: list[str] = [] + if RG: + cmd = [RG, "-n", "-i", "--no-heading", "--no-messages", "--color", "never"] + for t in terms: + cmd += ["-e", re.escape(t)] + for part in sorted(IGNORE_PARTS): + cmd += ["--glob", f"!{part}"] + proc = subprocess.run( + cmd, cwd=corpus.root, capture_output=True, text=True, + encoding="utf-8", errors="replace", + ) + for line in proc.stdout.splitlines(): + path, _, rest = line.partition(":") + lineno, _, _text = rest.partition(":") + rel = _norm(path) + if not corpus.is_text(rel) or not lineno.isdigit(): + continue + hits.setdefault(rel, []).append(int(lineno)) + listing.append(line) + else: # pragma: no cover - only without ripgrep + pats = [re.compile(re.escape(t), re.I) for t in terms] + for p in corpus.root.rglob("*"): + if not p.is_file(): + continue + rel = _norm(str(p.relative_to(corpus.root))) + if not corpus.is_text(rel): + continue + for i, text in enumerate(corpus.lines(rel), 1): + if any(pat.search(text) for pat in pats): + hits.setdefault(rel, []).append(i) + listing.append(f"{rel}:{i}:{text}") + return hits, listing, (time.perf_counter() - start) * 1000.0 + + +def run_rg_window(corpus: Corpus, q: EvalQuery) -> BaselineOutcome: + terms = salient_terms(q.query) + hits, listing, elapsed = _rg_hits(corpus, terms) + + # Rank by match density: the heuristic a grep agent applies when scanning + # `rg` output ("this file lights up the most"). + ranked = [rel for rel, _ in sorted(hits.items(), key=lambda kv: (-len(kv[1]), kv[0]))] + + reads: dict[str, list[tuple[int, int]]] = {} + for rel in ranked[:TOP_K]: + lines = hits[rel] + center = lines[len(lines) // 2] + reads.setdefault(rel, []).append((center - WINDOW // 2, center + WINDOW // 2)) + + listing_text = "\n".join(listing[:LISTING_CAP]) + packet_tokens = count_tokens(listing_text) + read_tokens = tokens_for_reads(corpus, reads) + return BaselineOutcome( + ranked_files=ranked, + tokens=packet_tokens + read_tokens, + packet_tokens=packet_tokens, + latency_ms=elapsed, + extra={ + "terms": terms, + "matched_files": len(hits), + "match_lines": sum(len(v) for v in hits.values()), + "read_tokens": read_tokens, + }, + ) + + +# --- repo-map style ------------------------------------------------------------ +@dataclass(frozen=True) +class MapEntry: + path: str + text: str + """Rendered map block: the path plus its definition signatures.""" + tokens: int + degree: int + names: frozenset[str] + + +def build_repo_map_entries(conn: sqlite3.Connection) -> list[MapEntry]: + """One block per file: path + top-level definition signatures, from the symbol table.""" + rows = conn.execute( + """ + SELECT f.path, s.name, s.kind, s.signature, s.in_degree, s.parent_id + FROM files f LEFT JOIN symbols s ON s.file_id = f.id + ORDER BY f.path, s.line_start + """ + ).fetchall() + by_file: dict[str, list[tuple]] = {} + for row in rows: + by_file.setdefault(_norm(row[0]), []).append(row) + + entries: list[MapEntry] = [] + for path, syms in by_file.items(): + lines = [f"{path}:"] + degree = 0 + names: set[str] = set() + for _p, name, kind, signature, in_degree, parent_id in syms: + if name is None: + continue + degree += int(in_degree or 0) + names.add(str(name).lower()) + if parent_id is None: # top-level definitions only, like a repo map + sig = (signature or f"{kind} {name}").strip().splitlines()[0] + lines.append(f" {sig}") + text = "\n".join(lines) + entries.append(MapEntry(path, text, count_tokens(text), degree, frozenset(names))) + return entries + + +def pack_repo_map( + entries: Sequence[MapEntry], + *, + budget: int, + query_terms: Sequence[str] = (), +) -> tuple[list[str], int]: + """Greedy pack of map blocks under `budget` tokens. + + Ordering: query-aware variant first ranks files whose definition names overlap + the question's identifiers, then by graph degree (a stand-in for the PageRank + a real repo map uses); the query-agnostic variant uses degree alone. + """ + terms = {t.lower() for t in query_terms} + + def overlap(e: MapEntry) -> int: + if not terms: + return 0 + return sum(1 for n in e.names if any(t in n or n in t for t in terms)) + + ordered = sorted(entries, key=lambda e: (-overlap(e), -e.degree, e.path)) + chosen: list[str] = [] + used = 0 + for e in ordered: + if used + e.tokens > budget: + continue + chosen.append(e.path) + used += e.tokens + return chosen, used + + +def run_repo_map( + entries: Sequence[MapEntry], q: EvalQuery, *, budget: int, query_aware: bool +) -> BaselineOutcome: + start = time.perf_counter() + terms = salient_terms(q.query) if query_aware else () + chosen, used = pack_repo_map(entries, budget=budget, query_terms=terms) + latency = (time.perf_counter() - start) * 1000.0 + # A repo map is a blob, not a ranking: `ranked_files` is the pack order, so + # hit@K on it says "is the answer even present in what the agent was handed". + return BaselineOutcome( + ranked_files=chosen, + tokens=used, + packet_tokens=used, + latency_ms=latency, + extra={"files_in_map": len(chosen), "budget": budget}, + ) + + +# --- scoring ------------------------------------------------------------------- +METRICS = ("hit@3", "recall@5", "MRR") + + +def score(outcome: BaselineOutcome, q: EvalQuery, *, present_only: bool = False) -> dict[str, float]: + rel = q.expected_files + ranked = outcome.ranked_files + if present_only: + # For a context blob, "present in the map" is the only meaningful hit. + present = 1.0 if any(p in set(ranked) for p in rel) else 0.0 + return {"present": present, "recall": metrics.recall_at_k(ranked, rel, len(ranked) or 1)} + return { + "hit@3": metrics.hit_rate_at_k(ranked, rel, 3), + "recall@5": metrics.recall_at_k(ranked, rel, 5), + "MRR": metrics.reciprocal_rank(ranked, rel), + } diff --git a/tests/eval/gen_queries.py b/tests/eval/gen_queries.py index 1e0d0c8..286f16b 100644 --- a/tests/eval/gen_queries.py +++ b/tests/eval/gen_queries.py @@ -70,7 +70,7 @@ # Files that may be *changed* by a commit but are never a useful retrieval answer. _ANSWER_DENY_RE = re.compile( r"(?:^|/)(?:" - r"CHANGELOG[^/]*|HISTORY[^/]*|NEWS[^/]*|" + r"CHANGELOG[^/]*|CHANGES[^/]*|HISTORY[^/]*|NEWS[^/]*|RELEASE[_-]?NOTES[^/]*|" r"package-lock\.json|yarn\.lock|poetry\.lock|Cargo\.lock|requirements\.lock|" r"go\.sum|Gemfile\.lock|pnpm-lock\.yaml" r")$", @@ -90,9 +90,13 @@ # Benchmark scaffolding: a commit touching it must not become a benchmark query. _SCAFFOLD_RE = re.compile(r"(?:^|/)tests/(?:eval|benchmark_)", re.I) -# Corpus documents that paraphrase commit subjects. `harness` applies these on top -# of its own excludes whenever a git-derived query set is evaluated. -CHANGELOG_EXCLUDES = ("CHANGELOG*", "HISTORY*", "NEWS*", "**/CHANGELOG*") +# Corpus documents that paraphrase commit subjects. `harness.build_corpus_index` +# applies these on top of its own excludes for every corpus, so a changelog can +# never be the top lexical hit for a query that is literally its own line item. +CHANGELOG_EXCLUDES = ( + "CHANGELOG*", "CHANGES*", "HISTORY*", "NEWS*", "RELEASE_NOTES*", "RELEASE-NOTES*", + "**/CHANGELOG*", "**/CHANGES*", "**/HISTORY*", "**/NEWS*", +) _WORD_RE = re.compile(r"[A-Za-z][A-Za-z0-9_-]*") diff --git a/tests/eval/harness.py b/tests/eval/harness.py index 58b9e2c..7802301 100644 --- a/tests/eval/harness.py +++ b/tests/eval/harness.py @@ -33,6 +33,7 @@ from codebase_index.storage.db import Database from . import metrics +from .gen_queries import CHANGELOG_EXCLUDES QUERY_DIR = Path(__file__).parent / "queries" DEFAULT_BUDGET = 1500 @@ -147,13 +148,19 @@ def validate_queries(queries: Iterable[EvalQuery], root: Path) -> list[str]: # scaffolding is excluded from the corpus it grades. CORPUS_EXCLUDES = ("tests/eval/**", "tests/benchmark_*", "tests/fixtures/expected_answers.yml") +# Changelog-style files paraphrase commit subjects, so for a git-derived query set +# they are the answer key in prose. They are never accepted as *answers* +# (`gen_queries._ANSWER_DENY_RE`) and, since 1.9.1, never indexed as *corpus* +# either: a changelog that outranks the implementation it describes is a +# measurement artefact, not a ranking signal. + def build_corpus_index(root: Path, db_path: Path) -> Database: """Build a fresh index for `root` at `db_path` and return the open handle.""" cfg = Config() cfg.root = str(root) cfg.embeddings.enabled = False - cfg.extra_ignore = [*cfg.extra_ignore, *CORPUS_EXCLUDES] + cfg.extra_ignore = [*cfg.extra_ignore, *CORPUS_EXCLUDES, *CHANGELOG_EXCLUDES] db = Database(db_path).open() build_index(cfg, db, root=root) return db diff --git a/tests/eval/results/2026-09-04-public-baselines.json b/tests/eval/results/2026-09-04-public-baselines.json new file mode 100644 index 0000000..50752d5 --- /dev/null +++ b/tests/eval/results/2026-09-04-public-baselines.json @@ -0,0 +1,14501 @@ +{ + "date": "2026-09-04", + "version": "1.9.0", + "tokenizer": "tiktoken/cl100k_base", + "ripgrep": "ripgrep 14.1.1 (rev 4649aa9700)", + "platform": "Windows 11 / Python 3.12.12", + "corpora": [ + { + "corpus": "flask", + "root": "C:\\Users\\dabin\\AppData\\Local\\Temp\\claude\\D--Projects-codebase-index\\7f35eb46-c8c9-49e7-9d02-4c510c6877d0\\scratchpad\\repos\\flask", + "head": "d318b683471101618febed18996405ad26462110", + "n_queries": 150, + "index_build_ms": 3240.377099999023, + "files_indexed": 229, + "symbols_indexed": 1622, + "edges_indexed": 4425, + "summary": { + "index": { + "hit@3": 0.4866666666666667, + "recall@5": 0.5033333333333333, + "MRR": 0.3832539682539683, + "tokens_mean": 3590.76, + "tokens_median": 3576.5, + "packet_tokens_mean": 2358.9133333333334, + "latency_p50_ms": 45.75900000054389 + }, + "index (uncapped reads)": { + "hit@3": 0.4866666666666667, + "recall@5": 0.5033333333333333, + "MRR": 0.3832539682539683, + "tokens_mean": 5723.6, + "tokens_median": 3776.5, + "packet_tokens_mean": 2350.5, + "latency_p50_ms": 44.97279999850434 + }, + "rg+window": { + "hit@3": 0.23333333333333334, + "recall@5": 0.245, + "MRR": 0.22882812025866897, + "tokens_mean": 3078.7733333333335, + "tokens_median": 3187.0, + "packet_tokens_mean": 1008.8533333333334, + "latency_p50_ms": 27.9039000015473 + }, + "repo-map 2k": { + "present": 0.31333333333333335, + "recall": 0.22444444444444442, + "tokens_mean": 1998.0, + "tokens_median": 1998.0, + "packet_tokens_mean": 1998.0, + "latency_p50_ms": 0.11129999984405003 + }, + "repo-map 2k query-aware": { + "present": 0.2733333333333333, + "recall": 0.20277777777777778, + "tokens_mean": 1999.0933333333332, + "tokens_median": 1999.0, + "packet_tokens_mean": 1999.0933333333332, + "latency_p50_ms": 0.8749000007810537 + }, + "repo-map 8k": { + "present": 1.0, + "recall": 1.0, + "tokens_mean": 7728.0, + "tokens_median": 7728.0, + "packet_tokens_mean": 7728.0, + "latency_p50_ms": 0.07610000102431513 + }, + "repo-map 8k query-aware": { + "present": 1.0, + "recall": 1.0, + "tokens_mean": 7728.0, + "tokens_median": 7728.0, + "packet_tokens_mean": 7728.0, + "latency_p50_ms": 0.8187000021280255 + } + }, + "per_query": { + "index": { + "hit@3": [ + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0 + ], + "recall@5": [ + 1.0, + 0.5, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 0.75, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.25, + 1.0, + 0.3333333333333333, + 0.5, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.5, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.3333333333333333, + 0.25, + 1.0, + 0.5, + 1.0, + 0.0, + 0.5, + 0.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.75, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0 + ], + "MRR": [ + 0.5, + 0.25, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.25, + 0.0, + 0.0, + 0.5, + 0.5, + 0.3333333333333333, + 1.0, + 0.16666666666666666, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.2, + 1.0, + 0.25, + 0.0, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.2, + 0.25, + 0.0, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.25, + 0.5, + 0.5, + 0.0, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 0.2, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.0, + 0.5, + 1.0, + 0.14285714285714285, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.5, + 0.14285714285714285, + 0.0, + 0.5, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.5, + 0.25, + 1.0, + 0.5, + 1.0, + 0.25, + 0.3333333333333333, + 0.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.25, + 0.0, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 0.0, + 0.5, + 0.25, + 0.0, + 0.0, + 0.14285714285714285, + 0.0, + 0.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.25, + 1.0, + 0.0, + 1.0, + 0.2, + 0.0, + 0.25, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 0.25, + 0.5, + 0.2, + 0.5, + 1.0, + 0.5, + 0.16666666666666666, + 0.0, + 1.0, + 0.14285714285714285, + 0.25, + 1.0, + 0.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.16666666666666666, + 0.0 + ] + }, + "index (uncapped reads)": { + "hit@3": [ + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0 + ], + "recall@5": [ + 1.0, + 0.5, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 0.75, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.25, + 1.0, + 0.3333333333333333, + 0.5, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.5, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.3333333333333333, + 0.25, + 1.0, + 0.5, + 1.0, + 0.0, + 0.5, + 0.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.75, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0 + ], + "MRR": [ + 0.5, + 0.25, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.25, + 0.0, + 0.0, + 0.5, + 0.5, + 0.3333333333333333, + 1.0, + 0.16666666666666666, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.2, + 1.0, + 0.25, + 0.0, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.2, + 0.25, + 0.0, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.25, + 0.5, + 0.5, + 0.0, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 0.2, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.0, + 0.5, + 1.0, + 0.14285714285714285, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.5, + 0.14285714285714285, + 0.0, + 0.5, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.5, + 0.25, + 1.0, + 0.5, + 1.0, + 0.25, + 0.3333333333333333, + 0.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.25, + 0.0, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 0.0, + 0.5, + 0.25, + 0.0, + 0.0, + 0.14285714285714285, + 0.0, + 0.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.25, + 1.0, + 0.0, + 1.0, + 0.2, + 0.0, + 0.25, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 0.25, + 0.5, + 0.2, + 0.5, + 1.0, + 0.5, + 0.16666666666666666, + 0.0, + 1.0, + 0.14285714285714285, + 0.25, + 1.0, + 0.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.16666666666666666, + 0.0 + ] + }, + "rg+window": { + "hit@3": [ + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0 + ], + "recall@5": [ + 0.0, + 0.5, + 0.5, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.3333333333333333, + 0.75, + 0.0, + 0.5, + 0.0, + 0.0, + 1.0, + 0.0, + 0.25, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.6666666666666666, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.25, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0 + ], + "MRR": [ + 0.08333333333333333, + 1.0, + 1.0, + 0.125, + 0.125, + 1.0, + 0.5, + 0.045454545454545456, + 0.023255813953488372, + 0.0, + 0.09090909090909091, + 0.05263157894736842, + 0.0, + 0.07142857142857142, + 0.07692307692307693, + 0.5, + 0.037037037037037035, + 1.0, + 0.1, + 0.0625, + 0.0, + 0.08333333333333333, + 0.0, + 0.1, + 0.1, + 1.0, + 0.05555555555555555, + 0.0, + 0.0, + 0.16666666666666666, + 0.0, + 0.5, + 0.125, + 1.0, + 0.02702702702702703, + 0.14285714285714285, + 1.0, + 0.125, + 0.0, + 0.125, + 0.047619047619047616, + 0.07692307692307693, + 1.0, + 0.2, + 0.5, + 0.01818181818181818, + 0.029411764705882353, + 0.0, + 0.0625, + 0.2, + 0.02631578947368421, + 0.5, + 0.0, + 0.06666666666666667, + 0.1111111111111111, + 0.0, + 0.0, + 0.041666666666666664, + 0.0, + 1.0, + 0.25, + 1.0, + 0.1, + 0.5, + 0.16666666666666666, + 0.0, + 1.0, + 0.019230769230769232, + 1.0, + 0.09090909090909091, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.5, + 0.16666666666666666, + 0.3333333333333333, + 0.3333333333333333, + 0.0, + 0.125, + 0.0, + 0.030303030303030304, + 0.1111111111111111, + 0.023809523809523808, + 0.25, + 0.0, + 0.3333333333333333, + 0.2, + 0.0, + 0.0, + 0.0, + 0.1111111111111111, + 0.3333333333333333, + 0.03333333333333333, + 0.09090909090909091, + 0.012987012987012988, + 0.08333333333333333, + 0.0625, + 0.0, + 0.0, + 0.0, + 0.06666666666666667, + 0.1, + 0.125, + 0.043478260869565216, + 1.0, + 0.043478260869565216, + 0.03333333333333333, + 0.14285714285714285, + 0.0, + 0.017241379310344827, + 0.125, + 1.0, + 1.0, + 0.09090909090909091, + 0.09090909090909091, + 0.0, + 0.25, + 0.0, + 0.25, + 0.125, + 0.043478260869565216, + 0.25, + 0.16666666666666666, + 0.0, + 0.2, + 0.058823529411764705, + 1.0, + 0.037037037037037035, + 0.25, + 0.14285714285714285, + 0.0, + 1.0, + 0.25, + 0.3333333333333333, + 0.5, + 0.05, + 1.0, + 0.006802721088435374, + 0.3333333333333333, + 0.0, + 1.0, + 0.05, + 0.0, + 0.0625 + ] + }, + "repo-map 2k": { + "present": [ + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "recall": [ + 0.0, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.5, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.5, + 0.0, + 0.5, + 1.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.0, + 0.6666666666666666, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.6666666666666666, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + }, + "repo-map 2k query-aware": { + "present": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "recall": [ + 1.0, + 0.5, + 0.5, + 1.0, + 1.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.6666666666666666, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.6666666666666666, + 0.5, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.25, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + }, + "repo-map 8k": { + "present": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "recall": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ] + }, + "repo-map 8k query-aware": { + "present": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "recall": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ] + } + }, + "tokens_per_query": { + "index": [ + 3714, + 3680, + 3425, + 1971, + 3112, + 3952, + 4956, + 6553, + 3511, + 1654, + 3373, + 3121, + 1494, + 2498, + 3704, + 5113, + 4314, + 3225, + 3568, + 3018, + 2828, + 1691, + 3880, + 3152, + 3154, + 3768, + 3877, + 1657, + 2358, + 5386, + 3997, + 2851, + 2669, + 4624, + 3511, + 4683, + 2753, + 3474, + 1656, + 3365, + 4509, + 4644, + 4011, + 4593, + 2887, + 3058, + 4687, + 1444, + 3599, + 5240, + 2577, + 3589, + 4648, + 3156, + 3740, + 5972, + 3145, + 4368, + 1625, + 3905, + 4239, + 4559, + 4217, + 4224, + 4096, + 3149, + 2649, + 2220, + 4157, + 6479, + 3321, + 4368, + 2425, + 6001, + 3128, + 3652, + 2650, + 2662, + 3747, + 3987, + 5200, + 3045, + 3552, + 4811, + 3806, + 5716, + 3045, + 3874, + 4270, + 1689, + 4136, + 4095, + 4081, + 4842, + 3305, + 4053, + 4120, + 2737, + 5518, + 2507, + 4077, + 2458, + 5359, + 4500, + 3741, + 3006, + 2859, + 4136, + 2563, + 3323, + 4764, + 2897, + 2515, + 4381, + 2225, + 2646, + 3073, + 3015, + 2913, + 4932, + 3027, + 2570, + 3177, + 3852, + 3711, + 4211, + 3519, + 2751, + 3851, + 3951, + 3085, + 3174, + 4214, + 2877, + 2454, + 2799, + 5621, + 3585, + 2683, + 4963, + 5104, + 1661, + 3538, + 3983, + 2455, + 4989, + 3951, + 2135, + 2385, + 3936 + ], + "index (uncapped reads)": [ + 3714, + 7191, + 3425, + 1957, + 3112, + 3952, + 6502, + 6553, + 3511, + 1640, + 3359, + 3121, + 1494, + 2498, + 15367, + 6659, + 4314, + 3211, + 10468, + 6852, + 2828, + 1691, + 3880, + 3152, + 3140, + 10634, + 3877, + 1657, + 2358, + 23927, + 3997, + 2851, + 2669, + 16301, + 3511, + 16360, + 2753, + 15151, + 1642, + 3365, + 8211, + 4644, + 5557, + 15016, + 2887, + 3058, + 15913, + 13121, + 3599, + 17246, + 2906, + 3589, + 11382, + 3142, + 3740, + 5972, + 3145, + 4368, + 1625, + 4336, + 15916, + 11579, + 4217, + 4213, + 15522, + 3149, + 2649, + 2220, + 4588, + 6479, + 3321, + 4368, + 2425, + 6863, + 3128, + 5198, + 2650, + 2662, + 3747, + 11007, + 21218, + 3045, + 3552, + 4811, + 3806, + 5716, + 3045, + 4203, + 11049, + 1689, + 4136, + 4081, + 10987, + 5185, + 3305, + 4053, + 4120, + 2737, + 17195, + 2507, + 4378, + 2458, + 17039, + 4500, + 3741, + 3006, + 2859, + 4136, + 2563, + 3323, + 16413, + 2897, + 2515, + 4367, + 2225, + 2646, + 3058, + 14427, + 6172, + 6092, + 14690, + 2570, + 3177, + 3852, + 3711, + 4211, + 3519, + 2751, + 3851, + 3951, + 3085, + 3174, + 8555, + 2877, + 2454, + 2799, + 5621, + 5117, + 2683, + 18172, + 20853, + 1661, + 3538, + 4312, + 2455, + 4989, + 3951, + 2135, + 2385, + 8184 + ], + "rg+window": [ + 2612, + 2868, + 3743, + 3534, + 2994, + 2853, + 3041, + 3067, + 3414, + 772, + 3668, + 3266, + 2860, + 2674, + 3196, + 3587, + 3493, + 3238, + 3612, + 3604, + 1272, + 3001, + 3089, + 3388, + 3370, + 3099, + 2582, + 3109, + 3089, + 3461, + 3535, + 2777, + 3459, + 3234, + 3199, + 3367, + 3040, + 3460, + 0, + 3105, + 3262, + 3188, + 2813, + 3385, + 2534, + 3314, + 3140, + 3518, + 3163, + 3594, + 2994, + 3014, + 2798, + 3002, + 3199, + 2162, + 3442, + 3189, + 2626, + 3529, + 3341, + 3284, + 3398, + 3342, + 2924, + 1853, + 2650, + 2649, + 3255, + 3411, + 2825, + 2866, + 2046, + 3399, + 3407, + 3255, + 3407, + 3202, + 3207, + 3284, + 2742, + 3292, + 3438, + 2966, + 3117, + 2242, + 2334, + 2499, + 2937, + 3706, + 3024, + 3175, + 3160, + 2773, + 3014, + 3209, + 3472, + 3091, + 3036, + 2813, + 2925, + 3344, + 3598, + 3321, + 3249, + 2903, + 3573, + 3430, + 3648, + 2468, + 3143, + 3340, + 3486, + 3123, + 2236, + 3445, + 3160, + 3347, + 3641, + 3714, + 3137, + 1225, + 3345, + 3057, + 3495, + 3041, + 3186, + 2884, + 2767, + 2695, + 2777, + 3511, + 3307, + 3516, + 3485, + 3090, + 3480, + 3435, + 2658, + 3600, + 3256, + 2859, + 3584, + 3386, + 2609, + 3073, + 3160, + 3249, + 2879, + 3333 + ], + "repo-map 2k": [ + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998 + ], + "repo-map 2k query-aware": [ + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 1998, + 1999, + 1999, + 1999, + 2000, + 1999, + 1999, + 1998, + 1998, + 2000, + 2000, + 1999, + 1999, + 1998, + 1999, + 1999, + 1998, + 2000, + 1998, + 1999, + 2000, + 2000, + 2000, + 1998, + 1998, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1998, + 1998, + 1999, + 1998, + 2000, + 1999, + 1998, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1998, + 2000, + 1999, + 1998, + 1999, + 1999, + 1999, + 1998, + 1999, + 1998, + 1999, + 1998, + 1999, + 2000, + 1998, + 1998, + 1998, + 1998, + 1999, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1998, + 2000, + 1998, + 1998, + 2000, + 1998, + 1998, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1998, + 1998, + 1999, + 2000, + 1999, + 1999, + 1998, + 2000, + 2000, + 2000, + 1999, + 1998, + 2000, + 1998, + 2000, + 1999, + 2000, + 2000, + 1998, + 1998, + 2000, + 1998, + 1999, + 2000, + 1999, + 1998, + 1998, + 2000, + 1998, + 2000, + 1999, + 1998, + 2000, + 1999, + 2000, + 1998, + 1999, + 1998, + 2000, + 1998, + 1998, + 1999, + 1999, + 1999, + 1998 + ], + "repo-map 8k": [ + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728 + ], + "repo-map 8k query-aware": [ + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728 + ] + }, + "packet_tokens_per_query": { + "index": [ + 2545, + 2189, + 2774, + 1505, + 2100, + 2468, + 2356, + 3003, + 2928, + 1434, + 2687, + 2602, + 1167, + 2155, + 2294, + 2611, + 2245, + 2680, + 2012, + 1349, + 2217, + 1474, + 2710, + 2648, + 2127, + 2134, + 2837, + 1508, + 2012, + 2790, + 2931, + 2225, + 2241, + 2455, + 2928, + 2436, + 2417, + 1952, + 1436, + 2537, + 2786, + 2553, + 2585, + 2140, + 2776, + 2866, + 2150, + 230, + 2564, + 2536, + 1343, + 2769, + 2410, + 2614, + 2632, + 3190, + 2603, + 2609, + 1418, + 2037, + 2429, + 2662, + 3019, + 2437, + 2268, + 2538, + 2554, + 2104, + 2409, + 3202, + 2718, + 2785, + 1814, + 2954, + 2658, + 2397, + 1970, + 2359, + 2769, + 2548, + 2357, + 2031, + 2473, + 2824, + 2182, + 2804, + 2412, + 2603, + 2647, + 1341, + 2651, + 2658, + 2298, + 2913, + 2668, + 2930, + 2920, + 1994, + 2935, + 2263, + 2617, + 2258, + 2556, + 2469, + 2740, + 2280, + 2116, + 2454, + 2199, + 2443, + 2512, + 1901, + 2174, + 2547, + 1804, + 2416, + 2420, + 1422, + 1381, + 2337, + 1708, + 1959, + 2369, + 2097, + 2473, + 2516, + 2288, + 2288, + 2604, + 3355, + 2473, + 2559, + 2427, + 2605, + 1878, + 2651, + 2700, + 2139, + 2472, + 2194, + 2473, + 1454, + 2256, + 1986, + 2047, + 2923, + 2604, + 1991, + 2193, + 2681 + ], + "index (uncapped reads)": [ + 2545, + 2161, + 2774, + 1491, + 2100, + 2468, + 2342, + 3003, + 2928, + 1420, + 2673, + 2602, + 1167, + 2155, + 2266, + 2569, + 2245, + 2666, + 1998, + 1335, + 2217, + 1474, + 2710, + 2648, + 2113, + 2120, + 2837, + 1508, + 2012, + 2748, + 2931, + 2225, + 2241, + 2441, + 2928, + 2422, + 2417, + 1938, + 1422, + 2537, + 2772, + 2553, + 2571, + 2111, + 2776, + 2866, + 2122, + 216, + 2564, + 2494, + 1315, + 2769, + 2396, + 2600, + 2632, + 3190, + 2603, + 2609, + 1418, + 2009, + 2415, + 2648, + 3019, + 2409, + 2254, + 2538, + 2554, + 2104, + 2381, + 3202, + 2718, + 2785, + 1814, + 2926, + 2658, + 2383, + 1970, + 2359, + 2769, + 2534, + 2329, + 2031, + 2473, + 2824, + 2182, + 2804, + 2412, + 2575, + 2633, + 1341, + 2651, + 2644, + 2284, + 2899, + 2668, + 2930, + 2920, + 1994, + 2921, + 2263, + 2561, + 2258, + 2528, + 2469, + 2740, + 2280, + 2116, + 2454, + 2199, + 2443, + 2470, + 1901, + 2174, + 2533, + 1804, + 2416, + 2405, + 1394, + 1353, + 2309, + 1680, + 1959, + 2369, + 2097, + 2473, + 2516, + 2288, + 2288, + 2604, + 3355, + 2473, + 2559, + 2413, + 2605, + 1878, + 2651, + 2700, + 2111, + 2472, + 2152, + 2431, + 1454, + 2256, + 1958, + 2047, + 2923, + 2604, + 1991, + 2193, + 2667 + ], + "rg+window": [ + 1041, + 1198, + 1141, + 1115, + 1176, + 1059, + 1080, + 977, + 1271, + 43, + 1133, + 1187, + 1088, + 875, + 994, + 1142, + 1216, + 1178, + 1089, + 1149, + 36, + 902, + 1026, + 1011, + 1157, + 1029, + 903, + 1166, + 1165, + 924, + 1131, + 983, + 1101, + 1092, + 1056, + 1094, + 1087, + 1037, + 0, + 1087, + 1129, + 1063, + 1147, + 1082, + 368, + 1051, + 1122, + 1147, + 1105, + 1111, + 1239, + 1094, + 1061, + 1102, + 1070, + 351, + 1115, + 1013, + 845, + 1110, + 1056, + 1076, + 924, + 995, + 1007, + 159, + 1038, + 884, + 1083, + 1151, + 1084, + 1170, + 123, + 1022, + 1030, + 1083, + 1030, + 1062, + 1099, + 1098, + 962, + 1113, + 1102, + 1165, + 1031, + 614, + 1013, + 1105, + 1215, + 1204, + 826, + 1073, + 973, + 853, + 877, + 771, + 1095, + 1080, + 1087, + 959, + 1208, + 1106, + 1262, + 1104, + 865, + 945, + 1129, + 1033, + 1155, + 914, + 1061, + 1126, + 1103, + 1123, + 556, + 1129, + 1030, + 1097, + 1153, + 1087, + 1046, + 44, + 1073, + 1019, + 1047, + 1126, + 1060, + 1052, + 1023, + 914, + 1026, + 1112, + 1058, + 1111, + 1113, + 1190, + 1103, + 1167, + 503, + 1154, + 1095, + 1078, + 1113, + 1255, + 367, + 1236, + 1094, + 1092, + 995, + 1120 + ], + "repo-map 2k": [ + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998, + 1998 + ], + "repo-map 2k query-aware": [ + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 1998, + 1999, + 1999, + 1999, + 2000, + 1999, + 1999, + 1998, + 1998, + 2000, + 2000, + 1999, + 1999, + 1998, + 1999, + 1999, + 1998, + 2000, + 1998, + 1999, + 2000, + 2000, + 2000, + 1998, + 1998, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1998, + 1998, + 1999, + 1998, + 2000, + 1999, + 1998, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1998, + 2000, + 1999, + 1998, + 1999, + 1999, + 1999, + 1998, + 1999, + 1998, + 1999, + 1998, + 1999, + 2000, + 1998, + 1998, + 1998, + 1998, + 1999, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1998, + 2000, + 1998, + 1998, + 2000, + 1998, + 1998, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1998, + 1998, + 1999, + 2000, + 1999, + 1999, + 1998, + 2000, + 2000, + 2000, + 1999, + 1998, + 2000, + 1998, + 2000, + 1999, + 2000, + 2000, + 1998, + 1998, + 2000, + 1998, + 1999, + 2000, + 1999, + 1998, + 1998, + 2000, + 1998, + 2000, + 1999, + 1998, + 2000, + 1999, + 2000, + 1998, + 1999, + 1998, + 2000, + 1998, + 1998, + 1999, + 1999, + 1999, + 1998 + ], + "repo-map 8k": [ + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728 + ], + "repo-map 8k query-aware": [ + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728, + 7728 + ] + } + }, + { + "corpus": "gson", + "root": "C:\\Users\\dabin\\AppData\\Local\\Temp\\claude\\D--Projects-codebase-index\\7f35eb46-c8c9-49e7-9d02-4c510c6877d0\\scratchpad\\repos\\gson", + "head": "b3f4ca20087f9066de4c340522ff84e0558e1ad1", + "n_queries": 150, + "index_build_ms": 5279.1104999996605, + "files_indexed": 311, + "symbols_indexed": 4193, + "edges_indexed": 22407, + "summary": { + "index": { + "hit@3": 0.5733333333333334, + "recall@5": 0.5233333333333333, + "MRR": 0.5034285714285714, + "tokens_mean": 4375.306666666666, + "tokens_median": 4376.0, + "packet_tokens_mean": 2418.1266666666666, + "latency_p50_ms": 112.98190000161412 + }, + "index (uncapped reads)": { + "hit@3": 0.5733333333333334, + "recall@5": 0.5233333333333333, + "MRR": 0.5034285714285714, + "tokens_mean": 10659.12, + "tokens_median": 8680.5, + "packet_tokens_mean": 2382.193333333333, + "latency_p50_ms": 114.68090000198572 + }, + "rg+window": { + "hit@3": 0.3466666666666667, + "recall@5": 0.33722222222222226, + "MRR": 0.2898899828278739, + "tokens_mean": 4117.386666666666, + "tokens_median": 4091.5, + "packet_tokens_mean": 1864.6533333333334, + "latency_p50_ms": 34.462199997506104 + }, + "repo-map 2k": { + "present": 0.48, + "recall": 0.3644444444444444, + "tokens_mean": 2000.0, + "tokens_median": 2000.0, + "packet_tokens_mean": 2000.0, + "latency_p50_ms": 0.1880999989225529 + }, + "repo-map 2k query-aware": { + "present": 0.6133333333333333, + "recall": 0.5516666666666666, + "tokens_mean": 1999.7333333333333, + "tokens_median": 2000.0, + "packet_tokens_mean": 1999.7333333333333, + "latency_p50_ms": 2.779500002361601 + }, + "repo-map 8k": { + "present": 1.0, + "recall": 1.0, + "tokens_mean": 6757.0, + "tokens_median": 6757.0, + "packet_tokens_mean": 6757.0, + "latency_p50_ms": 0.12520000018412247 + }, + "repo-map 8k query-aware": { + "present": 1.0, + "recall": 1.0, + "tokens_mean": 6757.0, + "tokens_median": 6757.0, + "packet_tokens_mean": 6757.0, + "latency_p50_ms": 2.672400001756614 + } + }, + "per_query": { + "index": { + "hit@3": [ + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0 + ], + "recall@5": [ + 0.25, + 0.5, + 0.25, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.6666666666666666, + 1.0, + 0.5, + 0.5, + 0.0, + 1.0, + 0.6666666666666666, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.5, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 0.5, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.5, + 0.6666666666666666, + 0.0, + 1.0, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.3333333333333333, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 1.0, + 1.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.25, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.25, + 1.0, + 1.0, + 1.0, + 0.5, + 0.6666666666666666, + 0.6666666666666666, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5 + ], + "MRR": [ + 0.5, + 1.0, + 0.2, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.14285714285714285, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.5, + 0.0, + 0.25, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.25, + 1.0, + 1.0, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.125, + 0.25, + 0.14285714285714285, + 0.25, + 1.0, + 1.0, + 0.0, + 0.2, + 1.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.14285714285714285, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.16666666666666666, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.5, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.5, + 1.0, + 1.0, + 1.0, + 0.25, + 0.2, + 0.125, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.25, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.3333333333333333, + 0.3333333333333333, + 0.3333333333333333, + 0.5, + 0.2, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.25, + 0.0, + 1.0, + 0.16666666666666666, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.25, + 1.0, + 0.25, + 1.0, + 0.5, + 1.0, + 1.0, + 0.14285714285714285, + 0.14285714285714285, + 1.0, + 0.0, + 0.0, + 1.0 + ] + }, + "index (uncapped reads)": { + "hit@3": [ + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0 + ], + "recall@5": [ + 0.25, + 0.5, + 0.25, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.6666666666666666, + 1.0, + 0.5, + 0.5, + 0.0, + 1.0, + 0.6666666666666666, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.5, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 0.5, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.5, + 0.6666666666666666, + 0.0, + 1.0, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.3333333333333333, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 1.0, + 1.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.25, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.25, + 1.0, + 1.0, + 1.0, + 0.5, + 0.6666666666666666, + 0.6666666666666666, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5 + ], + "MRR": [ + 0.5, + 1.0, + 0.2, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.14285714285714285, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.5, + 0.0, + 0.25, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.25, + 1.0, + 1.0, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.125, + 0.25, + 0.14285714285714285, + 0.25, + 1.0, + 1.0, + 0.0, + 0.2, + 1.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.14285714285714285, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.16666666666666666, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.5, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.5, + 1.0, + 1.0, + 1.0, + 0.25, + 0.2, + 0.125, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.25, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.3333333333333333, + 0.3333333333333333, + 0.3333333333333333, + 0.5, + 0.2, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.25, + 0.0, + 1.0, + 0.16666666666666666, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.25, + 1.0, + 0.25, + 1.0, + 0.5, + 1.0, + 1.0, + 0.14285714285714285, + 0.14285714285714285, + 1.0, + 0.0, + 0.0, + 1.0 + ] + }, + "rg+window": { + "hit@3": [ + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0 + ], + "recall@5": [ + 0.5, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.5, + 0.5, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 0.25, + 0.0, + 1.0, + 0.5, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.5, + 1.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.5, + 1.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.25, + 1.0, + 0.0, + 0.0, + 0.25 + ], + "MRR": [ + 1.0, + 0.5, + 0.0, + 0.2, + 0.0, + 0.09090909090909091, + 0.3333333333333333, + 0.00819672131147541, + 0.0625, + 0.16666666666666666, + 0.3333333333333333, + 0.5, + 1.0, + 0.2, + 1.0, + 0.047619047619047616, + 0.0, + 0.2, + 1.0, + 0.0625, + 0.5, + 0.5, + 0.14285714285714285, + 0.3333333333333333, + 0.25, + 0.025, + 0.14285714285714285, + 1.0, + 0.1111111111111111, + 0.5, + 0.09090909090909091, + 0.25, + 1.0, + 0.5, + 0.5, + 0.0, + 0.07692307692307693, + 1.0, + 0.037037037037037035, + 0.02702702702702703, + 0.25, + 0.25, + 0.038461538461538464, + 0.3333333333333333, + 0.3333333333333333, + 0.0, + 0.1111111111111111, + 0.14285714285714285, + 0.0, + 0.0, + 0.25, + 0.08333333333333333, + 0.25, + 0.022222222222222223, + 0.09090909090909091, + 0.14285714285714285, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.017857142857142856, + 0.0, + 0.015151515151515152, + 1.0, + 0.0, + 0.14285714285714285, + 0.25, + 0.02631578947368421, + 1.0, + 0.08333333333333333, + 0.07142857142857142, + 0.125, + 1.0, + 1.0, + 0.0625, + 0.015625, + 0.14285714285714285, + 0.007352941176470588, + 0.0, + 0.009900990099009901, + 0.008064516129032258, + 1.0, + 0.041666666666666664, + 0.006944444444444444, + 0.009615384615384616, + 1.0, + 0.04, + 1.0, + 1.0, + 0.0625, + 0.3333333333333333, + 0.009009009009009009, + 0.037037037037037035, + 0.04, + 0.0, + 0.02631578947368421, + 0.07142857142857142, + 1.0, + 0.1, + 0.07142857142857142, + 0.07142857142857142, + 0.25, + 0.3333333333333333, + 0.125, + 0.1111111111111111, + 0.05263157894736842, + 0.02631578947368421, + 0.0625, + 0.05555555555555555, + 1.0, + 0.5, + 0.5, + 0.5, + 0.5, + 0.2, + 0.3333333333333333, + 0.058823529411764705, + 1.0, + 0.09090909090909091, + 0.08333333333333333, + 0.5, + 0.5, + 0.3333333333333333, + 0.25, + 0.021739130434782608, + 0.2, + 0.3333333333333333, + 0.0, + 0.25, + 0.06666666666666667, + 0.08333333333333333, + 1.0, + 0.09090909090909091, + 0.5, + 0.0, + 0.1111111111111111, + 0.1, + 0.3333333333333333, + 0.008620689655172414, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.25, + 0.5, + 0.045454545454545456, + 0.0, + 0.5 + ] + }, + "repo-map 2k": { + "present": [ + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0 + ], + "recall": [ + 0.75, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5, + 0.5, + 0.5, + 0.6666666666666666, + 0.5, + 1.0, + 0.5, + 1.0, + 0.0, + 0.6666666666666666, + 0.75, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.6666666666666666, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.5, + 1.0, + 1.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.3333333333333333, + 0.6666666666666666, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.5, + 0.5, + 0.5, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.25, + 0.3333333333333333, + 0.0, + 0.0, + 0.5, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.5, + 0.0, + 0.0, + 0.25, + 0.0, + 0.0, + 0.0, + 0.5, + 0.6666666666666666, + 1.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.5 + ] + }, + "repo-map 2k query-aware": { + "present": [ + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "recall": [ + 0.75, + 1.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.6666666666666666, + 0.75, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.0, + 1.0, + 1.0, + 0.5, + 0.5, + 1.0, + 1.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.6666666666666666, + 0.0, + 0.0, + 1.0, + 0.6666666666666666, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.3333333333333333, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.75, + 1.0, + 1.0, + 1.0, + 0.5, + 0.0, + 0.5, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.25, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.75, + 1.0, + 1.0, + 1.0, + 1.0 + ] + }, + "repo-map 8k": { + "present": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "recall": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ] + }, + "repo-map 8k query-aware": { + "present": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "recall": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ] + } + }, + "tokens_per_query": { + "index": [ + 5059, + 6027, + 5470, + 5315, + 2845, + 4899, + 4064, + 3211, + 3556, + 3229, + 5950, + 4194, + 4752, + 5396, + 5761, + 5036, + 2164, + 4169, + 5272, + 2082, + 4745, + 4663, + 3110, + 2763, + 6487, + 4221, + 4616, + 4613, + 5390, + 5055, + 5095, + 4401, + 3568, + 4796, + 4955, + 3266, + 4726, + 4372, + 3179, + 5473, + 4149, + 4738, + 4650, + 6351, + 6625, + 4801, + 2803, + 4290, + 1766, + 4009, + 4654, + 4633, + 3589, + 4466, + 4924, + 4184, + 4352, + 4797, + 3755, + 2343, + 2540, + 4948, + 5321, + 2980, + 4310, + 4904, + 4930, + 3857, + 5260, + 4380, + 6344, + 5862, + 4099, + 6937, + 3969, + 4519, + 3972, + 5546, + 3074, + 5727, + 2940, + 4454, + 3396, + 4332, + 4814, + 3823, + 4463, + 5026, + 2696, + 5898, + 3750, + 5432, + 3815, + 4639, + 5007, + 5306, + 5482, + 2818, + 3864, + 3989, + 2649, + 5385, + 4119, + 4947, + 3331, + 3980, + 3371, + 3941, + 4323, + 5761, + 4847, + 3617, + 3100, + 6650, + 4958, + 5801, + 6244, + 4111, + 6039, + 5785, + 3828, + 4676, + 4939, + 3517, + 3676, + 5798, + 4984, + 3101, + 3756, + 4784, + 5186, + 4224, + 5030, + 2607, + 3577, + 2391, + 3946, + 3617, + 2314, + 4002, + 3491, + 3765, + 4161, + 4120, + 3085, + 3992, + 3763, + 2702, + 5863, + 5195 + ], + "index (uncapped reads)": [ + 22367, + 22931, + 5693, + 19077, + 2845, + 17752, + 4064, + 3211, + 9842, + 9487, + 22034, + 7241, + 8158, + 11626, + 16251, + 7752, + 2164, + 5046, + 5850, + 14935, + 4731, + 9332, + 23396, + 2763, + 30997, + 15156, + 8130, + 19436, + 18229, + 10227, + 13026, + 4401, + 17932, + 8878, + 13901, + 6720, + 10341, + 7689, + 3179, + 20130, + 12746, + 15659, + 15096, + 26637, + 22457, + 37777, + 3605, + 12940, + 1766, + 4009, + 4654, + 4619, + 16332, + 21370, + 13529, + 4184, + 17191, + 4783, + 3755, + 2343, + 2540, + 4934, + 13943, + 15833, + 18670, + 6488, + 4888, + 14812, + 7330, + 12702, + 32247, + 12832, + 4057, + 6909, + 20859, + 10777, + 3958, + 7993, + 3060, + 24230, + 9324, + 10253, + 3396, + 4318, + 6154, + 3809, + 4449, + 21421, + 2696, + 15474, + 6791, + 11920, + 11105, + 4639, + 5007, + 8483, + 9401, + 2825, + 7323, + 6031, + 4232, + 9583, + 12769, + 19642, + 7665, + 3952, + 4159, + 10665, + 19096, + 6535, + 26262, + 3603, + 3086, + 19647, + 15165, + 27304, + 7829, + 4083, + 10090, + 6448, + 3828, + 37801, + 9956, + 12167, + 12378, + 11110, + 13606, + 3101, + 3742, + 5161, + 15056, + 7602, + 11053, + 2607, + 3577, + 2377, + 13749, + 12237, + 2300, + 12621, + 6931, + 6142, + 12696, + 4120, + 3533, + 6028, + 6652, + 2660, + 18509, + 17500 + ], + "rg+window": [ + 4994, + 4821, + 4521, + 4553, + 4075, + 4559, + 4299, + 3396, + 4603, + 4064, + 4670, + 3943, + 3745, + 3587, + 3495, + 3364, + 4877, + 3833, + 3776, + 5068, + 3658, + 4628, + 5341, + 3372, + 4463, + 4167, + 4302, + 4442, + 4316, + 4210, + 3342, + 3790, + 4054, + 3897, + 3407, + 4114, + 4384, + 2136, + 4460, + 4118, + 4081, + 3204, + 4573, + 4961, + 3333, + 3567, + 3951, + 4093, + 3098, + 4640, + 3501, + 4694, + 3366, + 4279, + 4502, + 4069, + 4534, + 3653, + 4043, + 3147, + 3600, + 4436, + 4191, + 3683, + 4477, + 3831, + 4165, + 3646, + 4801, + 4165, + 3515, + 3359, + 4702, + 4540, + 3943, + 5681, + 3429, + 6258, + 4677, + 4483, + 4309, + 4153, + 3308, + 3841, + 3588, + 4194, + 4196, + 4068, + 3589, + 4287, + 3743, + 4310, + 3846, + 4691, + 3511, + 4500, + 3938, + 3411, + 3774, + 3700, + 4071, + 4287, + 3871, + 4124, + 5321, + 4071, + 4070, + 4492, + 4358, + 4342, + 3884, + 3508, + 4035, + 4735, + 3917, + 3374, + 5515, + 4699, + 4558, + 3683, + 4090, + 4685, + 3960, + 3701, + 4313, + 4295, + 3707, + 4420, + 4910, + 4141, + 3909, + 3894, + 4170, + 3870, + 4153, + 2889, + 4696, + 5305, + 3825, + 3807, + 6496, + 4648, + 3850, + 2969, + 4059, + 3713, + 4233, + 4331, + 2838, + 4144 + ], + "repo-map 2k": [ + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000 + ], + "repo-map 2k query-aware": [ + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999 + ], + "repo-map 8k": [ + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757 + ], + "repo-map 8k query-aware": [ + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757 + ] + }, + "packet_tokens_per_query": { + "index": [ + 2657, + 2973, + 2537, + 2120, + 2012, + 2719, + 2557, + 2512, + 1539, + 2353, + 2879, + 2088, + 2699, + 2888, + 2526, + 2833, + 2078, + 2005, + 2803, + 587, + 2564, + 2525, + 1309, + 2677, + 2933, + 2707, + 2780, + 2536, + 2398, + 2706, + 2948, + 2545, + 1522, + 2839, + 3159, + 2364, + 2619, + 2482, + 2564, + 2519, + 2421, + 1551, + 2352, + 2395, + 2934, + 2577, + 1254, + 2165, + 1696, + 2667, + 2585, + 2546, + 2358, + 2240, + 2981, + 2569, + 2595, + 2597, + 2784, + 1376, + 2454, + 2949, + 2525, + 1742, + 2260, + 2199, + 2183, + 2035, + 2734, + 2836, + 2684, + 2791, + 2640, + 2540, + 1677, + 2390, + 3033, + 2674, + 2885, + 2648, + 1442, + 2389, + 2688, + 2449, + 2953, + 2357, + 2627, + 2861, + 1915, + 2960, + 2048, + 2273, + 877, + 2544, + 2730, + 2967, + 2816, + 1633, + 2695, + 2489, + 1170, + 2376, + 2755, + 2840, + 1920, + 2446, + 2265, + 1968, + 2222, + 2584, + 2422, + 2323, + 2746, + 2781, + 2778, + 2876, + 2296, + 2652, + 2612, + 2940, + 2390, + 2567, + 2522, + 2211, + 1435, + 2469, + 2304, + 2268, + 2780, + 2457, + 2843, + 2171, + 2914, + 1861, + 2789, + 2202, + 2166, + 2290, + 2227, + 2709, + 2569, + 1773, + 2791, + 2527, + 2283, + 2363, + 2017, + 2616, + 2672, + 2265 + ], + "index (uncapped reads)": [ + 2587, + 2917, + 2509, + 2050, + 2012, + 2705, + 2557, + 2512, + 1525, + 2311, + 2823, + 2032, + 2629, + 2818, + 2442, + 2805, + 2078, + 1949, + 2747, + 573, + 2550, + 2483, + 1295, + 2677, + 2891, + 2679, + 2738, + 2480, + 2370, + 2664, + 2906, + 2545, + 1466, + 2755, + 3075, + 2322, + 2577, + 2412, + 2564, + 2449, + 2393, + 1509, + 2296, + 2381, + 2878, + 2535, + 1240, + 2151, + 1696, + 2667, + 2585, + 2532, + 2330, + 2184, + 2953, + 2569, + 2567, + 2583, + 2784, + 1376, + 2454, + 2935, + 2483, + 1728, + 2204, + 2129, + 2141, + 2007, + 2706, + 2794, + 2600, + 2721, + 2598, + 2512, + 1607, + 2348, + 3019, + 2618, + 2871, + 2592, + 1428, + 2361, + 2688, + 2435, + 2939, + 2343, + 2613, + 2819, + 1915, + 2904, + 1978, + 2231, + 821, + 2544, + 2730, + 2925, + 2718, + 1577, + 2639, + 2433, + 1156, + 2320, + 2741, + 2784, + 1906, + 2418, + 2237, + 1912, + 2166, + 2542, + 2352, + 2309, + 2732, + 2697, + 2708, + 2848, + 2282, + 2624, + 2570, + 2884, + 2390, + 2525, + 2480, + 2197, + 1407, + 2371, + 2262, + 2268, + 2766, + 2443, + 2773, + 2073, + 2858, + 1861, + 2789, + 2188, + 2082, + 2276, + 2213, + 2695, + 2513, + 1689, + 2735, + 2527, + 2269, + 2321, + 1989, + 2574, + 2616, + 2195 + ], + "rg+window": [ + 1959, + 2230, + 1527, + 2040, + 2362, + 1660, + 1616, + 1296, + 2294, + 2040, + 2407, + 1742, + 1692, + 1489, + 1643, + 1473, + 2309, + 1865, + 1579, + 2414, + 1358, + 2216, + 3044, + 1351, + 2369, + 2058, + 2527, + 2066, + 2225, + 1896, + 1534, + 1141, + 1919, + 1880, + 1544, + 2058, + 1508, + 279, + 2577, + 1988, + 1958, + 1190, + 2245, + 2477, + 1286, + 1984, + 1896, + 2058, + 784, + 1617, + 1795, + 2130, + 1477, + 1613, + 2197, + 1097, + 1565, + 1654, + 2333, + 1111, + 1740, + 1527, + 1972, + 1360, + 2168, + 1935, + 2058, + 1714, + 2234, + 2058, + 1533, + 1267, + 1512, + 1763, + 1866, + 2594, + 1723, + 2875, + 1736, + 1946, + 2190, + 1855, + 1109, + 1771, + 1478, + 1078, + 1900, + 1904, + 1420, + 1881, + 1765, + 2203, + 1824, + 1545, + 1730, + 2748, + 1916, + 1446, + 1718, + 1763, + 2150, + 1862, + 1631, + 1763, + 3202, + 2116, + 2213, + 2556, + 1878, + 2244, + 1725, + 1455, + 1998, + 2787, + 1951, + 1085, + 3112, + 2038, + 2193, + 1841, + 1688, + 2129, + 1531, + 1784, + 2215, + 2237, + 1735, + 1640, + 1817, + 2075, + 1790, + 1756, + 1855, + 1701, + 2046, + 1055, + 2604, + 2688, + 1589, + 1730, + 3121, + 2490, + 1743, + 797, + 2062, + 1585, + 2200, + 1851, + 221, + 1696 + ], + "repo-map 2k": [ + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000 + ], + "repo-map 2k query-aware": [ + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999 + ], + "repo-map 8k": [ + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757 + ], + "repo-map 8k query-aware": [ + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757, + 6757 + ] + } + }, + { + "corpus": "fastify", + "root": "C:\\Users\\dabin\\AppData\\Local\\Temp\\claude\\D--Projects-codebase-index\\7f35eb46-c8c9-49e7-9d02-4c510c6877d0\\scratchpad\\repos\\fastify", + "head": "15ebc8e2fe6932d3e91afa84709523804f8a2537", + "n_queries": 150, + "index_build_ms": 1983.2537000002048, + "files_indexed": 389, + "symbols_indexed": 1185, + "edges_indexed": 34297, + "summary": { + "index": { + "hit@3": 0.58, + "recall@5": 0.5472222222222222, + "MRR": 0.4817698412698413, + "tokens_mean": 3539.4733333333334, + "tokens_median": 3561.5, + "packet_tokens_mean": 2427.64, + "latency_p50_ms": 45.60020000280929 + }, + "index (uncapped reads)": { + "hit@3": 0.58, + "recall@5": 0.5472222222222222, + "MRR": 0.4817698412698413, + "tokens_mean": 4026.7066666666665, + "tokens_median": 3611.0, + "packet_tokens_mean": 2423.0666666666666, + "latency_p50_ms": 45.35319999922649 + }, + "rg+window": { + "hit@3": 0.3333333333333333, + "recall@5": 0.2972222222222222, + "MRR": 0.26914749481448863, + "tokens_mean": 3290.1866666666665, + "tokens_median": 3312.5, + "packet_tokens_mean": 1178.4266666666667, + "latency_p50_ms": 38.79890000098385 + }, + "repo-map 2k": { + "present": 0.36, + "recall": 0.20277777777777778, + "tokens_mean": 2000.0, + "tokens_median": 2000.0, + "packet_tokens_mean": 2000.0, + "latency_p50_ms": 0.1922999981616158 + }, + "repo-map 2k query-aware": { + "present": 0.36666666666666664, + "recall": 0.22333333333333333, + "tokens_mean": 1999.7266666666667, + "tokens_median": 2000.0, + "packet_tokens_mean": 1999.7266666666667, + "latency_p50_ms": 0.8546999997633975 + }, + "repo-map 8k": { + "present": 0.9733333333333334, + "recall": 0.9033333333333333, + "tokens_mean": 8000.0, + "tokens_median": 8000.0, + "packet_tokens_mean": 8000.0, + "latency_p50_ms": 0.1354000014543999 + }, + "repo-map 8k query-aware": { + "present": 0.8933333333333333, + "recall": 0.8188888888888889, + "tokens_mean": 7999.14, + "tokens_median": 8000.0, + "packet_tokens_mean": 7999.14, + "latency_p50_ms": 0.8130999995046295 + } + }, + "per_query": { + "index": { + "hit@3": [ + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0 + ], + "recall@5": [ + 1.0, + 0.5, + 0.6666666666666666, + 0.0, + 0.75, + 0.5, + 0.0, + 0.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.25, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 1.0, + 0.5, + 0.5, + 0.5, + 0.0, + 0.5, + 0.6666666666666666, + 0.5, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.5, + 0.0, + 1.0, + 0.6666666666666666, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.5, + 1.0, + 1.0, + 0.25, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.25, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.25, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.5 + ], + "MRR": [ + 1.0, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.2, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.25, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.16666666666666666, + 0.0, + 0.2, + 0.0, + 1.0, + 0.25, + 0.5, + 1.0, + 0.0, + 0.5, + 1.0, + 0.25, + 0.0, + 0.5, + 0.25, + 0.5, + 1.0, + 0.3333333333333333, + 0.5, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.5, + 0.2, + 0.14285714285714285, + 0.16666666666666666, + 0.5, + 1.0, + 0.0, + 0.14285714285714285, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.125, + 0.5, + 1.0, + 0.25, + 0.25, + 0.0, + 0.0, + 0.16666666666666666, + 0.0, + 0.3333333333333333, + 0.2, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.3333333333333333, + 0.2, + 1.0, + 1.0, + 1.0, + 0.5, + 0.14285714285714285, + 0.16666666666666666, + 0.5, + 0.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5, + 0.14285714285714285, + 0.3333333333333333, + 1.0, + 1.0, + 0.5, + 0.14285714285714285, + 0.25, + 0.0, + 0.14285714285714285, + 0.5, + 1.0, + 1.0, + 1.0, + 0.2, + 1.0 + ] + }, + "index (uncapped reads)": { + "hit@3": [ + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0 + ], + "recall@5": [ + 1.0, + 0.5, + 0.6666666666666666, + 0.0, + 0.75, + 0.5, + 0.0, + 0.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.25, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.0, + 1.0, + 0.5, + 0.5, + 0.5, + 0.0, + 0.5, + 0.6666666666666666, + 0.5, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.5, + 0.0, + 1.0, + 0.6666666666666666, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.5, + 1.0, + 1.0, + 0.25, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.25, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.25, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.5 + ], + "MRR": [ + 1.0, + 0.3333333333333333, + 0.5, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.3333333333333333, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.2, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.25, + 1.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.16666666666666666, + 0.0, + 0.2, + 0.0, + 1.0, + 0.25, + 0.5, + 1.0, + 0.0, + 0.5, + 1.0, + 0.25, + 0.0, + 0.5, + 0.25, + 0.5, + 1.0, + 0.3333333333333333, + 0.5, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.5, + 0.2, + 0.14285714285714285, + 0.16666666666666666, + 0.5, + 1.0, + 0.0, + 0.14285714285714285, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.125, + 0.5, + 1.0, + 0.25, + 0.25, + 0.0, + 0.0, + 0.16666666666666666, + 0.0, + 0.3333333333333333, + 0.2, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.0, + 0.5, + 1.0, + 0.3333333333333333, + 0.2, + 1.0, + 1.0, + 1.0, + 0.5, + 0.14285714285714285, + 0.16666666666666666, + 0.5, + 0.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5, + 0.14285714285714285, + 0.3333333333333333, + 1.0, + 1.0, + 0.5, + 0.14285714285714285, + 0.25, + 0.0, + 0.14285714285714285, + 0.5, + 1.0, + 1.0, + 1.0, + 0.2, + 1.0 + ] + }, + "rg+window": { + "hit@3": [ + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0 + ], + "recall@5": [ + 0.0, + 0.5, + 0.3333333333333333, + 0.0, + 0.0, + 0.25, + 0.0, + 1.0, + 0.5, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.6666666666666666, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.5, + 1.0, + 1.0, + 0.0, + 1.0, + 0.5, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.5, + 0.6666666666666666, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.25, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 1.0, + 0.25, + 1.0, + 1.0, + 0.3333333333333333, + 0.0, + 0.5 + ], + "MRR": [ + 0.125, + 0.5, + 0.3333333333333333, + 0.14285714285714285, + 0.14285714285714285, + 0.3333333333333333, + 0.023255813953488372, + 0.5, + 1.0, + 0.0058823529411764705, + 0.5, + 0.125, + 0.07692307692307693, + 0.3333333333333333, + 0.09090909090909091, + 0.058823529411764705, + 0.0, + 0.5, + 0.125, + 0.03333333333333333, + 0.5, + 0.5, + 0.125, + 0.02702702702702703, + 0.011904761904761904, + 0.125, + 0.1, + 1.0, + 1.0, + 0.07142857142857142, + 0.07142857142857142, + 0.3333333333333333, + 1.0, + 0.058823529411764705, + 0.2, + 0.034482758620689655, + 0.008547008547008548, + 0.0, + 0.1, + 0.14285714285714285, + 1.0, + 1.0, + 0.034482758620689655, + 0.07692307692307693, + 0.25, + 0.1, + 0.2, + 0.1, + 0.3333333333333333, + 0.0, + 1.0, + 0.14285714285714285, + 0.5, + 0.16666666666666666, + 0.5, + 0.0, + 0.1111111111111111, + 0.045454545454545456, + 0.1111111111111111, + 0.3333333333333333, + 0.03225806451612903, + 0.14285714285714285, + 0.16666666666666666, + 0.125, + 0.5, + 0.058823529411764705, + 1.0, + 0.024390243902439025, + 0.009615384615384616, + 0.25, + 0.2, + 0.14285714285714285, + 0.2, + 0.02631578947368421, + 0.07692307692307693, + 0.02564102564102564, + 0.04, + 0.045454545454545456, + 0.01694915254237288, + 1.0, + 0.5, + 0.5, + 0.3333333333333333, + 0.0, + 0.3333333333333333, + 0.3333333333333333, + 0.16666666666666666, + 0.05263157894736842, + 0.5, + 0.047619047619047616, + 0.2, + 0.03571428571428571, + 0.011363636363636364, + 0.14285714285714285, + 0.038461538461538464, + 0.05, + 0.5, + 0.0, + 0.058823529411764705, + 0.0, + 1.0, + 0.02857142857142857, + 0.16666666666666666, + 0.16666666666666666, + 0.5, + 0.5, + 1.0, + 0.017241379310344827, + 0.0, + 0.09090909090909091, + 0.5, + 0.03125, + 0.016129032258064516, + 0.14285714285714285, + 0.058823529411764705, + 0.0, + 1.0, + 0.5, + 0.0, + 0.023809523809523808, + 0.043478260869565216, + 0.014925373134328358, + 0.1111111111111111, + 0.0, + 0.01282051282051282, + 0.07692307692307693, + 0.08333333333333333, + 0.2, + 0.047619047619047616, + 0.0, + 0.08333333333333333, + 1.0, + 0.03225806451612903, + 0.034482758620689655, + 0.029411764705882353, + 0.5, + 0.3333333333333333, + 0.2, + 1.0, + 0.5, + 0.2, + 0.5, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 0.04, + 1.0 + ] + }, + "repo-map 2k": { + "present": [ + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0 + ], + "recall": [ + 0.5, + 0.0, + 0.3333333333333333, + 0.0, + 0.75, + 0.25, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.5, + 0.5, + 0.5, + 0.0, + 0.5, + 0.0, + 1.0, + 0.5, + 0.5, + 0.5, + 1.0, + 0.5, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.5, + 0.0, + 0.5, + 0.3333333333333333, + 1.0, + 0.5, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.5, + 0.5, + 0.0, + 0.5, + 0.6666666666666666, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.75, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.25, + 1.0, + 1.0, + 0.6666666666666666, + 0.5, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.5, + 0.6666666666666666, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.5, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.25, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5 + ] + }, + "repo-map 2k query-aware": { + "present": [ + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0 + ], + "recall": [ + 0.5, + 0.0, + 0.3333333333333333, + 0.0, + 0.25, + 0.25, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.5, + 0.5, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 0.5, + 0.5, + 0.25, + 0.5, + 0.5, + 0.5, + 1.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.5, + 0.0, + 0.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.0, + 0.5, + 0.3333333333333333, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.75, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.0, + 0.0, + 0.0, + 0.3333333333333333, + 0.0, + 0.5, + 0.25, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 0.6666666666666666, + 1.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.0, + 0.6666666666666666, + 0.5, + 0.0, + 0.0, + 0.0, + 0.0, + 0.5, + 0.0, + 0.0, + 0.5, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.25, + 0.0, + 0.0, + 0.0, + 1.0, + 0.5 + ] + }, + "repo-map 8k": { + "present": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0 + ], + "recall": [ + 1.0, + 0.5, + 1.0, + 0.6666666666666666, + 1.0, + 0.75, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.75, + 1.0, + 1.0, + 0.5, + 0.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.6666666666666666, + 1.0, + 0.5, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.6666666666666666, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.5 + ] + }, + "repo-map 8k query-aware": { + "present": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "recall": [ + 1.0, + 0.5, + 1.0, + 0.6666666666666666, + 1.0, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.75, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.6666666666666666, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 0.5, + 1.0, + 0.5, + 0.5, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 0.0, + 0.6666666666666666, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.3333333333333333, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 0.75, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 0.5, + 1.0, + 0.5, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 0.5 + ] + } + }, + "tokens_per_query": { + "index": [ + 4752, + 3963, + 3235, + 2620, + 3410, + 4526, + 4066, + 3907, + 2873, + 3324, + 4146, + 4758, + 4769, + 4478, + 3009, + 4731, + 1811, + 2891, + 4752, + 4904, + 3965, + 3181, + 4752, + 3609, + 2906, + 1098, + 3373, + 1602, + 3397, + 4635, + 890, + 1423, + 3947, + 4220, + 4103, + 2308, + 4554, + 5202, + 3567, + 3671, + 3216, + 3008, + 2991, + 3934, + 4115, + 2646, + 3519, + 3442, + 4910, + 3681, + 2581, + 2726, + 2578, + 3396, + 4143, + 2125, + 3637, + 2685, + 2319, + 2880, + 4193, + 3821, + 4096, + 5243, + 3689, + 2729, + 3276, + 4323, + 3937, + 3903, + 4672, + 4415, + 3665, + 4613, + 5390, + 3233, + 3613, + 2700, + 3297, + 3061, + 3760, + 3512, + 4516, + 2671, + 1201, + 2400, + 3514, + 5102, + 2825, + 2402, + 3478, + 2798, + 2704, + 3277, + 3938, + 4324, + 3071, + 3932, + 4020, + 369, + 3728, + 3365, + 4696, + 4300, + 4056, + 2834, + 4978, + 2685, + 3196, + 3459, + 1400, + 2734, + 4304, + 3329, + 3556, + 3718, + 2799, + 3449, + 2781, + 4346, + 5105, + 3206, + 3943, + 4433, + 4555, + 4269, + 4965, + 2331, + 3469, + 3494, + 2835, + 3503, + 4064, + 4302, + 3415, + 4383, + 1974, + 7514, + 4803, + 5023, + 2962, + 4249, + 2529, + 3951, + 3902, + 1576, + 620, + 2089, + 4546, + 3655 + ], + "index (uncapped reads)": [ + 4752, + 3963, + 3235, + 2620, + 3396, + 4515, + 4066, + 3907, + 2873, + 3324, + 4146, + 4758, + 4769, + 6536, + 3009, + 10609, + 1811, + 2891, + 4752, + 4904, + 3965, + 3181, + 4752, + 3609, + 2906, + 1098, + 3345, + 1588, + 3397, + 4621, + 890, + 1423, + 3947, + 4220, + 9981, + 2308, + 11715, + 5202, + 3567, + 4739, + 3216, + 3008, + 2991, + 3900, + 4231, + 2646, + 3519, + 3442, + 4896, + 3675, + 2581, + 2726, + 2578, + 3396, + 4143, + 2125, + 3637, + 2685, + 2319, + 2880, + 4485, + 3821, + 4096, + 5237, + 8541, + 2715, + 6407, + 9161, + 8789, + 3903, + 4672, + 4409, + 3665, + 4613, + 13387, + 3233, + 3613, + 2700, + 3297, + 3061, + 3760, + 3512, + 4516, + 2671, + 1201, + 2400, + 3514, + 5102, + 2811, + 2402, + 3478, + 2798, + 2704, + 3277, + 3938, + 9176, + 3071, + 3932, + 4020, + 369, + 3728, + 3365, + 5789, + 7431, + 4172, + 2834, + 4978, + 2685, + 3196, + 3459, + 1400, + 2734, + 4290, + 6474, + 3556, + 3718, + 2799, + 3449, + 2781, + 4346, + 5105, + 3206, + 3943, + 4433, + 4541, + 4269, + 4965, + 2331, + 3469, + 3494, + 2835, + 3503, + 8916, + 4288, + 3415, + 9235, + 1974, + 7514, + 4803, + 5023, + 2962, + 4249, + 2529, + 3951, + 3902, + 1576, + 620, + 2089, + 6604, + 4737 + ], + "rg+window": [ + 2855, + 3604, + 2856, + 3407, + 3366, + 2340, + 3020, + 3380, + 3074, + 3089, + 2942, + 2855, + 2471, + 3606, + 4411, + 3212, + 1729, + 3242, + 2855, + 3351, + 2820, + 3733, + 2855, + 3450, + 3499, + 2276, + 3905, + 350, + 3076, + 3802, + 2590, + 3131, + 3810, + 2456, + 3351, + 3894, + 2935, + 1164, + 3491, + 3602, + 3113, + 2951, + 3846, + 3315, + 3278, + 3229, + 3034, + 2682, + 3611, + 4194, + 3306, + 3186, + 3632, + 4434, + 3580, + 3292, + 3145, + 3812, + 3520, + 3453, + 4198, + 3449, + 4232, + 3418, + 3499, + 4052, + 3625, + 3724, + 2701, + 3948, + 2899, + 3494, + 2921, + 4025, + 4644, + 4252, + 4236, + 3072, + 3117, + 3619, + 3511, + 3561, + 3578, + 3808, + 2873, + 3357, + 3464, + 3756, + 3787, + 2911, + 2759, + 3252, + 3310, + 4036, + 4076, + 3530, + 3551, + 2966, + 3066, + 3207, + 3303, + 2594, + 3597, + 3445, + 3060, + 3435, + 3442, + 3119, + 2798, + 3005, + 2894, + 2786, + 2869, + 3008, + 2758, + 3842, + 3973, + 3603, + 2483, + 3243, + 3631, + 3712, + 2493, + 3178, + 3671, + 3457, + 2848, + 2429, + 3027, + 2825, + 3706, + 3237, + 3770, + 3565, + 3872, + 3774, + 3052, + 3056, + 3707, + 3366, + 2947, + 3689, + 2770, + 3227, + 2748, + 3231, + 3224, + 3387, + 2898, + 3822 + ], + "repo-map 2k": [ + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000 + ], + "repo-map 2k query-aware": [ + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000 + ], + "repo-map 8k": [ + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000 + ], + "repo-map 8k query-aware": [ + 8000, + 7996, + 7998, + 7999, + 8000, + 7998, + 8000, + 7999, + 8000, + 7999, + 8000, + 8000, + 8000, + 7999, + 8000, + 8000, + 8000, + 7999, + 8000, + 7997, + 7999, + 8000, + 8000, + 8000, + 7998, + 8000, + 7997, + 8000, + 8000, + 7996, + 8000, + 8000, + 7999, + 7997, + 8000, + 8000, + 8000, + 8000, + 8000, + 7998, + 8000, + 7996, + 8000, + 7999, + 7998, + 7996, + 8000, + 8000, + 8000, + 8000, + 7998, + 8000, + 7998, + 7999, + 7997, + 8000, + 8000, + 7998, + 7999, + 8000, + 8000, + 7997, + 8000, + 7999, + 7998, + 8000, + 7998, + 8000, + 8000, + 7997, + 8000, + 7999, + 8000, + 7999, + 7999, + 8000, + 7997, + 8000, + 7998, + 8000, + 7996, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 7998, + 8000, + 7997, + 7997, + 8000, + 8000, + 8000, + 7996, + 8000, + 8000, + 8000, + 7997, + 7998, + 7996, + 7998, + 8000, + 8000, + 8000, + 7996, + 8000, + 8000, + 7996, + 7998, + 8000, + 8000, + 8000, + 8000, + 8000, + 7999, + 8000, + 7999, + 7997, + 7997, + 7999, + 7997, + 8000, + 8000, + 8000, + 8000, + 7999, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 7996, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 7999, + 8000, + 7999, + 8000, + 8000 + ] + }, + "packet_tokens_per_query": { + "index": [ + 2946, + 2841, + 2765, + 2475, + 2622, + 2672, + 2718, + 2769, + 2322, + 2859, + 2913, + 2952, + 3035, + 2590, + 2341, + 2684, + 1469, + 2653, + 2946, + 2707, + 2703, + 2693, + 2946, + 2512, + 2664, + 970, + 2238, + 1529, + 2577, + 2629, + 380, + 1106, + 2662, + 2714, + 2440, + 1486, + 2797, + 2694, + 2775, + 2543, + 2308, + 2825, + 2753, + 2637, + 2942, + 2560, + 2888, + 2875, + 2551, + 2324, + 2106, + 2528, + 2314, + 2043, + 2538, + 1162, + 2482, + 1642, + 1792, + 2409, + 2605, + 2451, + 2400, + 2610, + 2036, + 2487, + 2239, + 2670, + 2294, + 2533, + 2666, + 2311, + 2751, + 2508, + 2853, + 2553, + 2653, + 2323, + 2073, + 2370, + 2863, + 2454, + 2506, + 2432, + 1201, + 2295, + 2906, + 2801, + 2095, + 1197, + 2870, + 2513, + 1859, + 2838, + 2732, + 2661, + 2701, + 2641, + 2839, + 369, + 2636, + 2669, + 2665, + 2464, + 2436, + 2128, + 2725, + 2443, + 2374, + 2320, + 1400, + 2238, + 2824, + 2130, + 2688, + 2411, + 2354, + 2769, + 2651, + 2620, + 2850, + 2753, + 3170, + 2589, + 2772, + 2655, + 2957, + 2025, + 2861, + 2886, + 2227, + 2559, + 2411, + 2622, + 2698, + 2730, + 1366, + 2584, + 2893, + 2526, + 2413, + 2702, + 2456, + 2762, + 2800, + 951, + 620, + 1194, + 2563, + 2431 + ], + "index (uncapped reads)": [ + 2946, + 2841, + 2765, + 2475, + 2608, + 2658, + 2718, + 2769, + 2322, + 2859, + 2913, + 2952, + 3035, + 2548, + 2341, + 2670, + 1469, + 2653, + 2946, + 2707, + 2703, + 2693, + 2946, + 2512, + 2664, + 970, + 2210, + 1515, + 2577, + 2615, + 380, + 1106, + 2662, + 2714, + 2426, + 1486, + 2783, + 2694, + 2775, + 2515, + 2308, + 2825, + 2753, + 2595, + 2928, + 2560, + 2888, + 2875, + 2537, + 2310, + 2106, + 2528, + 2314, + 2043, + 2538, + 1162, + 2482, + 1642, + 1792, + 2409, + 2591, + 2451, + 2400, + 2596, + 2022, + 2473, + 2211, + 2642, + 2280, + 2533, + 2666, + 2297, + 2751, + 2508, + 2825, + 2553, + 2653, + 2323, + 2073, + 2370, + 2863, + 2454, + 2506, + 2432, + 1201, + 2295, + 2906, + 2801, + 2081, + 1197, + 2870, + 2513, + 1859, + 2838, + 2732, + 2647, + 2701, + 2641, + 2839, + 369, + 2636, + 2669, + 2637, + 2436, + 2422, + 2128, + 2725, + 2443, + 2374, + 2320, + 1400, + 2238, + 2810, + 2116, + 2688, + 2411, + 2354, + 2769, + 2651, + 2620, + 2850, + 2753, + 3170, + 2589, + 2758, + 2655, + 2957, + 2025, + 2861, + 2886, + 2227, + 2559, + 2397, + 2608, + 2698, + 2716, + 1366, + 2584, + 2893, + 2526, + 2413, + 2702, + 2456, + 2762, + 2800, + 951, + 620, + 1194, + 2521, + 2417 + ], + "rg+window": [ + 1047, + 1626, + 1208, + 1119, + 1244, + 926, + 1107, + 876, + 1258, + 1194, + 1242, + 1047, + 848, + 1333, + 1235, + 1095, + 246, + 1545, + 1047, + 1360, + 1149, + 1417, + 1047, + 1329, + 1383, + 996, + 1201, + 30, + 1216, + 974, + 1123, + 1010, + 1436, + 995, + 1040, + 1056, + 1326, + 37, + 1021, + 990, + 1221, + 1247, + 1218, + 862, + 1250, + 1184, + 1148, + 962, + 1043, + 1103, + 1345, + 1391, + 1121, + 1149, + 1034, + 1518, + 1401, + 1089, + 1503, + 1384, + 1214, + 1134, + 1348, + 867, + 1015, + 1507, + 1806, + 1227, + 1115, + 1320, + 1239, + 832, + 1118, + 1394, + 1987, + 1181, + 1442, + 1106, + 1262, + 1813, + 1863, + 1002, + 1159, + 1827, + 1022, + 1034, + 1756, + 940, + 1282, + 1089, + 1176, + 1157, + 1177, + 1310, + 985, + 1009, + 1067, + 1390, + 1229, + 1475, + 1513, + 866, + 990, + 1664, + 1135, + 1656, + 1357, + 880, + 1302, + 1374, + 958, + 1126, + 1145, + 1172, + 994, + 1184, + 1157, + 1075, + 181, + 1171, + 1116, + 1375, + 912, + 1382, + 1330, + 1116, + 1314, + 830, + 1095, + 1197, + 1541, + 1048, + 1260, + 1134, + 1045, + 1154, + 1233, + 979, + 973, + 894, + 1157, + 1184, + 1113, + 1389, + 941, + 1675, + 547, + 1513, + 878, + 2111 + ], + "repo-map 2k": [ + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000 + ], + "repo-map 2k query-aware": [ + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 1999, + 1999, + 1999, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 2000, + 1999, + 1999, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 2000, + 2000, + 1999, + 2000, + 2000, + 1999, + 2000, + 2000 + ], + "repo-map 8k": [ + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000 + ], + "repo-map 8k query-aware": [ + 8000, + 7996, + 7998, + 7999, + 8000, + 7998, + 8000, + 7999, + 8000, + 7999, + 8000, + 8000, + 8000, + 7999, + 8000, + 8000, + 8000, + 7999, + 8000, + 7997, + 7999, + 8000, + 8000, + 8000, + 7998, + 8000, + 7997, + 8000, + 8000, + 7996, + 8000, + 8000, + 7999, + 7997, + 8000, + 8000, + 8000, + 8000, + 8000, + 7998, + 8000, + 7996, + 8000, + 7999, + 7998, + 7996, + 8000, + 8000, + 8000, + 8000, + 7998, + 8000, + 7998, + 7999, + 7997, + 8000, + 8000, + 7998, + 7999, + 8000, + 8000, + 7997, + 8000, + 7999, + 7998, + 8000, + 7998, + 8000, + 8000, + 7997, + 8000, + 7999, + 8000, + 7999, + 7999, + 8000, + 7997, + 8000, + 7998, + 8000, + 7996, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 7998, + 8000, + 7997, + 7997, + 8000, + 8000, + 8000, + 7996, + 8000, + 8000, + 8000, + 7997, + 7998, + 7996, + 7998, + 8000, + 8000, + 8000, + 7996, + 8000, + 8000, + 7996, + 7998, + 8000, + 8000, + 8000, + 8000, + 8000, + 7999, + 8000, + 7999, + 7997, + 7997, + 7999, + 7997, + 8000, + 8000, + 8000, + 8000, + 7999, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 7996, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 8000, + 7999, + 8000, + 7999, + 8000, + 8000 + ] + } + } + ], + "pooled": { + "n_queries": 450, + "summary": { + "index": { + "hit@3": 0.5466666666666666, + "recall@5": 0.5246296296296297, + "MRR": 0.45615079365079364, + "tokens_mean": 3835.18, + "tokens_median": 3860.5, + "packet_tokens_mean": 2401.56 + }, + "index (uncapped reads)": { + "hit@3": 0.5466666666666666, + "recall@5": 0.5246296296296297, + "MRR": 0.45615079365079364, + "tokens_mean": 6803.142222222222, + "tokens_median": 4207.0, + "packet_tokens_mean": 2385.2533333333336 + }, + "rg+window": { + "hit@3": 0.30444444444444446, + "recall@5": 0.2931481481481481, + "MRR": 0.2626218659670105, + "tokens_mean": 3495.448888888889, + "tokens_median": 3442.0, + "packet_tokens_mean": 1350.6444444444444 + }, + "repo-map 2k": { + "present": 0.3844444444444444, + "recall": 0.2638888888888889, + "tokens_mean": 1999.3333333333333, + "tokens_median": 2000.0, + "packet_tokens_mean": 1999.3333333333333 + }, + "repo-map 2k query-aware": { + "present": 0.4177777777777778, + "recall": 0.3259259259259259, + "tokens_mean": 1999.5177777777778, + "tokens_median": 2000.0, + "packet_tokens_mean": 1999.5177777777778 + }, + "repo-map 8k": { + "present": 0.9911111111111112, + "recall": 0.9677777777777777, + "tokens_mean": 7495.0, + "tokens_median": 7728.0, + "packet_tokens_mean": 7495.0 + }, + "repo-map 8k query-aware": { + "present": 0.9644444444444444, + "recall": 0.9396296296296296, + "tokens_mean": 7494.713333333333, + "tokens_median": 7728.0, + "packet_tokens_mean": 7494.713333333333 + } + }, + "index_vs_rg": { + "hit@3": { + "delta": 0.24222222222222223, + "ci_lo": 0.18666666666666668, + "ci_hi": 0.29777777777777775, + "p": 0.0 + }, + "recall@5": { + "delta": 0.23148148148148148, + "ci_lo": 0.18296296296296308, + "ci_hi": 0.2779629629629629, + "p": 0.0 + }, + "MRR": { + "delta": 0.19352892768378316, + "ci_lo": 0.15148047527838937, + "ci_hi": 0.23548311127822746, + "p": 0.0 + }, + "tokens": { + "delta": 339.7311111111111, + "ci_lo": 238.13111111111112, + "ci_hi": 439.5511111111111, + "p": 0.0 + } + } + } +} \ No newline at end of file diff --git a/tests/eval/results/2026-09-04-public-baselines.md b/tests/eval/results/2026-09-04-public-baselines.md new file mode 100644 index 0000000..856a769 --- /dev/null +++ b/tests/eval/results/2026-09-04-public-baselines.md @@ -0,0 +1,104 @@ +# Index vs grep-style and repo-map-style baselines on public repositories + +- **Date:** 2026-09-04 +- **codebase-index:** 1.9.0 +- **Tokenizer (both sides):** tiktoken/cl100k_base +- **ripgrep:** ripgrep 14.1.1 (rev 4649aa9700) +- **Platform:** Windows 11 / Python 3.12.12 +- **Read model:** top-3 follow-through reads, 80-line grep windows, `rg` listing capped at 50 lines; repo maps packed under 2000 / 8000 tokens +- **Ground truth:** commit subject → files that commit changed (`tests/eval/gen_queries.py`), newest N localised commits per repository; changelog-style files excluded from both answers and corpus + +## Corpora + +| corpus | commit | files indexed | symbols | edges | queries | index build | +|---|---|---:|---:|---:|---:|---:| +| flask | `d318b6834711` | 229 | 1622 | 4425 | 150 | 3.2 s | +| gson | `b3f4ca20087f` | 311 | 4193 | 22407 | 150 | 5.3 s | +| fastify | `15ebc8e2fe69` | 389 | 1185 | 34297 | 150 | 2.0 s | + +### Pooled — ranked retrieval (n=450) + +| method | hit@3 | recall@5 | MRR | packet / listing tokens | + top-3 reads (mean) | + top-3 reads (median) | +|---|---:|---:|---:|---:|---:|---:| +| index | 0.547 | 0.525 | 0.456 | 2,402 | 3,835 | 3,860 | +| index (uncapped reads) | 0.547 | 0.525 | 0.456 | 2,385 | 6,803 | 4,207 | +| rg+window | 0.304 | 0.293 | 0.263 | 1,351 | 3,495 | 3,442 | + +### Pooled — repo-map-style context (n=450) + +| method | answer present in map | tokens/query | +|---|---:|---:| +| repo-map 2k | 0.384 | 1,999 | +| repo-map 2k query-aware | 0.418 | 2,000 | +| repo-map 8k | 0.991 | 7,495 | +| repo-map 8k query-aware | 0.964 | 7,495 | + +### Index vs rg+window, paired significance (pooled) + +| metric | delta (index − rg) | 95% CI | p | significant | +|---|---:|---:|---:|---| +| hit@3 | +0.242 | [+0.187, +0.298] | 0.000 | yes | +| recall@5 | +0.231 | [+0.183, +0.278] | 0.000 | yes | +| MRR | +0.194 | [+0.151, +0.235] | 0.000 | yes | +| tokens | 340 | [238, 440] | 0.000 | yes | + +### flask — ranked retrieval (n=150) + +| method | hit@3 | recall@5 | MRR | packet / listing tokens | + top-3 reads (mean) | + top-3 reads (median) | +|---|---:|---:|---:|---:|---:|---:| +| index | 0.487 | 0.503 | 0.383 | 2,359 | 3,591 | 3,576 | +| index (uncapped reads) | 0.487 | 0.503 | 0.383 | 2,350 | 5,724 | 3,776 | +| rg+window | 0.233 | 0.245 | 0.229 | 1,009 | 3,079 | 3,187 | + +### flask — repo-map-style context (n=150) + +| method | answer present in map | tokens/query | +|---|---:|---:| +| repo-map 2k | 0.313 | 1,998 | +| repo-map 2k query-aware | 0.273 | 1,999 | +| repo-map 8k | 1.000 | 7,728 | +| repo-map 8k query-aware | 1.000 | 7,728 | + +### gson — ranked retrieval (n=150) + +| method | hit@3 | recall@5 | MRR | packet / listing tokens | + top-3 reads (mean) | + top-3 reads (median) | +|---|---:|---:|---:|---:|---:|---:| +| index | 0.573 | 0.523 | 0.503 | 2,418 | 4,375 | 4,376 | +| index (uncapped reads) | 0.573 | 0.523 | 0.503 | 2,382 | 10,659 | 8,680 | +| rg+window | 0.347 | 0.337 | 0.290 | 1,865 | 4,117 | 4,092 | + +### gson — repo-map-style context (n=150) + +| method | answer present in map | tokens/query | +|---|---:|---:| +| repo-map 2k | 0.480 | 2,000 | +| repo-map 2k query-aware | 0.613 | 2,000 | +| repo-map 8k | 1.000 | 6,757 | +| repo-map 8k query-aware | 1.000 | 6,757 | + +### fastify — ranked retrieval (n=150) + +| method | hit@3 | recall@5 | MRR | packet / listing tokens | + top-3 reads (mean) | + top-3 reads (median) | +|---|---:|---:|---:|---:|---:|---:| +| index | 0.580 | 0.547 | 0.482 | 2,428 | 3,539 | 3,562 | +| index (uncapped reads) | 0.580 | 0.547 | 0.482 | 2,423 | 4,027 | 3,611 | +| rg+window | 0.333 | 0.297 | 0.269 | 1,178 | 3,290 | 3,312 | + +### fastify — repo-map-style context (n=150) + +| method | answer present in map | tokens/query | +|---|---:|---:| +| repo-map 2k | 0.360 | 2,000 | +| repo-map 2k query-aware | 0.367 | 2,000 | +| repo-map 8k | 0.973 | 8,000 | +| repo-map 8k query-aware | 0.893 | 7,999 | + +## How to read this + +- `hit@3` is the metric that matters for an agent that opens the top three files. `recall@5` credits multi-file answers. `MRR` rewards putting the answer first. +- `packet / listing tokens` is what the agent sees before opening anything: the full JSON search payload for the index (snippets included), the first 50 `rg` lines for grep. `+ top-3 reads` adds the follow-through: the index's top-3 `recommended_reads` ranges in full, or three 80-line grep windows. Same tokenizer on every side. +- The index packet already carries budgeted snippets, so an agent that answers from the packet pays the first column only; an agent that re-reads every recommended range pays the second. Grep has no first-column equivalent — the listing alone rarely answers anything. +- A repo map is not a ranking, so it is scored on whether the answer file is *present* at all — an upper bound on what an agent could do with it. +- Ground truth is one commit's files; a query can have a correct answer the commit did not touch. Absolute numbers understate every method equally; read the *deltas*, and read the CI before the delta. +- Latency is deliberately not tabulated: the index runs in-process, ripgrep is a separate binary; neither number is a fair claim against the other. +- Not measured here: an LLM-driven agent exploring the repository. That needs model calls and is tracked as future work in `docs/BENCHMARKS.md`. diff --git a/tests/eval/run_baselines.py b/tests/eval/run_baselines.py new file mode 100644 index 0000000..ee305be --- /dev/null +++ b/tests/eval/run_baselines.py @@ -0,0 +1,348 @@ +#!/usr/bin/env python3 +"""Compare the index against grep-style and repo-map-style baselines on public repositories. + + # the pinned public corpora (cloned into --workdir at fixed commits) + python tests/eval/run_baselines.py --clone --workdir .tmp-baselines + + # any local git repository instead + python tests/eval/run_baselines.py --repo ../some-service + + # write raw JSON + Markdown next to the other logged runs + python tests/eval/run_baselines.py --clone --out tests/eval/results/public-baselines + +Why this exists +--------------- +`run_eval.py` measures the index against *its own earlier versions*; it can say +a ranking change helped, never that an index beats not having one. This script +asks the second question on repositories anyone can clone, at commits recorded +in the output, with ground truth mined from git history (`gen_queries.py`) so +neither the queries nor the answers are chosen by hand or by the retriever. + +What is compared (see `baselines.py` for the exact read model) +------------------------------------------------------------- +* index — `search()` with the shipped default tuning +* index, uncapped — same ranking with `max_read_lines=0` (the pre-1.9.1 read plan) +* rg + window — salient-term ripgrep, density-ranked, 80-line windows +* repo-map (2k/8k) — signature map packed under a budget, degree-ranked +* repo-map, query-aware — same map, files matching the question's identifiers first + +Every side is charged with the same tokenizer for the text that enters context. +Latency is context only. Significance is reported per metric with a paired +bootstrap CI and a paired permutation p-value against the rg baseline. +""" + +from __future__ import annotations + +import argparse +import json +import platform +import statistics +import subprocess +import sys +import tempfile +import time +from datetime import datetime, timezone +from pathlib import Path + +if sys.platform == "win32": # the report uses arrows/dashes; never die on a cp1251 console + sys.stdout.reconfigure(encoding="utf-8") # type: ignore[union-attr] + +REPO_ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(REPO_ROOT / "tests")) +sys.path.insert(0, str(REPO_ROOT / "src")) + +from codebase_index import __version__ # noqa: E402 +from eval import baselines, gen_queries, harness, metrics # noqa: E402 + +# Public corpora: three languages, three sizes, pinned so a re-run sees the same +# tree and the same history. Update the SHA deliberately and re-log the run. +PUBLIC_CORPORA = { + "flask": ("https://github.com/pallets/flask.git", "d318b683471101618febed18996405ad26462110"), + "gson": ("https://github.com/google/gson.git", "b3f4ca20087f9066de4c340522ff84e0558e1ad1"), + "fastify": ("https://github.com/fastify/fastify.git", "15ebc8e2fe6932d3e91afa84709523804f8a2537"), +} + + +def _git(*args: str, cwd: Path | None = None) -> str: + return subprocess.run( + ["git", *args], cwd=cwd, capture_output=True, text=True, + encoding="utf-8", errors="replace", check=True, + ).stdout.strip() + + +def _clone_pinned(name: str, url: str, sha: str, workdir: Path) -> Path: + dest = workdir / name + if not (dest / ".git").is_dir(): + print(f"cloning {url} -> {dest}", flush=True) + _git("clone", "--quiet", url, str(dest)) + head = _git("rev-parse", "HEAD", cwd=dest) + if head != sha: + _git("fetch", "--quiet", "origin", sha, cwd=dest) + _git("checkout", "--quiet", sha, cwd=dest) + return dest + + +def _sig(base: list[float], cand: list[float], *, resamples: int) -> dict[str, float]: + deltas = [c - b for b, c in zip(base, cand)] + delta, lo, hi = metrics.paired_bootstrap_ci(deltas, resamples=resamples) + p = metrics.paired_permutation_p(deltas, resamples=resamples) + return {"delta": delta, "ci_lo": lo, "ci_hi": hi, "p": p} + + +def run_corpus( + name: str, + root: Path, + queries: list[harness.EvalQuery], + *, + tmp: Path, +) -> dict: + corpus = baselines.Corpus(root) + t0 = time.perf_counter() + db = harness.build_corpus_index(root, tmp / f"{name}.sqlite") + build_ms = (time.perf_counter() - t0) * 1000.0 + try: + conn = db.conn + n_files = conn.execute("SELECT COUNT(*) FROM files").fetchone()[0] + n_symbols = conn.execute("SELECT COUNT(*) FROM symbols").fetchone()[0] + n_edges = conn.execute("SELECT COUNT(*) FROM edges").fetchone()[0] + map_entries = baselines.build_repo_map_entries(conn) + + per_side: dict[str, dict[str, list[float]]] = {} + tokens: dict[str, list[int]] = {} + packet: dict[str, list[int]] = {} + latency: dict[str, list[float]] = {} + + def record(side: str, out: baselines.BaselineOutcome, scores: dict[str, float]) -> None: + for k, v in scores.items(): + per_side.setdefault(side, {}).setdefault(k, []).append(v) + tokens.setdefault(side, []).append(out.tokens) + packet.setdefault(side, []).append(out.packet_tokens) + latency.setdefault(side, []).append(out.latency_ms) + + for q in queries: + idx = baselines.run_index(conn, corpus, q) + record("index", idx, baselines.score(idx, q)) + # Same ranking, read plan uncapped (pre-1.9.1 behaviour): isolates what + # the `max_read_lines` cap saves without touching quality. + raw = baselines.run_index(conn, corpus, q, max_read_lines=0) + record("index (uncapped reads)", raw, baselines.score(raw, q)) + rg = baselines.run_rg_window(corpus, q) + record("rg+window", rg, baselines.score(rg, q)) + for budget in baselines.REPO_MAP_BUDGETS: + for aware in (False, True): + label = f"repo-map {budget // 1000}k" + (" query-aware" if aware else "") + rm = baselines.run_repo_map(map_entries, q, budget=budget, query_aware=aware) + record(label, rm, baselines.score(rm, q, present_only=True)) + finally: + db.close() + + summary = {} + for side, cols in per_side.items(): + summary[side] = {k: statistics.fmean(v) for k, v in cols.items()} + summary[side]["tokens_mean"] = statistics.fmean(tokens[side]) + summary[side]["tokens_median"] = statistics.median(tokens[side]) + summary[side]["packet_tokens_mean"] = statistics.fmean(packet[side]) + summary[side]["latency_p50_ms"] = metrics.percentile(latency[side], 50) + return { + "corpus": name, + "root": str(root), + "head": _git("rev-parse", "HEAD", cwd=root), + "n_queries": len(queries), + "index_build_ms": build_ms, + "files_indexed": n_files, + "symbols_indexed": n_symbols, + "edges_indexed": n_edges, + "summary": summary, + "per_query": per_side, + "tokens_per_query": tokens, + "packet_tokens_per_query": packet, + } + + +def pooled(corpora: list[dict], *, resamples: int) -> dict: + sides = list(corpora[0]["per_query"].keys()) + per_side: dict[str, dict[str, list[float]]] = {} + tokens: dict[str, list[int]] = {} + packet: dict[str, list[int]] = {} + for c in corpora: + for side in sides: + for k, v in c["per_query"][side].items(): + per_side.setdefault(side, {}).setdefault(k, []).extend(v) + tokens.setdefault(side, []).extend(c["tokens_per_query"][side]) + packet.setdefault(side, []).extend(c["packet_tokens_per_query"][side]) + summary = {} + for side, cols in per_side.items(): + summary[side] = {k: statistics.fmean(v) for k, v in cols.items()} + summary[side]["tokens_mean"] = statistics.fmean(tokens[side]) + summary[side]["tokens_median"] = statistics.median(tokens[side]) + summary[side]["packet_tokens_mean"] = statistics.fmean(packet[side]) + significance = { + k: _sig(per_side["rg+window"][k], per_side["index"][k], resamples=resamples) + for k in baselines.METRICS + } + significance["tokens"] = _sig( + [float(t) for t in tokens["rg+window"]], [float(t) for t in tokens["index"]], + resamples=resamples, + ) + return {"n_queries": sum(c["n_queries"] for c in corpora), "summary": summary, + "index_vs_rg": significance} + + +def render_markdown(report: dict) -> str: + L: list[str] = [] + L.append("# Index vs grep-style and repo-map-style baselines on public repositories\n") + L.append(f"- **Date:** {report['date']}") + L.append(f"- **codebase-index:** {report['version']}") + L.append(f"- **Tokenizer (both sides):** {report['tokenizer']}") + L.append(f"- **ripgrep:** {report['ripgrep'] or 'not found — Python scan fallback (latency not comparable)'}") + L.append(f"- **Platform:** {report['platform']}") + L.append(f"- **Read model:** top-{baselines.TOP_K} follow-through reads, " + f"{baselines.WINDOW}-line grep windows, `rg` listing capped at " + f"{baselines.LISTING_CAP} lines; repo maps packed under " + f"{' / '.join(str(b) for b in baselines.REPO_MAP_BUDGETS)} tokens") + L.append("- **Ground truth:** commit subject → files that commit changed " + "(`tests/eval/gen_queries.py`), newest N localised commits per repository; " + "changelog-style files excluded from both answers and corpus\n") + + L.append("## Corpora\n") + L.append("| corpus | commit | files indexed | symbols | edges | queries | index build |") + L.append("|---|---|---:|---:|---:|---:|---:|") + for c in report["corpora"]: + L.append(f"| {c['corpus']} | `{c['head'][:12]}` | {c['files_indexed']} | " + f"{c['symbols_indexed']} | {c['edges_indexed']} | {c['n_queries']} | " + f"{c['index_build_ms'] / 1000:.1f} s |") + L.append("") + + def ranked_table(summary: dict, n: int, title: str) -> None: + L.append(f"### {title} — ranked retrieval (n={n})\n") + L.append("| method | hit@3 | recall@5 | MRR | packet / listing tokens | " + "+ top-3 reads (mean) | + top-3 reads (median) |") + L.append("|---|---:|---:|---:|---:|---:|---:|") + for side in ("index", "index (uncapped reads)", "rg+window"): + s = summary[side] + L.append(f"| {side} | {s['hit@3']:.3f} | {s['recall@5']:.3f} | {s['MRR']:.3f} | " + f"{s['packet_tokens_mean']:,.0f} | {s['tokens_mean']:,.0f} | " + f"{s['tokens_median']:,.0f} |") + L.append("") + L.append(f"### {title} — repo-map-style context (n={n})\n") + L.append("| method | answer present in map | tokens/query |") + L.append("|---|---:|---:|") + for side, s in summary.items(): + if side.startswith("repo-map"): + L.append(f"| {side} | {s['present']:.3f} | {s['tokens_mean']:,.0f} |") + L.append("") + + ranked_table(report["pooled"]["summary"], report["pooled"]["n_queries"], "Pooled") + L.append("### Index vs rg+window, paired significance (pooled)\n") + L.append("| metric | delta (index − rg) | 95% CI | p | significant |") + L.append("|---|---:|---:|---:|---|") + for k, s in report["pooled"]["index_vs_rg"].items(): + fmt = ",.0f" if k == "tokens" else "+.3f" + L.append(f"| {k} | {s['delta']:{fmt}} | [{s['ci_lo']:{fmt}}, {s['ci_hi']:{fmt}}] | " + f"{s['p']:.3f} | {'yes' if s['p'] < 0.05 else 'no'} |") + L.append("") + for c in report["corpora"]: + ranked_table(c["summary"], c["n_queries"], c["corpus"]) + + L.append("## How to read this\n") + L.append("- `hit@3` is the metric that matters for an agent that opens the top three files. " + "`recall@5` credits multi-file answers. `MRR` rewards putting the answer first.") + L.append("- `packet / listing tokens` is what the agent sees before opening anything: " + "the full JSON search payload for the index (snippets included), the " + f"first {baselines.LISTING_CAP} `rg` lines for grep. `+ top-3 reads` adds the " + "follow-through: the index's top-3 `recommended_reads` ranges in full, " + f"or three {baselines.WINDOW}-line grep windows. Same tokenizer on every side.") + L.append("- The index packet already carries budgeted snippets, so an agent that answers " + "from the packet pays the first column only; an agent that re-reads every " + "recommended range pays the second. Grep has no first-column equivalent — the " + "listing alone rarely answers anything.") + L.append("- A repo map is not a ranking, so it is scored on whether the answer file is " + "*present* at all — an upper bound on what an agent could do with it.") + L.append("- Ground truth is one commit's files; a query can have a correct answer the " + "commit did not touch. Absolute numbers understate every method equally; " + "read the *deltas*, and read the CI before the delta.") + L.append("- Latency is deliberately not tabulated: the index runs in-process, ripgrep " + "is a separate binary; neither number is a fair claim against the other.") + L.append("- Not measured here: an LLM-driven agent exploring the repository. That " + "needs model calls and is tracked as future work in `docs/BENCHMARKS.md`.") + return "\n".join(L) + "\n" + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--repo", action="append", default=[], + help="local git repository to benchmark (repeatable)") + ap.add_argument("--clone", action="store_true", + help="clone the pinned public corpora into --workdir") + ap.add_argument("--only", action="append", default=[], + help="restrict --clone to these corpus names") + ap.add_argument("--workdir", type=Path, default=Path(".tmp-baselines")) + ap.add_argument("--queries-per-repo", type=int, default=150, + help="newest N git-derived queries per repository") + ap.add_argument("--max-files", type=int, default=4) + ap.add_argument("--resamples", type=int, default=5000) + ap.add_argument("--out", type=Path, default=None, + help="write .json and .md") + args = ap.parse_args(argv) + + roots: list[tuple[str, Path]] = [] + if args.clone: + args.workdir.mkdir(parents=True, exist_ok=True) + for name, (url, sha) in PUBLIC_CORPORA.items(): + if args.only and name not in args.only: + continue + roots.append((name, _clone_pinned(name, url, sha, args.workdir))) + for r in args.repo: + p = Path(r).resolve() + roots.append((p.name, p)) + if not roots: + ap.error("pass --clone and/or --repo") + + corpora: list[dict] = [] + with tempfile.TemporaryDirectory() as tmp: + for name, root in roots: + records = gen_queries.harvest( + root, max_commits=4000, max_files=args.max_files, min_words=3, + )[: args.queries_per_repo] + queries = [ + harness.EvalQuery( + query=r["query"], category=r["category"], + expected_files=tuple(r["expected_files"]), + ) + for r in records + ] + problems = harness.validate_queries(queries, root) + if problems: + print(f"ground truth invalid for {name}: {problems[:3]}", file=sys.stderr) + return 2 + print(f"{name}: {len(queries)} queries; indexing + running...", flush=True) + corpora.append(run_corpus(name, root, queries, tmp=Path(tmp))) + s = corpora[-1]["summary"] + print(f" index hit@3={s['index']['hit@3']:.3f} tokens={s['index']['tokens_mean']:.0f}") + print(f" rg+window hit@3={s['rg+window']['hit@3']:.3f} tokens={s['rg+window']['tokens_mean']:.0f}") + + report = { + "date": datetime.now(timezone.utc).strftime("%Y-%m-%d"), + "version": __version__, + "tokenizer": baselines.TOKENIZER, + "ripgrep": baselines.rg_version(), + "platform": f"{platform.system()} {platform.release()} / Python {platform.python_version()}", + "corpora": corpora, + "pooled": pooled(corpora, resamples=args.resamples), + } + md = render_markdown(report) + print() + print(md) + if args.out: + args.out.parent.mkdir(parents=True, exist_ok=True) + slim = json.loads(json.dumps(report)) + # Per-query vectors are what a re-run compares against; keep them in the raw file. + args.out.with_suffix(".json").write_text(json.dumps(slim, indent=2), encoding="utf-8") + args.out.with_suffix(".md").write_text(md, encoding="utf-8") + print(f"wrote {args.out.with_suffix('.json')} and {args.out.with_suffix('.md')}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_eval_baselines.py b/tests/test_eval_baselines.py new file mode 100644 index 0000000..79f62fc --- /dev/null +++ b/tests/test_eval_baselines.py @@ -0,0 +1,68 @@ +"""Unit checks for the baseline models in tests/eval/baselines.py. + +These pin the read model the public baseline benchmark relies on, so a change to +salient-term extraction or map packing cannot silently move a logged number. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +from eval import baselines # noqa: E402 +from eval.harness import EvalQuery # noqa: E402 + + +def test_salient_terms_drop_stopwords_and_keep_identifiers(): + terms = baselines.salient_terms("fix how the SessionInterface saves a cookie when not set") + assert "SessionInterface" in terms + assert "cookie" in terms + assert not {"fix", "how", "the", "when", "not", "set"} & {t.lower() for t in terms} + + +def test_salient_terms_are_bounded_and_longest_first(): + terms = baselines.salient_terms("alpha betagamma de epsilonzeta eta theta iota kappa", max_terms=3) + assert len(terms) == 3 + assert terms[0] == "epsilonzeta" + + +def test_tokens_for_reads_merges_overlapping_ranges(tmp_path): + (tmp_path / "a.py").write_text("\n".join(f"line {i}" for i in range(1, 101)), encoding="utf-8") + corpus = baselines.Corpus(tmp_path) + once = baselines.tokens_for_reads(corpus, {"a.py": [(1, 50)]}) + twice = baselines.tokens_for_reads(corpus, {"a.py": [(1, 50), (10, 40)]}) + assert once == twice, "overlapping windows must be charged once" + assert baselines.tokens_for_reads(corpus, {"a.py": [(500, 600)]}) == 0 + + +def test_pack_repo_map_respects_budget_and_query_awareness(): + entries = [ + baselines.MapEntry("a.py", "a.py:\n def alpha()", 10, degree=5, names=frozenset({"alpha"})), + baselines.MapEntry("b.py", "b.py:\n def beta()", 10, degree=50, names=frozenset({"beta"})), + baselines.MapEntry("c.py", "c.py:\n def gamma()", 10, degree=1, names=frozenset({"gamma"})), + ] + chosen, used = baselines.pack_repo_map(entries, budget=20) + assert chosen == ["b.py", "a.py"] and used == 20 + chosen, _ = baselines.pack_repo_map(entries, budget=20, query_terms=["gamma"]) + assert chosen[0] == "c.py" + + +def test_score_ranked_and_present_only(): + q = EvalQuery(query="x", category="c", expected_files=("b.py",)) + out = baselines.BaselineOutcome(["a.py", "b.py"], tokens=0, packet_tokens=0, latency_ms=0) + assert baselines.score(out, q) == {"hit@3": 1.0, "recall@5": 1.0, "MRR": 0.5} + assert baselines.score(out, q, present_only=True)["present"] == 1.0 + + +@pytest.mark.skipif(baselines.RG is None, reason="ripgrep not available") +def test_rg_window_ranks_by_density(tmp_path): + (tmp_path / "dense.py").write_text("token\n" * 30, encoding="utf-8") + (tmp_path / "sparse.py").write_text("token\n" + "x\n" * 30, encoding="utf-8") + corpus = baselines.Corpus(tmp_path) + out = baselines.run_rg_window(corpus, EvalQuery("find the token handling", "c", ("dense.py",))) + assert out.ranked_files[0] == "dense.py" + assert out.tokens > out.packet_tokens > 0 diff --git a/tests/test_gen_queries.py b/tests/test_gen_queries.py index e6e0ebe..6fbec83 100644 --- a/tests/test_gen_queries.py +++ b/tests/test_gen_queries.py @@ -150,3 +150,18 @@ def test_strip_prefix_handles_unprefixed_and_breaking_subjects(): "perf", "speed up walking", ) + + +def test_changes_style_files_are_never_answers(): + """Flask keeps its changelog in CHANGES.rst; it paraphrases commit subjects + exactly like CHANGELOG.md and must be refused as an answer too.""" + for path in ("CHANGES.rst", "docs/CHANGES.md", "HISTORY.rst", "RELEASE_NOTES.md"): + assert not gen_queries._is_answerable(path), path + assert gen_queries._is_answerable("src/flask/sessions.py") + + +def test_changelog_excludes_reach_the_harness(): + from eval import harness + + assert harness.CHANGELOG_EXCLUDES is gen_queries.CHANGELOG_EXCLUDES + assert "CHANGES*" in gen_queries.CHANGELOG_EXCLUDES diff --git a/tests/test_mcp_golden.py b/tests/test_mcp_golden.py index 6c76a3c..de04c64 100644 --- a/tests/test_mcp_golden.py +++ b/tests/test_mcp_golden.py @@ -17,13 +17,11 @@ import pytest from typer.testing import CliRunner -try: - from codebase_index.mcp import server as mcp_server - MCP_AVAILABLE = True -except ImportError: - MCP_AVAILABLE = False - -pytestmark = pytest.mark.skipif(not MCP_AVAILABLE, reason="mcp extra not installed") +# Skip only when the `mcp` SDK itself is absent. Our own server module must import +# cleanly against whatever SDK is installed; wrapping *that* import in a skip once +# hid a full CI run in which every MCP test was silently skipped on mcp 2.x. +pytest.importorskip("mcp", reason="mcp extra not installed") +from codebase_index.mcp import server as mcp_server # noqa: E402 from codebase_index.cli import app # noqa: E402 (after the skip guard) from tests.golden_utils import assert_matches_golden # noqa: E402 diff --git a/tests/test_mcp_server.py b/tests/test_mcp_server.py index fd47394..e98412f 100644 --- a/tests/test_mcp_server.py +++ b/tests/test_mcp_server.py @@ -9,13 +9,11 @@ import pytest -try: - from codebase_index.mcp import server as mcp_server - MCP_AVAILABLE = True -except ImportError: - MCP_AVAILABLE = False - -pytestmark = pytest.mark.skipif(not MCP_AVAILABLE, reason="mcp extra not installed") +# Skip only when the `mcp` SDK itself is absent. Our own server module must import +# cleanly against whatever SDK is installed; wrapping *that* import in a skip once +# hid a full CI run in which every MCP test was silently skipped on mcp 2.x. +pytest.importorskip("mcp", reason="mcp extra not installed") +from codebase_index.mcp import server as mcp_server # noqa: E402 # ── helpers ────────────────────────────────────────────────────────────────── @@ -213,3 +211,30 @@ def test_explain_code_accepts_raw_parameter(): """explain_code accepts raw without raising TypeError.""" result = _with_missing_db(lambda: _call(mcp_server.explain_code, query="foo", raw=True)) assert "error" in result + + +def test_mcp_subcommand_accepts_root_option(monkeypatch, tmp_path): + """Every client template writes `codebase-index mcp --root `; the global + --root lives on the Typer callback, so the subcommand must accept it too or the + documented config fails with "No such option: --root".""" + import os + + from typer.testing import CliRunner + + from codebase_index import cli + + captured: dict = {} + + class FakeServer: + def run(self, transport): # noqa: D401 - mimics FastMCP.run + captured["transport"] = transport + captured["root"] = os.environ.get("CBX_ROOT") + + import codebase_index.mcp.server as server_mod + + monkeypatch.setattr(server_mod, "mcp", FakeServer()) + monkeypatch.delenv("CBX_ROOT", raising=False) + result = CliRunner().invoke(cli.app, ["mcp", "--root", str(tmp_path)]) + assert result.exit_code == 0, result.output + assert captured["transport"] == "stdio" + assert captured["root"] == str(tmp_path.resolve()) diff --git a/tests/test_plugin_wrappers.py b/tests/test_plugin_wrappers.py index 7a3bc97..b92d379 100644 --- a/tests/test_plugin_wrappers.py +++ b/tests/test_plugin_wrappers.py @@ -1,5 +1,6 @@ from __future__ import annotations +import re import shutil import subprocess from pathlib import Path @@ -64,3 +65,26 @@ def test_cbx_execs_venv_cli_via_pointer(tmp_path): ) assert res.returncode == 0, res.stderr assert "CBXSTUB search foo" in res.stdout + + +def _whitelist_sh(path: Path) -> set[str]: + m = re.search(r'^ALLOWED="([^"]+)"', path.read_text(encoding="utf-8"), re.M) + assert m, f"no ALLOWED= line in {path}" + return set(m.group(1).split()) + + +def _whitelist_ps1(path: Path) -> set[str]: + m = re.search(r"\$allowed = @\(([^)]+)\)", path.read_text(encoding="utf-8"), re.S) + assert m, f"no $allowed line in {path}" + return {tok.strip().strip('"') for tok in m.group(1).split(",")} + + +def test_plugin_wrapper_whitelist_matches_skill_template(): + """The plugin `bin/` wrappers must allow exactly what the skill template's + `cbx` allows, or plugin users silently lose commands the SKILL.md advertises + (this drifted once: architecture/diff-impact/path/describe were missing).""" + template = ROOT / "src/codebase_index/skill_template/scripts" + expected = _whitelist_sh(template / "cbx") + assert _whitelist_ps1(template / "cbx.ps1") == expected + assert _whitelist_sh(ROOT / "bin/cbx") == expected + assert _whitelist_ps1(ROOT / "bin/cbx.ps1") == expected diff --git a/tests/test_read_cap.py b/tests/test_read_cap.py new file mode 100644 index 0000000..176c426 --- /dev/null +++ b/tests/test_read_cap.py @@ -0,0 +1,61 @@ +"""`recommended_reads` entries are capped at `max_read_lines` (additive fields only).""" + +from __future__ import annotations + +from codebase_index.retrieval.pipeline import _bounded_read + + +def test_short_entry_is_untouched(): + entry = {"path": "a.py", "line_start": 10, "line_end": 40} + assert _bounded_read(entry, 120) is entry + + +def test_long_entry_is_capped_and_annotated(): + entry = {"path": "Big.java", "line_start": 1, "line_end": 1500} + out = _bounded_read(entry, 120) + assert out == { + "path": "Big.java", + "line_start": 1, + "line_end": 120, + "line_end_full": 1500, + "truncated": True, + } + assert entry["line_end"] == 1500, "input must not be mutated" + + +def test_zero_disables_the_cap(): + entry = {"path": "Big.java", "line_start": 1, "line_end": 1500} + assert _bounded_read(entry, 0) is entry + + +def test_search_applies_cap_through_config(tmp_path): + """End to end on a synthetic file whose only symbol spans 300 lines.""" + from pathlib import Path + + from codebase_index.config import Config + from codebase_index.indexer.pipeline import build_index + from codebase_index.retrieval.pipeline import search + from codebase_index.storage.db import Database + + root = tmp_path / "repo" + (root / "src").mkdir(parents=True) + body = "\n".join(f" x{i} = {i} # frobnicate widget" for i in range(298)) + (root / "src" / "widget.py").write_text( + f"class WidgetFrobnicator:\n '''Frobnicates widgets.'''\n{body}\n", encoding="utf-8" + ) + cfg = Config(root=str(root)) + cfg.embeddings.enabled = False + db = Database(tmp_path / "idx.sqlite").open() + try: + build_index(cfg, db, root=Path(root)) + # A budget of 1 token guarantees the hit lands in recommended_reads, not a snippet. + capped = search(db.conn, "WidgetFrobnicator", mode="hybrid", limit=5, + token_budget=1, no_fallback=True, max_read_lines=50) + full = search(db.conn, "WidgetFrobnicator", mode="hybrid", limit=5, + token_budget=1, no_fallback=True, max_read_lines=0) + finally: + db.close() + capped_reads = [r for r in capped["recommended_reads"] if r.get("truncated")] + assert capped_reads, capped["recommended_reads"] + assert all(r["line_end"] - r["line_start"] + 1 == 50 for r in capped_reads) + assert not any(r.get("truncated") for r in full["recommended_reads"]) diff --git a/tests/test_release_scripts.py b/tests/test_release_scripts.py new file mode 100644 index 0000000..0ae2a23 --- /dev/null +++ b/tests/test_release_scripts.py @@ -0,0 +1,172 @@ +"""Unit tests for the release/CI hygiene scripts under scripts/.""" + +from __future__ import annotations + +import importlib.util +import json +from pathlib import Path + +import pytest + +REPO = Path(__file__).resolve().parents[1] + + +def _load(name: str): + spec = importlib.util.spec_from_file_location(name, REPO / "scripts" / f"{name}.py") + mod = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(mod) + return mod + + +release_notes = _load("release_notes") +check_versions = _load("check_versions") +check_links = _load("check_links") + + +CHANGELOG = """# Changelog + +## [Unreleased] + +## [1.9.0] - 2026-09-02 + +### Added + +- Thing one → arrow. + +## [1.8.0] - 2026-09-02 + +### Changed + +- Older thing. + +## [1.7.0] - 2026-07-29 + +[Unreleased]: https://github.com/denfry/codebase-index/compare/v1.8.0...HEAD +[1.8.0]: https://github.com/denfry/codebase-index/compare/v1.7.0...v1.8.0 +""" + + +# --- release_notes ----------------------------------------------------------- +def test_release_notes_extracts_section_and_synthesizes_compare_link(): + out = release_notes.render(CHANGELOG, "1.9.0") + assert "Thing one" in out + assert "Older thing" not in out + assert "pip install codebase-index==1.9.0" in out + # No link definition for 1.9.0 -> derived from the previous heading. + assert "compare/v1.8.0...v1.9.0" in out + + +def test_release_notes_prefers_existing_link_definition(): + out = release_notes.render(CHANGELOG, "1.8.0") + assert "compare/v1.7.0...v1.8.0" in out + assert "[1.8.0]:" not in out # link defs are stripped from the body + + +def test_release_notes_fails_on_empty_or_missing_section(): + with pytest.raises(SystemExit): + release_notes.render(CHANGELOG, "1.7.0") # heading present, body empty + with pytest.raises(KeyError): + release_notes.render(CHANGELOG, "9.9.9") + + +def test_release_notes_cli_writes_file(tmp_path): + changelog = tmp_path / "CHANGELOG.md" + changelog.write_text(CHANGELOG, encoding="utf-8") + out = tmp_path / "notes.md" + rc = release_notes.main(["1.9.0", "--changelog", str(changelog), "--out", str(out)]) + assert rc == 0 + assert "Thing one" in out.read_text(encoding="utf-8") + assert release_notes.main(["9.9.9", "--changelog", str(changelog)]) == 1 + + +# --- check_versions ---------------------------------------------------------- +def _fake_repo(tmp_path: Path, version: str, *, plugin=None, lock=None, stamps=None, + changelog=CHANGELOG) -> Path: + tmp_path.mkdir(parents=True, exist_ok=True) + (tmp_path / "src/codebase_index").mkdir(parents=True) + (tmp_path / "src/codebase_index/__init__.py").write_text( + f'__version__ = "{version}"\n', encoding="utf-8" + ) + (tmp_path / ".claude-plugin").mkdir() + (tmp_path / ".claude-plugin/plugin.json").write_text( + json.dumps({"version": plugin or version}), encoding="utf-8" + ) + (tmp_path / "requirements.lock").write_text( + f"codebase-index @ https://github.com/denfry/codebase-index/archive/refs/tags/" + f"v{lock or version}.tar.gz\n", + encoding="utf-8", + ) + for rel in check_versions.STAMPS: + (tmp_path / rel).parent.mkdir(parents=True, exist_ok=True) + (tmp_path / rel).write_text(f"{stamps or version}\n", encoding="utf-8") + (tmp_path / "CHANGELOG.md").write_text(changelog, encoding="utf-8") + return tmp_path + + +def test_check_versions_consistent(tmp_path): + assert check_versions.check(_fake_repo(tmp_path, "1.9.0")) == [] + + +def test_check_versions_reports_every_mismatch(tmp_path): + repo = _fake_repo(tmp_path, "1.9.0", plugin="1.8.0", lock="1.8.0", stamps="1.8.0") + problems = check_versions.check(repo) + assert any("plugin.json" in p for p in problems) + assert any("requirements.lock" in p for p in problems) + assert sum(".skill_version" in p for p in problems) == 3 + + +def test_check_versions_allows_bumped_version_with_unreleased(tmp_path): + # main between releases: version ahead of the newest heading, notes under Unreleased. + assert check_versions.check(_fake_repo(tmp_path / "ahead", "1.10.0")) == [] + # but a version that is neither released nor newer is a mistake + problems = check_versions.check(_fake_repo(tmp_path / "behind", "1.8.5")) + assert any("CHANGELOG" in p for p in problems) + + +def test_check_versions_real_repo_is_consistent(): + assert check_versions.check(REPO) == [] + + +# --- check_links ------------------------------------------------------------- +def test_check_links_detects_broken_and_ignores_external(tmp_path, monkeypatch): + docs = tmp_path / "docs" + docs.mkdir() + (docs / "ok.md").write_text("target\n", encoding="utf-8") + (tmp_path / "img.png").write_bytes(b"\x89PNG") + md = tmp_path / "README.md" + md.write_text( + "\n".join([ + "[good](docs/ok.md#section)", + "[external](https://example.com/x.md)", + "[mail](mailto:a@b.c)", + "[anchor](#top)", + '', + "![shot](assets/missing.png)", + "[bad](docs/nope.md)", + "```", + "[fenced](docs/not-checked.md)", + "```", + ]), + encoding="utf-8", + ) + monkeypatch.setattr(check_links, "REPO", tmp_path) + broken = check_links.find_broken(md, tmp_path) + assert [t for _, t in broken] == ["assets/missing.png", "docs/nope.md"] + assert broken[0][0] == 6 and broken[1][0] == 7 + + +def test_check_links_cli_exit_codes(tmp_path, monkeypatch, capsys): + (tmp_path / "a.md").write_text("[x](b.md)\n", encoding="utf-8") + monkeypatch.setattr(check_links, "REPO", tmp_path) + assert check_links.main([str(tmp_path)]) == 1 + assert "a.md:1: b.md" in capsys.readouterr().out + (tmp_path / "b.md").write_text("ok\n", encoding="utf-8") + assert check_links.main([str(tmp_path)]) == 0 + + +def test_check_links_skips_venv(tmp_path, monkeypatch): + (tmp_path / ".venv").mkdir() + (tmp_path / ".venv/x.md").write_text("[x](missing.md)\n", encoding="utf-8") + monkeypatch.setattr(check_links, "REPO", tmp_path) + assert list(check_links.iter_markdown([tmp_path], tmp_path)) == []