diff --git a/.forgejo/workflows/ci.yml b/.forgejo/workflows/ci.yml index b600112..c4b46d7 100644 --- a/.forgejo/workflows/ci.yml +++ b/.forgejo/workflows/ci.yml @@ -11,15 +11,16 @@ jobs: run: apt-get update -qq && apt-get install -y -qq python3 python3-pip - name: Install dependencies run: | - if [ -f requirements.txt ]; then pip install --break-system-packages -r requirements.txt || true; fi - if [ -f pyproject.toml ]; then pip install --break-system-packages -e ".[dev]" 2>/dev/null || true; fi + if [ -f requirements.txt ]; then pip install --break-system-packages -r requirements.txt; fi + pip install --break-system-packages -e ".[dev]" - name: Lint run: | - pip install --break-system-packages ruff 2>/dev/null - ruff check . || true + pip install --break-system-packages ruff==0.16.10 + ruff check . + ruff format --check . - name: Test run: | if [ -f pytest.ini ] || [ -f setup.cfg ] || [ -d tests ]; then - pip install --break-system-packages pytest 2>/dev/null - pytest || true + pip install --break-system-packages pytest + pytest fi diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a3c4e50..718d562 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -39,6 +39,16 @@ jobs: run: | devarch demo --force --build-db + - name: Install and verify built wheel outside checkout + if: matrix.python-version == '3.12' + run: | + python -m pip install build twine + python -m build + python -m twine check dist/* + python -m pip uninstall -y devarch-framework + python -m pip install --no-deps --force-reinstall dist/devarch_framework-0.4.1-py3-none-any.whl + python scripts/check_installed_wheel.py + lint: runs-on: ubuntu-latest steps: @@ -57,8 +67,12 @@ jobs: python -m pip install --upgrade pip pip install -e ".[dev]" + - name: Lint and format + run: | + ruff check . + ruff format --check . + - name: Check Python syntax run: | python -m py_compile archaeology/*.py python -m py_compile archaeology/**/*.py - continue-on-error: true diff --git a/.ruff.toml b/.ruff.toml new file mode 100644 index 0000000..95d5c7d --- /dev/null +++ b/.ruff.toml @@ -0,0 +1,14 @@ +target-version = "py310" +line-length = 100 + +[lint] +select = ["E4", "E7", "E9", "F", "I"] + +# Repository-only content-generation utilities require Python 3.12. +[per-file-target-version] +"scripts/**/*.py" = "py312" + +[lint.per-file-ignores] +# These standalone utilities import project modules after explicit sys.path setup. +"scripts/data/refresh_data.py" = ["E402"] +"scripts/sync/generate-bridge.py" = ["E402"] diff --git a/CHANGELOG.md b/CHANGELOG.md index 57911a2..2def157 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,11 @@ +# 0.4.1 — Unreleased readiness candidate + +- Bind analysis, reports and visualizations to exact inputs; reject stale A-to-B runs. +- Disable unsafe legacy dashboard/static publishing, cascade and opportunity profiling before side effects; remove the active Markdown viewer and wildcard API CORS. +- Validate installed history HTML without a source-relative Node dependency. +- Correct package/documentation metadata to the existing Apache-2.0 LICENSE; no relicensing. +- Add installed-wheel checks to native Python3.12 CI and remove suppressed CI failures. + # 0.4.0 — Evidence integrity Fix full-history mining for bare repositories/worktrees, reject shallow coverage, bind extraction artifacts, reconcile CSV/SQLite metrics, and generate portable measured visualizations. Replace unsupported agent/ML/quality assertions with explicit uncertainty. See [release notes](docs/RELEASE_0.4.0.md) for output compatibility and validation. diff --git a/README.md b/README.md index b5e7b72..be9622c 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ [![PyPI version](https://img.shields.io/pypi/v/devarch-framework.svg)](https://pypi.org/project/devarch-framework/) [![Python](https://img.shields.io/pypi/pyversions/devarch-framework.svg)](https://pypi.org/project/devarch-framework/) -[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) +[![License: Apache-2.0](https://img.shields.io/badge/License-Apache--2.0-blue.svg)](LICENSE) [![GitHub stars](https://img.shields.io/github/stars/KyaniteLabs/devarch-framework.svg?style=social)](https://github.com/KyaniteLabs/devarch-framework) **Git repository archaeology framework.** Mine commit history, detect development signals, run 6 analysis vectors, and generate engineering narrative reports — from any git repository, fully local, no external services. @@ -133,7 +133,7 @@ devarch audit my-project --fail-on MEDIUM ### Analysis - `devarch analyze [--vector] [--prompts]` -- Run analysis vectors -- `devarch cascade [--dry-run] [--skip-mine]` -- Cascade era labels across repos +- Legacy `cascade`, `opportunity`, `dashboard` and `publish-static` commands are disabled pending safety/evidence redesign. ### Visualization & Reporting - `devarch visualize ` -- Generate HTML visualization @@ -302,7 +302,7 @@ _config/ -- Developer profile templates ## License -MIT License -- See LICENSE file for details. +Apache-2.0 — see the existing LICENSE file. Metadata was corrected; the repository has not been relicensed. ## Support @@ -411,3 +411,9 @@ Issues and PRs welcome on the canonical remote. Keep public docs free of secrets See [LICENSE](LICENSE) in this repository (or package metadata if license is package-only). + +### 0.4.1 readiness candidate (unreleased) + +The supported measured-history path is `init → mine → build-db → signals → analyze → visualize → validate → export-report → audit`. Analysis and reports bind to exact local input hashes. After changing history or configuration, rerun the affected stages; stale outputs fail audit/export. Bindings detect drift, not malicious coordinated rewriting or remote authenticity. The validator checks the generated history document and binding; it is not a general browser, accessibility or HTML conformance audit. + +Legacy network dashboard/static publishing, cascade and opportunity profiling are explicitly disabled before mutation. Open reviewed local HTML directly. Existing scripts depending on these commands receive a nonzero error; there is no unsafe override. No personal/medical profile inference is supported. diff --git a/archaeology/__init__.py b/archaeology/__init__.py index 353897b..4ca5ae6 100644 --- a/archaeology/__init__.py +++ b/archaeology/__init__.py @@ -1,3 +1,3 @@ """DevArch Framework - forensic mining of software development history.""" -__version__ = "0.4.0" +__version__ = "0.4.1" diff --git a/archaeology/analysis_runner.py b/archaeology/analysis_runner.py index 070450e..adfbb8f 100644 --- a/archaeology/analysis_runner.py +++ b/archaeology/analysis_runner.py @@ -95,25 +95,57 @@ def status(count: int, low: float, high: float) -> str: practices = [ ("CI/CD Pipeline", ci_cd, 0.01, 0.04, "Run local/GitHub quality gates automatically"), ("Test Coverage", tests, 0.05, 0.15, "Keep behavior tests above the agreed threshold"), - ("Refactoring Discipline", refactor, 0.02, 0.08, "Reserve explicit simplification cycles"), - ("Security Review", security, 0.005, 0.025, "Keep security findings tied to a verification gate"), - ("Documentation Hygiene", docs, 0.02, 0.08, "Synchronize public claims with canonical metrics"), + ( + "Refactoring Discipline", + refactor, + 0.02, + 0.08, + "Reserve explicit simplification cycles", + ), + ( + "Security Review", + security, + 0.005, + 0.025, + "Keep security findings tied to a verification gate", + ), + ( + "Documentation Hygiene", + docs, + 0.02, + 0.08, + "Synchronize public claims with canonical metrics", + ), ] gaps = [] for practice, rows, low, high, recommendation in practices: practice_status = status(len(rows), low, high) - severity = "MEDIUM" if practice_status == "UNVERIFIED" else "MEDIUM" if practice_status == "EMERGING" else "LOW" + severity = ( + "MEDIUM" + if practice_status == "UNVERIFIED" + else "MEDIUM" + if practice_status == "EMERGING" + else "LOW" + ) gaps.append( { "practice": practice, "status": practice_status, "confidence": "LOW", "interpretation": "Commit-keyword frequency only; not verified presence, absence or coverage", - "evidence": [{"sample": rows[:5], "result_count": len(rows), "ratio": f"{(len(rows) / total_commits if total_commits else 0):.1%}"}], + "evidence": [ + { + "sample": rows[:5], + "result_count": len(rows), + "ratio": f"{(len(rows) / total_commits if total_commits else 0):.1%}", + } + ], "severity": severity, "effort_to_implement": 3 if severity == "HIGH" else 2, "expected_impact": 5 if severity == "HIGH" else 3, - "roi": round((5 if severity == "HIGH" else 3) / (3 if severity == "HIGH" else 2), 2), + "roi": round( + (5 if severity == "HIGH" else 3) / (3 if severity == "HIGH" else 2), 2 + ), "recommendation": recommendation, } ) @@ -125,7 +157,10 @@ def status(count: int, low: float, high: float) -> str: "summary": { "total_gaps": len(gaps), "critical_gaps": sum(1 for g in gaps if g["severity"] == "CRITICAL"), - "top_3_roi": [g["practice"] for g in sorted(gaps, key=lambda row: row["roi"], reverse=True)[:3]], + "top_3_roi": [ + g["practice"] + for g in sorted(gaps, key=lambda row: row["roi"], reverse=True)[:3] + ], }, } @@ -133,11 +168,41 @@ def run_ml_pattern_mapper(self) -> dict[str, Any]: """Map intuitive code/commit language to formal ML patterns.""" self._log("Running ML Pattern Mapper...") patterns = [ - ("scoring system", "Weighted Multi-Criteria Decision Analysis", ["score", "rank", "weight", "threshold"], False, None), - ("evolution loop", "Evolutionary Strategy / Quality-Diversity Search", ["evolve", "mutate", "fitness", "diversity", "map-elites"], True, "DEAP or pymoo"), - ("model routing", "Contextual Bandit / Mixture-of-Experts Routing", ["router", "route", "model", "provider"], False, None), - ("critic ensemble", "Ensemble Evaluation / Multi-Critic Reward Modeling", ["critic", "aesthetic", "evaluator", "judge"], False, None), - ("retrieval memory", "Retrieval-Augmented Generation", ["rag", "retrieval", "archive", "memory", "semantic"], False, None), + ( + "scoring system", + "Weighted Multi-Criteria Decision Analysis", + ["score", "rank", "weight", "threshold"], + False, + None, + ), + ( + "evolution loop", + "Evolutionary Strategy / Quality-Diversity Search", + ["evolve", "mutate", "fitness", "diversity", "map-elites"], + True, + "DEAP or pymoo", + ), + ( + "model routing", + "Contextual Bandit / Mixture-of-Experts Routing", + ["router", "route", "model", "provider"], + False, + None, + ), + ( + "critic ensemble", + "Ensemble Evaluation / Multi-Critic Reward Modeling", + ["critic", "aesthetic", "evaluator", "judge"], + False, + None, + ), + ( + "retrieval memory", + "Retrieval-Augmented Generation", + ["rag", "retrieval", "archive", "memory", "semantic"], + False, + None, + ), ] mappings = [] for intuitive, formal, keywords, reinvention, library in patterns: @@ -174,7 +239,9 @@ def _approximate_sessions(self) -> list[dict]: Groups commits into sessions using a 2-hour inactivity gap heuristic. Falls back to daily grouping if timestamps lack time components. """ - tables = {r["name"] for r in self._query_db("SELECT name FROM sqlite_master WHERE type='table'")} + tables = { + r["name"] for r in self._query_db("SELECT name FROM sqlite_master WHERE type='table'") + } if "sessions" in tables: return self._query_db("SELECT session_id, timestamp FROM sessions ORDER BY timestamp") @@ -183,9 +250,10 @@ def _approximate_sessions(self) -> list[dict]: return [] from datetime import datetime as dt + GAP_HOURS = 2 sessions: list[dict] = [] - session_start = None + _session_start = None prev_ts = None for row in commits: @@ -205,7 +273,7 @@ def _approximate_sessions(self) -> list[dict]: if prev_ts is None or (ts - prev_ts).total_seconds() > GAP_HOURS * 3600: session_id = ts.strftime("%Y%m%d-%H%M%S") sessions.append({"session_id": session_id, "timestamp": ts.isoformat()}) - session_start = ts + _session_start = ts prev_ts = ts @@ -215,7 +283,9 @@ def run_agentic_workflow(self) -> dict[str, Any]: """Analyze AI agent interaction patterns.""" self._log("Running Agentic Workflow Analyzer...") hooks = self._like_commits(["hook", "pre-commit", "post-commit", "automation"], 50) - authors = self._query_db("SELECT author, COUNT(*) as cnt FROM commits GROUP BY author ORDER BY cnt DESC") + authors = self._query_db( + "SELECT author, COUNT(*) as cnt FROM commits GROUP BY author ORDER BY cnt DESC" + ) return { "project": self.project_name, "analysis_date": datetime.now().isoformat(), @@ -235,7 +305,11 @@ def run_formal_terms_mapper(self) -> dict[str, Any]: ("CompostMill", "Content Processing Pipeline / Creative Memory Store", ["compost"]), ("RalphLoop", "Generate-Evaluate-Improve Control Loop", ["ralph", "loop", "iterate"]), ("Swarm", "Multi-Agent Ensemble / Debate", ["swarm", "agent", "collaboration"]), - ("Quality Gate", "Verification Gate / Acceptance Criterion", ["quality gate", "guardrail", "validation"]), + ( + "Quality Gate", + "Verification Gate / Acceptance Criterion", + ["quality gate", "guardrail", "validation"], + ), ("Archive", "Event-Sourced Knowledge Store", ["archive", "event", "sqlite"]), ] dictionary = [] @@ -257,30 +331,58 @@ def run_formal_terms_mapper(self) -> dict[str, Any]: "analysis_date": datetime.now().isoformat(), "term_dictionary": dictionary, "naming_trajectory": "Unmeasured; keyword candidates require source validation.", - "learning_opportunities": ["Control theory", "Quality-diversity algorithms", "Event sourcing", "Multi-agent evaluation"], - "summary": {"terms_mapped": len(dictionary), "high_confidence": sum(1 for t in dictionary if t["similarity_score"] == "CLOSE")}, + "learning_opportunities": [ + "Control theory", + "Quality-diversity algorithms", + "Event sourcing", + "Multi-agent evaluation", + ], + "summary": { + "terms_mapped": len(dictionary), + "high_confidence": sum(1 for t in dictionary if t["similarity_score"] == "CLOSE"), + }, } def run_source_archaeologist(self) -> dict[str, Any]: """Mine commit history for code quality trajectory and hotspots.""" self._log("Running Source Code Archaeologist...") quality = self._like_commits(["fix", "test", "refactor", "security", "lint", "type"], None) - large_change = self._like_commits(["split", "extract", "monolith", "decompose", "simplify"], 100) + large_change = self._like_commits( + ["split", "extract", "monolith", "decompose", "simplify"], 100 + ) todo = self._like_commits(["todo", "stub", "placeholder", "not implemented"], 100) by_month: Counter[str] = Counter() for row in quality: date = str(row.get("date", ""))[:7] if date: by_month[date] += 1 - hotspots = self._query_db("SELECT message, COUNT(*) as cnt FROM commits GROUP BY message ORDER BY cnt DESC LIMIT 10") + hotspots = self._query_db( + "SELECT message, COUNT(*) as cnt FROM commits GROUP BY message ORDER BY cnt DESC LIMIT 10" + ) improvements = self._derive_improvements(quality, large_change, todo, hotspots) return { - "analysis_metadata": {"timestamp": datetime.now().isoformat(), "analyst": "Automated Source Code Archaeologist", "project": self.project_name, "commit_count": self._commit_count()}, - "quality_trajectory": {"assessment": "UNVERIFIED keyword activity; no quality direction established", "evidence_count": len(quality), "by_month": dict(sorted(by_month.items()))}, - "architecture_drift": {"large_change_signals": large_change[:10], "todo_or_stub_signals": todo[:10]}, + "analysis_metadata": { + "timestamp": datetime.now().isoformat(), + "analyst": "Automated Source Code Archaeologist", + "project": self.project_name, + "commit_count": self._commit_count(), + }, + "quality_trajectory": { + "assessment": "UNVERIFIED keyword activity; no quality direction established", + "evidence_count": len(quality), + "by_month": dict(sorted(by_month.items())), + }, + "architecture_drift": { + "large_change_signals": large_change[:10], + "todo_or_stub_signals": todo[:10], + }, "hotspots": hotspots, "improvements": improvements, - "summary": {"quality_signal_count": len(quality), "large_change_signal_count": len(large_change), "todo_signal_count": len(todo)}, + "summary": { + "quality_signal_count": len(quality), + "large_change_signal_count": len(large_change), + "todo_signal_count": len(todo), + }, } def _derive_improvements( @@ -297,51 +399,83 @@ def _derive_improvements( flapping = [h for h in hotspots if h.get("cnt", 0) >= 3] if flapping: top_msg = str(flapping[0].get("message", ""))[:60] - items.append(( - 100, - f"Investigate repeated message (may be merge/cherry-pick duplication): {top_msg}", - "M", "HIGH", - )) + items.append( + ( + 100, + f"Investigate repeated message (may be merge/cherry-pick duplication): {top_msg}", + "M", + "HIGH", + ) + ) # Unresolved stubs / TODOs if todo: - items.append(( - 90 if len(todo) >= 5 else 70, - f"Check whether {len(todo)} historical stub/placeholder mentions remain unresolved", - "S", "HIGH" if len(todo) >= 5 else "MEDIUM", - )) + items.append( + ( + 90 if len(todo) >= 5 else 70, + f"Check whether {len(todo)} historical stub/placeholder mentions remain unresolved", + "S", + "HIGH" if len(todo) >= 5 else "MEDIUM", + ) + ) # Decomposition momentum: carry it through if large_change: - items.append(( - 60, - f"Review {len(large_change)} historical decomposition signals before proposing more splits", - "L", "MEDIUM", - )) + items.append( + ( + 60, + f"Review {len(large_change)} historical decomposition signals before proposing more splits", + "L", + "MEDIUM", + ) + ) # Quality signal density: low fix/test ratio suggests coverage gaps commit_count = self._commit_count() or 1 quality_ratio = len(quality) / commit_count if quality_ratio < 0.10: - items.append(( - 80, - f"Boost quality signal density — fix/test ratio at {quality_ratio:.0%} (target ≥10%)", - "M", "HIGH", - )) + items.append( + ( + 80, + f"Boost quality signal density — fix/test ratio at {quality_ratio:.0%} (target ≥10%)", + "M", + "HIGH", + ) + ) elif quality_ratio < 0.20: - items.append(( - 50, - f"Maintain quality signal density — currently at {quality_ratio:.0%}", - "S", "LOW", - )) + items.append( + ( + 50, + f"Maintain quality signal density — currently at {quality_ratio:.0%}", + "S", + "LOW", + ) + ) # No issues found: project is healthy if not items: - items.append((10, "No keyword-derived candidates; source review still required", "S", "LOW")) + items.append( + (10, "No keyword-derived candidates; source review still required", "S", "LOW") + ) items.sort(key=lambda x: x[0], reverse=True) return [ - {"rank": i + 1, "title": title, "effort": effort, "impact": impact, "status": "UNVERIFIED investigation candidate", "evidence": (todo[:3] if "placeholder" in title else large_change[:3] if "decomposition" in title else hotspots[:3] if "repeated" in title else quality[:3])} + { + "rank": i + 1, + "title": title, + "effort": effort, + "impact": impact, + "status": "UNVERIFIED investigation candidate", + "evidence": ( + todo[:3] + if "placeholder" in title + else large_change[:3] + if "decomposition" in title + else hotspots[:3] + if "repeated" in title + else quality[:3] + ), + } for i, (_, title, effort, impact) in enumerate(items) ] @@ -353,9 +487,17 @@ def run_youtube_correlator(self) -> dict[str, Any]: yt_topics = self._load_json("data/youtube-topic-classification.json") or {} canonical = self._load_json("deliverables/canonical-metrics.json") or {} correlations = yt_corr.get("key_correlations") or yt_corr.get("correlations") or [] - creators = yt_creators.get("creators") if isinstance(yt_creators, dict) else yt_creators if isinstance(yt_creators, list) else [] + creators = ( + yt_creators.get("creators") + if isinstance(yt_creators, dict) + else yt_creators + if isinstance(yt_creators, list) + else [] + ) topics = yt_topics.get("categories") if isinstance(yt_topics, dict) else [] - smoking_guns = [row for row in correlations if isinstance(row, dict) and row.get("is_smoking_gun")] + smoking_guns = [ + row for row in correlations if isinstance(row, dict) and row.get("is_smoking_gun") + ] return { "project": self.project_name, "analysis_date": datetime.now().isoformat(), @@ -393,13 +535,25 @@ def run_all(self, vectors: list[str] | None = None) -> dict[str, str]: self.deliverables_dir.mkdir(parents=True, exist_ok=True) analysis_dir = self.deliverables_dir / "analysis" analysis_dir.mkdir(parents=True, exist_ok=True) + from .provenance import bind_artifact, snapshot + + inputs = snapshot(self.project_dir) for vector_name in target: runner_func = runners[vector_name] try: output_path = analysis_dir / f"analysis-{vector_name}.json" result = runner_func() - result["methodology"] = {"basis": "commit-message heuristics", "source_inspection": False, "causal_inference": False, "limitations": "Requires source/PR validation. Missing keyword evidence does not establish absence. Repeated commits across refs are not necessarily recurring defects."} + if snapshot(self.project_dir) != inputs: + raise ValueError("Analysis inputs changed during execution; retry") + result["evidence_binding"] = inputs + result["methodology"] = { + "basis": "commit-message heuristics", + "source_inspection": False, + "causal_inference": False, + "limitations": "Requires source/PR validation. Missing keyword evidence does not establish absence. Repeated commits across refs are not necessarily recurring defects.", + } atomic_write(output_path, json.dumps(result, indent=2, ensure_ascii=False) + "\n") + bind_artifact(self.project_dir, output_path) results[vector_name] = str(output_path) print(f" [analysis] {vector_name}: {output_path}") except (OSError, ValueError, KeyError, sqlite3.OperationalError) as exc: @@ -408,7 +562,9 @@ def run_all(self, vectors: list[str] | None = None) -> dict[str, str]: return results -def run_analysis_vectors(project_name: str, verbose: bool = False, vectors: list[str] | None = None) -> dict[str, str]: +def run_analysis_vectors( + project_name: str, verbose: bool = False, vectors: list[str] | None = None +) -> dict[str, str]: """Public entry point to run analysis vectors.""" project_dir = os.path.join("projects", project_name) if not os.path.isdir(project_dir): diff --git a/archaeology/api.py b/archaeology/api.py index 4b2ce28..1f28326 100644 --- a/archaeology/api.py +++ b/archaeology/api.py @@ -17,7 +17,6 @@ import re import sqlite3 from datetime import datetime -from http.server import BaseHTTPRequestHandler from pathlib import Path from urllib.parse import urlparse @@ -30,7 +29,7 @@ def _validate_project_name(name): """Reject path traversal and invalid characters in project names.""" - if not name or not re.match(r'^[a-zA-Z0-9._-]+$', name): + if not name or not re.match(r"^[a-zA-Z0-9._-]+$", name): return False # Double-check no path traversal resolved = (PROJECTS_DIR / name).resolve() @@ -44,7 +43,6 @@ def _validate_project_name(name): def _json_response(handler, data, status=200): handler.send_response(status) handler.send_header("Content-Type", "application/json") - handler.send_header("Access-Control-Allow-Origin", "*") payload = json.dumps(data, indent=2, default=str).encode() handler.send_header("Content-Length", str(len(payload))) handler.end_headers() @@ -159,7 +157,7 @@ def _parse_swot(project_dir): pattern = rf"{quadrant}.*?(\d+)\s+found" m = re.search(pattern, matrix_text, re.IGNORECASE) if m: - result.setdefault(f"{quadrangle if quadrant == 'strengths' else quadrant}_count", int(m.group(1))) + result.setdefault(f"{quadrant}_count", int(m.group(1))) return result @@ -189,14 +187,20 @@ def _parse_wardley(project_dir): table_text = sections.get("Component Evolution Table", "") components = [] for line in table_text.splitlines(): - if "|" in line and not line.strip().startswith("|-") and not line.strip().startswith("| Component"): + if ( + "|" in line + and not line.strip().startswith("|-") + and not line.strip().startswith("| Component") + ): cells = [c.strip() for c in line.split("|") if c.strip()] if len(cells) >= 3: - components.append({ - "component": cells[0], - "stage": cells[1], - "evidence": cells[2] if len(cells) > 2 else "", - }) + components.append( + { + "component": cells[0], + "stage": cells[1], + "evidence": cells[2] if len(cells) > 2 else "", + } + ) result["components"] = components # Parse recommendations @@ -277,20 +281,26 @@ def _parse_bcg(project_dir): components = [] table_text = sections.get("Component Classification", "") for line in table_text.splitlines(): - if "|" in line and not line.strip().startswith("|-") and not line.strip().startswith("| Component"): + if ( + "|" in line + and not line.strip().startswith("|-") + and not line.strip().startswith("| Component") + ): cells = [c.strip() for c in line.split("|") if c.strip()] if len(cells) >= 4: - components.append({ - "component": cells[0], - "commits": cells[1], - "share": cells[2], - "quadrant": cells[3], - }) + components.append( + { + "component": cells[0], + "commits": cells[1], + "share": cells[2], + "quadrant": cells[3], + } + ) result["components"] = components for quadrant_key in ("Stars", "Cash Cows", "Question Marks", "Dogs"): section = sections.get(f"{quadrant_key}", "") - count_match = re.search(rf"\((\d+)\)", sections.get("Quadrant Analysis", "")) + _count_match = re.search(r"\((\d+)\)", sections.get("Quadrant Analysis", "")) result[quadrant_key.lower().replace(" ", "_")] = _parse_list_items(section) result["recommendations"] = _parse_list_items(sections.get("Strategic Recommendations", "")) @@ -437,12 +447,15 @@ def _compute_health_score(metrics, swot, wardley, value_chain): def handle_health(handler): """GET /api/health""" projects = _discover_projects() - _json_response(handler, { - "status": "ok", - "version": __version__, - "projects": len(projects), - "bridge_available": BRIDGE_PATH.exists(), - }) + _json_response( + handler, + { + "status": "ok", + "version": __version__, + "projects": len(projects), + "bridge_available": BRIDGE_PATH.exists(), + }, + ) def handle_projects(handler): @@ -473,18 +486,21 @@ def handle_insights(handler, project_name): pipeline = _get_pipeline_status(pdir) health = _compute_health_score(metrics, swot, wardley, value_chain) - _json_response(handler, { - "project": project_name, - "health_score": health, - "metrics": metrics, - "swot": swot, - "wardley": wardley, - "value_chain": value_chain, - "bcg": bcg, - "ansoff": ansoff, - "blue_ocean": blue_ocean, - "pipeline": pipeline, - }) + _json_response( + handler, + { + "project": project_name, + "health_score": health, + "metrics": metrics, + "swot": swot, + "wardley": wardley, + "value_chain": value_chain, + "bcg": bcg, + "ansoff": ansoff, + "blue_ocean": blue_ocean, + "pipeline": pipeline, + }, + ) def handle_swot(handler, project_name): @@ -614,7 +630,9 @@ def _generate_bridge(): "weaknesses": len(swot.get("weaknesses", [])) if swot else 0, "opportunities": len(swot.get("opportunities", [])) if swot else 0, "threats": len(swot.get("threats", [])) if swot else 0, - } if swot else None, + } + if swot + else None, "wardley_maturity": wardley.get("maturity") if wardley else None, "value_chain_margin": value_chain.get("margin_score") if value_chain else None, "bcg_velocity_trend": bcg.get("velocity_trend") if bcg else None, @@ -634,13 +652,29 @@ def _generate_bridge(): "cross_repo": { "total_repos": len(projects_data), "total_commits": sum(p["total_commits"] for p in projects_data.values()), - "frameworks_available": ["swot", "wardley", "value-chain", "bcg", "ansoff", "blue-ocean"], + "frameworks_available": [ + "swot", + "wardley", + "value-chain", + "bcg", + "ansoff", + "blue-ocean", + ], "opportunity_features": [ - "learning-velocity", "frustration-to-automation", "knowledge-gap", - "token-efficiency", "session-quality", "ai-agent-mastery", - "creative-dna", "neurodivergent-profile", "model-selection-advisor", - "before-after-snapshot", "cross-repo-transfer", "youtube-learning-graph", - "architecture-timelapse", "commit-cognitive-load", + "learning-velocity", + "frustration-to-automation", + "knowledge-gap", + "token-efficiency", + "session-quality", + "ai-agent-mastery", + "creative-dna", + "neurodivergent-profile", + "model-selection-advisor", + "before-after-snapshot", + "cross-repo-transfer", + "youtube-learning-graph", + "architecture-timelapse", + "commit-cognitive-load", ], }, } @@ -677,7 +711,9 @@ def generate_bridge_file(): def _load_opportunity(project_name, feature): """Load a single opportunity analysis JSON.""" - path = PROJECTS_DIR / project_name / "deliverables" / "opportunity" / f"opportunity-{feature}.json" + path = ( + PROJECTS_DIR / project_name / "deliverables" / "opportunity" / f"opportunity-{feature}.json" + ) return _load_json(path) @@ -690,7 +726,9 @@ def handle_opportunity_feature(handler, project_name, feature): return _error_response(handler, f"Project '{project_name}' not found") data = _load_opportunity(project_name, feature) if not data: - return _error_response(handler, f"Opportunity '{feature}' not generated for '{project_name}'", 404) + return _error_response( + handler, f"Opportunity '{feature}' not generated for '{project_name}'", 404 + ) _json_response(handler, data) @@ -703,12 +741,15 @@ def handle_opportunity_all(handler, project_name): for feature in OPPORTUNITY_FEATURES: data = _load_opportunity(project_name, feature) results[feature] = data - _json_response(handler, { - "project": project_name, - "features": OPPORTUNITY_FEATURES, - "data": results, - "generated_at": datetime.now().isoformat(), - }) + _json_response( + handler, + { + "project": project_name, + "features": OPPORTUNITY_FEATURES, + "data": results, + "generated_at": datetime.now().isoformat(), + }, + ) def handle_opportunity_index(handler, project_name): @@ -719,17 +760,22 @@ def handle_opportunity_index(handler, project_name): available = [] for feature in OPPORTUNITY_FEATURES: data = _load_opportunity(project_name, feature) - available.append({ - "feature": feature, - "available": data is not None, - "analysis_type": data.get("analysis_type") if data else None, - }) - _json_response(handler, { - "project": project_name, - "total_features": len(OPPORTUNITY_FEATURES), - "available_count": sum(1 for f in available if f["available"]), - "features": available, - }) + available.append( + { + "feature": feature, + "available": data is not None, + "analysis_type": data.get("analysis_type") if data else None, + } + ) + _json_response( + handler, + { + "project": project_name, + "total_features": len(OPPORTUNITY_FEATURES), + "available_count": sum(1 for f in available if f["available"]), + "features": available, + }, + ) # ── Router ───────────────────────────────────────────────────── @@ -750,20 +796,76 @@ def handle_opportunity_index(handler, project_name): (r"^/api/health-trend/(.+)$", handle_health_trend, True), # Opportunity endpoints (14 features) — specific routes FIRST, generic LAST (r"^/api/opportunity/all/(.+)$", lambda h, p: handle_opportunity_all(h, p), True), - (r"^/api/opportunity/learning-velocity/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "learning-velocity"), True), - (r"^/api/opportunity/frustration-to-automation/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "frustration-to-automation"), True), - (r"^/api/opportunity/knowledge-gap/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "knowledge-gap"), True), - (r"^/api/opportunity/token-efficiency/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "token-efficiency"), True), - (r"^/api/opportunity/session-quality/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "session-quality"), True), - (r"^/api/opportunity/ai-agent-mastery/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "ai-agent-mastery"), True), - (r"^/api/opportunity/creative-dna/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "creative-dna"), True), - (r"^/api/opportunity/neurodivergent-profile/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "neurodivergent-profile"), True), - (r"^/api/opportunity/model-selection-advisor/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "model-selection-advisor"), True), - (r"^/api/opportunity/before-after-snapshot/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "before-after-snapshot"), True), - (r"^/api/opportunity/cross-repo-transfer/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "cross-repo-transfer"), True), - (r"^/api/opportunity/youtube-learning-graph/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "youtube-learning-graph"), True), - (r"^/api/opportunity/architecture-timelapse/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "architecture-timelapse"), True), - (r"^/api/opportunity/commit-cognitive-load/(.+)$", lambda h, p: handle_opportunity_feature(h, p, "commit-cognitive-load"), True), + ( + r"^/api/opportunity/learning-velocity/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "learning-velocity"), + True, + ), + ( + r"^/api/opportunity/frustration-to-automation/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "frustration-to-automation"), + True, + ), + ( + r"^/api/opportunity/knowledge-gap/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "knowledge-gap"), + True, + ), + ( + r"^/api/opportunity/token-efficiency/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "token-efficiency"), + True, + ), + ( + r"^/api/opportunity/session-quality/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "session-quality"), + True, + ), + ( + r"^/api/opportunity/ai-agent-mastery/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "ai-agent-mastery"), + True, + ), + ( + r"^/api/opportunity/creative-dna/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "creative-dna"), + True, + ), + ( + r"^/api/opportunity/neurodivergent-profile/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "neurodivergent-profile"), + True, + ), + ( + r"^/api/opportunity/model-selection-advisor/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "model-selection-advisor"), + True, + ), + ( + r"^/api/opportunity/before-after-snapshot/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "before-after-snapshot"), + True, + ), + ( + r"^/api/opportunity/cross-repo-transfer/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "cross-repo-transfer"), + True, + ), + ( + r"^/api/opportunity/youtube-learning-graph/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "youtube-learning-graph"), + True, + ), + ( + r"^/api/opportunity/architecture-timelapse/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "architecture-timelapse"), + True, + ), + ( + r"^/api/opportunity/commit-cognitive-load/(.+)$", + lambda h, p: handle_opportunity_feature(h, p, "commit-cognitive-load"), + True, + ), (r"^/api/opportunity/(.+)$", lambda h, p: handle_opportunity_index(h, p), True), ] diff --git a/archaeology/audit.py b/archaeology/audit.py index 73ace4e..aa89797 100644 --- a/archaeology/audit.py +++ b/archaeology/audit.py @@ -8,7 +8,6 @@ from __future__ import annotations -import json import re import sqlite3 from dataclasses import dataclass @@ -17,7 +16,6 @@ from .utils import _load_json - SEVERITY_ORDER = {"CRITICAL": 0, "HIGH": 1, "MEDIUM": 2, "LOW": 3, "INFO": 4} PUBLISHABLE_SUFFIXES = {".md", ".html", ".json", ".j2"} SENSITIVE_NAME_PATTERNS = [ @@ -126,7 +124,14 @@ def check_canonical_consistency(project_name: str, root: Path) -> list[AuditFind eras = _load_json(eras_path) or {} if not canonical: - findings.append(AuditFinding("CRITICAL", "CANONICAL_MISSING", "canonical-metrics.json is missing or invalid", _rel(canonical_path, root))) + findings.append( + AuditFinding( + "CRITICAL", + "CANONICAL_MISSING", + "canonical-metrics.json is missing or invalid", + _rel(canonical_path, root), + ) + ) return findings expected_commits = _as_int(canonical.get("total_commits")) @@ -136,16 +141,31 @@ def check_canonical_consistency(project_name: str, root: Path) -> list[AuditFind project_overrides = project.get("overrides", {}) if isinstance(project, dict) else {} project_timeline = project.get("timeline", {}) if isinstance(project, dict) else {} checks = [ - ("total_commits", expected_commits, project_overrides.get("total_commits"), project_json_path), + ( + "total_commits", + expected_commits, + project_overrides.get("total_commits"), + project_json_path, + ), ("span_days", expected_days, project_timeline.get("total_days"), project_json_path), ("active_days", expected_active, project_overrides.get("active_days"), project_json_path), ] for metric, expected, observed, path in checks: observed_int = _as_int(observed) if expected is not None and observed_int is not None and expected != observed_int: - findings.append(AuditFinding("HIGH", "PROJECT_DRIFT", f"project.json {metric} does not match canonical metrics", _rel(path, root), f"canonical={expected}, project={observed_int}")) - - tv_meta = data.get("telemetry_visualizations", {}).get("meta", {}) if isinstance(data, dict) else {} + findings.append( + AuditFinding( + "HIGH", + "PROJECT_DRIFT", + f"project.json {metric} does not match canonical metrics", + _rel(path, root), + f"canonical={expected}, project={observed_int}", + ) + ) + + tv_meta = ( + data.get("telemetry_visualizations", {}).get("meta", {}) if isinstance(data, dict) else {} + ) data_checks = [ ("total_commits", expected_commits, tv_meta.get("total_commits")), ("span_days", expected_days, tv_meta.get("lifespan_days") or tv_meta.get("span_days")), @@ -154,26 +174,77 @@ def check_canonical_consistency(project_name: str, root: Path) -> list[AuditFind for metric, expected, observed in data_checks: observed_int = _as_int(observed) if expected is not None and observed_int is not None and expected != observed_int: - findings.append(AuditFinding("HIGH", "DATA_DRIFT", f"data.json {metric} does not match canonical metrics", _rel(data_json_path, root), f"canonical={expected}, data={observed_int}")) + findings.append( + AuditFinding( + "HIGH", + "DATA_DRIFT", + f"data.json {metric} does not match canonical metrics", + _rel(data_json_path, root), + f"canonical={expected}, data={observed_int}", + ) + ) if isinstance(eras, dict): era_commits = _as_int(eras.get("total_commits")) - if expected_commits is not None and era_commits is not None and era_commits != expected_commits: - findings.append(AuditFinding("HIGH", "ERA_DRIFT", "commit-eras.json total_commits does not match canonical metrics", _rel(eras_path, root), f"canonical={expected_commits}, commit-eras={era_commits}")) + if ( + expected_commits is not None + and era_commits is not None + and era_commits != expected_commits + ): + findings.append( + AuditFinding( + "HIGH", + "ERA_DRIFT", + "commit-eras.json total_commits does not match canonical metrics", + _rel(eras_path, root), + f"canonical={expected_commits}, commit-eras={era_commits}", + ) + ) canonical_eras = _as_int(project_overrides.get("era_count")) era_count = len(eras.get("eras", [])) if isinstance(eras.get("eras"), list) else None if canonical_eras is not None and era_count is not None and canonical_eras != era_count: - findings.append(AuditFinding("HIGH", "ERA_COUNT_DRIFT", "commit-eras.json era count does not match project override", _rel(eras_path, root), f"project={canonical_eras}, commit-eras={era_count}")) + findings.append( + AuditFinding( + "HIGH", + "ERA_COUNT_DRIFT", + "commit-eras.json era count does not match project override", + _rel(eras_path, root), + f"project={canonical_eras}, commit-eras={era_count}", + ) + ) db_commits = _db_count(db_path, "commits") if expected_commits is not None and db_commits is not None and db_commits != expected_commits: - findings.append(AuditFinding("HIGH", "DB_COMMIT_DRIFT", "SQLite commits table does not match canonical metrics", _rel(db_path, root), f"canonical={expected_commits}, db={db_commits}")) + findings.append( + AuditFinding( + "HIGH", + "DB_COMMIT_DRIFT", + "SQLite commits table does not match canonical metrics", + _rel(db_path, root), + f"canonical={expected_commits}, db={db_commits}", + ) + ) db_eras = _db_count(db_path, "eras") canonical_eras = _as_int(project_overrides.get("era_count")) if canonical_eras is not None and db_eras is not None and db_eras != canonical_eras: - findings.append(AuditFinding("HIGH", "DB_ERA_DRIFT", "SQLite eras table does not match project era count", _rel(db_path, root), f"project={canonical_eras}, db={db_eras}")) + findings.append( + AuditFinding( + "HIGH", + "DB_ERA_DRIFT", + "SQLite eras table does not match project era count", + _rel(db_path, root), + f"project={canonical_eras}, db={db_eras}", + ) + ) if db_path.exists() and not _table_exists(db_path, "pipeline_runs"): - findings.append(AuditFinding("MEDIUM", "PIPELINE_TABLE_MISSING", "pipeline_runs table is absent; pipeline history queries cannot work", _rel(db_path, root))) + findings.append( + AuditFinding( + "MEDIUM", + "PIPELINE_TABLE_MISSING", + "pipeline_runs table is absent; pipeline history queries cannot work", + _rel(db_path, root), + ) + ) return findings @@ -191,7 +262,10 @@ def walk(obj: Any, path: str = "", excluded: bool = False) -> Iterable[tuple[str provenance = obj.get("provenance") if isinstance(provenance, dict): status = str(provenance.get("status", "")).lower() - if status in {"placeholder_excluded", "excluded", "historical_raw"} or provenance.get("publishable") is False: + if ( + status in {"placeholder_excluded", "excluded", "historical_raw"} + or provenance.get("publishable") is False + ): current_excluded = True yield path, obj, current_excluded if isinstance(obj, dict): @@ -213,13 +287,45 @@ def walk(obj: Any, path: str = "", excluded: bool = False) -> Iterable[tuple[str (excluded_mpc_paths if excluded else mpc_paths).append(path) if len(zero_total_paths) >= 3: - findings.append(AuditFinding("HIGH", "PLACEHOLDER_COAUTHORSHIP", "Repeated all-zero co-authorship rows look placeholder-derived", _rel(data_json_path, root), f"examples={zero_total_paths[:5]}")) + findings.append( + AuditFinding( + "HIGH", + "PLACEHOLDER_COAUTHORSHIP", + "Repeated all-zero co-authorship rows look placeholder-derived", + _rel(data_json_path, root), + f"examples={zero_total_paths[:5]}", + ) + ) elif excluded_zero_total_paths: - findings.append(AuditFinding("INFO", "PLACEHOLDER_COAUTHORSHIP_EXCLUDED", "Co-authorship placeholder rows are explicitly marked non-publishable", _rel(data_json_path, root), f"count={len(excluded_zero_total_paths)}")) + findings.append( + AuditFinding( + "INFO", + "PLACEHOLDER_COAUTHORSHIP_EXCLUDED", + "Co-authorship placeholder rows are explicitly marked non-publishable", + _rel(data_json_path, root), + f"count={len(excluded_zero_total_paths)}", + ) + ) if len(mpc_paths) >= 3: - findings.append(AuditFinding("MEDIUM", "PLACEHOLDER_SESSION_DEPTH", "Repeated messages_per_commit=1.0 rows look placeholder-derived", _rel(data_json_path, root), f"examples={mpc_paths[:5]}")) + findings.append( + AuditFinding( + "MEDIUM", + "PLACEHOLDER_SESSION_DEPTH", + "Repeated messages_per_commit=1.0 rows look placeholder-derived", + _rel(data_json_path, root), + f"examples={mpc_paths[:5]}", + ) + ) elif excluded_mpc_paths: - findings.append(AuditFinding("INFO", "PLACEHOLDER_SESSION_DEPTH_EXCLUDED", "Session-depth placeholder rows are explicitly marked non-publishable", _rel(data_json_path, root), f"count={len(excluded_mpc_paths)}")) + findings.append( + AuditFinding( + "INFO", + "PLACEHOLDER_SESSION_DEPTH_EXCLUDED", + "Session-depth placeholder rows are explicitly marked non-publishable", + _rel(data_json_path, root), + f"count={len(excluded_mpc_paths)}", + ) + ) return findings @@ -236,8 +342,20 @@ def check_sensitive_artifacts(project_name: str, root: Path) -> list[AuditFindin if sensitive_paths: manifest = project_root / "PRIVACY-MANIFEST.md" severity = "INFO" if manifest.exists() else "MEDIUM" - message = "Private/raw data artifacts are present and governed by the project privacy manifest" if manifest.exists() else "Private/raw data artifacts are present in the project tree" - findings.append(AuditFinding(severity, "SENSITIVE_ARTIFACTS", message, _rel(project_root, root), "examples=" + ", ".join(sensitive_paths[:8]))) + message = ( + "Private/raw data artifacts are present and governed by the project privacy manifest" + if manifest.exists() + else "Private/raw data artifacts are present in the project tree" + ) + findings.append( + AuditFinding( + severity, + "SENSITIVE_ARTIFACTS", + message, + _rel(project_root, root), + "examples=" + ", ".join(sensitive_paths[:8]), + ) + ) for path in _iter_publishable_files([project_root / "data", project_root / "deliverables"]): try: @@ -246,7 +364,14 @@ def check_sensitive_artifacts(project_name: str, root: Path) -> list[AuditFindin continue for pattern in SECRET_PATTERNS: if pattern.search(text): - findings.append(AuditFinding("CRITICAL", "SECRET_PATTERN", "Potential secret/private key pattern found", _rel(path, root))) + findings.append( + AuditFinding( + "CRITICAL", + "SECRET_PATTERN", + "Potential secret/private key pattern found", + _rel(path, root), + ) + ) break return findings @@ -256,13 +381,34 @@ def check_project_config(project_name: str, root: Path) -> list[AuditFinding]: project_json_path = _project_dir(project_name, root) / "project.json" project = _load_json(project_json_path) if not isinstance(project, dict): - findings.append(AuditFinding("CRITICAL", "PROJECT_JSON_INVALID", "project.json is missing or invalid", _rel(project_json_path, root))) + findings.append( + AuditFinding( + "CRITICAL", + "PROJECT_JSON_INVALID", + "project.json is missing or invalid", + _rel(project_json_path, root), + ) + ) return findings for key in ("name", "description", "repo_url"): if not str(project.get(key, "")).strip(): - findings.append(AuditFinding("MEDIUM", "PROJECT_FIELD_EMPTY", f"project.json field '{key}' is empty", _rel(project_json_path, root))) + findings.append( + AuditFinding( + "MEDIUM", + "PROJECT_FIELD_EMPTY", + f"project.json field '{key}' is empty", + _rel(project_json_path, root), + ) + ) if project.get("repo_url") and not str(project["repo_url"]).startswith("https://github.com/"): - findings.append(AuditFinding("MEDIUM", "PROJECT_REPO_URL", "repo_url should be a GitHub HTTPS URL", _rel(project_json_path, root))) + findings.append( + AuditFinding( + "MEDIUM", + "PROJECT_REPO_URL", + "repo_url should be a GitHub HTTPS URL", + _rel(project_json_path, root), + ) + ) return findings @@ -300,12 +446,15 @@ def check_era_references(project_name: str, root: Path) -> list[AuditFinding]: severity = "MEDIUM" code = "ERA_STALE_COUNT" - findings.append(AuditFinding( - severity, code, - f"Stale era reference: {ref.old_value} (expected: {ref.expected})", - path=_rel(ref.file, root), - detail=f"line {ref.line}, kind={ref.kind}", - )) + findings.append( + AuditFinding( + severity, + code, + f"Stale era reference: {ref.old_value} (expected: {ref.expected})", + path=_rel(ref.file, root), + detail=f"line {ref.line}, kind={ref.kind}", + ) + ) return findings @@ -315,11 +464,19 @@ def run_audit(project_name: str, root: str | Path = ".") -> list[AuditFinding]: project_root = _project_dir(project_name, root_path) findings: list[AuditFinding] = [] if not project_root.exists(): - return [AuditFinding("CRITICAL", "PROJECT_MISSING", f"Project '{project_name}' does not exist", _rel(project_root, root_path))] + return [ + AuditFinding( + "CRITICAL", + "PROJECT_MISSING", + f"Project '{project_name}' does not exist", + _rel(project_root, root_path), + ) + ] for check in ( check_project_config, check_mined_history, + check_derived_evidence, check_canonical_consistency, check_placeholder_data, check_sensitive_artifacts, @@ -327,7 +484,9 @@ def run_audit(project_name: str, root: str | Path = ".") -> list[AuditFinding]: ): findings.extend(check(project_name, root_path)) - return sorted(findings, key=lambda f: (SEVERITY_ORDER.get(f.severity, 99), f.code, f.path or "")) + return sorted( + findings, key=lambda f: (SEVERITY_ORDER.get(f.severity, 99), f.code, f.path or "") + ) def has_blocking_findings(findings: Iterable[AuditFinding], fail_on: str = "HIGH") -> bool: @@ -346,48 +505,95 @@ def check_mined_history(project_name: str, root: Path) -> list[AuditFinding]: """Reconcile extraction identities and byte bindings when mining evidence exists.""" import csv import hashlib + project = _project_dir(project_name, root) - coverage_path = project / 'data' / 'coverage.json' + coverage_path = project / "data" / "coverage.json" coverage = _load_json(coverage_path) - config = _load_json(project / 'project.json') or {} - if not coverage_path.exists() and not config.get('mined_history_manifest_required'): + config = _load_json(project / "project.json") or {} + if not coverage_path.exists() and not config.get("mined_history_manifest_required"): return [] # Legacy imported datasets never claimed a mining manifest. - if not isinstance(coverage, dict) or not isinstance(coverage.get('artifact_sha256'), dict): - return [AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Mining coverage manifest missing or invalid')] + if not isinstance(coverage, dict) or not isinstance(coverage.get("artifact_sha256"), dict): + return [ + AuditFinding( + "HIGH", "MINING_MANIFEST_INVALID", "Mining coverage manifest missing or invalid" + ) + ] findings = [] - expected_names = {'github-commits.csv', 'github-commits-with-stats.txt'} - if set(coverage['artifact_sha256']) != expected_names: - findings.append(AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Mining artifact bindings incomplete')) - for name, expected in coverage['artifact_sha256'].items(): + expected_names = {"github-commits.csv", "github-commits-with-stats.txt"} + if set(coverage["artifact_sha256"]) != expected_names: + findings.append( + AuditFinding("HIGH", "MINING_MANIFEST_INVALID", "Mining artifact bindings incomplete") + ) + for name, expected in coverage["artifact_sha256"].items(): if name not in expected_names: - findings.append(AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Unexpected artifact name')) + findings.append( + AuditFinding("HIGH", "MINING_MANIFEST_INVALID", "Unexpected artifact name") + ) continue - path = project / 'data' / name + path = project / "data" / name if not path.exists() or hashlib.sha256(path.read_bytes()).hexdigest() != expected: - findings.append(AuditFinding('HIGH', 'MINING_ARTIFACT_DRIFT', f'Mined artifact changed: {name}')) - csv_path = project / 'data' / 'github-commits.csv' - db_path = project / 'data' / 'archaeology.db' + findings.append( + AuditFinding("HIGH", "MINING_ARTIFACT_DRIFT", f"Mined artifact changed: {name}") + ) + csv_path = project / "data" / "github-commits.csv" + db_path = project / "data" / "archaeology.db" try: - with csv_path.open(encoding='utf-8', newline='') as handle: + with csv_path.open(encoding="utf-8", newline="") as handle: rows = list(csv.DictReader(handle)) - keys = ('hash', 'date', 'message', 'author') + keys = ("hash", "date", "message", "author") source = [tuple(row[key] for key in keys) for row in rows] - hashes = [row['hash'] for row in rows] + hashes = [row["hash"] for row in rows] if not db_path.exists(): - raise ValueError('Mined history database is missing') + raise ValueError("Mined history database is missing") with sqlite3.connect(db_path) as conn: - stored = list(conn.execute('SELECT hash,date,message,author FROM commits')) - if len(hashes) != coverage.get('commit_count') or len(set(hashes)) != len(hashes) or sorted(source) != sorted(stored): - raise ValueError('Git manifest, CSV and SQLite commit records do not reconcile') + stored = list(conn.execute("SELECT hash,date,message,author FROM commits")) + if ( + len(hashes) != coverage.get("commit_count") + or len(set(hashes)) != len(hashes) + or sorted(source) != sorted(stored) + ): + raise ValueError("Git manifest, CSV and SQLite commit records do not reconcile") from .metrics import calculate_metrics + measured = calculate_metrics(rows) - canonical = _load_json(project / 'deliverables' / 'canonical-metrics.json') or {} + canonical = _load_json(project / "deliverables" / "canonical-metrics.json") or {} if any(canonical.get(key) != value for key, value in measured.items()): - raise ValueError('Canonical metrics differ from source commit measurements') - visual = _load_json(project / 'deliverables' / 'data.json') or {} - meta = visual.get('telemetry_visualizations', {}).get('meta', {}) - if any(visual.get(key) != value or meta.get(key) != value for key, value in measured.items()): - raise ValueError('Visualization metrics differ from source commit measurements') + raise ValueError("Canonical metrics differ from source commit measurements") + visual = _load_json(project / "deliverables" / "data.json") or {} + meta = visual.get("telemetry_visualizations", {}).get("meta", {}) + if any( + visual.get(key) != value or meta.get(key) != value for key, value in measured.items() + ): + raise ValueError("Visualization metrics differ from source commit measurements") except (OSError, ValueError, KeyError, TypeError, sqlite3.Error) as exc: - findings.append(AuditFinding('HIGH', 'MINING_HISTORY_DRIFT', str(exc))) + findings.append(AuditFinding("HIGH", "MINING_HISTORY_DRIFT", str(exc))) return findings + + +def check_derived_evidence(project_name: str, root: Path) -> list[AuditFinding]: + from .provenance import requires_binding, verify_analyses, verify_artifact + + project = _project_dir(project_name, root) + try: + if not requires_binding(project): + return [] + verify_analyses(project) + deliverables = project / "deliverables" + artifacts = { + Path(str(p)[: -len(".provenance.json")]) + for p in deliverables.rglob("*.provenance.json") + } + artifacts.update( + p + for p in [ + deliverables / "reports/ARCHAEOLOGY-REPORT.md", + deliverables / "visuals/report.html", + deliverables / "visuals/archaeology.html", + ] + if p.exists() + ) + for path in artifacts: + verify_artifact(project, path) + except (OSError, ValueError, KeyError, TypeError) as exc: + return [AuditFinding("HIGH", "DERIVED_EVIDENCE_STALE", str(exc))] + return [] diff --git a/archaeology/classifiers/era_detector.py b/archaeology/classifiers/era_detector.py index 58a8621..10bb8d2 100644 --- a/archaeology/classifiers/era_detector.py +++ b/archaeology/classifiers/era_detector.py @@ -23,7 +23,11 @@ "min_gap_days": 3, "velocity_shift_factor": 2.0, "scope_change_keywords": [ - "refactor", "rewrite", "restructure", "migration", "architecture", + "refactor", + "rewrite", + "restructure", + "migration", + "architecture", ], "cross_repo_activation_threshold": 3, } @@ -87,10 +91,7 @@ def detect(self) -> dict: scope_signals = self._detect_scope_changes(commits) repo_signals = self._detect_cross_repo(commits) - all_signals = ( - gap_signals + velocity_signals + author_signals - + scope_signals + repo_signals - ) + all_signals = gap_signals + velocity_signals + author_signals + scope_signals + repo_signals all_signals.sort(key=lambda s: (s["index"], s["type"])) clusters = self._build_clusters(commits) @@ -101,9 +102,7 @@ def detect(self) -> dict: "first": commits[0]["date"][:10] if commits[0]["date"] else "", "last": commits[-1]["date"][:10] if commits[-1]["date"] else "", }, - "active_days": len({ - c["date"][:10] for c in commits if len(c["date"]) >= 10 - }), + "active_days": len({c["date"][:10] for c in commits if len(c["date"]) >= 10}), "signals": all_signals, "cluster_summary": clusters, } @@ -121,6 +120,7 @@ def save(self, output_path: str | Path, result: dict | None = None) -> Path: if result is None: result = self.detect() from ..utils import atomic_write + output_path = Path(output_path) atomic_write(output_path, json.dumps(result, indent=2, ensure_ascii=False)) return output_path @@ -143,9 +143,7 @@ def _load_commits(self) -> list[dict]: ) has_repo = True except sqlite3.OperationalError: - cursor = conn.execute( - "SELECT date, author, message FROM commits ORDER BY date ASC" - ) + cursor = conn.execute("SELECT date, author, message FROM commits ORDER BY date ASC") has_repo = False rows = cursor.fetchall() except sqlite3.OperationalError: @@ -158,12 +156,14 @@ def _load_commits(self) -> list[dict]: commits = [] for row in rows: - commits.append({ - "date": str(row["date"]) if row["date"] else "", - "author": str(row["author"]) if row["author"] else "", - "message": str(row["message"]) if row["message"] else "", - "repo": (str(row["repo"]) if has_repo and row["repo"] else ""), - }) + commits.append( + { + "date": str(row["date"]) if row["date"] else "", + "author": str(row["author"]) if row["author"] else "", + "message": str(row["message"]) if row["message"] else "", + "repo": (str(row["repo"]) if has_repo and row["repo"] else ""), + } + ) return commits # ------------------------------------------------------------------ @@ -184,13 +184,15 @@ def _detect_gaps(self, commits: list[dict]) -> list[dict]: continue gap = (curr_date - prev_date).days if gap >= self.min_gap_days: - signals.append({ - "index": i, - "date": commits[i]["date"][:10] if len(commits[i]["date"]) >= 10 else "", - "type": "gap", - "detail": f"{gap}-day gap since previous commit", - "strength": "strong" if gap >= 7 else "moderate", - }) + signals.append( + { + "index": i, + "date": commits[i]["date"][:10] if len(commits[i]["date"]) >= 10 else "", + "type": "gap", + "detail": f"{gap}-day gap since previous commit", + "strength": "strong" if gap >= 7 else "moderate", + } + ) return signals # ------------------------------------------------------------------ @@ -223,8 +225,10 @@ def _detect_velocity_shifts(self, commits: list[dict]) -> list[dict]: flagged_days = set() for i in range(window, len(days_sorted)): - before_avg = sum(day_counts.get(d, 0) for d in days_sorted[i - window:i]) / window - after_avg = sum(day_counts.get(d, 0) for d in days_sorted[i:i + window]) / max(1, min(window, len(days_sorted) - i)) + before_avg = sum(day_counts.get(d, 0) for d in days_sorted[i - window : i]) / window + after_avg = sum(day_counts.get(d, 0) for d in days_sorted[i : i + window]) / max( + 1, min(window, len(days_sorted) - i) + ) if before_avg == 0: continue @@ -232,7 +236,9 @@ def _detect_velocity_shifts(self, commits: list[dict]) -> list[dict]: ratio = after_avg / before_avg day = days_sorted[i] - if (ratio >= self.velocity_shift_factor or ratio <= 1.0 / self.velocity_shift_factor) and day not in flagged_days: + if ( + ratio >= self.velocity_shift_factor or ratio <= 1.0 / self.velocity_shift_factor + ) and day not in flagged_days: flagged_days.add(day) direction = "up" if ratio > 1 else "down" display_ratio = max(ratio, 1.0 / ratio) @@ -241,13 +247,15 @@ def _detect_velocity_shifts(self, commits: list[dict]) -> list[dict]: (j for j, c in enumerate(commits) if c["date"][:10] == day), 0, ) - signals.append({ - "index": idx, - "date": day, - "type": "velocity", - "detail": f"{display_ratio:.1f}x {direction}shift ({before_avg:.0f}→{after_avg:.0f} commits/day avg)", - "strength": "strong" if display_ratio >= 4.0 else "moderate", - }) + signals.append( + { + "index": idx, + "date": day, + "type": "velocity", + "detail": f"{display_ratio:.1f}x {direction}shift ({before_avg:.0f}→{after_avg:.0f} commits/day avg)", + "strength": "strong" if display_ratio >= 4.0 else "moderate", + } + ) return signals @@ -282,13 +290,15 @@ def primary_author(start: int, end: int) -> str: signals = [] for idx, before, after in deduped: - signals.append({ - "index": idx, - "date": commits[idx]["date"][:10] if len(commits[idx]["date"]) >= 10 else "", - "type": "author", - "detail": f"primary author shifts from {before} to {after}", - "strength": "strong", - }) + signals.append( + { + "index": idx, + "date": commits[idx]["date"][:10] if len(commits[idx]["date"]) >= 10 else "", + "type": "author", + "detail": f"primary author shifts from {before} to {after}", + "strength": "strong", + } + ) return signals # ------------------------------------------------------------------ @@ -331,13 +341,15 @@ def _detect_scope_changes(self, commits: list[dict]) -> list[dict]: day = commits[center]["date"][:10] if len(commits[center]["date"]) >= 10 else "" if day not in flagged: flagged.add(day) - signals.append({ - "index": center, - "date": day, - "type": "scope", - "detail": f"concentrated {', '.join(sorted(keywords))} burst ({count} in {window} commits)", - "strength": "strong" if count >= 6 else "moderate", - }) + signals.append( + { + "index": center, + "date": day, + "type": "scope", + "detail": f"concentrated {', '.join(sorted(keywords))} burst ({count} in {window} commits)", + "strength": "strong" if count >= 6 else "moderate", + } + ) return signals @@ -368,13 +380,17 @@ def _detect_cross_repo(self, commits: list[dict]) -> list[dict]: i, ) if first_appearance > 0: - signals.append({ - "index": first_appearance, - "date": commits[first_appearance]["date"][:10] if len(commits[first_appearance]["date"]) >= 10 else "", - "type": "repo_activation", - "detail": f"repo {repo} reaches {self.cross_repo_threshold} commits", - "strength": "moderate", - }) + signals.append( + { + "index": first_appearance, + "date": commits[first_appearance]["date"][:10] + if len(commits[first_appearance]["date"]) >= 10 + else "", + "type": "repo_activation", + "detail": f"repo {repo} reaches {self.cross_repo_threshold} commits", + "strength": "moderate", + } + ) return signals @@ -414,23 +430,21 @@ def _build_clusters(self, commits: list[dict]) -> list[dict]: if gap >= self.min_gap_days: # Close current cluster - clusters.append(self._summarize_cluster( - commits, cluster_days, day_commits - )) + clusters.append(self._summarize_cluster(commits, cluster_days, day_commits)) cluster_days = [days_sorted[i]] else: cluster_days.append(days_sorted[i]) # Close last cluster if cluster_days: - clusters.append(self._summarize_cluster( - commits, cluster_days, day_commits - )) + clusters.append(self._summarize_cluster(commits, cluster_days, day_commits)) return clusters def _summarize_cluster( - self, commits: list[dict], days: list[str], + self, + commits: list[dict], + days: list[str], day_commits: dict[str, list[int]], ) -> dict: """Build a summary dict for a cluster of active days.""" @@ -450,9 +464,7 @@ def _summarize_cluster( "commit_count": len(cluster_commits), "primary_author": authors.most_common(1)[0][0] if authors else "", "dominant_repo": repos.most_common(1)[0][0] if repos else "", - "daily_breakdown": { - d: len(day_commits.get(d, [])) for d in days - }, + "daily_breakdown": {d: len(day_commits.get(d, [])) for d in days}, } # ------------------------------------------------------------------ @@ -477,6 +489,7 @@ def _deduplicate(boundaries: list[tuple], min_gap: int = 5) -> list[tuple]: # Convenience function (called by CLI) # ---------------------------------------------------------------------- + def detect_signals(project_name: str, config: dict | None = None) -> dict: """Detect signals for a project by name. @@ -524,6 +537,7 @@ def detect_signals(project_name: str, config: dict | None = None) -> dict: # CLI entry point # ---------------------------------------------------------------------- + def main() -> None: """CLI entry point for standalone signal detection. @@ -531,9 +545,7 @@ def main() -> None: python -m archaeology.classifiers.era_detector --project python -m archaeology.classifiers.era_detector --db [--output ] """ - parser = argparse.ArgumentParser( - description="Detect development signals from commit history" - ) + parser = argparse.ArgumentParser(description="Detect development signals from commit history") parser.add_argument( "--project", help="Project name (resolves to projects//data/archaeology.db)", diff --git a/archaeology/cli.py b/archaeology/cli.py index b6dacac..36efd7b 100644 --- a/archaeology/cli.py +++ b/archaeology/cli.py @@ -2,7 +2,6 @@ import json import os - import subprocess import sys import tempfile @@ -34,8 +33,12 @@ def main(): @main.command() @click.argument("project_name") -@click.option("--description", default="Draft archaeology project", help="Human-readable project description") -@click.option("--repo-url", default="https://github.com/example/example", help="GitHub repository URL") +@click.option( + "--description", default="Draft archaeology project", help="Human-readable project description" +) +@click.option( + "--repo-url", default="https://github.com/example/example", help="GitHub repository URL" +) def init(project_name, description, repo_url): """Create a new project directory with default config.""" project_dir = os.path.join("projects", project_name) @@ -65,9 +68,13 @@ def init(project_name, description, repo_url): @main.command() -@click.option("--project", "project_name", default="demo-archaeology", help="Demo project name to create") +@click.option( + "--project", "project_name", default="demo-archaeology", help="Demo project name to create" +) @click.option("--force", is_flag=True, help="Overwrite an existing demo project") -@click.option("--build-db", is_flag=True, help="Build the demo SQLite database after creating files") +@click.option( + "--build-db", is_flag=True, help="Build the demo SQLite database after creating files" +) def demo(project_name, force, build_db): """Create a sanitized demo archaeology project.""" from .demo import create_demo_project @@ -84,7 +91,9 @@ def demo(project_name, force, build_db): cmd = [sys.executable, "-m", "archaeology.db.builder", "--project-root", str(project_root)] _env = os.environ.copy() _pkg_root = str(Path(__file__).parent.parent) - _env["PYTHONPATH"] = _pkg_root + ((os.pathsep + _env["PYTHONPATH"]) if _env.get("PYTHONPATH") else "") + _env["PYTHONPATH"] = _pkg_root + ( + (os.pathsep + _env["PYTHONPATH"]) if _env.get("PYTHONPATH") else "" + ) result = subprocess.run(cmd, check=True, timeout=300, env=_env) if result.returncode != 0: raise click.exceptions.Exit(result.returncode) @@ -96,7 +105,7 @@ def demo(project_name, force, build_db): @click.option("--verbose", "-v", is_flag=True) def mine(repo_path, project, verbose): """Phase 1: Extract data from a git repository.""" - from .extractors.git import extract_git_log, extract_git_log_with_stats, repository_coverage + from .extractors.git import extract_git_log, extract_git_log_with_stats project_dir = _project_dir(project) data_dir = os.path.join(project_dir, "data") @@ -111,7 +120,9 @@ def mine(repo_path, project, verbose): except RuntimeError as exc: raise click.ClickException(str(exc)) from exc if coverage["shallow"]: - raise click.ClickException("Shallow history: fetch complete history before mining; no fetch performed") + raise click.ClickException( + "Shallow history: fetch complete history before mining; no fetch performed" + ) click.echo(f"Extracting git log from {repo_path}...") @@ -132,9 +143,14 @@ def mine(repo_path, project, verbose): if repository_coverage(repo_path) != coverage: raise click.ClickException("Repository refs changed during mining; retry before analysis") - from .utils import atomic_write import hashlib - coverage["artifact_sha256"] = {Path(path).name: hashlib.sha256(Path(path).read_bytes()).hexdigest() for path in (csv_path, stats_path)} + + from .utils import atomic_write + + coverage["artifact_sha256"] = { + Path(path).name: hashlib.sha256(Path(path).read_bytes()).hexdigest() + for path in (csv_path, stats_path) + } atomic_write(Path(data_dir) / "coverage.json", json.dumps(coverage, indent=2)) config_path = Path(project_dir) / "project.json" config = json.loads(config_path.read_text(encoding="utf-8")) @@ -153,14 +169,15 @@ def build_db(project_name, verbose): project_dir = _project_dir(project_name) db_path = os.path.join(project_dir, "data", "archaeology.db") - cmd = [sys.executable, "-m", "archaeology.db.builder", - "--project-root", project_dir] + cmd = [sys.executable, "-m", "archaeology.db.builder", "--project-root", project_dir] if verbose: cmd.append("--verbose") _env = os.environ.copy() _pkg_root = str(Path(__file__).parent.parent) - _env["PYTHONPATH"] = _pkg_root + ((os.pathsep + _env["PYTHONPATH"]) if _env.get("PYTHONPATH") else "") + _env["PYTHONPATH"] = _pkg_root + ( + (os.pathsep + _env["PYTHONPATH"]) if _env.get("PYTHONPATH") else "" + ) result = subprocess.run(cmd, check=True, timeout=300, env=_env) if result.returncode == 0 and os.path.exists(db_path): click.echo(f"Database built at {db_path}") @@ -172,14 +189,20 @@ def build_db(project_name, verbose): @main.command() @click.argument("project_name") @click.option("--port", default=8001, help="Port for Datasette server") -@click.option("--unsafe-cors", is_flag=True, help="Enable Datasette CORS headers. Off by default for local data safety.") +@click.option( + "--unsafe-cors", + is_flag=True, + help="Enable Datasette CORS headers. Off by default for local data safety.", +) def serve(project_name, port, unsafe_cors): """Launch Datasette for a project.""" project_dir = _project_dir(project_name) db_path = os.path.join(project_dir, "data", "archaeology.db") if not os.path.exists(db_path): - click.echo(f"Database not found at {db_path}. Run 'archaeology build-db {project_name}' first.") + click.echo( + f"Database not found at {db_path}. Run 'archaeology build-db {project_name}' first." + ) sys.exit(1) # Load project config for display name @@ -193,9 +216,7 @@ def serve(project_name, port, unsafe_cors): click.echo(f"Error: Invalid JSON in {config_path}: {e}", err=True) sys.exit(1) - display_name = project_config.get("visualization", {}).get( - "title", project_name.upper() - ) + display_name = project_config.get("visualization", {}).get("title", project_name.upper()) # Use project-specific metadata if it exists, otherwise default project_metadata = os.path.join(project_dir, "datasette-metadata.yaml") @@ -219,8 +240,7 @@ def serve(project_name, port, unsafe_cors): tmp.close() metadata_src = tmp.name - cmd = ["datasette", db_path, "--port", str(port), - "--setting", "sql_time_limit_ms,5000"] + cmd = ["datasette", db_path, "--port", str(port), "--setting", "sql_time_limit_ms,5000"] if unsafe_cors: cmd.append("--cors") if metadata_src: @@ -265,8 +285,10 @@ def signals(project_name, config_path, min_gap_days, verbose): result = detect_signals(project_name, config=config or None) if result.get("signals"): - click.echo(f"Detected {len(result['signals'])} signals " - f"across {len(result['cluster_summary'])} clusters.") + click.echo( + f"Detected {len(result['signals'])} signals " + f"across {len(result['cluster_summary'])} clusters." + ) else: click.echo("No significant patterns detected in the commit history.") @@ -280,8 +302,7 @@ def extract_sessions(project_name, sessions_dir, verbose): project_dir = _project_dir(project_name) output_path = os.path.join(project_dir, "data", "raw-sessions.md") - cmd = [sys.executable, "-m", "archaeology.extractors.sessions", - "--output", output_path] + cmd = [sys.executable, "-m", "archaeology.extractors.sessions", "--output", output_path] if sessions_dir: cmd.extend(["--sessions-dir", sessions_dir]) cmd.extend(["--project", project_name]) @@ -290,15 +311,24 @@ def extract_sessions(project_name, sessions_dir, verbose): if result.returncode == 0: click.echo(f"Sessions extracted to {output_path}") else: - click.echo(f"Session extraction failed", err=True) + click.echo("Session extraction failed", err=True) sys.exit(result.returncode) @main.command() @click.argument("project_name") -@click.option("--vector", "-v", "vectors", multiple=True, - help="Run specific analysis vector(s). Repeat for multiple.") -@click.option("--prompts", is_flag=True, help="Show legacy prompt-template instructions instead of running automation") +@click.option( + "--vector", + "-v", + "vectors", + multiple=True, + help="Run specific analysis vector(s). Repeat for multiple.", +) +@click.option( + "--prompts", + is_flag=True, + help="Show legacy prompt-template instructions instead of running automation", +) @click.option("--verbose", is_flag=True, help="Print vector execution detail") def analyze(project_name, vectors, prompts, verbose): """Phase 3: Run automated analysis vectors against project data.""" @@ -338,53 +368,77 @@ def analyze(project_name, vectors, prompts, verbose): @main.command() @click.argument("project_name") -@click.option("--analyzers", "-a", multiple=True, help="Specific analyzer(s) to run (default: all 14)") +@click.option( + "--analyzers", "-a", multiple=True, help="Specific analyzer(s) to run (default: all 14)" +) @click.option("--verbose", "-v", is_flag=True, help="Show detailed output") def opportunity(project_name, analyzers, verbose): - """Run all 14 opportunity analyzers against project data.""" - from .opportunity_analyzers import OpportunityAnalyzer - - target = list(analyzers) if analyzers else None - project_dir = _project_dir(project_name) - if not os.path.isdir(project_dir): - click.echo(f"Project '{project_name}' not found", err=True) - raise click.exceptions.Exit(1) - - click.echo(f"Running opportunity analyzers for '{project_name}'") - if target: - click.echo(f"Analyzers: {', '.join(target)}") - - runner = OpportunityAnalyzer(project_name, project_dir, verbose=verbose) - results = runner.run_all(analyzers=target) - - ok = sum(1 for v in results.values() if v == "OK") - err = sum(1 for v in results.values() if v.startswith("ERROR")) - click.echo(f"\n {ok} OK, {err} errors out of {len(results)} analyzers") - if err: - raise click.exceptions.Exit(1) + """Disabled pending evidence and data-exposure hardening.""" + raise click.ClickException( + "opportunity is disabled pending readiness hardening. Use init, mine, build-db, signals, analyze, visualize, export-report and audit." + ) @main.command("public-case-study") -@click.option("--output", "output_dir", default="public-case-study", help="Output directory for the sanitized public case study") -@click.option("--project", "project_name", default="demo-archaeology", help="Temporary/generated sanitized demo project name") -@click.option("--force", is_flag=True, default=True, help="Overwrite existing generated demo project") +@click.option( + "--output", + "output_dir", + default="public-case-study", + help="Output directory for the sanitized public case study", +) +@click.option( + "--project", + "project_name", + default="demo-archaeology", + help="Temporary/generated sanitized demo project name", +) +@click.option( + "--force", is_flag=True, default=True, help="Overwrite existing generated demo project" +) def public_case_study(output_dir, project_name, force): """Generate a sanitized public case-study showroom.""" from .report import export_public_case_study - path = export_public_case_study(Path.cwd(), output_dir=output_dir, project_name=project_name, force=force) + path = export_public_case_study( + Path.cwd(), output_dir=output_dir, project_name=project_name, force=force + ) click.echo(f"Public case study exported to {path}") @main.command("local-pipeline") -@click.option("--repo", "repo_name", default="dev-archaeology", help="Repository name or owner/name to inspect") -@click.option("--pipeline-dir", default=None, help="Path to the local GITHUB_pipeline workspace (defaults to ARCHAEOLOGY_PIPELINE_ROOT)") -@click.option("--repos-dir", default=None, help="Directory containing local repositories (defaults to ARCHAEOLOGY_REPOS_DIR)") -@click.option("--top-repos", default=20, type=int, help="Number of active repos to review when --run is used") -@click.option("--review-days", default=30, type=int, help="Commit lookback window when --run is used") -@click.option("--run", "run_first", is_flag=True, help="Run the local pipeline before reading latest.json") -@click.option("--fail-on-issues", is_flag=True, help="Exit nonzero if the repo has any local-pipeline findings") -def local_pipeline(repo_name, pipeline_dir, repos_dir, top_repos, review_days, run_first, fail_on_issues): +@click.option( + "--repo", + "repo_name", + default="dev-archaeology", + help="Repository name or owner/name to inspect", +) +@click.option( + "--pipeline-dir", + default=None, + help="Path to the local GITHUB_pipeline workspace (defaults to ARCHAEOLOGY_PIPELINE_ROOT)", +) +@click.option( + "--repos-dir", + default=None, + help="Directory containing local repositories (defaults to ARCHAEOLOGY_REPOS_DIR)", +) +@click.option( + "--top-repos", default=20, type=int, help="Number of active repos to review when --run is used" +) +@click.option( + "--review-days", default=30, type=int, help="Commit lookback window when --run is used" +) +@click.option( + "--run", "run_first", is_flag=True, help="Run the local pipeline before reading latest.json" +) +@click.option( + "--fail-on-issues", + is_flag=True, + help="Exit nonzero if the repo has any local-pipeline findings", +) +def local_pipeline( + repo_name, pipeline_dir, repos_dir, top_repos, review_days, run_first, fail_on_issues +): """Read or run the local GITHUB_pipeline verification status.""" from .local_pipeline import read_local_pipeline_status, run_local_pipeline, status_lines @@ -408,7 +462,12 @@ def local_pipeline(repo_name, pipeline_dir, repos_dir, top_repos, review_days, r "Please set it to the directory containing your local repositories, " "or use --repos-dir option." ) - run_local_pipeline(pipeline_dir=pipeline_dir, repos_dir=repos_dir, top_repos=top_repos, review_days=review_days) + run_local_pipeline( + pipeline_dir=pipeline_dir, + repos_dir=repos_dir, + top_repos=top_repos, + review_days=review_days, + ) status = read_local_pipeline_status(pipeline_dir, repo_name) for line in status_lines(status): click.echo(line) @@ -418,14 +477,27 @@ def local_pipeline(repo_name, pipeline_dir, repos_dir, top_repos, review_days, r @main.command("export-report") @click.argument("project_name") -@click.option("--format", "fmt", type=click.Choice(["markdown", "md", "html"]), default="markdown", help="Report format to export") -@click.option("--output", "output_path", help="Output path. Defaults to project deliverables/ARCHAEOLOGY-REPORT.") +@click.option( + "--format", + "fmt", + type=click.Choice(["markdown", "md", "html"]), + default="markdown", + help="Report format to export", +) +@click.option( + "--output", + "output_path", + help="Output path. Defaults to project deliverables/ARCHAEOLOGY-REPORT.", +) def export_report_cmd(project_name, fmt, output_path): """Export an archaeology report from analysis outputs.""" from .report import export_report project_dir = _project_dir(project_name) - path = export_report(project_name, project_dir, output_path=output_path, fmt=fmt) + try: + path = export_report(project_name, project_dir, output_path=output_path, fmt=fmt) + except (OSError, ValueError) as exc: + raise click.ClickException(str(exc)) from exc click.echo(f"Report exported to {path}") @@ -434,6 +506,7 @@ def export_report_cmd(project_name, fmt, output_path): def visualize(project_name): """Phase 4: Generate visualization HTML from template.""" from .visualization.history import render_history + try: output = render_history(Path(_project_dir(project_name))) except (OSError, ValueError) as exc: @@ -457,7 +530,9 @@ def ingest_pipeline(project_name, logs_dir, verbose): db_path = os.path.join(project_dir, "data", "archaeology.db") if not os.path.exists(db_path): - click.echo(f"Database not found. Run 'archaeology build-db {project_name}' first.", err=True) + click.echo( + f"Database not found. Run 'archaeology build-db {project_name}' first.", err=True + ) sys.exit(1) # Auto-detect pipeline logs dir @@ -479,7 +554,9 @@ def ingest_pipeline(project_name, logs_dir, verbose): click.echo(f"Ingesting pipeline logs from {logs_dir}...") stats = ingest_directory(Path(db_path), Path(logs_dir), verbose=verbose) - click.echo(f" Ingested: {stats['ingested']}, Skipped: {stats['skipped']}, Errors: {len(stats['errors'])}") + click.echo( + f" Ingested: {stats['ingested']}, Skipped: {stats['skipped']}, Errors: {len(stats['errors'])}" + ) for err in stats["errors"]: click.echo(f" ERROR: {err}", err=True) @@ -489,135 +566,20 @@ def ingest_pipeline(project_name, logs_dir, verbose): @click.option("--dry-run", is_flag=True, help="Show what would change without writing") @click.option("--skip-mine", is_flag=True, help="Skip git mining (use existing data)") def cascade(project_name, dry_run, skip_mine): - """Full pipeline: mine → build-db → signals → era cascade → sync → audit.""" - from .era_cascade import cascade as run_cascade - from .extractors.git import extract_git_log, extract_git_log_with_stats, repository_coverage - from .classifiers.era_detector import detect_signals - - project_dir = Path(_project_dir(project_name)) - project_json_path = project_dir / "project.json" - eras_path = project_dir / "data" / "commit-eras.json" - data_dir = project_dir / "data" - - # Load project config for repo path - repo_path = None - if project_json_path.exists(): - try: - pj = json.loads(project_json_path.read_text()) - repo_path = pj.get("repo_path") - if repo_path: - repo_path = os.path.expanduser(repo_path) - except json.JSONDecodeError as e: - click.echo(f"Error: Invalid JSON in {project_json_path}: {e}", err=True) - sys.exit(1) - - # ── Step 1: Mine fresh git data ── - if not skip_mine: - if not repo_path or not os.path.isdir(repo_path): - click.echo(f" SKIP: repo_path not found ({repo_path}). Use --skip-mine to skip mining.") - else: - click.echo(f"\n[1/7] Mining git data from {repo_path}...") - csv_path = data_dir / "github-commits.csv" - try: - count = extract_git_log(repo_path, str(csv_path)) - click.echo(f" Extracted {count} commits") - except (RuntimeError, Exception) as e: - click.echo(f" Error: Git extraction failed: {e}", err=True) - sys.exit(1) - stats_path = data_dir / "github-commits-with-stats.txt" - try: - extract_git_log_with_stats(repo_path, str(stats_path)) - except (RuntimeError, Exception) as e: - click.echo(f" Error: Git stats extraction failed: {e}", err=True) - sys.exit(1) - else: - click.echo(f"\n[1/7] Mining — SKIPPED (--skip-mine)") - - # ── Step 2: Build database ── - click.echo(f"\n[2/7] Building database...") - db_path = data_dir / "archaeology.db" - cmd = [sys.executable, "-m", "archaeology.db.builder", - "--project-root", str(project_dir)] - _env = os.environ.copy() - _pkg_root = str(Path(__file__).parent.parent) - _env["PYTHONPATH"] = _pkg_root + ((os.pathsep + _env["PYTHONPATH"]) if _env.get("PYTHONPATH") else "") - result = subprocess.run(cmd, capture_output=True, text=True, timeout=300, env=_env) - if result.returncode == 0: - click.echo(f" Database built ({db_path})") - else: - click.echo(f" Build failed: {result.stderr}", err=True) - - # ── Step 3: Detect signals ── - click.echo(f"\n[3/7] Detecting signals...") - sig_result = detect_signals(project_name) - n_signals = len(sig_result.get("signals", [])) - n_clusters = len(sig_result.get("cluster_summary", [])) - click.echo(f" {n_signals} signals across {n_clusters} clusters") - - # ── Step 4: Era cascade ── - click.echo(f"\n[4/7] Running era cascade...") - if not eras_path.exists(): - click.echo(f" ERROR: No commit-eras.json found at {eras_path}", err=True) - sys.exit(1) - - if dry_run: - click.echo(" (dry run — no files will be written)") - - cascade_result = run_cascade(project_dir, eras_path, dry_run=dry_run) - click.echo(f" Files scanned: {cascade_result.files_scanned}") - click.echo(f" Files changed: {cascade_result.files_changed}") - click.echo(f" Era fields remapped: {cascade_result.era_fields_remapped}") - click.echo(f" Stale refs remaining: {cascade_result.stale_refs_remaining}") - - # ── Step 5: Sync derived deliverables ── - click.echo(f"\n[5/7] Syncing derived deliverables...") - sync_script = Path(__file__).parent.parent / "scripts" / "sync" / "sync_derived_deliverables.py" - if sync_script.exists(): - sync_cmd = [sys.executable, str(sync_script)] - if dry_run: - sync_cmd.append("--check") - sync_result = subprocess.run(sync_cmd, capture_output=True, text=True, timeout=120) - click.echo(f" {sync_result.stdout.strip()}") - else: - click.echo(" SKIP: sync script not found") - - # ── Step 6: Audit ── - click.echo(f"\n[6/7] Running audit...") - from .audit import has_blocking_findings, run_audit, summarize - findings = run_audit(project_name, root=Path.cwd()) - summary = summarize(findings) - blocking = [f for f in findings if f.severity in ("CRITICAL", "HIGH")] - - # ── Step 7: Opportunity analyzers ── - click.echo(f"\n[7/7] Running opportunity analyzers...") - from .opportunity_analyzers import OpportunityAnalyzer - opp_runner = OpportunityAnalyzer(project_name, str(project_dir), verbose=False) - opp_results = opp_runner.run_all() - opp_ok = sum(1 for v in opp_results.values() if v == "OK") - click.echo(f" {opp_ok}/{len(opp_results)} opportunity analyzers OK") - - if blocking: - click.echo(f" FAIL: {len(blocking)} HIGH/CRITICAL findings") - for f in blocking: - click.echo(f" {f.format()}") - else: - info_count = sum(1 for f in findings if f.severity == "INFO") - click.echo(f" PASS: {info_count} info-only findings") - - if cascade_result.stale_refs_remaining > 0: - click.echo(f"\n WARNING: {cascade_result.stale_refs_remaining} stale era references remain") - if not dry_run: - raise click.exceptions.Exit(1) - elif blocking: - if not dry_run: - raise click.exceptions.Exit(1) - else: - click.echo(f"\n ✓ Pipeline complete. All {cascade_result.files_scanned} deliverables consistent.") + """Disabled pending evidence and data-exposure hardening.""" + raise click.ClickException( + "cascade is disabled pending readiness hardening. Use init, mine, build-db, signals, analyze, visualize, export-report and audit." + ) @main.command() @click.argument("project_name") -@click.option("--fail-on", type=click.Choice(["CRITICAL", "HIGH", "MEDIUM", "LOW"]), default="HIGH", help="Lowest severity that causes nonzero exit") +@click.option( + "--fail-on", + type=click.Choice(["CRITICAL", "HIGH", "MEDIUM", "LOW"]), + default="HIGH", + help="Lowest severity that causes nonzero exit", +) def audit(project_name, fail_on): """Run forensic audit quality gate.""" from .audit import has_blocking_findings, run_audit, summarize @@ -638,23 +600,20 @@ def audit(project_name, fail_on): @main.command() @click.argument("project_name") def validate(project_name): - """Run HTML validation checks.""" - project_dir = _project_dir(project_name) - html_path = os.path.join(project_dir, "deliverables", "archaeology.html") - validator = os.path.join("archaeology", "validators", "validate_html.cjs") + """Validate the generated history HTML and its evidence binding.""" + from .validators.history import validate_history - if not os.path.exists(html_path): - click.echo(f"No archaeology.html found at {html_path}") - sys.exit(1) - - subprocess.run(["node", validator, html_path, "--project-dir", project_dir], check=True, timeout=120) + try: + validate_history(Path(_project_dir(project_name))) + except (OSError, ValueError) as exc: + raise click.ClickException(str(exc)) from exc + click.echo("PASS: generated history HTML and evidence binding") def _aggregate_global(targets, profile, verbose=False): """Merge per-project data into global/ for cross-project narrative.""" import csv import sqlite3 - from datetime import datetime global_dir = os.path.join("global", "data") os.makedirs(global_dir, exist_ok=True) @@ -697,15 +656,16 @@ def _aggregate_global(targets, profile, verbose=False): conn.row_factory = sqlite3.Row try: row = conn.execute( - "SELECT COUNT(*) as cnt, MIN(date) as first, MAX(date) as last " - "FROM commits" + "SELECT COUNT(*) as cnt, MIN(date) as first, MAX(date) as last FROM commits" ).fetchone() - project_summaries.append({ - "name": proj_name, - "total_commits": row["cnt"], - "first_commit": row["first"], - "last_commit": row["last"], - }) + project_summaries.append( + { + "name": proj_name, + "total_commits": row["cnt"], + "first_commit": row["first"], + "last_commit": row["last"], + } + ) except sqlite3.OperationalError: project_summaries.append({"name": proj_name, "total_commits": 0}) finally: @@ -723,14 +683,18 @@ def _aggregate_global(targets, profile, verbose=False): writer = csv.DictWriter(f, fieldnames=fields, extrasaction="ignore") writer.writeheader() writer.writerows(all_commits) - click.echo(f" {len(all_commits)} commits across {len(targets)} projects → global-commits.csv") + click.echo( + f" {len(all_commits)} commits across {len(targets)} projects → global-commits.csv" + ) # Write global signals JSON if all_eras: signals_path = os.path.join(global_dir, "global-signals.json") with open(signals_path, "w", encoding="utf-8") as f: json.dump(all_eras, f, indent=2) - click.echo(f" {len(all_eras)} signal reports across {len(targets)} projects → global-signals.json") + click.echo( + f" {len(all_eras)} signal reports across {len(targets)} projects → global-signals.json" + ) # Write project summaries summaries_path = os.path.join(global_dir, "project-summaries.json") @@ -748,7 +712,9 @@ def _aggregate_global(targets, profile, verbose=False): try: subprocess.run( ["sqlite-utils", "insert", global_db, "commits", tmp_csv, "--csv"], - capture_output=True, check=True, timeout=300, + capture_output=True, + check=True, + timeout=300, ) except (FileNotFoundError, subprocess.CalledProcessError): conn = sqlite3.connect(global_db, timeout=30) @@ -756,9 +722,7 @@ def _aggregate_global(targets, profile, verbose=False): with open(tmp_csv, newline="", encoding="utf-8") as f: reader = csv.DictReader(f) cols = reader.fieldnames - conn.execute( - f"CREATE TABLE commits ({', '.join(c + ' TEXT' for c in cols)})" - ) + conn.execute(f"CREATE TABLE commits ({', '.join(c + ' TEXT' for c in cols)})") for row in reader: placeholders = ", ".join("?" for _ in cols) conn.execute( @@ -769,12 +733,17 @@ def _aggregate_global(targets, profile, verbose=False): finally: conn.close() - click.echo(f" Global DB → global.db") + click.echo(" Global DB → global.db") @main.command() -@click.option("--project", "-p", "projects", multiple=True, - help="Sync specific project(s) only. Defaults to all in profile.json.") +@click.option( + "--project", + "-p", + "projects", + multiple=True, + help="Sync specific project(s) only. Defaults to all in profile.json.", +) @click.option("--skip-mine", is_flag=True, help="Skip git extraction (use cached data)") @click.option("--skip-signals", is_flag=True, help="Skip signal detection") @click.option("--verbose", "-v", is_flag=True) @@ -784,7 +753,10 @@ def sync(projects, skip_mine, skip_signals, verbose): if not os.path.exists(profile_path): profile_path = "config/profile.json" if not os.path.exists(profile_path): - click.echo("No profile.json found. Create one at config/profile.json with your project list.", err=True) + click.echo( + "No profile.json found. Create one at config/profile.json with your project list.", + err=True, + ) sys.exit(1) try: @@ -868,17 +840,24 @@ def sync(projects, skip_mine, skip_signals, verbose): db_path = os.path.join("projects", proj_name, "data", "archaeology.db") click.echo(f" Building DB for {proj_name}...") - cmd = [sys.executable, "-m", "archaeology.db.builder", - "--project-root", os.path.join("projects", proj_name)] + cmd = [ + sys.executable, + "-m", + "archaeology.db.builder", + "--project-root", + os.path.join("projects", proj_name), + ] if verbose: cmd.append("--verbose") _env = os.environ.copy() _pkg_root = str(Path(__file__).parent.parent) - _env["PYTHONPATH"] = _pkg_root + ((os.pathsep + _env["PYTHONPATH"]) if _env.get("PYTHONPATH") else "") + _env["PYTHONPATH"] = _pkg_root + ( + (os.pathsep + _env["PYTHONPATH"]) if _env.get("PYTHONPATH") else "" + ) result = subprocess.run(cmd, capture_output=not verbose, check=True, timeout=300, env=_env) if result.returncode == 0 and os.path.exists(db_path): - click.echo(f" DB built") + click.echo(" DB built") else: click.echo(f" DB build failed (exit {result.returncode})", err=True) @@ -916,7 +895,12 @@ def sync(projects, skip_mine, skip_signals, verbose): @main.command("global-viz") -@click.option("--output", "output_dir", default="global/deliverables", help="Output directory for the global visualization") +@click.option( + "--output", + "output_dir", + default="global/deliverables", + help="Output directory for the global visualization", +) @click.option("--top", "top_n", type=int, help="Limit to top N repos by commit count") @click.option("--year", type=int, help="Only include repos updated in this year") @click.option("--verbose", "-v", is_flag=True) @@ -930,7 +914,10 @@ def global_viz(output_dir, top_n, year, verbose): commits_csv = os.path.join(data_dir, "global-commits.csv") if not os.path.exists(commits_csv) and not os.path.exists(github_json): - click.echo("No global data found. Run 'archaeology fetch-github' or 'archaeology sync' first.", err=True) + click.echo( + "No global data found. Run 'archaeology fetch-github' or 'archaeology sync' first.", + err=True, + ) sys.exit(1) # Build visualization data @@ -953,7 +940,9 @@ def global_viz(output_dir, top_n, year, verbose): html = f.read() # Inline the data JSON - safe_data = json.dumps(viz_data).replace("<", "\\u003c").replace(">", "\\u003e").replace("&", "\\u0026") + safe_data = ( + json.dumps(viz_data).replace("<", "\\u003c").replace(">", "\\u003e").replace("&", "\\u0026") + ) # Replace the placeholder lines inside the script block # Template has: "// GLOBAL_DATA_PLACEHOLDER\nwindow.GLOBAL_DATA = {};" @@ -974,11 +963,18 @@ def global_viz(output_dir, top_n, year, verbose): click.echo(f"Global visualization generated at {output_path}") meta = viz_data.get("meta", {}) - click.echo(f" {meta.get('total_commits', '?')} commits across {meta.get('total_repos', '?')} repos") + click.echo( + f" {meta.get('total_commits', '?')} commits across {meta.get('total_repos', '?')} repos" + ) @main.command("multi-project-dashboard") -@click.option("--output", "output_dir", default="global/deliverables", help="Output directory for the dashboard") +@click.option( + "--output", + "output_dir", + default="global/deliverables", + help="Output directory for the dashboard", +) @click.option("--top", "top_n", type=int, help="Limit to top N repos by commit count") @click.option("--year", type=int, help="Only include repos updated in this year") @click.option("--verbose", "-v", is_flag=True) @@ -1015,7 +1011,12 @@ def multi_project_dashboard(output_dir, top_n, year, verbose): html = f.read() # Inline the data JSON - safe_data = json.dumps(dashboard_data).replace("<", "\\u003c").replace(">", "\\u003e").replace("&", "\\u0026") + safe_data = ( + json.dumps(dashboard_data) + .replace("<", "\\u003c") + .replace(">", "\\u003e") + .replace("&", "\\u0026") + ) # Replace the placeholder old_placeholder = "// DATA_PLACEHOLDER\nwindow.DASHBOARD_DATA = {};" @@ -1035,13 +1036,19 @@ def multi_project_dashboard(output_dir, top_n, year, verbose): click.echo(f"Multi-project dashboard generated at {output_path}") meta = dashboard_data.get("meta", {}) - click.echo(f" {meta.get('total_commits', '?')} commits across {meta.get('total_repos', '?')} repos") + click.echo( + f" {meta.get('total_commits', '?')} commits across {meta.get('total_repos', '?')} repos" + ) click.echo(f" Period: {meta.get('first_date', '?')} to {meta.get('last_date', '?')}") @main.command("fetch-github") -@click.option("--owner", default=os.environ.get("ARCHAEOLOGY_GITHUB_OWNER", ""), help="GitHub username/org") -@click.option("--output", "output_path", default="global/data/github-repos.json", help="Output JSON path") +@click.option( + "--owner", default=os.environ.get("ARCHAEOLOGY_GITHUB_OWNER", ""), help="GitHub username/org" +) +@click.option( + "--output", "output_path", default="global/data/github-repos.json", help="Output JSON path" +) def fetch_github(owner, output_path): """Fetch repo metadata from GitHub API for all repos (no cloning).""" from .visualization.github_fetcher import save_github_data @@ -1075,217 +1082,20 @@ def benchmark(project_name): @main.command("dashboard") @click.option("--port", default=8080, help="Port to serve on") @click.option("--no-open", is_flag=True, help="Don't open browser automatically") -def serve(port, no_open): - """Start local dashboard server for all project deliverables. - - Generates the master dashboard and serves all projects over HTTP. - Accessible from any device on your Tailscale network. - """ - import http.server - import threading - import webbrowser - - from .visualization.dashboard import discover_projects, generate_master_dashboard, generate_project_index, load_api_repos, generate_global_section - - root = Path.cwd() - projects_dir = root / "projects" - global_data_dir = root / "global" / "data" - - # Generate master dashboard - projects = discover_projects(projects_dir) - if not projects: - click.echo("No projects found. Run 'archaeology mine ' first.", err=True) - sys.exit(1) - - # Load API-only repos (no cloning needed) - api_repos = load_api_repos(global_data_dir) if global_data_dir.exists() else [] - # Deduplicate: remove API repos already present as mined projects - mined_names = {p["name"].lower().replace("-", "").replace("_", "") for p in projects} - api_repos = [r for r in api_repos if r["name"].lower().replace("-", "").replace("_", "") not in mined_names] - print(f" After dedup: {len(api_repos)} API-only repos") - owner_labels = {} # populated from ARCHAEOLOGY_GITHUB_OWNER env or left empty - api_section_html = generate_global_section(api_repos, owner_labels) if api_repos else "" - - dashboard_html = generate_master_dashboard(projects, api_section_html=api_section_html, api_repos=api_repos) - - # Symlink global visualizations if they exist - site_dir = root / ".serve" - site_dir.mkdir(exist_ok=True) - global_deliverables = root / "global" / "deliverables" - if global_deliverables.exists(): - for html_file in global_deliverables.glob("*.html"): - link_path = site_dir / html_file.name - if link_path.is_symlink() or link_path.exists(): - link_path.unlink() - link_path.symlink_to(html_file.resolve()) - (site_dir / "index.html").write_text(dashboard_html, encoding="utf-8") - - # Generate per-project index pages and symlink all deliverable files - for proj in projects: - proj_site_dir = site_dir / proj["name"] - proj_site_dir.mkdir(exist_ok=True) - - # Generate project index page - proj_index_html = generate_project_index(proj) - (proj_site_dir / "index.html").write_text(proj_index_html, encoding="utf-8") - - # Symlink ALL deliverable files from all subdirectories - deliverables_dir = projects_dir / proj["name"] / "deliverables" - if deliverables_dir.exists(): - # Symlink top-level data files (data.json, canonical-metrics.json) - for data_file in deliverables_dir.glob("*.json"): - link_path = proj_site_dir / data_file.name - if link_path.is_symlink() or link_path.exists(): - link_path.unlink() - link_path.symlink_to(data_file.resolve()) - - # Symlink all files from each deliverable subdirectory - for sub_dir in deliverables_dir.iterdir(): - if not sub_dir.is_dir(): - continue - target_dir = proj_site_dir / sub_dir.name - target_dir.mkdir(exist_ok=True) - for f in sub_dir.iterdir(): - if f.is_dir(): - continue - link_path = target_dir / f.name - if link_path.is_symlink() or link_path.exists(): - link_path.unlink() - link_path.symlink_to(f.resolve()) - - # Symlink global deliverables for cross-repo analysis - global_deliverables_dir = root / "global" / "deliverables" - if global_deliverables_dir.exists(): - global_site_dir = site_dir / "global" - global_site_dir.mkdir(exist_ok=True) - for f in global_deliverables_dir.rglob("*"): - if f.is_dir(): - continue - rel = f.relative_to(global_deliverables_dir) - link_path = global_site_dir / rel - link_path.parent.mkdir(parents=True, exist_ok=True) - if link_path.is_symlink() or link_path.exists(): - link_path.unlink() - link_path.symlink_to(f.resolve()) - - # Copy md-viewer.html to serve directory - md_viewer_src = root / "archaeology" / "templates" / "md-viewer.html" - if md_viewer_src.exists(): - md_viewer_dst = site_dir / "md-viewer.html" - if md_viewer_dst.exists(): - md_viewer_dst.unlink() - import shutil - shutil.copy2(md_viewer_src, md_viewer_dst) - - total_deliverables = sum(p.get("total_deliverables", 0) for p in projects) - click.echo(f" Master dashboard: {len(projects)} projects") - click.echo(f" Total deliverables: {total_deliverables}") - - # Custom handler: /api/* routes to JSON API, everything else is static files - import functools - from .api import route as api_route - - class DevArchHandler(http.server.SimpleHTTPRequestHandler): - def do_GET(self): - if self.path.startswith("/api/"): - api_route(self) - else: - super().do_GET() - - def log_message(self, fmt, *args): - # Suppress per-request logging for static files, keep for API - if self.path.startswith("/api/"): - click.echo(f" API: {self.path}") - - handler = functools.partial(DevArchHandler, directory=str(site_dir)) - - server = http.server.HTTPServer(("0.0.0.0", port), handler) - url = f"http://localhost:{port}" - - click.echo(f"\n Serving at {url}") - click.echo(f" Tailscale: http://100.115.175.18:{port}") - click.echo(f" Press Ctrl+C to stop\n") - - if not no_open: - threading.Timer(0.5, lambda: webbrowser.open(url)).start() - - try: - server.serve_forever() - except KeyboardInterrupt: - click.echo("\n Server stopped.") - server.server_close() +def dashboard(port, no_open): + """Disabled pending evidence and data-exposure hardening.""" + raise click.ClickException( + "dashboard is disabled pending readiness hardening. Use init, mine, build-db, signals, analyze, visualize, export-report and audit." + ) @main.command("publish-static") @click.option("--output", "output_dir", default="site", help="Output directory for the static site") def publish_static(output_dir): - """Generate a static site for deployment (GitHub Pages, nginx, etc.).""" - import shutil - - from .visualization.dashboard import discover_projects, generate_master_dashboard, generate_project_index, load_api_repos, generate_global_section - - root = Path.cwd() - projects_dir = root / "projects" - site = root / output_dir - - # Clean output directory - if site.exists(): - shutil.rmtree(site) - site.mkdir(parents=True) - - # Generate master dashboard - projects = discover_projects(projects_dir) - if not projects: - click.echo("No projects found.", err=True) - sys.exit(1) - - # Load and deduplicate API repos - global_data_dir = root / "global" / "data" - api_repos = load_api_repos(global_data_dir) if global_data_dir.exists() else [] - mined_names = {p["name"].lower().replace("-", "").replace("_", "") for p in projects} - api_repos = [r for r in api_repos if r["name"].lower().replace("-", "").replace("_", "") not in mined_names] - owner_labels = {} # populated from ARCHAEOLOGY_GITHUB_OWNER env or left empty - api_section_html = generate_global_section(api_repos, owner_labels) if api_repos else "" - - dashboard_html = generate_master_dashboard(projects, api_section_html=api_section_html, api_repos=api_repos) - (site / "index.html").write_text(dashboard_html, encoding="utf-8") - - click.echo(f" Master dashboard: {len(projects)} projects, {len(api_repos)} API repos") - - # Copy global deliverables (dashboard.html, global.html) - global_deliverables = root / "global" / "deliverables" - if global_deliverables.exists(): - for html_file in global_deliverables.glob("*.html"): - shutil.copy2(html_file, site / html_file.name) - click.echo(f" Global visualizations copied") - - # Generate per-project pages - for proj in projects: - proj_site_dir = site / proj["name"] - proj_site_dir.mkdir() - - # Project index - proj_index_html = generate_project_index(proj) - (proj_site_dir / "index.html").write_text(proj_index_html, encoding="utf-8") - - # Copy HTML files from deliverables - deliverables_dir = projects_dir / proj["name"] / "deliverables" - visuals_dir = deliverables_dir / "visuals" - source_dir = visuals_dir if visuals_dir.exists() else deliverables_dir - - for html_file in source_dir.glob("*.html"): - shutil.copy2(html_file, proj_site_dir / html_file.name) - - # Copy data.json - data_json = deliverables_dir / "data.json" - if data_json.exists(): - shutil.copy2(data_json, proj_site_dir / "data.json") - - click.echo(f" {proj['name']}: {len(proj['visuals'])} pages") - - total = sum(len(p["visuals"]) for p in projects) + len(projects) + 1 - click.echo(f"\n Static site generated at {site}/ ({total} pages)") - click.echo(f" Deploy with: rsync -avz {site}/ user@host:/var/www/archaeology/") + """Disabled pending evidence and data-exposure hardening.""" + raise click.ClickException( + "publish-static is disabled pending readiness hardening. Use init, mine, build-db, signals, analyze, visualize, export-report and audit." + ) if __name__ == "__main__": diff --git a/archaeology/db/builder.py b/archaeology/db/builder.py index b5a3671..2f60735 100644 --- a/archaeology/db/builder.py +++ b/archaeology/db/builder.py @@ -26,7 +26,6 @@ from ..utils import _script_dir - # --------------------------------------------------------------------------- # Default table registry (fallback when no project.json or defaults.json) # --------------------------------------------------------------------------- @@ -49,7 +48,10 @@ "telemetry_github_full": {"file": "metrics-github-full.json", "format": "json_nested"}, "telemetry_repo_depth": {"file": "metrics-repo-depth.json", "format": "json_nested"}, "telemetry_visualizations": {"file": "metrics-visualizations.json", "format": "json_nested"}, - "youtube_topic_classification": {"file": "youtube-topic-classification.json", "format": "json_nested"}, + "youtube_topic_classification": { + "file": "youtube-topic-classification.json", + "format": "json_nested", + }, "youtube_engagement": {"file": "youtube-engagement-heuristics.json", "format": "json"}, "youtube_transcript_analysis": {"file": "youtube-transcript-analysis.json", "format": "json"}, "context_management": {"file": "context-management-analysis.json", "format": "json"}, @@ -132,6 +134,7 @@ # Helpers # --------------------------------------------------------------------------- + def log(msg: str, verbose: bool = False) -> None: if verbose: print(f" {msg}") @@ -222,7 +225,9 @@ def extract_nested(data: dict, key: str) -> list[dict] | None: if isinstance(value, list): return value if isinstance(value, dict): - return [{"_key": k, **(v if isinstance(v, dict) else {"value": v})} for k, v in value.items()] + return [ + {"_key": k, **(v if isinstance(v, dict) else {"value": v})} for k, v in value.items() + ] return None @@ -251,6 +256,7 @@ def _import_mapping(db: Path, data: dict, mapping: dict[str, str], verbose: bool # Config loading # --------------------------------------------------------------------------- + def load_table_registry(project_root: Path, verbose: bool = False) -> dict[str, dict]: """Load table registry from project.json, then defaults.json, then fallback. @@ -290,7 +296,9 @@ def load_table_registry(project_root: Path, verbose: bool = False) -> dict[str, return DEFAULT_TABLE_REGISTRY.copy() -def load_nested_key_mappings(project_root: Path, verbose: bool = False) -> dict[str, dict[str, str]]: +def load_nested_key_mappings( + project_root: Path, verbose: bool = False +) -> dict[str, dict[str, str]]: """Load nested key-to-table mappings from project config. Priority: @@ -333,7 +341,10 @@ def resolve_project_root(args: argparse.Namespace) -> Path: # Specialized importers # --------------------------------------------------------------------------- -def import_commit_eras(db: Path, data_dir: Path, era_file: str = "commit-eras.json", verbose: bool = False) -> int: + +def import_commit_eras( + db: Path, data_dir: Path, era_file: str = "commit-eras.json", verbose: bool = False +) -> int: data = load_json(data_dir / era_file, verbose) if data is None: return 0 @@ -345,12 +356,17 @@ def import_commit_eras(db: Path, data_dir: Path, era_file: str = "commit-eras.js return total -def import_derived_patterns(db: Path, data_dir: Path, patterns_file: str = "derived-patterns.json", verbose: bool = False) -> int: +def import_derived_patterns( + db: Path, data_dir: Path, patterns_file: str = "derived-patterns.json", verbose: bool = False +) -> int: data = load_json(data_dir / patterns_file, verbose) if data is None: return 0 total = 0 - named = {"frustration_to_automation_latency": "frustration_patterns", "co_authorship_gap_analysis": "co_authorship_gaps"} + named = { + "frustration_to_automation_latency": "frustration_patterns", + "co_authorship_gap_analysis": "co_authorship_gaps", + } for key, table in named.items(): section = data.get(key) if section is None: @@ -368,7 +384,9 @@ def import_derived_patterns(db: Path, data_dir: Path, patterns_file: str = "deri return total -def import_telemetry_sessions(db: Path, data_dir: Path, sessions_file: str = "metrics-sessions.json", verbose: bool = False) -> int: +def import_telemetry_sessions( + db: Path, data_dir: Path, sessions_file: str = "metrics-sessions.json", verbose: bool = False +) -> int: data = load_json(data_dir / sessions_file, verbose) if data is None: return 0 @@ -407,9 +425,14 @@ def import_audit_files(db: Path, data_dir: Path, verbose: bool = False) -> int: # Registry-driven import (generalized) # --------------------------------------------------------------------------- -def import_from_registry(db: Path, data_dir: Path, registry: dict[str, dict], - nested_mappings: dict[str, dict[str, str]], - verbose: bool = False) -> int: + +def import_from_registry( + db: Path, + data_dir: Path, + registry: dict[str, dict], + nested_mappings: dict[str, dict[str, str]], + verbose: bool = False, +) -> int: """Import all tables defined in the registry. Format handling: @@ -473,16 +496,19 @@ def import_from_registry(db: Path, data_dir: Path, registry: dict[str, dict], # Indexes & FTS # --------------------------------------------------------------------------- + def _validate_table_name(table: str) -> str: """Reject table names that could enable SQL injection.""" - if not re.match(r'^[a-zA-Z_][a-zA-Z0-9_]*$', table): + if not re.match(r"^[a-zA-Z_][a-zA-Z0-9_]*$", table): raise ValueError(f"Invalid table name: {table!r}") return table def table_exists(db: Path, table: str) -> bool: table = _validate_table_name(table) - result = run_su(["query", str(db), f"SELECT name FROM sqlite_master WHERE type='table' AND name='{table}'"]) + result = run_su( + ["query", str(db), f"SELECT name FROM sqlite_master WHERE type='table' AND name='{table}'"] + ) if result.returncode == 0 and result.stdout.strip(): try: rows = json.loads(result.stdout) @@ -504,8 +530,9 @@ def table_columns(db: Path, table: str) -> set[str]: return {row.get("name") for row in rows if row.get("name")} -def create_indexes(db: Path, indexes: list[tuple[str, list[str]]] | None = None, - verbose: bool = False) -> None: +def create_indexes( + db: Path, indexes: list[tuple[str, list[str]]] | None = None, verbose: bool = False +) -> None: if indexes is None: indexes = DEFAULT_INDEXES for table, columns in indexes: @@ -517,11 +544,23 @@ def create_indexes(db: Path, indexes: list[tuple[str, list[str]]] | None = None, if col not in available: log(f"SKIP index {table}.{col} (column not found)", verbose) continue - run_su(["create-index", str(db), table, col, "--name", f"idx_{table}_{col}", "--if-not-exists"], verbose) + run_su( + [ + "create-index", + str(db), + table, + col, + "--name", + f"idx_{table}_{col}", + "--if-not-exists", + ], + verbose, + ) -def create_fts(db: Path, fts_config: list[tuple[str, list[str]]] | None = None, - verbose: bool = False) -> None: +def create_fts( + db: Path, fts_config: list[tuple[str, list[str]]] | None = None, verbose: bool = False +) -> None: if fts_config is None: fts_config = DEFAULT_FTS for table, columns in fts_config: @@ -539,6 +578,7 @@ def create_fts(db: Path, fts_config: list[tuple[str, list[str]]] | None = None, # Summary # --------------------------------------------------------------------------- + def print_summary(db: Path, verbose: bool = False) -> None: result = run_su(["tables", str(db), "--counts"], verbose) if result.returncode == 0 and result.stdout.strip(): @@ -550,7 +590,14 @@ def print_summary(db: Path, verbose: bool = False) -> None: print("\nIndexes:") for line in idx.stdout.strip().splitlines(): print(f" {line}") - fts = run_su(["query", str(db), "SELECT name FROM sqlite_master WHERE type='table' AND name LIKE '%_fts%'"], verbose) + fts = run_su( + [ + "query", + str(db), + "SELECT name FROM sqlite_master WHERE type='table' AND name LIKE '%_fts%'", + ], + verbose, + ) if fts.returncode == 0 and fts.stdout.strip(): try: tables = json.loads(fts.stdout) @@ -566,6 +613,7 @@ def print_summary(db: Path, verbose: bool = False) -> None: # Main # --------------------------------------------------------------------------- + def _assert_commits_ingested(db_path: Path, data_dir: Path) -> None: """Fail loud if github-commits.csv has rows but the commits table is empty. @@ -582,7 +630,9 @@ def _assert_commits_ingested(db_path: Path, data_dir: Path) -> None: conn = sqlite3.connect(db_path) try: tables = {r[0] for r in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")} - db_rows = conn.execute("SELECT COUNT(*) FROM commits").fetchone()[0] if "commits" in tables else 0 + db_rows = ( + conn.execute("SELECT COUNT(*) FROM commits").fetchone()[0] if "commits" in tables else 0 + ) finally: conn.close() if db_rows != csv_rows: @@ -660,6 +710,7 @@ def build_db(project_root: Path, output: Path | None = None, verbose: bool = Fal # query helpers can distinguish "no runs" from "schema missing". try: from .pipeline_ingest import ensure_tables + ensure_tables(db_path) except (OSError, sqlite3.Error) as exc: # pragma: no cover - defensive CLI guard print(f" WARNING: Failed to ensure pipeline tables: {exc}", file=sys.stderr) @@ -671,16 +722,25 @@ def build_db(project_root: Path, output: Path | None = None, verbose: bool = Fal create_fts(db_path, fts_config, verbose) from ..metrics import write_metrics + write_metrics(project_root, db_path) print_summary(db_path, verbose) print(f"\nDone. Database: {db_path}") def main() -> None: - parser = argparse.ArgumentParser(description="Build archaeology SQLite database from data files") + parser = argparse.ArgumentParser( + description="Build archaeology SQLite database from data files" + ) parser.add_argument("--project", help="Project name (resolves to projects//)") - parser.add_argument("--project-root", default=".", help="Direct path to project root (default: .)") - parser.add_argument("--output", default=None, help="Output DB path (default: /data/archaeology.db)") + parser.add_argument( + "--project-root", default=".", help="Direct path to project root (default: .)" + ) + parser.add_argument( + "--output", + default=None, + help="Output DB path (default: /data/archaeology.db)", + ) parser.add_argument("--verbose", action="store_true", help="Print detailed progress") args = parser.parse_args() diff --git a/archaeology/db/pipeline_ingest.py b/archaeology/db/pipeline_ingest.py index 0089056..b68dbc2 100644 --- a/archaeology/db/pipeline_ingest.py +++ b/archaeology/db/pipeline_ingest.py @@ -33,12 +33,10 @@ import json import sqlite3 -import sys from datetime import datetime from pathlib import Path from typing import Optional - SCHEMA = """ CREATE TABLE IF NOT EXISTS pipeline_runs ( id INTEGER PRIMARY KEY AUTOINCREMENT, @@ -128,7 +126,9 @@ def ingest_run(db_path: Path, run_json: dict, source_file: str = "") -> int: conn = sqlite3.connect(str(db_path), timeout=30) try: # Support both old (timestamp) and new (run_timestamp) field names - ts = run_json.get("run_timestamp") or run_json.get("timestamp", datetime.utcnow().isoformat()) + ts = run_json.get("run_timestamp") or run_json.get( + "timestamp", datetime.utcnow().isoformat() + ) # Map new status values to old format for backward compatibility raw_status = run_json.get("status", "unknown") status = _normalize_status(raw_status) @@ -146,13 +146,25 @@ def ingest_run(db_path: Path, run_json: dict, source_file: str = "") -> int: for repo in run_json.get("repos", []): issues = repo.get("issues", []) - issues_count = len(issues) if isinstance(issues, list) else sum(issues.values()) if isinstance(issues, dict) else 0 + issues_count = ( + len(issues) + if isinstance(issues, list) + else sum(issues.values()) + if isinstance(issues, dict) + else 0 + ) issues_json = json.dumps(issues) if isinstance(issues, (list, dict)) else "[]" fixes_raw = repo.get("fixes_applied", 0) - fixes_count = len(fixes_raw) if isinstance(fixes_raw, list) else fixes_raw if isinstance(fixes_raw, int) else 0 + fixes_count = ( + len(fixes_raw) + if isinstance(fixes_raw, list) + else fixes_raw + if isinstance(fixes_raw, int) + else 0 + ) # Extract repo name from multiple possible fields - repo_name = (repo.get("full_name") or repo.get("path") or repo.get("name", "unknown")) + repo_name = repo.get("full_name") or repo.get("path") or repo.get("name", "unknown") conn.execute( "INSERT INTO pipeline_repo_results (run_id, repo_name, tier, status, issues_count, fixes_applied, issues_json) " @@ -186,7 +198,9 @@ def ingest_directory(db_path: Path, logs_dir: Path, verbose: bool = False) -> di try: existing = { row[0] - for row in conn.execute("SELECT source_file FROM pipeline_runs WHERE source_file != ''").fetchall() + for row in conn.execute( + "SELECT source_file FROM pipeline_runs WHERE source_file != ''" + ).fetchall() } finally: conn.close() @@ -224,7 +238,9 @@ def ingest_directory(db_path: Path, logs_dir: Path, verbose: bool = False) -> di return stats -def get_pipeline_history(db_path: Path, repo_name: Optional[str] = None, limit: int = 50) -> list[dict]: +def get_pipeline_history( + db_path: Path, repo_name: Optional[str] = None, limit: int = 50 +) -> list[dict]: """Query pipeline run history, optionally filtered by repo.""" conn = sqlite3.connect(str(db_path), timeout=30) conn.row_factory = sqlite3.Row diff --git a/archaeology/db/queries.py b/archaeology/db/queries.py index e13848b..9884912 100644 --- a/archaeology/db/queries.py +++ b/archaeology/db/queries.py @@ -5,7 +5,6 @@ from pathlib import Path from typing import Optional - # Allowed table names for FTS queries (whitelist validation) _ALLOWED_FTS_TABLES = {"commits", "sessions", "eras"} @@ -16,7 +15,7 @@ def _validate_table_name(table: str) -> str: Only allows alphanumeric characters and underscores. Raises ValueError if invalid. """ - if not re.match(r'^[a-zA-Z_][a-zA-Z0-9_]*$', table): + if not re.match(r"^[a-zA-Z_][a-zA-Z0-9_]*$", table): raise ValueError(f"Invalid table name: {table}") return table @@ -27,7 +26,7 @@ def _validate_order_by(col: str) -> str: parts = col.split() if len(parts) > 2: raise ValueError(f"Invalid order_by: {col!r}") - if not re.match(r'^[a-zA-Z_][a-zA-Z0-9_]*$', parts[0]): + if not re.match(r"^[a-zA-Z_][a-zA-Z0-9_]*$", parts[0]): raise ValueError(f"Invalid column name in order_by: {parts[0]!r}") if len(parts) == 2 and parts[1].upper() not in ("ASC", "DESC"): raise ValueError(f"Invalid sort direction: {parts[1]!r}") @@ -126,14 +125,15 @@ def get_fts_results(db_path: str, table: str, query_text: str, limit: int = 50) """ # Validate table name against whitelist to prevent SQL injection if table not in _ALLOWED_FTS_TABLES: - raise ValueError(f"Table '{table}' not allowed for FTS queries. Allowed: {sorted(_ALLOWED_FTS_TABLES)}") + raise ValueError( + f"Table '{table}' not allowed for FTS queries. Allowed: {sorted(_ALLOWED_FTS_TABLES)}" + ) conn = get_connection(db_path) try: fts_table = f"{table}_fts" rows = conn.execute( - f"SELECT * FROM {fts_table} WHERE {fts_table} MATCH ? LIMIT ?", - [query_text, limit] + f"SELECT * FROM {fts_table} WHERE {fts_table} MATCH ? LIMIT ?", [query_text, limit] ).fetchall() return [dict(r) for r in rows] finally: @@ -144,7 +144,9 @@ def get_table_list(db_path: str) -> list[str]: """Get all table names in the database.""" conn = get_connection(db_path) try: - rows = conn.execute("SELECT name FROM sqlite_master WHERE type='table' ORDER BY name").fetchall() + rows = conn.execute( + "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name" + ).fetchall() return [r[0] for r in rows] finally: conn.close() @@ -164,10 +166,12 @@ def get_table_count(db_path: str, table: str) -> int: def get_pipeline_runs(db_path: str, repo_name: str | None = None, limit: int = 50) -> list[dict]: """Query pipeline run history from the pipeline_runs table.""" from .pipeline_ingest import get_pipeline_history + return get_pipeline_history(Path(db_path), repo_name=repo_name, limit=limit) def get_repo_quality_trend(db_path: str, repo_name: str, limit: int = 30) -> list[dict]: """Get quality trend for a repo across pipeline runs.""" from .pipeline_ingest import get_repo_quality_trend as _trend + return _trend(Path(db_path), repo_name=repo_name, limit=limit) diff --git a/archaeology/demo.py b/archaeology/demo.py index 1849d8c..754a815 100644 --- a/archaeology/demo.py +++ b/archaeology/demo.py @@ -6,7 +6,6 @@ import json from pathlib import Path - DEMO_PROJECT = "demo-archaeology" @@ -15,7 +14,9 @@ def _write_json(path: Path, data: object) -> None: path.write_text(json.dumps(data, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") -def create_demo_project(root: str | Path = ".", project_name: str = DEMO_PROJECT, force: bool = False) -> Path: +def create_demo_project( + root: str | Path = ".", project_name: str = DEMO_PROJECT, force: bool = False +) -> Path: """Create a small sanitized demo project under projects/. The demo uses invented commit/session data. It contains no raw private logs, @@ -24,7 +25,9 @@ def create_demo_project(root: str | Path = ".", project_name: str = DEMO_PROJECT root = Path(root) project_root = root / "projects" / project_name if project_root.exists() and not force: - raise FileExistsError(f"Demo project already exists at {project_root}. Use force=True to overwrite.") + raise FileExistsError( + f"Demo project already exists at {project_root}. Use force=True to overwrite." + ) data_dir = project_root / "data" deliverables_dir = project_root / "deliverables" @@ -55,12 +58,27 @@ def create_demo_project(root: str | Path = ".", project_name: str = DEMO_PROJECT ) commits = [ - ["demo001", "2026-01-01 09:00:00 +0000", "docs: write initial product intent", "Demo Developer"], + [ + "demo001", + "2026-01-01 09:00:00 +0000", + "docs: write initial product intent", + "Demo Developer", + ], ["demo002", "2026-01-01 11:00:00 +0000", "feat: scaffold prototype", "Agent"], ["demo003", "2026-01-02 15:30:00 +0000", "fix: wire prototype output", "Agent"], ["demo004", "2026-01-03 10:15:00 +0000", "test: add behavior checks", "Agent"], - ["demo005", "2026-01-05 13:00:00 +0000", "refactor: extract audit boundary", "Demo Developer"], - ["demo006", "2026-01-05 16:45:00 +0000", "docs: publish remediation notes", "Demo Developer"], + [ + "demo005", + "2026-01-05 13:00:00 +0000", + "refactor: extract audit boundary", + "Demo Developer", + ], + [ + "demo006", + "2026-01-05 16:45:00 +0000", + "docs: publish remediation notes", + "Demo Developer", + ], ] with (data_dir / "github-commits.csv").open("w", newline="", encoding="utf-8") as handle: writer = csv.writer(handle) @@ -70,8 +88,16 @@ def create_demo_project(root: str | Path = ".", project_name: str = DEMO_PROJECT _write_json( data_dir / "human-messages.json", [ - {"session_id": "demo-session-1", "timestamp": "2026-01-01T09:00:00Z", "messages": "We need a prototype that proves the core loop."}, - {"session_id": "demo-session-2", "timestamp": "2026-01-03T10:00:00Z", "messages": "The audit should catch wiring gaps before launch."}, + { + "session_id": "demo-session-1", + "timestamp": "2026-01-01T09:00:00Z", + "messages": "We need a prototype that proves the core loop.", + }, + { + "session_id": "demo-session-2", + "timestamp": "2026-01-03T10:00:00Z", + "messages": "The audit should catch wiring gaps before launch.", + }, ], ) @@ -82,9 +108,30 @@ def create_demo_project(root: str | Path = ".", project_name: str = DEMO_PROJECT "lifespan": "5 days (2026-01-01 to 2026-01-05)", "total_commits": 6, "eras": [ - {"id": 1, "name": "Intent", "dates": "2026-01-01", "commits": 1, "description": "The project goal is written down.", "narrative_arc": "A clear intent appears before code."}, - {"id": 2, "name": "Prototype", "dates": "2026-01-01 to 2026-01-02", "commits": 2, "description": "The prototype is scaffolded and wired.", "narrative_arc": "Implementation pressure exposes the first integration gap."}, - {"id": 3, "name": "Hardening", "dates": "2026-01-03 to 2026-01-05", "commits": 3, "description": "Tests and audit boundaries are added.", "narrative_arc": "The project shifts from making claims to proving them."}, + { + "id": 1, + "name": "Intent", + "dates": "2026-01-01", + "commits": 1, + "description": "The project goal is written down.", + "narrative_arc": "A clear intent appears before code.", + }, + { + "id": 2, + "name": "Prototype", + "dates": "2026-01-01 to 2026-01-02", + "commits": 2, + "description": "The prototype is scaffolded and wired.", + "narrative_arc": "Implementation pressure exposes the first integration gap.", + }, + { + "id": 3, + "name": "Hardening", + "dates": "2026-01-03 to 2026-01-05", + "commits": 3, + "description": "Tests and audit boundaries are added.", + "narrative_arc": "The project shifts from making claims to proving them.", + }, ], }, ) diff --git a/archaeology/era_cascade.py b/archaeology/era_cascade.py index b2b5502..6b289a8 100644 --- a/archaeology/era_cascade.py +++ b/archaeology/era_cascade.py @@ -13,23 +13,28 @@ import json import re -from dataclasses import dataclass, field +from dataclasses import dataclass from pathlib import Path -from .era_mapper import EraDef, load_eras, remap_json_era_fields, era_count, get_current_era_names +from .era_mapper import EraDef, era_count, get_current_era_names, load_eras, remap_json_era_fields from .era_scanner import scan_deliverables - # Known old era names that may appear in deliverables KNOWN_OLD_NAMES = { - "The Acceleration", "The Crusade", "The Hardening", - "The Threshold", "The Surface", "The Return", + "The Acceleration", + "The Crusade", + "The Hardening", + "The Threshold", + "The Surface", + "The Return", } # Files exempt from era name replacement (historical mapping docs) -EXEMPT_FILES: frozenset[str] = frozenset({ - "ERA_UPDATE_SUMMARY.md", -}) +EXEMPT_FILES: frozenset[str] = frozenset( + { + "ERA_UPDATE_SUMMARY.md", + } +) @dataclass @@ -129,8 +134,9 @@ def _sync_project_json( # Fix era_colors — trim to n_eras entries viz = pj.setdefault("visualization", {}) colors = viz.setdefault("era_colors", {}) - trimmed = {f"era-{i+1:02d}": colors.get(f"era-{i+1:02d}", _default_color(i)) - for i in range(n_eras)} + trimmed = { + f"era-{i + 1:02d}": colors.get(f"era-{i + 1:02d}", _default_color(i)) for i in range(n_eras) + } if len(colors) != len(trimmed) or colors != trimmed: viz["era_colors"] = trimmed changed = True @@ -144,15 +150,21 @@ def _sync_project_json( def _default_color(index: int) -> str: """Default era color palette.""" palette = [ - "#4ade80", "#f87171", "#fb923c", "#60a5fa", "#a78bfa", - "#34d399", "#fbbf24", "#f472b6", "#c084fc", "#9ca3af", + "#4ade80", + "#f87171", + "#fb923c", + "#60a5fa", + "#a78bfa", + "#34d399", + "#fbbf24", + "#f472b6", + "#c084fc", + "#9ca3af", ] return palette[index % len(palette)] -def _remap_data_json( - data_json: Path, eras: list[EraDef], dry_run: bool -) -> int: +def _remap_data_json(data_json: Path, eras: list[EraDef], dry_run: bool) -> int: """Remap era fields in data.json using date-based calculation.""" data = json.loads(data_json.read_text()) changes = remap_json_era_fields(data, eras) @@ -210,6 +222,7 @@ def _fix_html_file( def _fix_css_vars(line: str, n_eras: int, result: CascadeResult) -> str: """Remove or fix era CSS variables beyond current count.""" + def _replace(m: re.Match) -> str: num = int(m.group(1)) if num > n_eras: @@ -222,6 +235,7 @@ def _replace(m: re.Match) -> str: def _fix_era_range(line: str, n_eras: int, result: CascadeResult) -> str: """Cap data-era-range upper bound to n_eras.""" + def _replace(m: re.Match) -> str: low = int(m.group(1)) high = int(m.group(2)) @@ -243,10 +257,22 @@ def _fix_era_count_text(line: str, n_eras: int, result: CascadeResult) -> str: return line word_map = { - 1: "One", 2: "Two", 3: "Three", 4: "Four", 5: "Five", - 6: "Six", 7: "Seven", 8: "Eight", 9: "Nine", 10: "Ten", - 11: "Eleven", 12: "Twelve", 13: "Thirteen", 14: "Fourteen", - 15: "Fifteen", 16: "Sixteen", + 1: "One", + 2: "Two", + 3: "Three", + 4: "Four", + 5: "Five", + 6: "Six", + 7: "Seven", + 8: "Eight", + 9: "Nine", + 10: "Ten", + 11: "Eleven", + 12: "Twelve", + 13: "Thirteen", + 14: "Fourteen", + 15: "Fifteen", + 16: "Sixteen", } target_word = word_map.get(n_eras, str(n_eras)) @@ -275,12 +301,10 @@ def _fix_era_number_text(line: str, n_eras: int, result: CascadeResult) -> str: return line -def _fix_embedded_json_eras( - content: str, eras: list[EraDef], result: CascadeResult -) -> str: +def _fix_embedded_json_eras(content: str, eras: list[EraDef], result: CascadeResult) -> str: """Fix 'era': N fields in embedded HTML data using nearby date context.""" lines = content.splitlines(keepends=True) - new_lines = [] + _new_lines = [] era_line_indices = [] for i, line in enumerate(lines): @@ -294,13 +318,17 @@ def _fix_embedded_json_eras( # Search backwards up to 20 lines for a date field found_date = None for j in range(max(0, idx - 20), idx + 1): - dm = re.search(r'"(?:date|first_expression|estimated_hook_commit)":\s*"(\d{4}-\d{2}-\d{2})', lines[j]) + dm = re.search( + r'"(?:date|first_expression|estimated_hook_commit)":\s*"(\d{4}-\d{2}-\d{2})', + lines[j], + ) if dm: found_date = dm.group(1) break if found_date: from .era_mapper import era_from_date + new_era = era_from_date(eras, found_date) if new_era is not None: lines[idx] = re.sub(r'"era":\s*\d+', f'"era": {new_era}', lines[idx]) diff --git a/archaeology/era_mapper.py b/archaeology/era_mapper.py index 740a590..7468dbb 100644 --- a/archaeology/era_mapper.py +++ b/archaeology/era_mapper.py @@ -71,6 +71,7 @@ def load_eras(eras_path: Path) -> list[EraDef]: if not eras_path.exists(): return [] import re as _re + raw = json.loads(eras_path.read_text()) year = _infer_year(raw) eras = [] @@ -96,13 +97,15 @@ def load_eras(eras_path: Path) -> list[EraDef]: if isinstance(commits, str): m = _re.search(r"(\d+)", commits) commits = int(m.group(1)) if m else 0 - eras.append(EraDef( - id=era["id"], - name=era["name"], - start=start, - end=end, - commits=commits, - )) + eras.append( + EraDef( + id=era["id"], + name=era["name"], + start=start, + end=end, + commits=commits, + ) + ) return eras @@ -120,9 +123,7 @@ def era_from_date(eras: list[EraDef], date_str: str) -> int | None: return None -def remap_json_era_fields( - data: Any, eras: list[EraDef] -) -> list[tuple[int, int, str]]: +def remap_json_era_fields(data: Any, eras: list[EraDef]) -> list[tuple[int, int, str]]: """Walk JSON structure and remap all 'era' fields based on their date fields. Returns list of (old_era, new_era, date_string) for each change made. @@ -133,17 +134,13 @@ def remap_json_era_fields( return changed -def _remap_walk( - obj: Any, eras: list[EraDef], changed: list[tuple[int, int, str]] -) -> None: +def _remap_walk(obj: Any, eras: list[EraDef], changed: list[tuple[int, int, str]]) -> None: """Recursively walk and remap era fields.""" if isinstance(obj, dict): if "era" in obj: # Try multiple date field names date_val = ( - obj.get("date") - or obj.get("first_expression") - or obj.get("estimated_hook_commit") + obj.get("date") or obj.get("first_expression") or obj.get("estimated_hook_commit") ) if date_val: new_era = era_from_date(eras, date_val) diff --git a/archaeology/era_scanner.py b/archaeology/era_scanner.py index acd55d7..1feebb7 100644 --- a/archaeology/era_scanner.py +++ b/archaeology/era_scanner.py @@ -12,18 +12,20 @@ from dataclasses import dataclass, field from pathlib import Path -from .era_mapper import EraDef, get_current_era_names, era_count - +from .era_mapper import EraDef, era_count, get_current_era_names # Files that legitimately contain historical era references (mapping docs) -HISTORICAL_FILES: frozenset[str] = frozenset({ - "ERA_UPDATE_SUMMARY.md", -}) +HISTORICAL_FILES: frozenset[str] = frozenset( + { + "ERA_UPDATE_SUMMARY.md", + } +) def _load_canonical_metrics(project_dir: Path) -> dict: """Load canonical metrics from project.json for semantic drift detection.""" import json + pjson = project_dir / "project.json" if not pjson.exists(): return {} @@ -43,9 +45,9 @@ def _load_canonical_metrics(project_dir: Path) -> dict: class EraRef: file: Path line: int - kind: str # "era_number" | "era_name" | "era_css_var" | "era_count" | "era_json_field" + kind: str # "era_number" | "era_name" | "era_css_var" | "era_count" | "era_json_field" old_value: str - expected: str # what it should be, or "N/A" if unmappable + expected: str # what it should be, or "N/A" if unmappable @dataclass @@ -59,9 +61,7 @@ def has_findings(self) -> bool: return len(self.refs) > 0 -def scan_deliverables( - project_dir: Path, eras: list[EraDef] -) -> ScanResult: +def scan_deliverables(project_dir: Path, eras: list[EraDef]) -> ScanResult: """Scan all deliverable files for stale era references.""" deliverables_dir = project_dir / "deliverables" if not deliverables_dir.exists(): @@ -101,7 +101,7 @@ def _scan_file( return rel = path.relative_to(path.parents[2]) # relative to project dir - is_historical = any(hf in str(rel) for hf in HISTORICAL_FILES) + _is_historical = any(hf in str(rel) for hf in HISTORICAL_FILES) # Build per-era commit count map from canonical source era_commits = {e.id: e.commits for e in eras} @@ -112,7 +112,8 @@ def _scan_file( # Skip historical/context lines if re.search( r"Original Claim|originally reported|was \d+ eras|Corrected To", - line, re.I, + line, + re.I, ): continue @@ -122,21 +123,29 @@ def _scan_file( for m in re.finditer(r"\bEra\s+(\d+)\b", line): num = int(m.group(1)) if num > n_eras: - result.refs.append(EraRef( - file=path, line=i, kind="era_number", - old_value=f"Era {num}", - expected=f"Era 1-{n_eras} (remap by date)", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_number", + old_value=f"Era {num}", + expected=f"Era 1-{n_eras} (remap by date)", + ) + ) # Check "era-NN" CSS variables where NN > n_eras for m in re.finditer(r"era-(\d{2})", line): num = int(m.group(1)) if num > n_eras: - result.refs.append(EraRef( - file=path, line=i, kind="era_css_var", - old_value=f"era-{m.group(1)}", - expected=f"era-01 through era-{n_eras:02d}", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_css_var", + old_value=f"era-{m.group(1)}", + expected=f"era-01 through era-{n_eras:02d}", + ) + ) # Check "N eras" count text — only flag if clearly claiming to be the total # Known stale total counts from previous structures: 10, 14, 15, 16 @@ -144,52 +153,79 @@ def _scan_file( for m in re.finditer(r"\b(\d+)\s+eras\b", line, re.I): num = int(m.group(1)) if num in known_stale_totals: - result.refs.append(EraRef( - file=path, line=i, kind="era_count", - old_value=f"{num} eras", - expected=f"{n_eras} eras", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_count", + old_value=f"{num} eras", + expected=f"{n_eras} eras", + ) + ) # Check "Ten Eras" / "Fourteen Eras" style word_map = { - "Ten": 10, "Eleven": 11, "Twelve": 12, "Thirteen": 13, - "Fourteen": 14, "Fifteen": 15, "Sixteen": 16, - "Seven": 7, "Eight": 8, "Nine": 9, + "Ten": 10, + "Eleven": 11, + "Twelve": 12, + "Thirteen": 13, + "Fourteen": 14, + "Fifteen": 15, + "Sixteen": 16, + "Seven": 7, + "Eight": 8, + "Nine": 9, } - for m in re.finditer(r"\b(Ten|Eleven|Twelve|Thirteen|Fourteen|Fifteen|Sixteen)\s+Eras\b", line): + for m in re.finditer( + r"\b(Ten|Eleven|Twelve|Thirteen|Fourteen|Fifteen|Sixteen)\s+Eras\b", line + ): num = word_map.get(m.group(1), 0) if num != n_eras: - result.refs.append(EraRef( - file=path, line=i, kind="era_count", - old_value=f"{m.group(1)} Eras", - expected=f"{_number_word(n_eras)} Eras", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_count", + old_value=f"{m.group(1)} Eras", + expected=f"{_number_word(n_eras)} Eras", + ) + ) # Check "era": N in JSON/JS where N > n_eras for m in re.finditer(r'"era":\s*(\d+)', line): num = int(m.group(1)) if num > n_eras or num == 0: - result.refs.append(EraRef( - file=path, line=i, kind="era_json_field", - old_value=f'"era": {num}', - expected=f'"era": 1-{n_eras} (remap by date)', - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_json_field", + old_value=f'"era": {num}', + expected=f'"era": 1-{n_eras} (remap by date)', + ) + ) # Check old era names (skip era names inside quotes that are # clearly part of a mapping table, sub-phase names, or blog titles) known_old_names = { - "The Acceleration", "The Crusade", "The Hardening", + "The Acceleration", + "The Crusade", + "The Hardening", "The Return", } for old_name in known_old_names: if old_name in line and old_name not in current_names: if "→" in line or "->" in line: continue - result.refs.append(EraRef( - file=path, line=i, kind="era_name", - old_value=old_name, - expected=", ".join(sorted(current_names)), - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_name", + old_value=old_name, + expected=", ".join(sorted(current_names)), + ) + ) # --- Semantic drift checks (new) --- @@ -204,11 +240,15 @@ def _scan_file( for m in re.finditer(r"\b(\d+)\s+[Cc]hapters?\b", line): num = int(m.group(1)) if num != n_eras: - result.refs.append(EraRef( - file=path, line=i, kind="era_count", - old_value=f"{num} chapters", - expected=f"{n_eras} (matches era count)", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_count", + old_value=f"{num} chapters", + expected=f"{n_eras} (matches era count)", + ) + ) # Check "N-day development" / "N days" against canonical span_days canonical_span = metrics.get("span_days") @@ -217,19 +257,27 @@ def _scan_file( for m in re.finditer(r"\b(\d+)[\s-]*day", line, re.I): num = int(m.group(1)) if num in known_stale_spans and num != canonical_span: - result.refs.append(EraRef( - file=path, line=i, kind="semantic_drift", - old_value=f"{num} day", - expected=f"{canonical_span} days (from project.json)", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="semantic_drift", + old_value=f"{num} day", + expected=f"{canonical_span} days (from project.json)", + ) + ) # Check "1,050 commits" / "1050 commits" (old Cluster 4 count) for m in re.finditer(r"\b1,?050\s+commits", line): - result.refs.append(EraRef( - file=path, line=i, kind="semantic_drift", - old_value="1,050 commits", - expected="972 commits (Eras 3-7) or era-specific count", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="semantic_drift", + old_value="1,050 commits", + expected="972 commits (Eras 3-7) or era-specific count", + ) + ) # Check per-era commit count drift: only direct "Era N: X commits" patterns # Avoid matching combined counts like "Era 3-4: 691 commits" or sub-periods @@ -241,21 +289,29 @@ def _scan_file( except ValueError: continue if era_num in era_commits and count != era_commits[era_num]: - result.refs.append(EraRef( - file=path, line=i, kind="semantic_drift", - old_value=f"Era {era_num}: {m.group(2)} commits", - expected=f"Era {era_num}: {era_commits[era_num]} commits", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="semantic_drift", + old_value=f"Era {era_num}: {m.group(2)} commits", + expected=f"Era {era_num}: {era_commits[era_num]} commits", + ) + ) # Check "N development eras" / "across N eras" against canonical for m in re.finditer(r"across\s+(\d+)\s+(?:development\s+)?eras?\b", line, re.I): num = int(m.group(1)) if num != n_eras: - result.refs.append(EraRef( - file=path, line=i, kind="era_count", - old_value=f"across {num} eras", - expected=f"across {n_eras} eras", - )) + result.refs.append( + EraRef( + file=path, + line=i, + kind="era_count", + old_value=f"across {num} eras", + expected=f"across {n_eras} eras", + ) + ) # Check JS/HTML script blocks for oversized era arrays if path.suffix in {".html", ".js"}: @@ -265,10 +321,22 @@ def _scan_file( def _number_word(n: int) -> str: """Convert a number to its English word form for era counts.""" words = { - 1: "One", 2: "Two", 3: "Three", 4: "Four", 5: "Five", - 6: "Six", 7: "Seven", 8: "Eight", 9: "Nine", 10: "Ten", - 11: "Eleven", 12: "Twelve", 13: "Thirteen", 14: "Fourteen", - 15: "Fifteen", 16: "Sixteen", + 1: "One", + 2: "Two", + 3: "Three", + 4: "Four", + 5: "Five", + 6: "Six", + 7: "Seven", + 8: "Eight", + 9: "Nine", + 10: "Ten", + 11: "Eleven", + 12: "Twelve", + 13: "Thirteen", + 14: "Fourteen", + 15: "Fifteen", + 16: "Sixteen", } return words.get(n, str(n)) @@ -301,15 +369,13 @@ def _scan_js_era_arrays( _check_era_name_arrays(path, lines, current_names, n_eras, result) -def _check_era_number_arrays( - path: Path, lines: list[str], n_eras: int, result: ScanResult -) -> None: +def _check_era_number_arrays(path: Path, lines: list[str], n_eras: int, result: ScanResult) -> None: """Find JS arrays containing { era: N } entries where N exceeds n_eras.""" # Track array start lines and collect era numbers within them in_array = False array_start = 0 era_numbers: list[int] = [] - brace_depth = 0 + _brace_depth = 0 for i, line in enumerate(lines, start=1): stripped = line.strip() @@ -317,33 +383,36 @@ def _check_era_number_arrays( if not in_array: # Detect array start that will contain era entries # Look for = [...] or const xyz = [ - if re.search(r'=\s*\[', stripped) and not stripped.startswith('//'): + if re.search(r"=\s*\[", stripped) and not stripped.startswith("//"): in_array = True array_start = i era_numbers = [] - brace_depth = 0 + _brace_depth = 0 if in_array: # Collect era: N entries - for m in re.finditer(r'\bera:\s*(\d+)', line): + for m in re.finditer(r"\bera:\s*(\d+)", line): era_numbers.append(int(m.group(1))) # Track if array closes - if ']' in stripped: + if "]" in stripped: # Check if this is the closing bracket (rough heuristic) - open_brackets = stripped.count('[') - close_brackets = stripped.count(']') + open_brackets = stripped.count("[") + close_brackets = stripped.count("]") if close_brackets > open_brackets: in_array = False # Evaluate collected era numbers if era_numbers and max(era_numbers) > n_eras: stale = [n for n in era_numbers if n > n_eras] - result.refs.append(EraRef( - file=path, line=array_start, - kind="js_era_array", - old_value=f"array with era entries {era_numbers} ({len(era_numbers)} entries, max={max(era_numbers)})", - expected=f"max {n_eras} entries (stale: {stale})", - )) + result.refs.append( + EraRef( + file=path, + line=array_start, + kind="js_era_array", + old_value=f"array with era entries {era_numbers} ({len(era_numbers)} entries, max={max(era_numbers)})", + expected=f"max {n_eras} entries (stale: {stale})", + ) + ) def _check_era_name_arrays( @@ -362,7 +431,7 @@ def _check_era_name_arrays( stripped = line.strip() if not in_array: - if re.search(r'=\s*\[', stripped) and not stripped.startswith('//'): + if re.search(r"=\s*\[", stripped) and not stripped.startswith("//"): in_array = True array_start = i era_names = [] @@ -372,9 +441,9 @@ def _check_era_name_arrays( for m in re.finditer(r"\bera:\s*['\"]([^'\"]+)['\"]", line): era_names.append(m.group(1)) - if ']' in stripped: - open_brackets = stripped.count('[') - close_brackets = stripped.count(']') + if "]" in stripped: + open_brackets = stripped.count("[") + close_brackets = stripped.count("]") if close_brackets > open_brackets: in_array = False if era_names and len(era_names) > n_eras: @@ -384,9 +453,12 @@ def _check_era_name_arrays( desc = f"array with {len(era_names)} era profiles (expected {n_eras})" if non_current: desc += f", non-current names: {non_current}" - result.refs.append(EraRef( - file=path, line=array_start, - kind="js_era_array", - old_value=desc, - expected=f"{n_eras} entries with names: {', '.join(sorted(current_names))}", - )) + result.refs.append( + EraRef( + file=path, + line=array_start, + kind="js_era_array", + old_value=desc, + expected=f"{n_eras} entries with names: {', '.join(sorted(current_names))}", + ) + ) diff --git a/archaeology/extractors/git.py b/archaeology/extractors/git.py index a143e97..76cf574 100644 --- a/archaeology/extractors/git.py +++ b/archaeology/extractors/git.py @@ -10,7 +10,11 @@ def _git(repo_path: str, *args: str) -> str: try: result = subprocess.run( ["git", "-C", str(Path(repo_path).expanduser()), *args], - capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=300, + capture_output=True, + text=True, + encoding="utf-8", + errors="replace", + timeout=300, ) except FileNotFoundError as exc: raise RuntimeError("git binary not found. Install git and ensure it's on PATH.") from exc @@ -34,7 +38,9 @@ def repository_coverage(repo_path: str) -> dict: "commit_count": count, "refs": [dict(zip(("name", "object"), line.split("\t", 1))) for line in refs.splitlines()], "roots": _git(repo_path, "rev-list", "--all", "--max-parents=0").splitlines(), - "gaps": ["Deleted, inaccessible and unfetched remote refs are outside this local snapshot."], + "gaps": [ + "Deleted, inaccessible and unfetched remote refs are outside this local snapshot." + ], } @@ -45,8 +51,9 @@ def extract_git_log(repo_path: str, output_path: str, verbose: bool = False) -> preserve those values rather than silently shifting or dropping CSV fields. Empty histories write a header, replacing any stale previous extraction. """ - from datetime import datetime, timezone import io + from datetime import datetime, timezone + from ..utils import atomic_write raw = _git(repo_path, "log", "-z", "--all", "--format=%H%x00%aI%x00%s%x00%an") @@ -59,8 +66,10 @@ def extract_git_log(repo_path: str, output_path: str, verbose: bool = False) -> writer = csv.writer(stream, lineterminator="\n") writer.writerow(["hash", "date", "message", "author"]) for i in range(0, len(fields), 4): - sha, date, subject, author = fields[i:i + 4] - normalized = datetime.fromisoformat(date.replace("Z", "+00:00")).astimezone(timezone.utc).isoformat() + sha, date, subject, author = fields[i : i + 4] + normalized = ( + datetime.fromisoformat(date.replace("Z", "+00:00")).astimezone(timezone.utc).isoformat() + ) writer.writerow([sha, normalized, subject, author]) count = len(fields) // 4 expected = int(_git(repo_path, "rev-list", "--all", "--count").strip()) @@ -74,12 +83,11 @@ def extract_git_log(repo_path: str, output_path: str, verbose: bool = False) -> def extract_git_log_with_stats(repo_path: str, output_path: str, verbose: bool = False) -> int: """Extract git log with file change stats.""" - cmd = [ - "git", "-C", repo_path, - "log", "--format=%H%x1f%ai%x1f%s%x1f%an", "--shortstat", "--all" - ] + cmd = ["git", "-C", repo_path, "log", "--format=%H%x1f%ai%x1f%s%x1f%an", "--shortstat", "--all"] try: - result = subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=300) + result = subprocess.run( + cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=300 + ) except FileNotFoundError: raise RuntimeError("git binary not found. Install git and ensure it's on PATH.") except subprocess.TimeoutExpired: @@ -103,7 +111,10 @@ def get_repo_list(repo_path: str) -> list[str]: try: result = subprocess.run( ["gh", "repo", "list", "--limit", "100", "--json", "name,url"], - capture_output=True, text=True, cwd=repo_path, timeout=60 + capture_output=True, + text=True, + cwd=repo_path, + timeout=60, ) except FileNotFoundError: raise RuntimeError("gh CLI not found. Install GitHub CLI and ensure it's on PATH.") @@ -112,6 +123,7 @@ def get_repo_list(repo_path: str) -> list[str]: if result.returncode == 0: import json + repos = json.loads(result.stdout) return [r["name"] for r in repos] return [] diff --git a/archaeology/extractors/sessions.py b/archaeology/extractors/sessions.py index 1e68700..e5d2ca4 100644 --- a/archaeology/extractors/sessions.py +++ b/archaeology/extractors/sessions.py @@ -5,12 +5,13 @@ V2: Better filtering of tool-result noise, focus on genuine dialogue. V3: Project-agnostic -- accepts --sessions-dir, --output, --config, --project args. """ + import argparse +import glob import json import os import re import sys -import glob from datetime import datetime # --------------------------------------------------------------------------- @@ -18,60 +19,153 @@ # --------------------------------------------------------------------------- DEFAULT_EMOTIONAL_KEYWORDS = [ - "frustrat", "excit", "breakthrough", "finally", "aha", "damn", "hell", - "love", "hate", "annoy", "stuck", "confus", "wow", "amazing", "beautiful", - "ugly", "terrible", "horrible", "incredible", "magic", "magical", - "inspired", "inspiration", "creative", "creativity", "art", "artist", - "philosophy", "philosophical", "meaning", "purpose", "vision", - "dream", "passion", "proud", "disappoint", "surpris", "shock", - "satisfy", "satisfying", "elegant", "inelegant", "kludge", "hack", - "eureka", "awesome", "disgust", "delight", "joy", "rage", "anger", - "happy", "sad", "nervous", "anxious", "worried", "relief", "celebrate", - "ship", "shipped", "done", "works", "working", "fixed", "broke", - "painful", "pain", "hard", "easy", "simple", "complicated", + "frustrat", + "excit", + "breakthrough", + "finally", + "aha", + "damn", + "hell", + "love", + "hate", + "annoy", + "stuck", + "confus", + "wow", + "amazing", + "beautiful", + "ugly", + "terrible", + "horrible", + "incredible", + "magic", + "magical", + "inspired", + "inspiration", + "creative", + "creativity", + "art", + "artist", + "philosophy", + "philosophical", + "meaning", + "purpose", + "vision", + "dream", + "passion", + "proud", + "disappoint", + "surpris", + "shock", + "satisfy", + "satisfying", + "elegant", + "inelegant", + "kludge", + "hack", + "eureka", + "awesome", + "disgust", + "delight", + "joy", + "rage", + "anger", + "happy", + "sad", + "nervous", + "anxious", + "worried", + "relief", + "celebrate", + "ship", + "shipped", + "done", + "works", + "working", + "fixed", + "broke", + "painful", + "pain", + "hard", + "easy", + "simple", + "complicated", ] DEFAULT_PHILOSOPHICAL_KEYWORDS = [ - "why are we", "what is the point", "the whole idea", "the vision", - "i want to", "i believe", "the goal is", "the dream", "the mission", - "this project is", "what i really", "the truth is", "honestly", - "at the end of the day", "the real question", "fundamentally", - "the deeper", "meta", "recursive", "self-", "emergent", "emergence", - "consciousness", "intelligence", "creative coding", "agent", - "ai should", "ai could", "ai is", "what if", "imagine", - "i think we should", "let's think about", "bigger picture", - "the story", "the narrative", "the journey", - "this is about", "the whole point", "it's not about", - "we're building", "we are building", "end goal", + "why are we", + "what is the point", + "the whole idea", + "the vision", + "i want to", + "i believe", + "the goal is", + "the dream", + "the mission", + "this project is", + "what i really", + "the truth is", + "honestly", + "at the end of the day", + "the real question", + "fundamentally", + "the deeper", + "meta", + "recursive", + "self-", + "emergent", + "emergence", + "consciousness", + "intelligence", + "creative coding", + "agent", + "ai should", + "ai could", + "ai is", + "what if", + "imagine", + "i think we should", + "let's think about", + "bigger picture", + "the story", + "the narrative", + "the journey", + "this is about", + "the whole point", + "it's not about", + "we're building", + "we are building", + "end goal", ] DEFAULT_REDACT_PATTERNS = [ - (r'sk-[a-zA-Z0-9]{20,}', '[REDACTED_API_KEY]'), - (r'key["\s:=]+[a-zA-Z0-9]{32,}', '[REDACTED_KEY]'), - (r'token["\s:=]+[a-zA-Z0-9]{20,}', '[REDACTED_TOKEN]'), - (r'password["\s:=]+\S+', '[REDACTED_PASSWORD]'), - (r'[\w.+-]+@[\w.-]+\.\w+', '[REDACTED_EMAIL]'), - (r'ghp_[a-zA-Z0-9]{36}', '[REDACTED_GITHUB_TOKEN]'), - (r'gho_[a-zA-Z0-9]{36}', '[REDACTED_GITHUB_TOKEN]'), - (r'github_pat_[a-zA-Z0-9_]{82}', '[REDACTED_GITHUB_TOKEN]'), - (r'AKIA[0-9A-Z]{16}', '[REDACTED_AWS_KEY]'), + (r"sk-[a-zA-Z0-9]{20,}", "[REDACTED_API_KEY]"), + (r'key["\s:=]+[a-zA-Z0-9]{32,}', "[REDACTED_KEY]"), + (r'token["\s:=]+[a-zA-Z0-9]{20,}', "[REDACTED_TOKEN]"), + (r'password["\s:=]+\S+', "[REDACTED_PASSWORD]"), + (r"[\w.+-]+@[\w.-]+\.\w+", "[REDACTED_EMAIL]"), + (r"ghp_[a-zA-Z0-9]{36}", "[REDACTED_GITHUB_TOKEN]"), + (r"gho_[a-zA-Z0-9]{36}", "[REDACTED_GITHUB_TOKEN]"), + (r"github_pat_[a-zA-Z0-9_]{82}", "[REDACTED_GITHUB_TOKEN]"), + (r"AKIA[0-9A-Z]{16}", "[REDACTED_AWS_KEY]"), ] # Label used for config-provided patterns that don't specify one. -_UNLABELLED_REDACT = '[REDACTED]' +_UNLABELLED_REDACT = "[REDACTED]" # --------------------------------------------------------------------------- # Config loading helpers # --------------------------------------------------------------------------- + def load_config(config_path): """Load a JSON config file and return the session_extraction section.""" if not config_path or not os.path.exists(config_path): return {} - with open(config_path, 'r', encoding='utf-8') as f: + with open(config_path, "r", encoding="utf-8") as f: data = json.load(f) - return data.get('session_extraction', {}) + return data.get("session_extraction", {}) def resolve_config(args): @@ -86,17 +180,17 @@ def resolve_config(args): if args.project and not config_path: repo_root = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) - candidate = os.path.join(repo_root, 'projects', args.project, 'project.json') + candidate = os.path.join(repo_root, "projects", args.project, "project.json") if os.path.exists(candidate): config_path = candidate cfg = load_config(config_path) - emotional = cfg.get('emotional_keywords', DEFAULT_EMOTIONAL_KEYWORDS) - philosophical = cfg.get('philosophical_keywords', DEFAULT_PHILOSOPHICAL_KEYWORDS) + emotional = cfg.get("emotional_keywords", DEFAULT_EMOTIONAL_KEYWORDS) + philosophical = cfg.get("philosophical_keywords", DEFAULT_PHILOSOPHICAL_KEYWORDS) # Redaction patterns: config may provide raw regex strings or [pattern, label] pairs. - raw_patterns = cfg.get('redaction_patterns', None) + raw_patterns = cfg.get("redaction_patterns", None) if raw_patterns is None: redact_patterns = list(DEFAULT_REDACT_PATTERNS) else: @@ -114,6 +208,7 @@ def resolve_config(args): # Text processing helpers # --------------------------------------------------------------------------- + def redact_text(text, redact_patterns=None): if redact_patterns is None: redact_patterns = DEFAULT_REDACT_PATTERNS @@ -134,13 +229,13 @@ def extract_human_text(content): texts = [] for block in content: if isinstance(block, dict): - if block.get('type') == 'text': - texts.append(block.get('text', '')) + if block.get("type") == "text": + texts.append(block.get("text", "")) # Skip tool_result blocks entirely - those are tool outputs, not human text # Skip tool_use blocks - those are agent tool calls elif isinstance(block, str): texts.append(block) - return '\n'.join(texts) + return "\n".join(texts) return str(content) @@ -153,12 +248,12 @@ def extract_assistant_text(content): texts = [] for block in content: if isinstance(block, dict): - if block.get('type') == 'text': - texts.append(block.get('text', '')) + if block.get("type") == "text": + texts.append(block.get("text", "")) # Skip tool_use blocks - we just want the narrative text elif isinstance(block, str): texts.append(block) - return '\n'.join(texts) + return "\n".join(texts) return str(content) @@ -170,38 +265,49 @@ def is_tool_result_noise(text): stripped = text.strip() # Pure file content dumps (start with line numbers) - if re.match(r'^\s*\d+[→|]', stripped): + if re.match(r"^\s*\d+[→|]", stripped): return True # Pure diff output - if stripped.startswith('diff --git') or stripped.startswith('--- a/') or stripped.startswith('+++ b/'): + if ( + stripped.startswith("diff --git") + or stripped.startswith("--- a/") + or stripped.startswith("+++ b/") + ): return True - if re.match(r'^\+[^+]', stripped) and len(stripped.split('\n')) > 5: + if re.match(r"^\+[^+]", stripped) and len(stripped.split("\n")) > 5: return True - if re.match(r'^-[^-]', stripped) and len(stripped.split('\n')) > 5: + if re.match(r"^-[^-]", stripped) and len(stripped.split("\n")) > 5: return True # Pure file path listings - if stripped.startswith('/') and len(stripped.split('\n')) > 5 and all(l.startswith('/') for l in stripped.split('\n')): + if ( + stripped.startswith("/") + and len(stripped.split("\n")) > 5 + and all(line.startswith("/") for line in stripped.split("\n")) + ): return True # Mostly code (over 60% looks like code) - lines = stripped.split('\n') + lines = stripped.split("\n") code_lines = 0 - for l in lines: - if re.match(r'^\s*(import |export |const |let |var |function |class |interface |type |if |for |while |return |}\s*$|{\s*$|\)\s*$|^\s*\d+[→|])', l): + for line in lines: + if re.match( + r"^\s*(import |export |const |let |var |function |class |interface |type |if |for |while |return |}\s*$|{\s*$|\)\s*$|^\s*\d+[→|])", + line, + ): code_lines += 1 if len(lines) > 3 and code_lines / len(lines) > 0.6: return True # Very short file update confirmations - if re.match(r'^The file .+ has been (updated|created|deleted) successfully\.?$', stripped): + if re.match(r"^The file .+ has been (updated|created|deleted) successfully\.?$", stripped): return True # Task notification XML blocks - if stripped.startswith(''): + if stripped.startswith(""): return True - if '' in stripped and '' in stripped: + if "" in stripped and "" in stripped: return True # Subagent completion summaries (auto-generated) @@ -209,22 +315,22 @@ def is_tool_result_noise(text): return True # Messages that are mostly task-notification content - if stripped.count(' 3: + if stripped.count(" 3: return True # Skill/plugin content dumps (Claude Code plugins injecting instructions) - if stripped.startswith('Base directory for this skill:') or 'skills/brainstorming' in stripped: + if stripped.startswith("Base directory for this skill:") or "skills/brainstorming" in stripped: return True - if stripped.startswith('# Brainstorming Ideas') or 'HARD-GATE' in stripped: + if stripped.startswith("# Brainstorming Ideas") or "HARD-GATE" in stripped: return True - if '' in stripped and len(stripped) > 500: + if "" in stripped and len(stripped) > 500: return True # Very long plan dumps (>2000 chars, mostly formatted plans not dialogue) # But keep shorter messages that are plans since they show intent - if len(stripped) > 3000 and stripped.count('\n') > 30: + if len(stripped) > 3000 and stripped.count("\n") > 30: # Check if it's mostly structural content (headers, lists, code blocks) - structural = stripped.count('\n#') + stripped.count('\n-') + stripped.count('\n```') + structural = stripped.count("\n#") + stripped.count("\n-") + stripped.count("\n```") if structural > 15: return True @@ -246,14 +352,15 @@ def is_interesting(text, keywords): # Session parsing # --------------------------------------------------------------------------- + def parse_session(filepath): - session_id = os.path.basename(filepath).replace('.jsonl', '') + session_id = os.path.basename(filepath).replace(".jsonl", "") messages = [] ai_title = None session_timestamp = None try: - with open(filepath, 'r', encoding='utf-8', errors='replace') as f: + with open(filepath, "r", encoding="utf-8", errors="replace") as f: for line in f: line = line.strip() if not line: @@ -263,77 +370,81 @@ def parse_session(filepath): except json.JSONDecodeError: continue - msg_type = d.get('type', '') + msg_type = d.get("type", "") - if session_timestamp is None and 'timestamp' in d: - session_timestamp = d.get('timestamp') + if session_timestamp is None and "timestamp" in d: + session_timestamp = d.get("timestamp") - if msg_type == 'ai-title': - ai_title = d.get('aiTitle', d.get('title', d.get('message', ''))) + if msg_type == "ai-title": + ai_title = d.get("aiTitle", d.get("title", d.get("message", ""))) - elif msg_type == 'user': - msg = d.get('message', {}) - content = extract_human_text(msg.get('content', '')) + elif msg_type == "user": + msg = d.get("message", {}) + content = extract_human_text(msg.get("content", "")) if content and content.strip(): - messages.append({ - 'type': 'user', - 'role': 'user', - 'content': content, - 'timestamp': d.get('timestamp', ''), - }) - - elif msg_type == 'assistant': - msg = d.get('message', {}) - content = extract_assistant_text(msg.get('content', '')) + messages.append( + { + "type": "user", + "role": "user", + "content": content, + "timestamp": d.get("timestamp", ""), + } + ) + + elif msg_type == "assistant": + msg = d.get("message", {}) + content = extract_assistant_text(msg.get("content", "")) if content and content.strip(): - messages.append({ - 'type': 'assistant', - 'role': 'assistant', - 'content': content, - 'timestamp': d.get('timestamp', ''), - }) + messages.append( + { + "type": "assistant", + "role": "assistant", + "content": content, + "timestamp": d.get("timestamp", ""), + } + ) except (json.JSONDecodeError, KeyError, TypeError) as e: return { - 'session_id': session_id, - 'error': str(e), - 'ai_title': ai_title, - 'timestamp': session_timestamp, + "session_id": session_id, + "error": str(e), + "ai_title": ai_title, + "timestamp": session_timestamp, } # Filter human messages to only genuine human dialogue genuine_human = [] for m in messages: - if m['role'] != 'user': + if m["role"] != "user": continue - text = m['content'].strip() + text = m["content"].strip() # Skip IDE notifications - if text.startswith('') or text.startswith(''): + if text.startswith("") or text.startswith(""): continue # Skip pure tool result noise if is_tool_result_noise(text): continue genuine_human.append(m) - assistant_msgs = [m for m in messages if m['role'] == 'assistant'] + assistant_msgs = [m for m in messages if m["role"] == "assistant"] # Generate a fallback title from first substantive human message fallback_title = None for m in genuine_human: - text = m['content'].strip() - if len(text) > 10 and not text.startswith('[Request interrupted'): - first_line = text.split('\n')[0].strip() + text = m["content"].strip() + if len(text) > 10 and not text.startswith("[Request interrupted"): + first_line = text.split("\n")[0].strip() fallback_title = first_line[:80] break return { - 'session_id': session_id, - 'ai_title': ai_title, - 'fallback_title': fallback_title, - 'timestamp': session_timestamp, - 'human_messages': genuine_human, - 'assistant_messages': assistant_msgs, - 'total_human': len(genuine_human), - 'total_assistant': len(assistant_msgs), + "session_id": session_id, + "ai_title": ai_title, + "fallback_title": fallback_title, + "timestamp": session_timestamp, + "human_messages": genuine_human, + "assistant_messages": assistant_msgs, + "total_human": len(genuine_human), + "total_assistant": len(assistant_msgs), } @@ -341,8 +452,10 @@ def parse_session(filepath): # Narrative extraction # --------------------------------------------------------------------------- -def extract_narrative(session_data, emotional_keywords=None, philosophical_keywords=None, - redact_patterns=None): + +def extract_narrative( + session_data, emotional_keywords=None, philosophical_keywords=None, redact_patterns=None +): if emotional_keywords is None: emotional_keywords = DEFAULT_EMOTIONAL_KEYWORDS if philosophical_keywords is None: @@ -350,23 +463,23 @@ def extract_narrative(session_data, emotional_keywords=None, philosophical_keywo if redact_patterns is None: redact_patterns = DEFAULT_REDACT_PATTERNS - sid = session_data['session_id'] - title = session_data.get('ai_title') or session_data.get('fallback_title') or 'Untitled Session' - timestamp = session_data.get('timestamp', '') - human_msgs = session_data.get('human_messages', []) - assistant_msgs = session_data.get('assistant_messages', []) + sid = session_data["session_id"] + title = session_data.get("ai_title") or session_data.get("fallback_title") or "Untitled Session" + timestamp = session_data.get("timestamp", "") + human_msgs = session_data.get("human_messages", []) + assistant_msgs = session_data.get("assistant_messages", []) - if session_data.get('error'): + if session_data.get("error"): return f"## Session: Error\n**ID**: `{sid}`\n**Error**: {session_data['error']}\n\n---\n\n" if not human_msgs and not assistant_msgs: return f"## Session: Empty\n**ID**: `{sid}`\n*No messages found.*\n\n---\n\n" - ts_display = '' + ts_display = "" if timestamp: try: - dt = datetime.fromisoformat(timestamp.replace('Z', '+00:00')) - ts_display = dt.strftime('%Y-%m-%d %H:%M') + dt = datetime.fromisoformat(timestamp.replace("Z", "+00:00")) + ts_display = dt.strftime("%Y-%m-%d %H:%M") except Exception: ts_display = timestamp[:16] @@ -379,7 +492,7 @@ def extract_narrative(session_data, emotional_keywords=None, philosophical_keywo md += "### Human Intent\n\n" first_real = None for m in human_msgs: - text = m['content'].strip() + text = m["content"].strip() if len(text) > 20: first_real = text break @@ -390,18 +503,18 @@ def extract_narrative(session_data, emotional_keywords=None, philosophical_keywo # --- ALL GENUINE HUMAN MESSAGES --- md += "### Human Messages\n\n" for i, m in enumerate(human_msgs): - text = m['content'].strip() + text = m["content"].strip() if not text: continue redacted = redact_text(text, redact_patterns) - md += f"**[{i+1}]** {truncate(redacted, 500)}\n\n" + md += f"**[{i + 1}]** {truncate(redacted, 500)}\n\n" # --- KEY ASSISTANT MOMENTS --- md += "### Key Assistant Responses\n\n" interesting_assistant = [] for idx, m in enumerate(assistant_msgs): - text = m['content'] + text = m["content"] if is_interesting(text, emotional_keywords) or is_interesting(text, philosophical_keywords): interesting_assistant.append(idx) @@ -417,63 +530,102 @@ def extract_narrative(session_data, emotional_keywords=None, philosophical_keywo for idx in sorted(indices_to_show): m = assistant_msgs[idx] - text = redact_text(truncate(m['content'], 500), redact_patterns) - md += f"**[Assistant #{idx+1}]** {text}\n\n" + text = redact_text(truncate(m["content"], 500), redact_patterns) + md += f"**[Assistant #{idx + 1}]** {text}\n\n" # --- EMOTIONAL MOMENTS --- - emotional_human = [m for m in human_msgs if is_interesting(m['content'], emotional_keywords)] - emotional_assistant = [m for m in assistant_msgs if is_interesting(m['content'], emotional_keywords)] + emotional_human = [m for m in human_msgs if is_interesting(m["content"], emotional_keywords)] + emotional_assistant = [ + m for m in assistant_msgs if is_interesting(m["content"], emotional_keywords) + ] if emotional_human or emotional_assistant: md += "### Emotional Moments\n\n" for m in emotional_human[:4]: - text = redact_text(truncate(m['content'], 400), redact_patterns) + text = redact_text(truncate(m["content"], 400), redact_patterns) md += f"**[Human]** {text}\n\n" for m in emotional_assistant[:4]: - text = redact_text(truncate(m['content'], 400), redact_patterns) + text = redact_text(truncate(m["content"], 400), redact_patterns) md += f"**[Agent]** {text}\n\n" # --- PHILOSOPHICAL MOMENTS --- - phil_human = [m for m in human_msgs if is_interesting(m['content'], philosophical_keywords)] - phil_assistant = [m for m in assistant_msgs if is_interesting(m['content'], philosophical_keywords)] + phil_human = [m for m in human_msgs if is_interesting(m["content"], philosophical_keywords)] + phil_assistant = [ + m for m in assistant_msgs if is_interesting(m["content"], philosophical_keywords) + ] if phil_human or phil_assistant: md += "### Philosophical Moments\n\n" for m in phil_human[:4]: - text = redact_text(truncate(m['content'], 400), redact_patterns) + text = redact_text(truncate(m["content"], 400), redact_patterns) md += f"**[Human]** {text}\n\n" for m in phil_assistant[:4]: - text = redact_text(truncate(m['content'], 400), redact_patterns) + text = redact_text(truncate(m["content"], 400), redact_patterns) md += f"**[Agent]** {text}\n\n" # --- CREATIVE DECISIONS --- decision_keywords = [ - "let's call", "named", "rename", "should we", "i think", "what about", - "how about", "instead of", "better to", "let's use", "let's go with", - "the name", "naming", "i prefer", "design decision", "architectural", - "i want", "i don't want", "make it", "this should", "this needs to", - "the idea is", "concept here", "the approach", + "let's call", + "named", + "rename", + "should we", + "i think", + "what about", + "how about", + "instead of", + "better to", + "let's use", + "let's go with", + "the name", + "naming", + "i prefer", + "design decision", + "architectural", + "i want", + "i don't want", + "make it", + "this should", + "this needs to", + "the idea is", + "concept here", + "the approach", ] - decision_msgs = [m for m in human_msgs if is_interesting(m['content'], decision_keywords)] + decision_msgs = [m for m in human_msgs if is_interesting(m["content"], decision_keywords)] if decision_msgs: md += "### Creative/Design Decisions\n\n" for m in decision_msgs[:5]: - text = redact_text(truncate(m['content'], 400), redact_patterns) + text = redact_text(truncate(m["content"], 400), redact_patterns) md += f"**[Human]** {text}\n\n" # --- TECHNICAL BREAKTHROUGHS --- breakthrough_keywords = [ - "finally works", "got it working", "this works", "test passes", - "all tests pass", "build passes", "success", "breakthrough", - "figured out", "solved", "the fix", "working now", - "it's alive", "that did it", "nailed it", "perfect", + "finally works", + "got it working", + "this works", + "test passes", + "all tests pass", + "build passes", + "success", + "breakthrough", + "figured out", + "solved", + "the fix", + "working now", + "it's alive", + "that did it", + "nailed it", + "perfect", + ] + tech_msgs = [ + m + for m in human_msgs + assistant_msgs + if is_interesting(m["content"], breakthrough_keywords) ] - tech_msgs = [m for m in human_msgs + assistant_msgs if is_interesting(m['content'], breakthrough_keywords)] if tech_msgs: md += "### Technical Breakthroughs\n\n" for m in tech_msgs[:4]: - role = "Human" if m['role'] == 'user' else "Agent" - text = redact_text(truncate(m['content'], 400), redact_patterns) + role = "Human" if m["role"] == "user" else "Agent" + text = redact_text(truncate(m["content"], 400), redact_patterns) md += f"**[{role}]** {text}\n\n" md += "---\n\n" @@ -484,25 +636,29 @@ def extract_narrative(session_data, emotional_keywords=None, philosophical_keywo # CLI entry point # --------------------------------------------------------------------------- + def parse_args(argv=None): parser = argparse.ArgumentParser( - description="Extract narrative material from Claude Code JSONL session files.") + description="Extract narrative material from Claude Code JSONL session files." + ) parser.add_argument( "--sessions-dir", default=os.path.expanduser("~/.claude/projects/"), - help="Directory containing .jsonl session files (default: ~/.claude/projects/)") + help="Directory containing .jsonl session files (default: ~/.claude/projects/)", + ) parser.add_argument( "--output", default=None, - help="Output markdown file (default: /data/raw-sessions.md)") + help="Output markdown file (default: /data/raw-sessions.md)", + ) parser.add_argument( - "--config", - default=None, - help="Path to a JSON config file with session_extraction settings") + "--config", default=None, help="Path to a JSON config file with session_extraction settings" + ) parser.add_argument( "--project", default=None, - help="Project name; loads projects//project.json for config") + help="Project name; loads projects//project.json for config", + ) return parser.parse_args(argv) @@ -530,13 +686,13 @@ def main(argv=None): sessions = [] for f in files: - session_id = os.path.basename(f).replace('.jsonl', '') + session_id = os.path.basename(f).replace(".jsonl", "") print(f" Parsing {session_id}...") data = parse_session(f) sessions.append(data) # Sort by timestamp - sessions.sort(key=lambda s: s.get('timestamp') or '') + sessions.sort(key=lambda s: s.get("timestamp") or "") # Derive project name for the header (from --project arg or cwd) project_name = args.project or os.path.basename(os.path.abspath(".")) @@ -545,12 +701,14 @@ def main(argv=None): output_parts.append(f"# {project_name.title()} Session Narratives\n\n") output_parts.append(f"Extracted from {len(files)} Claude Code session logs.\n") output_parts.append(f"Generated: {datetime.now().strftime('%Y-%m-%d %H:%M')}\n") - output_parts.append(f"Note: Tool-result noise (file contents, diffs, code dumps) filtered out. ") + output_parts.append("Note: Tool-result noise (file contents, diffs, code dumps) filtered out. ") output_parts.append("Only genuine human dialogue and agent narrative text included.\n\n") output_parts.append("---\n\n") for data in sessions: - print(f" Extracting narrative for {data['session_id']} ({data['total_human']} human msgs)...") + print( + f" Extracting narrative for {data['session_id']} ({data['total_human']} human msgs)..." + ) narrative = extract_narrative( data, emotional_keywords=emotional_kw, @@ -559,20 +717,20 @@ def main(argv=None): ) output_parts.append(narrative) - full_output = ''.join(output_parts) - with open(output_file, 'w', encoding='utf-8') as f: + full_output = "".join(output_parts) + with open(output_file, "w", encoding="utf-8") as f: f.write(full_output) print(f"\nDone! Written to {output_file}") print(f"Total size: {len(full_output):,} characters") - total_human = sum(s.get('total_human', 0) for s in sessions) - total_assistant = sum(s.get('total_assistant', 0) for s in sessions) - sessions_with_content = sum(1 for s in sessions if s.get('total_human', 0) > 0) + total_human = sum(s.get("total_human", 0) for s in sessions) + total_assistant = sum(s.get("total_assistant", 0) for s in sessions) + sessions_with_content = sum(1 for s in sessions if s.get("total_human", 0) > 0) print(f"Sessions with human dialogue: {sessions_with_content}/{len(sessions)}") print(f"Total genuine human messages: {total_human}") print(f"Total assistant messages: {total_assistant}") -if __name__ == '__main__': +if __name__ == "__main__": main() diff --git a/archaeology/local_pipeline.py b/archaeology/local_pipeline.py index 1e42f3d..32692d2 100644 --- a/archaeology/local_pipeline.py +++ b/archaeology/local_pipeline.py @@ -16,10 +16,7 @@ def _get_dir(env_var: str, label: str) -> Path: env_val = os.environ.get(env_var, "") if env_val: return Path(env_val) - raise OSError( - f"{env_var} environment variable not set. " - f"Please set it to {label}." - ) + raise OSError(f"{env_var} environment variable not set. Please set it to {label}.") DEFAULT_PIPELINE_DIR: Path | None = None @@ -94,16 +91,22 @@ def read_local_pipeline_status(pipeline_dir: str | Path, repo_name: str) -> Loca if isinstance(mission, dict): mission_repo = mission.get("repo", "") # Match against owner/repo or just repo name - if (mission_repo == repo_name or - mission_repo.endswith("/" + repo_name) or - repo_name.endswith("/" + mission_repo.split("/")[-1])): + if ( + mission_repo == repo_name + or mission_repo.endswith("/" + repo_name) + or repo_name.endswith("/" + mission_repo.split("/")[-1]) + ): target = mission break # Fall back to repos array (old format) if target is None: for repo in payload.get("repos", []): - names = {str(repo.get("name", "")), str(repo.get("full_name", "")), str(repo.get("path", ""))} + names = { + str(repo.get("name", "")), + str(repo.get("full_name", "")), + str(repo.get("path", "")), + } if repo_name in names or repo_name.endswith("/" + str(repo.get("name", ""))): target = repo break @@ -119,7 +122,9 @@ def read_local_pipeline_status(pipeline_dir: str | Path, repo_name: str) -> Loca if not reviewed_repos: reviewed_repos = [repo.get("name", "") for repo in payload.get("repos", [])] reviewed = ", ".join(str(r) for r in reviewed_repos) - raise ValueError(f"Repo '{repo_name}' not found in latest local pipeline reviewed repos. Reviewed: {reviewed}") + raise ValueError( + f"Repo '{repo_name}' not found in latest local pipeline reviewed repos. Reviewed: {reviewed}" + ) summary = payload.get("summary", {}) # Normalize issues to dict - pipeline may return list or dict @@ -131,10 +136,10 @@ def read_local_pipeline_status(pipeline_dir: str | Path, repo_name: str) -> Loca # Extract repo name from multiple possible fields repo_full_name = ( - target.get("repo") or # new format - target.get("full_name") or # old format - target.get("path") or - target.get("name", "") + target.get("repo") # new format + or target.get("full_name") # old format + or target.get("path") + or target.get("name", "") ) # Extract health/verdict from mission or repo data diff --git a/archaeology/mcp_server/project_utils.py b/archaeology/mcp_server/project_utils.py index 5186685..d805049 100644 --- a/archaeology/mcp_server/project_utils.py +++ b/archaeology/mcp_server/project_utils.py @@ -47,14 +47,14 @@ def validate_project_name(name: str) -> None: raise ValueError(f"Project name too long (max 100 chars): {len(name)}") # Only allow alphanumeric, dot, underscore, hyphen - if not re.match(r'^[a-zA-Z0-9._-]+$', name): + if not re.match(r"^[a-zA-Z0-9._-]+$", name): raise ValueError(f"Project name contains invalid characters: {name!r}") # Reject path traversal attempts - if '..' in name: + if ".." in name: raise ValueError(f"Project name cannot contain '..': {name!r}") - if name.startswith('/'): + if name.startswith("/"): raise ValueError(f"Project name cannot start with '/': {name!r}") @@ -103,15 +103,17 @@ def list_projects() -> list[dict[str, Any]]: if not project_dir.is_dir() or project_dir.name.startswith((".", "_")): continue config = get_project_config(project_dir.name) - projects.append({ - "name": project_dir.name, - "path": str(project_dir), - "has_data": (project_dir / "data").exists(), - "has_deliverables": (project_dir / "deliverables").exists(), - "has_database": (project_dir / "data" / "archaeology.db").exists(), - "description": config.get("description", "") if config else "", - "repo_url": config.get("repo_url", "") if config else "", - }) + projects.append( + { + "name": project_dir.name, + "path": str(project_dir), + "has_data": (project_dir / "data").exists(), + "has_deliverables": (project_dir / "deliverables").exists(), + "has_database": (project_dir / "data" / "archaeology.db").exists(), + "description": config.get("description", "") if config else "", + "repo_url": config.get("repo_url", "") if config else "", + } + ) return projects diff --git a/archaeology/mcp_server/tools.py b/archaeology/mcp_server/tools.py index b3aa902..1b649b3 100644 --- a/archaeology/mcp_server/tools.py +++ b/archaeology/mcp_server/tools.py @@ -9,7 +9,6 @@ import json import os from contextlib import contextmanager -from pathlib import Path from typing import Any from .project_utils import ( @@ -249,6 +248,7 @@ def devarch_visualize(project_name: str) -> dict[str, Any]: projects = [] if projects_dir.exists(): from archaeology.visualization.dashboard import discover_projects + projects = discover_projects(projects_dir) # Find the matching project @@ -284,7 +284,7 @@ def devarch_report(project_name: str, fmt: str = "html") -> dict[str, Any]: if not project_dir.exists(): return {"error": f"Project '{project_name}' not found"} - from archaeology.report import export_markdown_report, _markdown_to_html + from archaeology.report import _markdown_to_html, export_markdown_report md_path = export_markdown_report(project_name, str(project_dir)) @@ -322,8 +322,7 @@ def devarch_audit(project_name: str, fail_on: str = "HIGH") -> dict[str, Any]: "passed": passed, "failed": failed, "findings": [ - {"check": f.code, "severity": f.severity, "message": f.message} - for f in findings + {"check": f.code, "severity": f.severity, "message": f.message} for f in findings ], } diff --git a/archaeology/metrics.py b/archaeology/metrics.py index ab8fb73..246403c 100644 --- a/archaeology/metrics.py +++ b/archaeology/metrics.py @@ -1,9 +1,11 @@ """Canonical metrics from the mined commits, reconciled against SQLite.""" + import csv import json import sqlite3 from collections import Counter from pathlib import Path + from .utils import _parse_date, atomic_write @@ -19,11 +21,19 @@ def write_metrics(project_root: Path, db_path: Path) -> dict: if len(set(hashes)) != len(hashes) or sorted(hashes) != sorted(stored): raise ValueError("CSV/SQLite commit identities differ or contain duplicates") metrics = calculate_metrics(rows) - atomic_write(project_root / "deliverables" / "canonical-metrics.json", json.dumps(metrics, indent=2)) - atomic_write(project_root / "deliverables" / "data.json", json.dumps({ - **metrics, - "telemetry_visualizations": {"meta": metrics}, - }, indent=2)) + atomic_write( + project_root / "deliverables" / "canonical-metrics.json", json.dumps(metrics, indent=2) + ) + atomic_write( + project_root / "deliverables" / "data.json", + json.dumps( + { + **metrics, + "telemetry_visualizations": {"meta": metrics}, + }, + indent=2, + ), + ) return metrics @@ -38,12 +48,14 @@ def calculate_metrics(rows: list[dict]) -> dict: dates = sorted(days) peak = min(days, key=lambda d: (-days[d], d)) if days else None metrics = { - "total_commits": len(rows), "active_days": len(days), + "total_commits": len(rows), + "active_days": len(days), "daily_commits": dict(sorted(days.items())), "first_commit_date": dates[0] if dates else None, "last_commit_date": dates[-1] if dates else None, "span_days": (_parse_date(dates[-1]) - _parse_date(dates[0])).days + 1 if dates else 0, - "peak_day": peak, "peak_day_commits": days[peak] if peak else 0, + "peak_day": peak, + "peak_day_commits": days[peak] if peak else 0, "date_basis": "author timestamps normalized to UTC; inclusive calendar span", "source_scope": "all locally mined refs, not proof of all remote history", } diff --git a/archaeology/opportunity_analyzers.py b/archaeology/opportunity_analyzers.py index 4153dbc..14b9d3d 100644 --- a/archaeology/opportunity_analyzers.py +++ b/archaeology/opportunity_analyzers.py @@ -1,1226 +1,14 @@ -"""14 missed-opportunity analyzers for DevArch Framework. - -Each analyzer mines existing data (SQLite DB + JSON artifacts) to produce -structured insights that the original pipeline didn't extract. - -Output directory: deliverables/opportunity/ -""" - -from __future__ import annotations - -import json -import math -import re -import sqlite3 -from collections import Counter, defaultdict -from datetime import datetime, timedelta -from pathlib import Path -from typing import Any - -from .utils import _load_json, _parse_date, atomic_write +"""Legacy opportunity inference is disabled: commit history cannot establish personal profiles.""" class OpportunityAnalyzer: - """Runs all 14 opportunity analyzers against a project database.""" - - ANALYZERS = [ - "learning-velocity", - "frustration-to-automation", - "knowledge-gap", - "token-efficiency", - "session-quality", - "ai-agent-mastery", - "creative-dna", - "neurodivergent-profile", - "model-selection-advisor", - "before-after-snapshot", - "cross-repo-transfer", - "youtube-learning-graph", - "architecture-timelapse", - "commit-cognitive-load", - ] - - def __init__(self, project_name: str, project_dir: str, verbose: bool = False): - self.project_name = project_name - self.project_dir = Path(project_dir) - self.verbose = verbose - self.data_dir = self.project_dir / "data" - self.deliverables_dir = self.project_dir / "deliverables" - self.db_path = self.data_dir / "archaeology.db" - self.output_dir = self.deliverables_dir / "opportunity" - self._conn: sqlite3.Connection | None = None - - def _log(self, msg: str) -> None: - if self.verbose: - print(f" [opportunity] {msg}") - - @property - def conn(self) -> sqlite3.Connection: - if self._conn is None: - self._conn = sqlite3.connect(str(self.db_path), timeout=30) - self._conn.row_factory = sqlite3.Row - return self._conn - - def _query(self, sql: str, params: tuple = ()) -> list[dict]: - try: - return [dict(r) for r in self.conn.execute(sql, params).fetchall()] - except sqlite3.OperationalError: - return [] + ANALYZERS = [] - def _query_one(self, sql: str, params: tuple = ()) -> dict | None: - rows = self._query(sql, params) - return rows[0] if rows else None - - def _load(self, rel_path: str) -> Any: - return _load_json(self.project_dir / rel_path) - - def _commit_count(self) -> int: - row = self._query_one("SELECT COUNT(*) as cnt FROM commits") - return int(row["cnt"]) if row else 0 - - def _like_commits(self, keywords: list[str], limit: int = 100) -> list[dict]: - if not keywords: - return [] - clauses = " OR ".join("LOWER(message) LIKE ?" for _ in keywords) - params = tuple(f"%{kw.lower()}%" for kw in keywords) + (limit,) - return self._query( - f"SELECT hash, date, message, author FROM commits WHERE {clauses} ORDER BY date DESC LIMIT ?", - params, + def __init__(self, *args, **kwargs): + raise RuntimeError( + "opportunity is disabled: unsupported personal profiles and measurements" ) - def _eras(self) -> list[dict]: - return self._query("SELECT * FROM eras ORDER BY id") - - def _max_era(self) -> int: - """Canonical max era number from the eras table.""" - eras = self._eras() - return max((e.get("id", 0) for e in eras), default=7) - - def _sanitize_eras(self, data: Any) -> Any: - """Remap era references > canonical max to the canonical range.""" - max_era = self._max_era() - if max_era <= 0: - return data - text = json.dumps(data, default=str) - # Remap "Era N" text where N > max_era - def remap_era_text(m): - n = int(m.group(1)) - return f"Era {min(n, max_era)}" if n > max_era else m.group(0) - text = re.sub(r'Era\s+(\d+)', remap_era_text, text) - # Remap "era": N JSON fields where N > max_era - def remap_era_json(m): - n = int(m.group(1)) - return f'"era": {min(n, max_era)}' if n > max_era else m.group(0) - text = re.sub(r'"era":\s*(\d+)', remap_era_json, text) - return json.loads(text) - - def _save(self, name: str, data: dict) -> Path: - self.output_dir.mkdir(parents=True, exist_ok=True) - path = self.output_dir / f"opportunity-{name}.json" - data["generated_at"] = datetime.now().isoformat() - data["project"] = self.project_name - # Sanitize era references before writing - data = self._sanitize_eras(data) - atomic_write(path, json.dumps(data, indent=2, ensure_ascii=False, default=str) + "\n") - self._log(f"{name}: {path}") - return path - - def close(self) -> None: - if self._conn is not None: - self._conn.close() - self._conn = None - - # ── Analyzer 1: Learning Velocity Tracker ───────────────────── - - def run_learning_velocity(self) -> dict: - """YouTube-to-commit latency + session deepening + era velocity.""" - self._log("Learning Velocity Tracker...") - - yt_corr = self._load("data/youtube-ai-correlation.json") or {} - lag = yt_corr.get("lag_analysis", {}) - sessions = self._query("SELECT session_id, human_message_count FROM sessions ORDER BY timestamp") - eras = self._eras() - commits = self._query("SELECT date, message FROM commits ORDER BY date") - - # Era velocity: commits per active day per era - era_velocity = [] - for era in eras: - commits_in_era = era.get("commits", 0) - active = era.get("active_days", 1) or 1 - velocity = round(commits_in_era / active, 1) - era_velocity.append({ - "era": era.get("name", f"Era {era.get('id')}"), - "commits": commits_in_era, - "active_days": active, - "velocity_per_day": velocity, - }) - - # Session deepening: average message count progression - session_depths = [] - if sessions: - chunk_size = max(1, len(sessions) // 5) - for i in range(0, len(sessions), chunk_size): - chunk = sessions[i : i + chunk_size] - avg_msgs = sum(s.get("human_message_count", 0) for s in chunk) / len(chunk) if chunk else 0 - session_depths.append({ - "phase": f"Sessions {i + 1}-{min(i + chunk_size, len(sessions))}", - "avg_message_count": round(avg_msgs, 1), - "session_count": len(chunk), - }) - - # YouTube learning lag trend - lag_trend = {} - if lag: - lag_trend = { - "initial_lag_days": lag.get("initial_lag_days"), - "current_lag_days": lag.get("current_lag_days"), - "improvement_pct": lag.get("improvement_pct"), - } - - # Cumulative commit velocity over time - monthly_velocity = self._query("SELECT * FROM monthly_velocity ORDER BY _key") - - result = { - "analysis_type": "learning-velocity", - "youtube_lag_trend": lag_trend, - "session_deepening_curve": session_depths, - "era_velocity": era_velocity, - "monthly_velocity": [ - {"month": r.get("_key"), "commits": r.get("value")} for r in monthly_velocity - ], - "summary": { - "total_eras": len(eras), - "peak_velocity_era": max(era_velocity, key=lambda e: e["velocity_per_day"])["era"] if era_velocity else None, - "learning_acceleration": round( - era_velocity[-1]["velocity_per_day"] / era_velocity[0]["velocity_per_day"], 2 - ) if len(era_velocity) >= 2 and era_velocity[0]["velocity_per_day"] > 0 else None, - "session_depth_trend": "deepening" if len(session_depths) >= 2 and session_depths[-1]["avg_message_count"] > session_depths[0]["avg_message_count"] else "stable", - }, - } - self._save("learning-velocity", result) - return result - - # ── Analyzer 2: Frustration-to-Automation Converter ─────────── - - def run_frustration_to_automation(self) -> dict: - """Map frustrations → hooks with conversion metrics.""" - self._log("Frustration-to-Automation Converter...") - - frustration_patterns = self._query("SELECT * FROM frustration_patterns ORDER BY latency_hours") - derived = self._load("data/derived-patterns.json") or {} - f2a = derived.get("frustration_to_automation_latency", {}) - context_mgmt = self._load("data/context-management-analysis.json") or {} - - conversions = [] - for fp in frustration_patterns: - conversions.append({ - "category": fp.get("category", ""), - "frustration_level": fp.get("frustration_level", ""), - "quote": fp.get("quote", ""), - "infrastructure_response": fp.get("infrastructure", ""), - "estimated_hook": fp.get("estimated_hook_commit", ""), - "latency_hours": fp.get("latency_hours"), - "converted": bool(fp.get("estimated_hook_commit")), - }) - - # Latency statistics - latencies = [c["latency_hours"] for c in conversions if c["latency_hours"] is not None] - avg_latency = sum(latencies) / len(latencies) if latencies else None - - # Frustration density timeline from context management analysis - frustration_timeline = context_mgmt.get("frustration_to_skill_timeline", []) - - result = { - "analysis_type": "frustration-to-automation", - "conversion_patterns": conversions, - "latency_stats": { - "avg_hours_to_automation": round(avg_latency, 1) if avg_latency else None, - "min_hours": min(latencies) if latencies else None, - "max_hours": max(latencies) if latencies else None, - "total_frustrations": len(conversions), - "converted_to_hooks": sum(1 for c in conversions if c["converted"]), - "conversion_rate": round(sum(1 for c in conversions if c["converted"]) / len(conversions) * 100, 1) if conversions else 0, - }, - "frustration_timeline": frustration_timeline[:20] if isinstance(frustration_timeline, list) else [], - "summary": { - "total_patterns": len(conversions), - "automation_rate": f"{sum(1 for c in conversions if c['converted'])}/{len(conversions)}", - "avg_cycle_hours": round(avg_latency, 1) if avg_latency else None, - "fastest_conversion": round(min(latencies), 1) if latencies else None, - }, - } - self._save("frustration-to-automation", result) - return result - - # ── Analyzer 3: Knowledge Gap Detector ───────────────────────── - - def run_knowledge_gap(self) -> dict: - """ML reinventions + formal term gaps → personalized curriculum.""" - self._log("Knowledge Gap Detector...") - - ml_mapper = self._load("deliverables/analysis/analysis-ml-pattern-mapper.json") or {} - formal_terms = self._load("deliverables/analysis/analysis-formal-terms-mapper.json") or {} - context_mgmt = self._load("data/context-management-analysis.json") or {} - - # Gaps from ML pattern reinventions - reinventions = ml_mapper.get("reinventions", []) if isinstance(ml_mapper, dict) else [] - gaps_from_reinvention = [] - for r in reinventions: - gaps_from_reinvention.append({ - "intuitive_name": r.get("intuitive_name", ""), - "formal_term": r.get("formal_term", ""), - "severity": "HIGH" if r.get("is_reinvention") else "MEDIUM", - "library_alternative": r.get("library_alternative"), - "estimated_waste_tokens": r.get("estimated_token_waste"), - }) - - # Gaps from formal term mapping - term_gaps = formal_terms.get("term_dictionary", []) if isinstance(formal_terms, dict) else [] - learning_opps = formal_terms.get("learning_opportunities", []) if isinstance(formal_terms, dict) else [] - - # Frustration-to-skill timeline shows where knowledge was missing - skill_timeline = context_mgmt.get("frustration_to_skill_timeline", []) - - # Build prioritized curriculum - curriculum = [] - priority = 1 - for gap in sorted(gaps_from_reinvention, key=lambda g: g.get("estimated_waste_tokens") or 0, reverse=True): - curriculum.append({ - "priority": priority, - "topic": gap["formal_term"], - "trigger": f"Reinvented as '{gap['intuitive_name']}'", - "severity": gap["severity"], - "recommended_resource": gap.get("library_alternative") or "Official documentation", - "status": "GAP", - }) - priority += 1 - - for topic in learning_opps: - curriculum.append({ - "priority": priority, - "topic": topic, - "trigger": "Formal terms mapper", - "severity": "MEDIUM", - "recommended_resource": "Academic course or textbook", - "status": "PARTIAL", - }) - priority += 1 - - result = { - "analysis_type": "knowledge-gap", - "reinvention_gaps": gaps_from_reinvention, - "term_mapping_gaps": [ - {"term": t.get("code_name"), "formal": t.get("formal_term"), "similarity": t.get("similarity_score")} - for t in term_gaps - ], - "prioritized_curriculum": curriculum, - "skill_timeline": skill_timeline[:15] if isinstance(skill_timeline, list) else [], - "summary": { - "total_gaps": len(gaps_from_reinvention) + len(learning_opps), - "high_severity": sum(1 for g in gaps_from_reinvention if g["severity"] == "HIGH"), - "estimated_token_waste": sum(g.get("estimated_waste_tokens") or 0 for g in gaps_from_reinvention), - "curriculum_items": len(curriculum), - }, - } - self._save("knowledge-gap", result) - return result - - # ── Analyzer 4: Token Efficiency Coach ───────────────────────── - - def run_token_efficiency(self) -> dict: - """Messages/commit ratio trajectory + context management rules.""" - self._log("Token Efficiency Coach...") - - sessions = self._query("SELECT session_id, human_message_count, timestamp FROM sessions ORDER BY timestamp") - commits = self._query("SELECT date, message FROM commits ORDER BY date") - context_mgmt = self._load("data/context-management-analysis.json") or {} - context_table = self._query("SELECT * FROM context_management") - derived = self._load("data/derived-patterns.json") or {} - - # Calculate messages-per-commit by era - eras = self._eras() - era_efficiency = [] - for era in eras: - era_name = era.get("name", f"Era {era.get('id')}") - era_commits = era.get("commits", 0) or 1 - # Estimate sessions in this era by date range - dates_str = era.get("dates", "") - msgs_in_era = 0 - if sessions and dates_str: - for s in sessions: - ts = s.get("timestamp", "") - if ts and dates_str.split(" - ")[0] <= str(ts)[:10] <= dates_str.split(" - ")[-1] if " - " in dates_str else False: - msgs_in_era += s.get("human_message_count", 0) - ratio = round(msgs_in_era / era_commits, 2) if era_commits > 0 else None - era_efficiency.append({ - "era": era_name, - "commits": era_commits, - "estimated_messages": msgs_in_era, - "messages_per_commit": ratio, - }) - - # Context management trajectory - cm_trajectory = context_mgmt.get("context_management_trajectory", {}) - tool_usage = context_mgmt.get("tool_usage_analysis", {}) - - # Token rules from token-efficiency-plan.md - token_plan = self._load("data/token-efficiency-plan.json") - rules = derived.get("commit_message_sentiment", {}) - - result = { - "analysis_type": "token-efficiency", - "era_efficiency": era_efficiency, - "context_management_trajectory": cm_trajectory, - "tool_usage_analysis": tool_usage, - "commit_message_sentiment": rules, - "summary": { - "efficiency_trend": "improving" if len(era_efficiency) >= 2 and ( - (era_efficiency[-1].get("messages_per_commit") or 0) < (era_efficiency[0].get("messages_per_commit") or 999) - ) else "stable", - "best_era": min((e for e in era_efficiency if e.get("messages_per_commit")), key=lambda e: e["messages_per_commit"])["era"] if any(e.get("messages_per_commit") for e in era_efficiency) else None, - "total_sessions": len(sessions), - "total_commits": len(commits), - "global_ratio": round(len(sessions) * 12 / max(len(commits), 1), 2), # rough estimate - }, - } - self._save("token-efficiency", result) - return result - - # ── Analyzer 5: Session Quality Scorer ────────────────────────── - - def run_session_quality(self) -> dict: - """Per-session quality rating with taxonomy classification.""" - self._log("Session Quality Scorer...") - - sessions = self._query("SELECT session_id, human_message_count, messages, timestamp FROM sessions ORDER BY timestamp") - frustration = self._query("SELECT * FROM frustration_patterns") - - # Session type keywords for classification - type_keywords = { - "SCAFFOLDING": ["scaffold", "initialize", "setup", "create", "new"], - "BUILDING": ["feat", "implement", "add", "build", "integrate"], - "DEBUGGING": ["fix", "debug", "error", "bug", "broken", "fail"], - "REFACTORING": ["refactor", "cleanup", "simplify", "extract", "split"], - "EXPLORING": ["explore", "investigate", "analyze", "understand", "research"], - "REVIEWING": ["review", "audit", "check", "verify", "lint"], - } - - scored_sessions = [] - for session in sessions: - msgs = session.get("messages", "") or "" - msg_count = session.get("human_message_count", 0) or 0 - - # Classify session type - type_scores = {} - for stype, keywords in type_keywords.items(): - count = sum(1 for kw in keywords if kw.lower() in msgs.lower()) - type_scores[stype] = count - session_type = max(type_scores, key=type_scores.get) if any(v > 0 for v in type_scores.values()) else "MIXED" - - # Quality scoring (0-10) - depth_score = min(10, msg_count / 3) if msg_count > 0 else 0 - diversity_score = min(10, sum(1 for v in type_scores.values() if v > 0) * 2.5) - frustration_hits = sum(1 for f in frustration if any(kw in msgs.lower() for kw in ["frustrat", "annoy", "broken", "stuck"])) - frustration_penalty = min(3, frustration_hits) - quality = max(1, round((depth_score * 0.4 + diversity_score * 0.3 + 5 * 0.3) - frustration_penalty, 1)) - - scored_sessions.append({ - "session_id": session.get("session_id"), - "timestamp": session.get("timestamp"), - "type": session_type, - "message_count": msg_count, - "quality_score": min(10, quality), - "productivity_score": round(min(10, msg_count * 0.5), 1), - "learning_value": round(min(10, diversity_score), 1), - }) - - # Aggregate stats - type_distribution = Counter(s["type"] for s in scored_sessions) - avg_quality = sum(s["quality_score"] for s in scored_sessions) / len(scored_sessions) if scored_sessions else 0 - - result = { - "analysis_type": "session-quality", - "sessions": scored_sessions, - "type_distribution": dict(type_distribution), - "summary": { - "total_sessions": len(scored_sessions), - "avg_quality": round(avg_quality, 2), - "top_quarter_avg": round( - sum(s["quality_score"] for s in sorted(scored_sessions, key=lambda x: x["quality_score"], reverse=True)[ - : max(1, len(scored_sessions) // 4) - ]) / max(1, len(scored_sessions) // 4), 2 - ) if scored_sessions else 0, - "dominant_type": type_distribution.most_common(1)[0][0] if type_distribution else None, - "quality_range": { - "min": min(s["quality_score"] for s in scored_sessions) if scored_sessions else 0, - "max": max(s["quality_score"] for s in scored_sessions) if scored_sessions else 0, - }, - }, - } - self._save("session-quality", result) - return result - - # ── Analyzer 6: AI Agent Mastery Score ────────────────────────── - - def run_ai_agent_mastery(self) -> dict: - """Scoring system for "how good are you at using AI?" 0-100.""" - self._log("AI Agent Mastery Score...") - - agent_comparison = self._query("SELECT * FROM agent_comparison") - co_auth_patterns = self._query("SELECT * FROM co_authorship_patterns") - adoption_lag = self._query("SELECT * FROM adoption_lag") - model_timeline = self._query("SELECT * FROM model_timeline") - sessions = self._query("SELECT COUNT(*) as cnt FROM sessions") - agentic = self._load("deliverables/analysis/analysis-agentic-workflow.json") or {} - - # Sub-scores (0-100 each) - - # 1. Autonomy: from co-authorship gap → higher gap = more autonomous AI usage - co_auth_gaps = self._query("SELECT era, gap_percentage FROM co_authorship_gaps ORDER BY era") - autonomy_scores = [g.get("gap_percentage", 0) for g in co_auth_gaps if g.get("gap_percentage") is not None] - latest_autonomy = autonomy_scores[-1] if autonomy_scores else 0 - autonomy_score = min(100, latest_autonomy) - - # 2. Tool breadth: number of distinct AI tools used - tools_used = set() - for row in agent_comparison: - for k in row.keys(): - if k != "_key": - val = row.get(k, 0) - if isinstance(val, (int, float)) and val > 0: - tools_used.add(k) - for row in model_timeline: - tool = row.get("tool", "") - if tool: - tools_used.add(tool) - breadth_score = min(100, len(tools_used) * 15) - - # 3. Adoption speed: inverse of adoption lag - lags = [r.get("adoption_lag_months") for r in adoption_lag if r.get("adoption_lag_months") is not None] - avg_lag = sum(lags) / len(lags) if lags else 12 - speed_score = max(0, min(100, 100 - (avg_lag * 10))) - - # 4. Delegation quality: session depth indicates good delegation - total_sessions = sessions[0]["cnt"] if sessions else 0 - delegation_score = min(100, total_sessions * 2) if total_sessions > 0 else 10 - - # 5. Session taxonomy from agentic analysis - taxonomy = agentic.get("session_taxonomy", {}) if isinstance(agentic, dict) else {} - building_ratio = taxonomy.get("BUILDING", 0) / max(sum(taxonomy.values()), 1) - delegation_score = min(100, delegation_score * (0.5 + building_ratio)) - - overall = round(autonomy_score * 0.25 + breadth_score * 0.20 + speed_score * 0.20 + delegation_score * 0.15 + 50 * 0.20, 1) - - # Mastery level - if overall >= 80: - level = "Expert" - elif overall >= 60: - level = "Advanced" - elif overall >= 40: - level = "Intermediate" - else: - level = "Beginner" - - result = { - "analysis_type": "ai-agent-mastery", - "overall_score": round(overall, 1), - "mastery_level": level, - "sub_scores": { - "autonomy": round(autonomy_score, 1), - "tool_breadth": round(breadth_score, 1), - "adoption_speed": round(speed_score, 1), - "delegation_quality": round(delegation_score, 1), - "baseline": 50, - }, - "tools_detected": sorted(tools_used), - "adoption_lag_avg_months": round(avg_lag, 1), - "autonomy_trajectory": autonomy_scores, - "agent_comparison": [dict(r) for r in agent_comparison], - "summary": { - "overall_score": round(overall, 1), - "level": level, - "strongest_dimension": max( - [("autonomy", autonomy_score), ("breadth", breadth_score), ("speed", speed_score), ("delegation", delegation_score)], - key=lambda x: x[1], - )[0], - "weakest_dimension": min( - [("autonomy", autonomy_score), ("breadth", breadth_score), ("speed", speed_score), ("delegation", delegation_score)], - key=lambda x: x[1], - )[0], - }, - } - self._save("ai-agent-mastery", result) - return result - - # ── Analyzer 7: Creative DNA Transfer Map ─────────────────────── - - def run_creative_dna(self) -> dict: - """Physical craft → code metaphor transfer map.""" - self._log("Creative DNA Transfer Map...") - - pre_history = self._load("data/pre-history-creative-journey.json") or {} - formal_terms = self._load("deliverables/analysis/analysis-formal-terms-mapper.json") or {} - ml_mapper = self._load("deliverables/analysis/analysis-ml-pattern-mapper.json") or {} - - # Extract creative phases - phases = pre_history.get("phases", []) if isinstance(pre_history, dict) else [] - creative_phases = [] - for phase in phases: - creative_phases.append({ - "period": phase.get("period", ""), - "medium": phase.get("medium", ""), - "skills": phase.get("skills", []), - "key_insight": phase.get("key_insight", ""), - }) - - # Map creative metaphors to code patterns - transfer_map = [ - { - "creative_source": "Ceramics / Glaze Chemistry", - "code_destination": "Creative evaluation systems", - "metaphor": "Glaze recipes → algorithmic parameter spaces", - "evidence": "CreativeEvaluator, UMF calculator, prediction systems", - "transfer_type": "MATERIAL_SCIENCE", - "strength": "HIGH", - }, - { - "creative_source": "Aquariums / Ecosystems", - "code_destination": "ForgettingCurve, learning retention", - "metaphor": "Water chemistry balance → parameter optimization", - "evidence": "ForgettingCurve implementation, multi-parameter systems", - "transfer_type": "SYSTEMS_THINKING", - "strength": "MEDIUM", - }, - { - "creative_source": "Music / Composition", - "code_destination": "Euclidean rhythms, Markov chains", - "metaphor": "Musical structure → generative algorithms", - "evidence": "Music theory engine commits", - "transfer_type": "PATTERN_RECOGNITION", - "strength": "HIGH", - }, - { - "creative_source": "Visual Art / Ceramics", - "code_destination": "VAE, generative visual systems", - "metaphor": "Kiln transformation → latent space variation", - "evidence": "P5Generator, ParticleSystem generators", - "transfer_type": "VISUAL_REASONING", - "strength": "HIGH", - }, - { - "creative_source": "ICM Methodology", - "code_destination": "Iterative development loops", - "metaphor": "Creative iteration → RalphLoop, quality gates", - "evidence": "RalphLoop, quality verification systems", - "transfer_type": "PROCESS_DESIGN", - "strength": "HIGH", - }, - ] - - # ICM catalysis - icm = pre_history.get("the_icm_catalysis", {}) if isinstance(pre_history, dict) else {} - - result = { - "analysis_type": "creative-dna", - "creative_phases": creative_phases, - "transfer_map": transfer_map, - "icm_catalysis": icm, - "summary": { - "total_creative_sources": len(set(t["creative_source"] for t in transfer_map)), - "total_code_transfers": len(transfer_map), - "strongest_transfers": [t for t in transfer_map if t["strength"] == "HIGH"], - "transfer_types": list(set(t["transfer_type"] for t in transfer_map)), - }, - } - self._save("creative-dna", result) - return result - - # ── Analyzer 8: Neurodivergent Developer Profile ─────────────── - - def run_neurodivergent_profile(self) -> dict: - """ADHD hyperfocus cycles, burst-recovery, working style profile.""" - self._log("Neurodivergent Developer Profile...") - - hourly = self._query("SELECT * FROM hourly_activity ORDER BY _key") - weekly = self._query("SELECT * FROM weekly_activity ORDER BY _key") - eras = self._eras() - lunar = self._query("SELECT * FROM lunar_phases ORDER BY date") - commits = self._query("SELECT date, message FROM commits ORDER BY date") - - # Hourly pattern: identify peak focus hours - hourly_pattern = {} - for row in hourly: - key = row.get("_key", "") - val = row.get("value", 0) - if key and key.isdigit(): - hourly_pattern[int(key)] = val - - peak_hours = sorted(hourly_pattern, key=hourly_pattern.get, reverse=True)[:5] if hourly_pattern else [] - quiet_hours = sorted(hourly_pattern, key=hourly_pattern.get)[:5] if hourly_pattern else [] - - # Burst-recovery pattern: detect high-commit days vs low-commit days - daily_commits: Counter[str] = Counter() - for c in commits: - date = str(c.get("date", ""))[:10] - if date: - daily_commits[date] += 1 - - burst_days = {d: c for d, c in daily_commits.items() if c > 50} - recovery_days = {d: c for d, c in daily_commits.items() if c <= 5} - - # Weekend vs weekday bias - weekday_pattern = {r.get("_key"): r.get("value") for r in weekly} - weekend_total = (weekday_pattern.get("Saturday", 0) or 0) + (weekday_pattern.get("Sunday", 0) or 0) - weekday_total = sum(v for k, v in weekday_pattern.items() if k not in ("Saturday", "Sunday")) - - # Lunar phase correlation (creative cycles) - lunar_correlation = [] - for lp in lunar[:10]: - lunar_correlation.append({ - "date": lp.get("date"), - "phase": lp.get("phase"), - "illumination": lp.get("illumination_percent"), - }) - - # Witching hour detection (late-night productivity) - witching_hour_commits = sum(hourly_pattern.get(h, 0) for h in range(21, 24)) + sum(hourly_pattern.get(h, 0) for h in range(0, 3)) - total_commits = sum(hourly_pattern.values()) or 1 - witching_ratio = round(witching_hour_commits / total_commits * 100, 1) - - result = { - "analysis_type": "neurodivergent-profile", - "hourly_pattern": hourly_pattern, - "peak_hours": peak_hours, - "quiet_hours": quiet_hours, - "burst_days": burst_days, - "recovery_days_count": len(recovery_days), - "weekday_vs_weekend": { - "weekday_commits": weekday_total, - "weekend_commits": weekend_total, - "weekend_bias": round(weekend_total / max(weekday_total, 1) * 100, 1), - }, - "witching_hour": { - "hours": "21:00-03:00", - "commits": witching_hour_commits, - "percentage_of_total": witching_ratio, - }, - "lunar_correlation": lunar_correlation, - "working_style": { - "pattern": "Burst-recovery with nocturnal peak", - "peak_productivity": f"{peak_hours[0]:02d}:00" if peak_hours else "unknown", - "avg_burst_size": round(sum(burst_days.values()) / len(burst_days), 1) if burst_days else 0, - "recovery_frequency": f"{len(recovery_days)} recovery days in {len(daily_commits)} active days", - }, - "summary": { - "profile_type": "ADHD-hyperfocus", - "peak_hour": f"{peak_hours[0]:02d}:00" if peak_hours else None, - "witching_hour_pct": witching_ratio, - "burst_days_count": len(burst_days), - "recommendations": [ - "Schedule complex architecture work during peak hours", - "Use burst days for feature development, recovery days for documentation", - "Protect the witching hour — it's the most productive period", - "Batch similar tasks to reduce context-switching overhead", - ], - }, - } - self._save("neurodivergent-profile", result) - return result - - # ── Analyzer 9: Model Selection Advisor ───────────────────────── - - def run_model_selection_advisor(self) -> dict: - """Which AI tool for which task — recommendation matrix.""" - self._log("Model Selection Advisor...") - - adoption = self._load("data/model-adoption-analysis.json") or {} - model_releases = self._query("SELECT * FROM model_releases ORDER BY _key") - model_mentions = self._query("SELECT * FROM model_mentions ORDER BY _key") - adoption_lag = self._query("SELECT * FROM adoption_lag ORDER BY adoption_lag_months") - model_timeline = self._query("SELECT * FROM model_timeline ORDER BY date") - co_auth_patterns = self._query("SELECT * FROM co_authorship_patterns") - - # Build tool capability matrix - tool_capabilities = {} - for row in model_timeline: - tool = row.get("tool", "unknown") - event = row.get("event", "") - if tool not in tool_capabilities: - tool_capabilities[tool] = {"events": [], "first_seen": row.get("date"), "use_count": 0} - tool_capabilities[tool]["events"].append(event) - tool_capabilities[tool]["use_count"] += 1 - - # Task-to-tool recommendation matrix - recommendations = [ - {"task_type": "Code generation", "recommended": "Claude Code / Cursor", "confidence": 0.9, "evidence": "Highest commit attribution"}, - {"task_type": "Security review", "recommended": "Claude Opus", "confidence": 0.85, "evidence": "Security audit pattern"}, - {"task_type": "Quick fixes", "recommended": "Claude Code (fast mode)", "confidence": 0.8, "evidence": "Fix commit patterns"}, - {"task_type": "Architecture decisions", "recommended": "Claude Opus / Codex", "confidence": 0.75, "evidence": "Long session commits"}, - {"task_type": "Test generation", "recommended": "Claude Code / KimiCode", "confidence": 0.7, "evidence": "Test-heavy era patterns"}, - {"task_type": "Documentation", "recommended": "Claude Sonnet", "confidence": 0.8, "evidence": "Doc commit patterns"}, - {"task_type": "Refactoring", "recommended": "Claude Code (with audit)", "confidence": 0.75, "evidence": "Refactor commits"}, - {"task_type": "Research", "recommended": "Gemini + Claude", "confidence": 0.6, "evidence": "Model adoption timeline"}, - ] - - # Adoption lag insights - lag_insights = [] - for row in adoption_lag: - lag_insights.append({ - "model": row.get("model", ""), - "lag_months": row.get("adoption_lag_months"), - "total_mentions": row.get("total_mentions", 0), - }) - - result = { - "analysis_type": "model-selection-advisor", - "recommendation_matrix": recommendations, - "tool_capabilities": tool_capabilities, - "adoption_lag": lag_insights, - "model_insights": adoption.get("insights", []) if isinstance(adoption, dict) else [], - "summary": { - "tools_evaluated": len(tool_capabilities), - "recommendations": len(recommendations), - "avg_adoption_lag_months": round(sum(r.get("lag_months", 0) or 0 for r in lag_insights) / max(len(lag_insights), 1), 1), - "fastest_adopter": min(lag_insights, key=lambda x: x.get("lag_months", 999))["model"] if lag_insights else None, - }, - } - self._save("model-selection-advisor", result) - return result - - # ── Analyzer 10: Before/After Snapshot ────────────────────────── - - def run_before_after_snapshot(self) -> dict: - """Proof of growth: early chaos vs late mastery.""" - self._log("Before/After Snapshot...") - - eras = self._eras() - co_auth_gaps = self._query("SELECT * FROM co_authorship_gaps ORDER BY era") - frustration = self._query("SELECT * FROM frustration_patterns") - derived = self._load("data/derived-patterns.json") or {} - sentiment = derived.get("commit_message_sentiment", {}) - - if len(eras) < 2: - return {"analysis_type": "before-after-snapshot", "error": "Not enough eras for comparison"} - - early = eras[0] - late = eras[-1] - - # Before metrics (Era 1) - before = { - "era": early.get("name"), - "dates": early.get("dates"), - "commits": early.get("commits", 0), - "active_days": early.get("active_days", 1), - "velocity": round(early.get("commits", 0) / max(early.get("active_days", 1), 1), 1), - "authors": early.get("authors", ""), - } - - # After metrics (last Era) - after = { - "era": late.get("name"), - "dates": late.get("dates"), - "commits": late.get("commits", 0), - "active_days": late.get("active_days", 1), - "velocity": round(late.get("commits", 0) / max(late.get("active_days", 1), 1), 1), - "authors": late.get("authors", ""), - } - - # Growth metrics - velocity_change = round(after["velocity"] / max(before["velocity"], 0.1), 2) - commit_change = round(after["commits"] / max(before["commits"], 1), 2) - - # Frustration conversion - converted = sum(1 for f in frustration if f.get("estimated_hook_commit")) - total_frustrations = len(frustration) - - # Attribution improvement from co-authorship gaps - gap_early = co_auth_gaps[0] if co_auth_gaps else {} - gap_late = co_auth_gaps[-1] if co_auth_gaps else {} - - result = { - "analysis_type": "before-after-snapshot", - "before": before, - "after": after, - "growth": { - "velocity_multiplier": velocity_change, - "commit_multiplier": commit_change, - "frustration_to_automation_rate": f"{converted}/{total_frustrations}", - "attribution_improvement": { - "early_gap_pct": gap_early.get("gap_percentage"), - "late_gap_pct": gap_late.get("gap_percentage"), - }, - }, - "narrative": f"From {before['era']} ({before['dates']}) to {after['era']} ({after['dates']}): velocity multiplied by {velocity_change}x, commits by {commit_change}x. {converted} of {total_frustrations} frustrations converted to automation.", - "summary": { - "velocity_change": f"{velocity_change}x", - "commit_change": f"{commit_change}x", - "growth_direction": "accelerating" if velocity_change > 1.5 else "maturing", - }, - } - self._save("before-after-snapshot", result) - return result - - # ── Analyzer 11: Cross-Repo Learning Transfer ─────────────────── - - def run_cross_repo_transfer(self) -> dict: - """Map of how learning transfers across repositories.""" - self._log("Cross-Repo Learning Transfer...") - - cross_repo = self._load("data/cross-repo-analysis.json") or {} - multi_repo = self._load("data/multi-repo-correlation.json") or {} - concurrent = self._query("SELECT * FROM concurrent_repos ORDER BY commit_count DESC") - github_repos = self._query("SELECT * FROM github_repos ORDER BY total_commits DESC") - cross_timeline = self._query("SELECT * FROM cross_repo_timeline ORDER BY _key") - - # Identify R&D labs (repos that feed into the main project) - rd_labs = [] - for repo in concurrent: - overlap = repo.get("overlap_with_demo_project_era", "") - rel = repo.get("relationship", "") - if "R&D" in rel or "research" in rel.lower() or "experiment" in rel.lower(): - rd_labs.append({ - "repo": repo.get("repo"), - "commits": repo.get("commit_count"), - "overlap_era": overlap, - "relationship": rel, - }) - - # Top repos by activity (top_repos is a dict {name: commits}) - raw_top_repos = cross_repo.get("top_repos", {}) if isinstance(cross_repo, dict) else {} - if isinstance(raw_top_repos, dict): - top_repos = [ - {"repo": k, "commits": v} for k, v in sorted(raw_top_repos.items(), key=lambda x: x[1], reverse=True) - ][:10] - elif isinstance(raw_top_repos, list): - top_repos = raw_top_repos[:10] - else: - top_repos = [] - - # Language evolution (may be dict or list) - raw_lang_evo = cross_repo.get("language_evolution", []) if isinstance(cross_repo, dict) else [] - if isinstance(raw_lang_evo, dict): - lang_evo = [{"language": k, "value": v} for k, v in sorted(raw_lang_evo.items(), key=lambda x: str(x[1]), reverse=True)][:10] - elif isinstance(raw_lang_evo, list): - lang_evo = raw_lang_evo[:10] - else: - lang_evo = [] - - # Cross-repo timeline for learning transfer detection - transfer_events = [] - for row in cross_timeline: - demo_project_commits = row.get("demo_project_commits", 0) - other_commits = row.get("other_repos", 0) - if demo_project_commits and other_commits: - transfer_events.append({ - "period": row.get("_key"), - "demo_project_commits": demo_project_commits, - "other_repo_commits": other_commits, - "note": row.get("note", ""), - }) - - result = { - "analysis_type": "cross-repo-transfer", - "rd_labs": rd_labs[:10], - "top_repos": top_repos, - "language_evolution": lang_evo, - "transfer_events": transfer_events[:20], - "concurrent_repos": [ - {"repo": r.get("repo"), "commits": r.get("commit_count"), "overlap": r.get("overlap_with_demo_project_era")} - for r in concurrent[:10] - ], - "summary": { - "total_repos": len(github_repos), - "rd_labs_identified": len(rd_labs), - "transfer_events": len(transfer_events), - "primary_languages": list(set(l.get("language", "") for l in top_repos[:5])) if top_repos else [], - }, - } - self._save("cross-repo-transfer", result) - return result - - # ── Analyzer 12: YouTube Learning Graph ───────────────────────── - - def run_youtube_learning_graph(self) -> dict: - """Learning diet visualization — who and what shaped the developer.""" - self._log("YouTube Learning Graph...") - - yt_corr = self._load("data/youtube-ai-correlation.json") or {} - yt_creators = self._load("data/youtube-creators.json") or {} - yt_topics = self._load("data/youtube-topic-classification.json") or {} - yt_engagement = self._query("SELECT * FROM youtube_engagement") - yt_categories = self._query("SELECT * FROM yt_categories ORDER BY value DESC") - yt_creator_influence = self._query("SELECT * FROM yt_creator_influence") - yt_monthly = self._query("SELECT * FROM yt_monthly ORDER BY _key") - - # Creator influence map - creators_list = yt_creators.get("creators", []) if isinstance(yt_creators, dict) else [] - top_creators = sorted(creators_list, key=lambda c: c.get("video_count", 0) if isinstance(c, dict) else 0, reverse=True)[:15] - - # Topic distribution - categories = yt_topics.get("categories", []) if isinstance(yt_topics, dict) else [] - distribution = yt_topics.get("distribution", {}) if isinstance(yt_topics, dict) else {} - - # Correlations (YouTube → commits) - correlations = yt_corr.get("key_correlations", [])[:20] if isinstance(yt_corr, dict) else [] - smoking_guns = [c for c in correlations if isinstance(c, dict) and c.get("is_smoking_gun")] - - # Monthly learning curve - monthly_curve = yt_corr.get("quarterly_learning_curve", {}) if isinstance(yt_corr, dict) else {} - - # Creator influence by archetype - creator_archetypes = [] - for row in yt_creator_influence: - archetypes = {k: v for k, v in row.items() if k != "_key" and k != "value"} - creator_archetypes.append({ - "period": row.get("_key"), - "total_videos": row.get("value"), - "archetypes": archetypes, - }) - - result = { - "analysis_type": "youtube-learning-graph", - "top_creators": [ - { - "name": c.get("channel_name", c.get("name", "")), - "videos": c.get("video_count", 0), - "category": c.get("category", ""), - } for c in top_creators if isinstance(c, dict) - ], - "topic_distribution": distribution, - "categories": [{"category": r.get("_key"), "count": r.get("value")} for r in yt_categories], - "smoking_guns": smoking_guns[:10], - "monthly_learning": [{"month": r.get("_key"), "videos": r.get("value")} for r in yt_monthly], - "quarterly_curve": monthly_curve, - "creator_archetypes": creator_archetypes, - "summary": { - "total_creators": len(creators_list), - "total_videos": yt_topics.get("total_videos", 0) if isinstance(yt_topics, dict) else 0, - "smoking_gun_correlations": len(smoking_guns), - "top_category": max(distribution, key=distribution.get) if distribution else None, - }, - } - self._save("youtube-learning-graph", result) - return result - - # ── Analyzer 13: Architecture Evolution Timelapse ─────────────── - - def run_architecture_timelapse(self) -> dict: - """Era-by-era codebase structure evolution.""" - self._log("Architecture Evolution Timelapse...") - - eras = self._eras() - module_emergence = self._query("SELECT * FROM module_emergence ORDER BY _key") - file_growth = self._query("SELECT * FROM file_growth ORDER BY _key") - codebase_langs = self._query("SELECT * FROM codebase_languages ORDER BY _key") - - # Restructuring commits per era - restructuring_keywords = ["split", "extract", "decompose", "refactor", "rename", "reorganize", "restructure"] - naming_keywords = ["rename", "migrate", "move to", "rebrand"] - - era_snapshots = [] - for era in eras: - era_name = era.get("name", f"Era {era.get('id')}") - # Get commits from this era's date range - dates_str = era.get("dates", "") - restructure_commits = self._like_commits(restructuring_keywords, 50) - naming_commits = self._like_commits(naming_keywords, 30) - - era_snapshots.append({ - "era": era_name, - "dates": dates_str, - "commits": era.get("commits", 0), - "active_days": era.get("active_days", 0), - "restructuring_signals": len(restructure_commits), - "naming_changes": len(naming_commits), - "key_events": era.get("key_events", [])[:3] if isinstance(era.get("key_events"), list) else [], - "narrative": era.get("narrative_arc", "")[:200] if era.get("narrative_arc") else "", - }) - - # Module emergence timeline - modules = [{"period": r.get("_key"), "data": {k: v for k, v in r.items() if k != "_key"}} for r in module_emergence] - - # File growth - growth = [{"metric": r.get("_key"), "value": r.get("value")} for r in file_growth] - - # Language evolution - languages = [] - for row in codebase_langs: - for lang, pct in row.items(): - if lang != "_key" and pct: - languages.append({"period": row.get("_key"), "language": lang, "percentage": pct}) - - result = { - "analysis_type": "architecture-timelapse", - "era_snapshots": era_snapshots, - "module_emergence": modules, - "file_growth": growth, - "language_evolution": languages, - "summary": { - "total_eras": len(eras), - "restructuring_events": sum(s["restructuring_signals"] for s in era_snapshots), - "naming_changes": sum(s["naming_changes"] for s in era_snapshots), - "growth_trajectory": "exponential" if len(era_snapshots) > 3 and era_snapshots[-1]["commits"] > era_snapshots[0]["commits"] * 2 else "linear", - }, - } - self._save("architecture-timelapse", result) - return result - - # ── Analyzer 14: Commit Message Cognitive Load Proxy ──────────── - - def run_commit_cognitive_load(self) -> dict: - """Classify work type from commit message length and content.""" - self._log("Commit Message Cognitive Load Proxy...") - - commits = self._query("SELECT hash, date, message, author FROM commits ORDER BY date") - hourly = self._query("SELECT * FROM hourly_activity ORDER BY _key") - eras = self._eras() - - # Classify by message length - load_categories = { - "TRIVIAL": {"max_len": 30, "description": "Quick fixes, typos, config tweaks"}, - "ROUTINE": {"min_len": 31, "max_len": 80, "description": "Standard features, small changes"}, - "MODERATE": {"min_len": 81, "max_len": 150, "description": "Feature implementation, refactoring"}, - "HIGH": {"min_len": 151, "max_len": 300, "description": "Architecture decisions, complex features"}, - "INTENSE": {"min_len": 301, "description": "Major rewrites, deep architectural work"}, - } - - classified = [] - for c in commits: - msg = c.get("message", "") - msg_len = len(msg) - - if msg_len <= 30: - load = "TRIVIAL" - elif msg_len <= 80: - load = "ROUTINE" - elif msg_len <= 150: - load = "MODERATE" - elif msg_len <= 300: - load = "HIGH" - else: - load = "INTENSE" - - # Time of day from date - date_str = str(c.get("date", "")) - hour = int(date_str[11:13]) if len(date_str) > 12 else -1 - - # Work type from keywords - msg_lower = msg.lower() - if any(kw in msg_lower for kw in ["fix", "bug", "error", "broken"]): - work_type = "FIX" - elif any(kw in msg_lower for kw in ["feat", "add", "implement", "create"]): - work_type = "FEATURE" - elif any(kw in msg_lower for kw in ["refactor", "clean", "simplify", "split"]): - work_type = "REFACTOR" - elif any(kw in msg_lower for kw in ["test", "spec", "coverage"]): - work_type = "TEST" - elif any(kw in msg_lower for kw in ["doc", "readme", "comment"]): - work_type = "DOCS" - elif any(kw in msg_lower for kw in ["ci", "deploy", "pipeline", "workflow"]): - work_type = "CI/CD" - else: - work_type = "OTHER" - - classified.append({ - "cognitive_load": load, - "work_type": work_type, - "message_length": msg_len, - "hour": hour, - }) - - # Aggregate statistics - load_distribution = Counter(c["cognitive_load"] for c in classified) - type_distribution = Counter(c["work_type"] for c in classified) - - # Cognitive load by hour of day - load_by_hour = defaultdict(lambda: {"total": 0, "count": 0, "high_or_intense": 0}) - for c in classified: - h = c["hour"] - if h >= 0: - load_by_hour[h]["total"] += c["message_length"] - load_by_hour[h]["count"] += 1 - if c["cognitive_load"] in ("HIGH", "INTENSE"): - load_by_hour[h]["high_or_intense"] += 1 - - hourly_load = [ - { - "hour": h, - "avg_message_length": round(d["total"] / d["count"], 1) if d["count"] else 0, - "high_cognitive_pct": round(d["high_or_intense"] / d["count"] * 100, 1) if d["count"] else 0, - } - for h, d in sorted(load_by_hour.items()) - ] - - # Find peak cognitive hours - peak_cognitive = sorted(hourly_load, key=lambda x: x["high_cognitive_pct"], reverse=True)[:5] - - result = { - "analysis_type": "commit-cognitive-load", - "load_distribution": dict(load_distribution), - "work_type_distribution": dict(type_distribution), - "load_by_hour": hourly_load, - "peak_cognitive_hours": peak_cognitive, - "load_definitions": load_categories, - "summary": { - "total_commits": len(classified), - "avg_message_length": round(sum(c["message_length"] for c in classified) / max(len(classified), 1), 1), - "high_cognitive_pct": round( - sum(1 for c in classified if c["cognitive_load"] in ("HIGH", "INTENSE")) / max(len(classified), 1) * 100, 1 - ), - "dominant_work_type": type_distribution.most_common(1)[0][0] if type_distribution else None, - "peak_cognitive_hour": peak_cognitive[0]["hour"] if peak_cognitive else None, - "insight": "Longer commit messages correlate with architecture decisions. Peak cognitive hours suggest when complex work happens.", - }, - } - self._save("commit-cognitive-load", result) - return result - - # ── Run all analyzers ─────────────────────────────────────────── - - def run_all(self, analyzers: list[str] | None = None) -> dict[str, str]: - """Execute selected opportunity analyzers and save JSON outputs.""" - results: dict[str, str] = {} - runners = { - "learning-velocity": self.run_learning_velocity, - "frustration-to-automation": self.run_frustration_to_automation, - "knowledge-gap": self.run_knowledge_gap, - "token-efficiency": self.run_token_efficiency, - "session-quality": self.run_session_quality, - "ai-agent-mastery": self.run_ai_agent_mastery, - "creative-dna": self.run_creative_dna, - "neurodivergent-profile": self.run_neurodivergent_profile, - "model-selection-advisor": self.run_model_selection_advisor, - "before-after-snapshot": self.run_before_after_snapshot, - "cross-repo-transfer": self.run_cross_repo_transfer, - "youtube-learning-graph": self.run_youtube_learning_graph, - "architecture-timelapse": self.run_architecture_timelapse, - "commit-cognitive-load": self.run_commit_cognitive_load, - } - target = analyzers or self.ANALYZERS - unknown = [a for a in target if a not in runners] - if unknown: - raise ValueError(f"Unknown analyzer(s): {', '.join(unknown)}") - - self.output_dir.mkdir(parents=True, exist_ok=True) - for name in target: - try: - result = runners[name]() - results[name] = "OK" - print(f" [opportunity] {name}: OK") - except Exception as exc: - results[name] = f"ERROR: {exc}" - print(f" [opportunity] {name}: ERROR: {exc}") - self.close() - return results - -def run_opportunity_analyzers( - project_name: str, verbose: bool = False, analyzers: list[str] | None = None -) -> dict[str, str]: - """Public entry point to run opportunity analyzers.""" - project_dir = f"projects/{project_name}" - import os - if not os.path.isdir(project_dir): - raise ValueError(f"Project '{project_name}' not found") - runner = OpportunityAnalyzer(project_name, project_dir, verbose) - return runner.run_all(analyzers=analyzers) +def run_opportunity_analyzers(*args, **kwargs): + raise RuntimeError("opportunity is disabled: unsupported personal profiles and measurements") diff --git a/archaeology/provenance.py b/archaeology/provenance.py new file mode 100644 index 0000000..c4041ac --- /dev/null +++ b/archaeology/provenance.py @@ -0,0 +1,93 @@ +"""Bind derived evidence to the exact local inputs, without claiming signed provenance.""" + +import hashlib +import json +from pathlib import Path + +from .utils import atomic_write + +INPUTS = ( + "project.json", + "data/coverage.json", + "data/github-commits.csv", + "data/github-commits-with-stats.txt", + "data/archaeology.db", + "data/detected-signals.json", + "data/commit-eras.json", + "deliverables/canonical-metrics.json", + "deliverables/data.json", +) + + +def digest(path): + return hashlib.sha256(path.read_bytes()).hexdigest() if path.is_file() else None + + +def snapshot(project): + project = Path(project) + return {name: digest(project / name) for name in INPUTS} + + +def requires_binding(project): + project = Path(project) + config_path = project / "project.json" + config = json.loads(config_path.read_text(encoding="utf-8")) if config_path.exists() else {} + return bool( + config.get("mined_history_manifest_required") or (project / "data/coverage.json").exists() + ) + + +def analysis_paths(project): + deliverables = Path(project) / "deliverables" + paths = set(deliverables.glob("analysis-*.json")) | set( + (deliverables / "analysis").glob("analysis-*.json") + ) + return sorted(p for p in paths if not p.name.endswith(".provenance.json")) + + +def verify_analyses(project, require=False): + if not requires_binding(project): + return + paths = analysis_paths(project) + if require and not paths: + raise ValueError("Missing bound analysis; run analyze before exporting") + current = snapshot(project) + for path in paths: + data = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(data, dict) or data.get("evidence_binding") != current: + raise ValueError(f"Stale or unbound analysis: {path.name}; rerun signals and analyze") + verify_artifact(project, path) + + +def bind_artifact(project, artifact, include_analysis=False): + project, artifact = Path(project), Path(artifact) + if not requires_binding(project): + return + binding = {"snapshot": snapshot(project), "artifact": artifact.name, "sha256": digest(artifact)} + if include_analysis: + binding["analysis"] = { + str(p.relative_to(project)): digest(p) for p in analysis_paths(project) + } + atomic_write(str(artifact) + ".provenance.json", json.dumps(binding, indent=2)) + + +def verify_artifact(project, artifact): + project, artifact = Path(project), Path(artifact) + if not requires_binding(project): + return + sidecar = Path(str(artifact) + ".provenance.json") + try: + binding = json.loads(sidecar.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + raise ValueError(f"Missing or invalid evidence binding: {artifact.name}") from exc + if ( + not isinstance(binding, dict) + or binding.get("artifact") != artifact.name + or binding.get("sha256") != digest(artifact) + or binding.get("snapshot") != snapshot(project) + ): + raise ValueError(f"Stale or modified artifact: {artifact.name}; regenerate it") + if "analysis" in binding: + current = {str(p.relative_to(project)): digest(p) for p in analysis_paths(project)} + if binding["analysis"] != current: + raise ValueError(f"Stale analysis inputs for artifact: {artifact.name}; re-export") diff --git a/archaeology/report.py b/archaeology/report.py index 4d8b9e2..85c492c 100644 --- a/archaeology/report.py +++ b/archaeology/report.py @@ -9,12 +9,11 @@ from typing import Any from archaeology.visualization.design_system import ( - head_bundle, - body_end_bundle, THEME_SWITCHER_HTML, + body_end_bundle, + head_bundle, ) - ANALYSIS_FILES = [ "analysis-sdlc-gap-finder.json", "analysis-ml-pattern-mapper.json", @@ -44,10 +43,15 @@ def _fmt_count(value: Any) -> str: return str(value) if value is not None else "unknown" -def export_markdown_report(project_name: str, project_root: str | Path, output_path: str | Path | None = None) -> Path: +def export_markdown_report( + project_name: str, project_root: str | Path, output_path: str | Path | None = None +) -> Path: """Export a concise Markdown report from canonical metrics + analysis JSON.""" project_root = Path(project_root) deliverables = project_root / "deliverables" + from .provenance import verify_analyses + + verify_analyses(project_root, require=True) analysis_dir = deliverables / "analysis" data_dir = project_root / "data" project = _load_json(project_root / "project.json") or {} @@ -72,7 +76,9 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p f"This report summarizes the `{project_name}` development archaeology from canonical project metrics, era data, and automated analysis vectors.\n\n" ) - out.append("Automated vectors are commit-keyword investigation leads, not verified source findings, causal explanations, session measurements or implementation-quality scores. Missing evidence is not absence.\n\n") + out.append( + "Automated vectors are commit-keyword investigation leads, not verified source findings, causal explanations, session measurements or implementation-quality scores. Missing evidence is not absence.\n\n" + ) out.append("## Canonical Metrics\n\n") metric_rows = [ ("Total commits", canonical.get("total_commits", eras.get("total_commits"))), @@ -89,7 +95,11 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p if era_list: out.append("## Development Eras\n\n") for era in era_list: - out.append(_bullet(f"**Era {era.get('id')}: {era.get('name')}** — {era.get('dates', 'unknown dates')}; {era.get('commits', 'unknown')} commits. {era.get('description') or era.get('narrative_arc') or ''}")) + out.append( + _bullet( + f"**Era {era.get('id')}: {era.get('name')}** — {era.get('dates', 'unknown dates')}; {era.get('commits', 'unknown')} commits. {era.get('description') or era.get('narrative_arc') or ''}" + ) + ) out.append("\n") sdlc = analyses.get("sdlc-gap-finder") or {} @@ -97,7 +107,11 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p if gaps: out.append("## SDLC / Process Gaps\n\n") for gap in gaps[:10]: - out.append(_bullet(f"**{gap.get('practice')}** — {gap.get('status')} ({gap.get('severity')}). {gap.get('recommendation')}")) + out.append( + _bullet( + f"**{gap.get('practice')}** — {gap.get('status')} ({gap.get('severity')}). {gap.get('recommendation')}" + ) + ) out.append("\n") ml = analyses.get("ml-pattern-mapper") or {} @@ -105,7 +119,11 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p if mappings: out.append("## Formal ML / Architecture Patterns\n\n") for mapping in mappings[:10]: - out.append(_bullet(f"**{mapping.get('intuitive_name')}** → {mapping.get('formal_term')} (confidence: {mapping.get('confidence')})")) + out.append( + _bullet( + f"**{mapping.get('intuitive_name')}** → {mapping.get('formal_term')} (confidence: {mapping.get('confidence')})" + ) + ) out.append("\n") formal = analyses.get("formal-terms-mapper") or {} @@ -113,7 +131,11 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p if terms: out.append("## Vocabulary Translation\n\n") for term in terms[:10]: - out.append(_bullet(f"**{term.get('code_name')}** → {term.get('formal_term')} ({term.get('similarity_score')})")) + out.append( + _bullet( + f"**{term.get('code_name')}** → {term.get('formal_term')} ({term.get('similarity_score')})" + ) + ) out.append("\n") source = analyses.get("source-archaeologist") or {} @@ -121,13 +143,19 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p if improvements: out.append("## Remediation Priorities\n\n") for item in improvements: - out.append(_bullet(f"P{item.get('rank')}: **{item.get('title')}** — effort {item.get('effort')}, impact {item.get('impact')}")) + out.append( + _bullet( + f"P{item.get('rank')}: **{item.get('title')}** — effort {item.get('effort')}, impact {item.get('impact')}" + ) + ) out.append("\n") youtube = analyses.get("youtube-correlator") or {} yt_summary = youtube.get("summary") or {} out.append("## Behavioral / External Data\n\n") - out.append(_bullet(f"YouTube/behavioral data available: {bool(yt_summary.get('data_available'))}")) + out.append( + _bullet(f"YouTube/behavioral data available: {bool(yt_summary.get('data_available'))}") + ) out.append(_bullet(f"Correlations found: {_fmt_count(yt_summary.get('correlation_count'))}")) out.append(_bullet(f"Creator count: {_fmt_count(yt_summary.get('creator_count'))}")) out.append("\n") @@ -142,6 +170,9 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p output = Path(output_path) output.parent.mkdir(parents=True, exist_ok=True) output.write_text("".join(out), encoding="utf-8") + from .provenance import bind_artifact + + bind_artifact(project_root, output, include_analysis=True) return output @@ -159,9 +190,10 @@ def close_list() -> None: def render_inline(text: str) -> str: """Convert **bold** and `code` inline markup to HTML.""" import re + text = html.escape(text) - text = re.sub(r'\*\*(.+?)\*\*', r'\1', text) - text = re.sub(r'`(.+?)`', r'\1', text) + text = re.sub(r"\*\*(.+?)\*\*", r"\1", text) + text = re.sub(r"`(.+?)`", r"\1", text) return text for raw_line in markdown.splitlines(): @@ -342,7 +374,9 @@ def render_inline(text: str) -> str: """ -def export_html_report(project_name: str, project_root: str | Path, output_path: str | Path | None = None) -> Path: +def export_html_report( + project_name: str, project_root: str | Path, output_path: str | Path | None = None +) -> Path: """Export a standalone HTML report from the Markdown report content.""" project_root = Path(project_root) deliverables = project_root / "deliverables" @@ -355,10 +389,18 @@ def export_html_report(project_name: str, project_root: str | Path, output_path: output = Path(output_path) output.parent.mkdir(parents=True, exist_ok=True) output.write_text(_markdown_to_html(markdown, f"{title} Archaeology Report"), encoding="utf-8") + from .provenance import bind_artifact + + bind_artifact(project_root, output, include_analysis=True) return output -def export_report(project_name: str, project_root: str | Path, output_path: str | Path | None = None, fmt: str = "markdown") -> Path: +def export_report( + project_name: str, + project_root: str | Path, + output_path: str | Path | None = None, + fmt: str = "markdown", +) -> Path: """Export a report in markdown or html format.""" if fmt in {"markdown", "md"}: return export_markdown_report(project_name, project_root, output_path=output_path) @@ -367,7 +409,12 @@ def export_report(project_name: str, project_root: str | Path, output_path: str raise ValueError(f"Unsupported report format: {fmt}") -def export_public_case_study(root: str | Path = ".", output_dir: str | Path = "public-case-study", project_name: str = "demo-archaeology", force: bool = True) -> Path: +def export_public_case_study( + root: str | Path = ".", + output_dir: str | Path = "public-case-study", + project_name: str = "demo-archaeology", + force: bool = True, +) -> Path: """Generate a sanitized public case-study showroom from invented demo data.""" from .analysis_runner import run_analysis_vectors from .db.builder import build_db @@ -387,10 +434,15 @@ def export_public_case_study(root: str | Path = ".", output_dir: str | Path = "p output.mkdir(parents=True, exist_ok=True) data_out = output / "data" data_out.mkdir(parents=True, exist_ok=True) - (output / "ARCHAEOLOGY-REPORT.md").write_text(md_report.read_text(encoding="utf-8"), encoding="utf-8") + (output / "ARCHAEOLOGY-REPORT.md").write_text( + md_report.read_text(encoding="utf-8"), encoding="utf-8" + ) (output / "index.html").write_text(html_report.read_text(encoding="utf-8"), encoding="utf-8") for src, dst in [ - (work_project / "deliverables" / "canonical-metrics.json", data_out / "canonical-metrics.json"), + ( + work_project / "deliverables" / "canonical-metrics.json", + data_out / "canonical-metrics.json", + ), (work_project / "data" / "commit-eras.json", data_out / "commit-eras.json"), (work_project / "data" / "github-commits.csv", data_out / "github-commits.csv"), ]: diff --git a/archaeology/templates/md-viewer.html b/archaeology/templates/md-viewer.html index f78d92d..54bef09 100644 --- a/archaeology/templates/md-viewer.html +++ b/archaeology/templates/md-viewer.html @@ -1,476 +1 @@ - - - - - -Document Viewer — DevArch - - - - - - - - - - - - - - - - -
-
Loading document
- -
- - - - +Viewer disabled

Legacy Markdown viewer disabled

Open reviewed local reports directly. Remote and query-selected document loading is unavailable.

diff --git a/archaeology/utils.py b/archaeology/utils.py index 74fdffa..dba906b 100644 --- a/archaeology/utils.py +++ b/archaeology/utils.py @@ -9,7 +9,6 @@ from pathlib import Path from typing import Any - _logger = logging.getLogger(__name__) diff --git a/archaeology/validators/history.py b/archaeology/validators/history.py new file mode 100644 index 0000000..f103a15 --- /dev/null +++ b/archaeology/validators/history.py @@ -0,0 +1,35 @@ +"""Validate the installed CLI's self-contained measured-history output.""" + +from html.parser import HTMLParser +from pathlib import Path + +from ..provenance import verify_artifact + + +class HistoryParser(HTMLParser): + def __init__(self): + super().__init__() + self.tags = set() + self.unsafe = False + + def handle_starttag(self, tag, attrs): + self.tags.add(tag) + if tag in {"script", "iframe", "object", "embed", "form"}: + self.unsafe = True + if any( + k.lower().startswith("on") or (v and v.strip().lower().startswith("javascript:")) + for k, v in attrs + ): + self.unsafe = True + + +def validate_history(project): + path = Path(project) / "deliverables/visuals/archaeology.html" + if not path.is_file(): + raise ValueError("Generated history HTML missing; run visualize first") + verify_artifact(project, path) + parser = HistoryParser() + parser.feed(path.read_text(encoding="utf-8")) + parser.close() + if parser.unsafe or not {"html", "title", "h1", "table", "caption"}.issubset(parser.tags): + raise ValueError("Generated history HTML is incomplete or contains active content") diff --git a/archaeology/visualization/agent_benchmark.py b/archaeology/visualization/agent_benchmark.py index f0166ac..d755bc2 100644 --- a/archaeology/visualization/agent_benchmark.py +++ b/archaeology/visualization/agent_benchmark.py @@ -7,10 +7,12 @@ import json import sqlite3 from pathlib import Path -from typing import Any, Dict, List +from typing import Any, Dict from archaeology.visualization.design_system import ( - head_bundle, body_end_bundle, THEME_SWITCHER_HTML + THEME_SWITCHER_HTML, + body_end_bundle, + head_bundle, ) @@ -49,28 +51,29 @@ def analyze_agent_benchmarks(db_path: str) -> Dict[str, Any]: cursor = conn.cursor() # Check if eras table exists - has_eras = cursor.execute( - "SELECT name FROM sqlite_master WHERE type='table' AND name='eras'" - ).fetchone() is not None + has_eras = ( + cursor.execute( + "SELECT name FROM sqlite_master WHERE type='table' AND name='eras'" + ).fetchone() + is not None + ) # Get era information (optional) eras = {} era_date_ranges = {} if has_eras: - eras_data = cursor.execute( - "SELECT id, name FROM eras ORDER BY id" - ).fetchall() + eras_data = cursor.execute("SELECT id, name FROM eras ORDER BY id").fetchall() eras = {row["id"]: row["name"] for row in eras_data} - era_ids = list(eras.keys()) + _era_ids = list(eras.keys()) # Build era date ranges for mapping commits era_date_ranges = {} if has_eras: for row in eras_data: era_id = row["id"] - dates_str = cursor.execute( - "SELECT dates FROM eras WHERE id = ?", (era_id,) - ).fetchone()["dates"] + dates_str = cursor.execute("SELECT dates FROM eras WHERE id = ?", (era_id,)).fetchone()[ + "dates" + ] era_date_ranges[era_id] = dates_str # Simpler approach: get all commits and map to eras in Python @@ -82,9 +85,7 @@ def analyze_agent_benchmarks(db_path: str) -> Dict[str, Any]: era_mappings = [] if has_eras: for era_id, era_name in eras.items(): - era_row = cursor.execute( - "SELECT dates FROM eras WHERE id = ?", (era_id,) - ).fetchone() + era_row = cursor.execute("SELECT dates FROM eras WHERE id = ?", (era_id,)).fetchone() dates_str = era_row["dates"] if era_row else None start_date = None @@ -97,12 +98,9 @@ def analyze_agent_benchmarks(db_path: str) -> Dict[str, Any]: start_date = f"{start_str}, 2026" end_date = f"{end_str}, 2026" - era_mappings.append({ - "id": era_id, - "name": era_name, - "start": start_date, - "end": end_date - }) + era_mappings.append( + {"id": era_id, "name": era_name, "start": start_date, "end": end_date} + ) # Map commits to eras import re @@ -111,7 +109,7 @@ def analyze_agent_benchmarks(db_path: str) -> Dict[str, Any]: def parse_abbreviated_date(date_str: str) -> datetime: """Parse dates like 'Feb 28, 2026' or 'Mar 19, 2026'.""" # Remove weekday names if present - date_str = re.sub(r'^[A-Z][a-z]{2}\s+', '', date_str) + date_str = re.sub(r"^[A-Z][a-z]{2}\s+", "", date_str) # Parse with format try: return datetime.strptime(date_str, "%b %d, %Y") @@ -122,7 +120,7 @@ def parse_abbreviated_date(date_str: str) -> datetime: def normalize_author(author: str) -> str: """Normalize author names to canonical agent names.""" author_lower = author.lower() - ai_agents = {"claude", "kai", "cursor", "kimicode", "codex"} + _ai_agents = {"claude", "kai", "cursor", "kimicode", "codex"} if "claude" in author_lower: return "Claude" elif author_lower == "kai": @@ -176,7 +174,7 @@ def normalize_author(author: str) -> str: "rework_commits": 0, "first_commit": commit_date_str, "last_commit": commit_date_str, - "all_dates": [] + "all_dates": [], } # Update stats @@ -193,7 +191,7 @@ def normalize_author(author: str) -> str: agent_stats[author]["message_lengths"].append(len(message)) # Track rework (fix/revert commits) - if re.search(r'\bfix|revert|oops|undo\b', message, re.IGNORECASE): + if re.search(r"\bfix|revert|oops|undo\b", message, re.IGNORECASE): agent_stats[author]["rework_commits"] += 1 # Update date range @@ -213,17 +211,21 @@ def normalize_author(author: str) -> str: for agent_name, stats in agent_stats.items(): avg_message_length = ( sum(stats["message_lengths"]) / len(stats["message_lengths"]) - if stats["message_lengths"] else 0 + if stats["message_lengths"] + else 0 ) rework_rate = ( - stats["rework_commits"] / stats["total_commits"] - if stats["total_commits"] > 0 else 0 + stats["rework_commits"] / stats["total_commits"] if stats["total_commits"] > 0 else 0 ) # Parse first and last commit dates for display try: - first_date = datetime.strptime(stats["first_commit"].split()[0], "%Y-%m-%d").strftime("%Y-%m-%d") - last_date = datetime.strptime(stats["last_commit"].split()[0], "%Y-%m-%d").strftime("%Y-%m-%d") + first_date = datetime.strptime(stats["first_commit"].split()[0], "%Y-%m-%d").strftime( + "%Y-%m-%d" + ) + last_date = datetime.strptime(stats["last_commit"].split()[0], "%Y-%m-%d").strftime( + "%Y-%m-%d" + ) except (ValueError, IndexError): first_date = stats["first_commit"][:10] last_date = stats["last_commit"][:10] @@ -236,7 +238,7 @@ def normalize_author(author: str) -> str: "rework_rate": round(rework_rate * 100, 1), # Percentage "avg_message_length": round(avg_message_length, 1), "first_commit": first_date, - "last_commit": last_date + "last_commit": last_date, } agents_list.append(agent_data) @@ -254,9 +256,9 @@ def normalize_author(author: str) -> str: date_range = "Unknown" if all_dates: try: - dates_sorted = sorted(set( - datetime.strptime(d.split()[0], "%Y-%m-%d") for d in all_dates - )) + dates_sorted = sorted( + set(datetime.strptime(d.split()[0], "%Y-%m-%d") for d in all_dates) + ) if dates_sorted: date_range = f"{dates_sorted[0].strftime('%Y-%m-%d')} to {dates_sorted[-1].strftime('%Y-%m-%d')}" except ValueError: @@ -268,8 +270,8 @@ def normalize_author(author: str) -> str: "meta": { "total_commits": total_commits, "total_agents": len(agents_list), - "date_range": date_range - } + "date_range": date_range, + }, } @@ -759,19 +761,19 @@ def generate_benchmark_html(benchmark_data: Dict[str, Any], project_name: str) -
-
{meta['total_commits']}
+
{meta["total_commits"]}
Total Commits
-
{meta['total_agents']}
+
{meta["total_agents"]}
Active Agents
-
{meta['date_range'].split(' to ')[0] if ' to ' in meta['date_range'] else meta['date_range']}
+
{meta["date_range"].split(" to ")[0] if " to " in meta["date_range"] else meta["date_range"]}
Project Start
-
{meta['date_range'].split(' to ')[1] if ' to ' in meta['date_range'] else ''}
+
{meta["date_range"].split(" to ")[1] if " to " in meta["date_range"] else ""}
Latest Activity
diff --git a/archaeology/visualization/dashboard.py b/archaeology/visualization/dashboard.py index c5f4103..b73baaf 100644 --- a/archaeology/visualization/dashboard.py +++ b/archaeology/visualization/dashboard.py @@ -13,7 +13,9 @@ from typing import Any from archaeology.visualization.design_system import ( - head_bundle, body_end_bundle, THEME_SWITCHER_HTML, THEME_SWITCHER_CSS + THEME_SWITCHER_HTML, + body_end_bundle, + head_bundle, ) # Deliverable categories with display metadata and colors @@ -31,7 +33,9 @@ } -def _discover_all_deliverables(deliverables_dir: Path, project_name: str) -> dict[str, list[dict[str, str]]]: +def _discover_all_deliverables( + deliverables_dir: Path, project_name: str +) -> dict[str, list[dict[str, str]]]: """Scan all deliverable subdirectories and categorize files.""" result: dict[str, list[dict[str, str]]] = {} for cat_name in CATEGORIES: @@ -80,15 +84,17 @@ def discover_projects(projects_dir: Path) -> list[dict[str, Any]]: deliverables = _discover_all_deliverables(deliverables_dir, project_dir.name) total_deliverables = sum(len(v) for v in deliverables.values()) - projects.append({ - "name": project_dir.name, - "slug": project_dir.name, - "meta": meta, - "visuals": visuals, - "deliverables": deliverables, - "total_deliverables": total_deliverables, - "has_data": (deliverables_dir / "data.json").exists() or (data_dir).exists(), - }) + projects.append( + { + "name": project_dir.name, + "slug": project_dir.name, + "meta": meta, + "visuals": visuals, + "deliverables": deliverables, + "total_deliverables": total_deliverables, + "has_data": (deliverables_dir / "data.json").exists() or (data_dir).exists(), + } + ) return projects @@ -107,7 +113,11 @@ def _load_project_meta(deliverables_dir: Path, data_dir: Path) -> dict[str, Any] meta["active_days"] = summary.get("active_days", 0) meta["span_days"] = summary.get("span_days", 0) meta["era_count"] = len(data.get("eras", [])) - meta["authors"] = list(data.get("authors", {}).keys()) if isinstance(data.get("authors"), dict) else [] + meta["authors"] = ( + list(data.get("authors", {}).keys()) + if isinstance(data.get("authors"), dict) + else [] + ) except (json.JSONDecodeError, OSError): pass @@ -155,6 +165,7 @@ def _load_git_metrics(data_dir: Path) -> dict[str, Any]: span = 0 if len(sorted_dates) >= 2: from datetime import date as _date + try: d0 = _date.fromisoformat(sorted_dates[0]) d1 = _date.fromisoformat(sorted_dates[-1]) @@ -246,7 +257,11 @@ def _project_description(name: str, meta: dict) -> str: return f"{commits:,} commits of development history" -def generate_master_dashboard(projects: list[dict[str, Any]], api_section_html: str = "", api_repos: list[dict[str, Any]] | None = None) -> str: +def generate_master_dashboard( + projects: list[dict[str, Any]], + api_section_html: str = "", + api_repos: list[dict[str, Any]] | None = None, +) -> str: """Generate the master dashboard HTML.""" now = datetime.utcnow().strftime("%Y-%m-%d %H:%M UTC") api_repos = api_repos or [] @@ -255,12 +270,13 @@ def generate_master_dashboard(projects: list[dict[str, Any]], api_section_html: if api_repos: print(f" API repos: {len(api_repos)} repos, {api_commits:,} commits") else: - print(f" WARNING: No API repos loaded") + print(" WARNING: No API repos loaded") # Sort projects by commit count (descending) — richest first def _proj_commits(p: dict) -> int: c = p["meta"].get("commits", 0) return c if isinstance(c, int) else 0 + projects_sorted = sorted(projects, key=_proj_commits, reverse=True) # Separate featured (top project) from the rest @@ -290,16 +306,16 @@ def _proj_commits(p: dict) -> int: featured_html = f"""
- + @@ -320,7 +336,9 @@ def _proj_commits(p: dict) -> int: commits_fmt = f"{commits:,}" if isinstance(commits, int) else str(commits) viz_links = "" for viz in proj["visuals"]: - viz_links += f'{viz["name"]}\n ' + viz_links += ( + f'{viz["name"]}\n ' + ) # Category pills cat_pills = "" @@ -331,15 +349,15 @@ def _proj_commits(p: dict) -> int: cat_pills += f'{cat_meta["label"]} {count}\n' project_cards += f""" - +
-

{proj['name'].upper()}

- {total_deliverables} {_pluralize(total_deliverables, 'deliverable')} +

{proj["name"].upper()}

+ {total_deliverables} {_pluralize(total_deliverables, "deliverable")}
-
{commits_fmt}{_pluralize(commits if isinstance(commits, int) else 0, 'commit')}
-
{eras if eras else '—'}{_pluralize(eras, 'era')}
-
{active_days}{_pluralize(active_days if isinstance(active_days, int) else 0, 'active day')}
+
{commits_fmt}{_pluralize(commits if isinstance(commits, int) else 0, "commit")}
+
{eras if eras else "—"}{_pluralize(eras, "era")}
+
{active_days}{_pluralize(active_days if isinstance(active_days, int) else 0, "active day")}
{cat_pills}