diff --git a/.bca-baseline.toml b/.bca-baseline.toml index 1c0dfad86..d09408f8e 100644 --- a/.bca-baseline.toml +++ b/.bca-baseline.toml @@ -5,7 +5,7 @@ # # Path keys are relative to this file's directory (the anchor), so the # baseline survives any `--paths` form (`.`, `src/`, `$PWD`, …). -version = 5 +version = 6 [provenance] tier = "soft" @@ -14,1224 +14,1043 @@ headroom = 0.95 [[entry]] path = "big-code-analysis-cli/src/baseline.rs" qualified = "Baseline::from_str" -start_line = 380 metric = "halstead.effort" -value = 56942.06556051136 +value = 58410.89968953587 [[entry]] path = "big-code-analysis-cli/src/baseline.rs" qualified = "Baseline::match_in_group" -start_line = 581 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/baseline_diff.rs" qualified = "BaselineDiff::column_widths" -start_line = 299 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/baseline_diff.rs" qualified = "BaselineDiff::compute" -start_line = 109 metric = "halstead.effort" -value = 116932.193453204 +value = 118801.67275912332 [[entry]] path = "big-code-analysis-cli/src/baseline_diff.rs" qualified = "BaselineDiff::compute" -start_line = 109 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/check_format.rs" qualified = "AggregatedFormat::dump" -start_line = 89 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/check_format.rs" qualified = "write_github_annotations" -start_line = 167 metric = "halstead.effort" value = 47834.17129195417 [[entry]] path = "big-code-analysis-cli/src/commands.rs" qualified = "run" -start_line = 101 metric = "halstead.effort" value = 136001.1105471366 [[entry]] path = "big-code-analysis-cli/src/commands/analyze.rs" qualified = "run_command_metrics" -start_line = 88 metric = "halstead.effort" value = 69608.24451083591 [[entry]] path = "big-code-analysis-cli/src/commands/analyze.rs" qualified = "write_aggregate" -start_line = 211 metric = "halstead.effort" value = 48544.21441032752 [[entry]] path = "big-code-analysis-cli/src/commands/analyze.rs" qualified = "write_aggregate" -start_line = 211 metric = "nargs" value = 7.0 -[[entry]] -path = "big-code-analysis-cli/src/commands/check.rs" -qualified = "emit_check_results" -start_line = 553 -metric = "nargs" -value = 8.0 - [[entry]] path = "big-code-analysis-cli/src/commands/check.rs" qualified = "filter_by_baseline" -start_line = 364 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/commands/check.rs" qualified = "run_check" -start_line = 13 metric = "halstead.effort" -value = 70554.29290465376 +value = 60852.18615151166 [[entry]] path = "big-code-analysis-cli/src/commands/check/effective_config.rs" qualified = "EffectiveConfig::from_resolved" -start_line = 192 metric = "nargs" value = 14.0 [[entry]] path = "big-code-analysis-cli/src/commands/check/effective_config.rs" qualified = "print_effective_config" -start_line = 23 metric = "nargs" value = 9.0 [[entry]] path = "big-code-analysis-cli/src/commands/diff_cmd.rs" qualified = "compute_since_diff" -start_line = 113 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/commands/init.rs" qualified = "run_command_init" -start_line = 186 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/commands/init.rs" qualified = "scaffold_baseline" -start_line = 118 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/commands/report.rs" qualified = "run_command_report" -start_line = 5 metric = "abc" value = 39.06404996924922 [[entry]] path = "big-code-analysis-cli/src/commands/report.rs" qualified = "run_command_report" -start_line = 5 metric = "halstead.effort" value = 63112.954051595756 [[entry]] path = "big-code-analysis-cli/src/diff.rs" qualified = "materialize_tree" -start_line = 477 metric = "halstead.effort" value = 60390.01093335322 [[entry]] path = "big-code-analysis-cli/src/diff.rs" qualified = "materialize_tree" -start_line = 477 metric = "nargs" value = 10.0 [[entry]] path = "big-code-analysis-cli/src/diff.rs" qualified = "materialize_tree" -start_line = 477 metric = "nexits" value = 9.0 [[entry]] path = "big-code-analysis-cli/src/diff.rs" qualified = "parse_ls_tree_record" -start_line = 568 metric = "nexits" value = 6.0 [[entry]] path = "big-code-analysis-cli/src/dispatch.rs" qualified = "dispatch_metrics" -start_line = 235 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/dispatch.rs" qualified = "dispatch_ops" -start_line = 294 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/html_report.rs" qualified = "write_table_classed" -start_line = 86 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/html_report.rs" qualified = "write_table_classed_with_tooltips" -start_line = 126 metric = "nargs" value = 9.0 [[entry]] path = "big-code-analysis-cli/src/html_report.rs" qualified = "write_table_core" -start_line = 144 metric = "halstead.effort" value = 91596.1159616176 [[entry]] path = "big-code-analysis-cli/src/html_report.rs" qualified = "write_table_core" -start_line = 144 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/html_report.rs" qualified = "write_table_with_tooltips" -start_line = 113 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/html_report/sections.rs" qualified = "" -start_line = 1 metric = "loc.ploc" value = 551.0 [[entry]] path = "big-code-analysis-cli/src/html_report/sections.rs" qualified = "write_language_section" -start_line = 664 metric = "halstead.effort" value = 58205.59941550624 [[entry]] path = "big-code-analysis-cli/src/html_report/sections.rs" qualified = "write_language_section" -start_line = 664 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/lib.rs" qualified = "legacy_hint" -start_line = 830 metric = "halstead.effort" value = 59351.805921378844 [[entry]] path = "big-code-analysis-cli/src/markdown_report.rs" qualified = "" -start_line = 1 metric = "loc.ploc" value = 601.0 [[entry]] path = "big-code-analysis-cli/src/markdown_report.rs" qualified = "extract_summaries_inner" -start_line = 128 metric = "abc" value = 39.153543900903784 [[entry]] path = "big-code-analysis-cli/src/markdown_report.rs" qualified = "write_language_section" -start_line = 992 metric = "halstead.effort" value = 52993.355167839094 [[entry]] path = "big-code-analysis-cli/src/markdown_report/hotspot.rs" qualified = "" -start_line = 1 metric = "loc.ploc" value = 650.0 [[entry]] path = "big-code-analysis-cli/src/markdown_report/sections.rs" qualified = "emit_section_md" -start_line = 68 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/metric_diff.rs" qualified = "MetricDiff::from_sets" -start_line = 269 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/threshold_suggestion.rs" qualified = "closest_names" -start_line = 77 metric = "halstead.effort" value = 55785.628204736706 [[entry]] path = "big-code-analysis-cli/src/threshold_suggestion.rs" qualified = "closest_names" -start_line = 77 metric = "nargs" value = 9.0 [[entry]] path = "big-code-analysis-cli/src/threshold_suggestion.rs" qualified = "edit_distance_with_cutoff" -start_line = 28 metric = "halstead.effort" value = 66737.04848699087 [[entry]] path = "big-code-analysis-cli/src/thresholds.rs" qualified = "ThresholdSet::evaluate_with_policy" -start_line = 802 metric = "halstead.effort" value = 57580.96697647807 [[entry]] path = "big-code-analysis-cli/src/vcs_command.rs" qualified = "" -start_line = 1 metric = "loc.ploc" value = 532.0 [[entry]] path = "big-code-analysis-cli/src/vcs_command.rs" qualified = "build_options" -start_line = 234 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-cli/src/vcs_command.rs" qualified = "rank" -start_line = 324 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-cli/src/vcs_command.rs" qualified = "write_table" -start_line = 475 metric = "nexits" value = 5.0 [[entry]] path = "big-code-analysis-cli/src/vcs_report.rs" qualified = "write_html_body" -start_line = 620 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-py/src/batch.rs" qualified = "PyAnalysisError::__repr__" -start_line = 166 metric = "nexits" value = 8.0 [[entry]] path = "big-code-analysis-py/src/batch.rs" qualified = "analyze_batch" -start_line = 369 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-py/src/batch.rs" qualified = "analyze_batch" -start_line = 369 metric = "nexits" value = 6.0 [[entry]] path = "big-code-analysis-py/src/batch.rs" qualified = "analyze_paths" -start_line = 625 metric = "nargs" value = 12.0 [[entry]] path = "big-code-analysis-py/src/batch.rs" qualified = "analyze_paths" -start_line = 625 metric = "nexits" value = 5.0 [[entry]] path = "big-code-analysis-py/src/lib.rs" qualified = "PyVcsOptions::py_new" -start_line = 607 metric = "nargs" value = 15.0 [[entry]] path = "big-code-analysis-py/src/lib.rs" qualified = "_native" -start_line = 838 metric = "nexits" value = 35.0 [[entry]] path = "big-code-analysis-py/src/lib.rs" qualified = "analyze" -start_line = 284 metric = "nargs" value = 8.0 [[entry]] path = "big-code-analysis-py/src/lib.rs" qualified = "extract_as_of" -start_line = 540 metric = "nexits" value = 5.0 [[entry]] path = "big-code-analysis-py/src/lib.rs" qualified = "register_vcs_submodule" -start_line = 901 metric = "nexits" value = 14.0 [[entry]] path = "big-code-analysis-py/src/lib.rs" qualified = "vcs_trend" -start_line = 754 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-py/src/node.rs" qualified = "PyNode" -start_line = 135 metric = "nom" value = 52.0 [[entry]] path = "big-code-analysis-py/src/node.rs" qualified = "PyNode" -start_line = 135 metric = "wmc" value = 59.0 [[entry]] path = "big-code-analysis-py/src/node.rs" qualified = "PyNode::span" -start_line = 237 metric = "nexits" value = 6.0 [[entry]] path = "big-code-analysis-py/src/sarif.rs" qualified = "collect_offenders" -start_line = 559 metric = "nexits" value = 6.0 [[entry]] path = "big-code-analysis-py/src/sarif.rs" qualified = "extract_line_number" -start_line = 367 metric = "nexits" value = 5.0 [[entry]] path = "big-code-analysis-py/src/sarif.rs" qualified = "resolve_thresholds" -start_line = 278 metric = "nexits" value = 6.0 [[entry]] path = "big-code-analysis-py/src/types_codegen.rs" qualified = "" -start_line = 1 metric = "loc.ploc" value = 745.0 [[entry]] path = "big-code-analysis-py/src/vcs.rs" qualified = "options_from" -start_line = 80 metric = "halstead.effort" value = 47661.70050580648 [[entry]] path = "big-code-analysis-py/src/vcs.rs" qualified = "options_from" -start_line = 80 metric = "nexits" value = 8.0 [[entry]] path = "big-code-analysis-web/src/web/server.rs" qualified = "run_parse" -start_line = 91 metric = "halstead.effort" value = 82478.1543040951 [[entry]] path = "big-code-analysis-web/src/web/server.rs" qualified = "run_with_timeout" -start_line = 244 metric = "nargs" value = 7.0 [[entry]] path = "big-code-analysis-web/src/web/server/routing.rs" qualified = "register_endpoints" -start_line = 101 metric = "abc" value = 91.0 [[entry]] path = "big-code-analysis-web/src/web/vcs.rs" qualified = "compute_vcs_jit" -start_line = 539 metric = "nexits" value = 5.0 [[entry]] path = "big-code-analysis-web/src/web/vcs.rs" qualified = "options_from" -start_line = 180 metric = "halstead.effort" value = 51846.28098938873 [[entry]] path = "big-code-analysis-web/src/web/vcs.rs" qualified = "options_from" -start_line = 180 metric = "nexits" value = 8.0 [[entry]] path = "src/alterator.rs" qualified = "" -start_line = 1 metric = "loc.ploc" value = 545.0 [[entry]] path = "src/ast.rs" qualified = "build" -start_line = 239 metric = "halstead.effort" value = 72581.41181251501 [[entry]] path = "src/c_macro.rs" qualified = "replace" -start_line = 288 metric = "halstead.effort" value = 82074.58162485641 [[entry]] path = "src/c_macro.rs" qualified = "step_normal" -start_line = 119 metric = "halstead.effort" value = 66515.67234293568 [[entry]] path = "src/c_macro.rs" qualified = "step_normal" -start_line = 119 metric = "nexits" value = 6.0 [[entry]] path = "src/getter/bash.rs" qualified = "BashCode::get_op_type" -start_line = 31 metric = "halstead.effort" value = 64170.027458982564 [[entry]] path = "src/getter/elixir.rs" qualified = "ElixirCode::get_op_type" -start_line = 142 metric = "halstead.effort" value = 56464.822908007496 [[entry]] path = "src/getter/irules.rs" qualified = "IrulesCode::get_op_type" -start_line = 19 metric = "halstead.effort" value = 53163.06867422894 [[entry]] path = "src/getter/perl.rs" qualified = "PerlCode::get_op_type" -start_line = 17 metric = "halstead.effort" value = 76538.42828561828 [[entry]] path = "src/getter/php.rs" qualified = "PhpCode::get_op_type" -start_line = 31 metric = "halstead.effort" value = 54691.60133906107 [[entry]] path = "src/getter/python.rs" qualified = "PythonCode::get_op_type" -start_line = 16 metric = "halstead.effort" value = 48346.66496041094 [[entry]] path = "src/getter/ruby.rs" qualified = "RubyCode::get_op_type" -start_line = 21 metric = "halstead.effort" value = 84039.86210807812 [[entry]] path = "src/metrics/abc/go.rs" qualified = "GoCode::compute" -start_line = 94 metric = "cyclomatic" value = 15.0 [[entry]] path = "src/metrics/abc/go.rs" qualified = "GoCode::compute" -start_line = 94 metric = "halstead.effort" value = 50933.14135135399 [[entry]] path = "src/metrics/abc/kotlin.rs" qualified = "KotlinCode::compute" -start_line = 156 metric = "cyclomatic" value = 16.0 [[entry]] path = "src/metrics/abc/perl.rs" qualified = "perl_inspect_container" -start_line = 55 metric = "halstead.effort" value = 49178.00338181524 [[entry]] path = "src/metrics/abc/rust.rs" qualified = "RustCode::compute" -start_line = 127 metric = "cyclomatic" value = 16.0 [[entry]] path = "src/metrics/cognitive.rs" qualified = "tcl_switch_decision_arms" -start_line = 534 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/cognitive.rs" qualified = "tcl_switch_decision_arms" -start_line = 534 metric = "nexits" value = 6.0 [[entry]] path = "src/metrics/cognitive/csharp.rs" qualified = "CsharpCode::compute" -start_line = 17 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/cognitive/kotlin.rs" qualified = "KotlinCode::compute" -start_line = 36 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/cognitive/php.rs" qualified = "PhpCode::compute" -start_line = 17 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/loc/c.rs" qualified = "CCode::compute" -start_line = 17 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/loc/cpp.rs" qualified = "CppCode::compute" -start_line = 17 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/loc/groovy.rs" qualified = "GroovyCode::compute" -start_line = 17 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/loc/mozcpp.rs" qualified = "MozcppCode::compute" -start_line = 17 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/loc/objc.rs" qualified = "ObjcCode::compute" -start_line = 17 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/loc/perl.rs" qualified = "PerlCode::compute" -start_line = 17 metric = "halstead.effort" value = 55025.91689557041 [[entry]] path = "src/metrics/loc/shared.rs" qualified = "add_multiline_string_ploc" -start_line = 171 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/nargs.rs" qualified = "" -start_line = 1 metric = "loc.ploc" value = 525.0 [[entry]] path = "src/metrics/nargs.rs" qualified = "ObjcCode::compute" -start_line = 325 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/npa/csharp.rs" qualified = "CsharpCode::compute" -start_line = 12 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/npa/go.rs" qualified = "GoCode::compute" -start_line = 42 metric = "nargs" value = 7.0 [[entry]] path = "src/metrics/npa/php.rs" qualified = "PhpCode::compute" -start_line = 12 metric = "halstead.effort" value = 52227.844233305164 [[entry]] path = "src/metrics/npa/php.rs" qualified = "PhpCode::compute" -start_line = 12 metric = "nargs" value = 10.0 [[entry]] path = "src/metrics/npa/python.rs" qualified = "python_self_attr_name_bytes" -start_line = 303 metric = "nexits" value = 6.0 [[entry]] path = "src/metrics/npm/go.rs" qualified = "GoCode::compute" -start_line = 12 metric = "halstead.effort" value = 65898.2477704053 [[entry]] path = "src/metrics/npm/go.rs" qualified = "GoCode::compute" -start_line = 12 metric = "nargs" value = 10.0 [[entry]] path = "src/node.rs" qualified = "Node<'a>" -start_line = 93 metric = "nom" value = 34.0 [[entry]] path = "src/ops.rs" qualified = "ops_inner" -start_line = 310 metric = "halstead.effort" -value = 70750.09027709182 +value = 67661.84803230887 [[entry]] path = "src/output/checkstyle.rs" qualified = "write_checkstyle" -start_line = 45 metric = "nexits" value = 7.0 [[entry]] path = "src/output/code_climate.rs" qualified = "write_code_climate" -start_line = 73 metric = "halstead.effort" value = 60471.77938561803 [[entry]] path = "src/output/dump_metrics.rs" qualified = "dump_metrics" -start_line = 148 metric = "nexits" value = 6.0 [[entry]] path = "src/output/dump_metrics.rs" qualified = "dump_space" -start_line = 104 metric = "nexits" value = 9.0 [[entry]] path = "src/output/dump_metrics.rs" qualified = "dump_value" -start_line = 241 metric = "nexits" value = 5.0 [[entry]] path = "src/output/dump_ops.rs" qualified = "dump_ops_values" -start_line = 140 metric = "nexits" value = 12.0 [[entry]] path = "src/output/dump_ops.rs" qualified = "dump_space" -start_line = 75 metric = "nexits" value = 9.0 [[entry]] path = "src/output/funcspace_row.rs" qualified = "metric_values" -start_line = 34 metric = "abc" value = 111.87939935484101 [[entry]] path = "src/output/funcspace_row.rs" qualified = "metric_values" -start_line = 34 metric = "halstead.effort" value = 80351.11907066956 [[entry]] path = "src/output/sarif.rs" qualified = "write_sarif_with_suppressed" -start_line = 168 metric = "nargs" value = 7.0 [[entry]] path = "src/parser.rs" qualified = "Parser::filters" -start_line = 176 metric = "halstead.effort" value = 53278.10526315789 [[entry]] path = "src/spaces/ast.rs" qualified = "Ast::from_path" -start_line = 101 metric = "nexits" value = 5.0 [[entry]] path = "src/spaces/code_metrics.rs" qualified = "CodeMetrics::fmt" -start_line = 33 metric = "nexits" value = 8.0 [[entry]] path = "src/spaces/compute.rs" qualified = "compute_per_node" -start_line = 248 metric = "halstead.effort" value = 63007.70094399036 [[entry]] path = "src/spaces/compute.rs" qualified = "compute_per_node" -start_line = 248 metric = "nargs" value = 7.0 [[entry]] path = "src/spaces/compute.rs" qualified = "metrics_inner" -start_line = 544 metric = "halstead.effort" -value = 125902.05368685763 +value = 122031.61766024183 [[entry]] path = "src/suppression.rs" qualified = "parse_native" -start_line = 380 metric = "nexits" -value = 8.0 +value = 6.0 [[entry]] path = "src/tools.rs" qualified = "read_file_with_eol" -start_line = 142 metric = "nexits" value = 7.0 [[entry]] path = "src/vcs/bus_factor.rs" qualified = "authors_of_file" -start_line = 322 metric = "nargs" value = 9.0 [[entry]] path = "src/vcs/bus_factor.rs" qualified = "compute" -start_line = 192 metric = "halstead.effort" value = 47694.803479864124 [[entry]] path = "src/vcs/bus_factor.rs" qualified = "compute" -start_line = 192 metric = "nargs" value = 10.0 [[entry]] path = "src/vcs/git/blame.rs" qualified = "PerFunctionBlame" -start_line = 255 metric = "nom" value = 30.0 [[entry]] path = "src/vcs/git/blame.rs" qualified = "PerFunctionBlame::blame_spans" -start_line = 391 metric = "nargs" value = 7.0 [[entry]] path = "src/vcs/git/blame.rs" qualified = "PerFunctionBlame::open" -start_line = 266 metric = "nexits" value = 6.0 [[entry]] path = "src/vcs/git/blame.rs" qualified = "PerFunctionBlame::resolve_commit" -start_line = 518 metric = "nexits" value = 6.0 [[entry]] path = "src/vcs/git/cached.rs" qualified = "build_cached" -start_line = 40 metric = "halstead.effort" value = 76740.04063247392 [[entry]] path = "src/vcs/git/cached.rs" qualified = "build_cached" -start_line = 40 metric = "nexits" value = 11.0 [[entry]] path = "src/vcs/git/cached.rs" qualified = "incremental_events" -start_line = 161 metric = "nargs" value = 12.0 [[entry]] path = "src/vcs/git/cached.rs" qualified = "load_candidates" -start_line = 216 metric = "nargs" value = 7.0 [[entry]] path = "src/vcs/git/diff_parse.rs" qualified = "unquote_git_path" -start_line = 401 metric = "halstead.effort" value = 71494.28221151498 [[entry]] path = "src/vcs/git/history.rs" qualified = "collect_events" -start_line = 49 metric = "halstead.effort" value = 48928.29416311421 [[entry]] path = "src/vcs/git/history.rs" qualified = "collect_events" -start_line = 49 metric = "nexits" value = 8.0 [[entry]] path = "src/vcs/git/history.rs" qualified = "diff_collect" -start_line = 200 metric = "nexits" value = 8.0 [[entry]] path = "src/vcs/git/history.rs" qualified = "diff_collect::" -start_line = 217 metric = "nexits" value = 6.0 [[entry]] path = "src/vcs/git/history.rs" qualified = "process_commit" -start_line = 128 metric = "nargs" value = 8.0 [[entry]] path = "src/vcs/git/history.rs" qualified = "process_commit" -start_line = 128 metric = "nexits" value = 9.0 [[entry]] path = "src/vcs/git/jit.rs" qualified = "collect_touched" -start_line = 208 metric = "nexits" value = 10.0 [[entry]] path = "src/vcs/git/jit.rs" qualified = "collect_touched::" -start_line = 227 metric = "nexits" value = 7.0 [[entry]] path = "src/vcs/git/jit.rs" qualified = "compute_features" -start_line = 164 metric = "nargs" value = 7.0 [[entry]] path = "src/vcs/git/jit.rs" qualified = "compute_features" -start_line = 164 metric = "nexits" value = 5.0 [[entry]] path = "src/vcs/git/jit.rs" qualified = "experience_features" -start_line = 478 metric = "nargs" value = 7.0 [[entry]] path = "src/vcs/git/jit.rs" qualified = "experience_features" -start_line = 478 metric = "nexits" value = 6.0 [[entry]] path = "src/vcs/git/jit.rs" qualified = "history_features" -start_line = 391 metric = "halstead.effort" value = 52560.02538757721 [[entry]] path = "src/vcs/git/jit.rs" qualified = "score_commit" -start_line = 67 metric = "nexits" value = 6.0 [[entry]] path = "src/vcs/git/mod.rs" qualified = "build" -start_line = 51 metric = "nexits" value = 6.0 [[entry]] path = "src/vcs/git/mod.rs" qualified = "resolve_anchor" -start_line = 96 metric = "nexits" value = 5.0 [[entry]] path = "src/vcs/git/repo.rs" qualified = "enumerate_target_files" -start_line = 127 metric = "nexits" value = 7.0 [[entry]] path = "src/vcs/git/trend.rs" qualified = "build_trend" -start_line = 42 metric = "nexits" value = 5.0 [[entry]] path = "src/vcs/options.rs" qualified = "parse_iso8601" -start_line = 393 metric = "nexits" value = 7.0 [[entry]] path = "src/vcs/options.rs" qualified = "parse_window" -start_line = 345 metric = "nexits" value = 5.0 [[entry]] path = "src/vcs/replay.rs" qualified = "fold_commit" -start_line = 89 metric = "nargs" value = 7.0 [[entry]] path = "src/vcs/score.rs" qualified = "apply_percentile" -start_line = 200 metric = "halstead.effort" value = 68450.87280037583 [[entry]] path = "src/vcs/score.rs" qualified = "apply_percentile" -start_line = 200 metric = "nargs" value = 13.0 [[entry]] path = "src/vcs/stats.rs" qualified = "Accumulator::finalize" -start_line = 238 metric = "halstead.effort" value = 79569.58843331039 [[entry]] path = "src/vcs/stats.rs" qualified = "Accumulator::finalize" -start_line = 238 metric = "nargs" value = 7.0 [[entry]] path = "src/vcs/stats.rs" qualified = "Accumulator::record" -start_line = 154 metric = "halstead.effort" value = 56240.27018612609 [[entry]] path = "src/wire.rs" qualified = "CodeMetrics::from" -start_line = 337 metric = "halstead.effort" value = 49790.15156224738 [[entry]] path = "src/wire/vcs.rs" qualified = "VcsTrend::from_trend" -start_line = 255 metric = "nargs" value = 8.0 diff --git a/.claude/hooks/bca-guidance.txt b/.claude/hooks/bca-guidance.txt index 776cc7576..1c5ad7fed 100644 --- a/.claude/hooks/bca-guidance.txt +++ b/.claude/hooks/bca-guidance.txt @@ -17,11 +17,12 @@ simpler -- not the number smaller. - When the complexity is essential, suppress with a reason. Some functions are irreducibly complex and clearest left whole -- a dispatch match, a hand-rolled parser table, an exhaustive state - machine. For these, add a suppression marker with a one-line - rationale: // bca: suppress() inside the function (canonical - names; it is nexits, never exit -- an unknown metric voids the whole - marker). A clear function with an honest suppression beats a - "compliant" tangle. + machine. For these, add a suppression marker with the rationale on the + same line: // bca: suppress() -- why, inside the function. + Anything after the metric list is free text. Use canonical names; it + is nexits, never exit -- an unknown name warns and is skipped, and the + names beside it still suppress. A clear function with an honest + suppression beats a "compliant" tangle. - Keep the fix where the violation is. The flag is scoped to the function you just edited. Fix it there, mention anything larger you noticed, and do not widen the change into a module rewrite to bring diff --git a/.claude/rules/tool-output.md b/.claude/rules/tool-output.md new file mode 100644 index 000000000..de3c6f428 --- /dev/null +++ b/.claude/rules/tool-output.md @@ -0,0 +1,50 @@ +# Tool Output Rule + +A truncated tool result is not the result. When output is too large to +inline, the harness persists it and shows a fragment headed +`Preview (first 2KB)` alongside the path to the full text. That fragment +is shaped exactly like a finished answer: no marker at the cut, no total +row count, nothing at the end saying more was dropped. + +## Why it yields a wrong answer rather than an error + +A preview is a prefix, so what survives is decided by output order, and +output order is rarely importance order. The commands used here most +often put the rows you need past the cut at least as reliably as before +it: + +- `sort | uniq -c` — the counts you are hunting are the largest, and + they sort **last** unless you passed `-rn`. +- `rg` over a tree — hits arrive in path order, so a sweep across + `src/languages/` or `src/getter/` shows the alphabetically early + languages and hides every one after them. +- `cargo test` / `make pre-commit` — the failure summary is at the end, + behind all the passing output. +- `git log`, `git diff --stat` — the last-listed file is the one cut. + +Reading a prefix of any of these produces a coherent, plausible, partial +answer. Nothing in it looks incomplete, which is the whole problem: the +tell that normally prompts a second look is absent. + +## What it cost here + +During #1127 an agent audited a `sort | uniq -c` tally from the preview +alone, missed two rows, and nearly shipped two tests that could no longer +fail. + +## How to apply + +- Read the persisted file in full before drawing a conclusion from it. A + preview establishes *that* there is output. It never establishes what + the output says. +- Better, do not generate output you will have to re-read. Aggregate in + the command instead: `| wc -l` for a count, `sort -rn | head` to bring + the interesting rows to the front, `rg -c` in place of `rg`, + `--name-only` in place of a full diff. +- Treat every "there are no other X" claim as requiring the full text. + Absence is precisely what a prefix cannot establish, and it is the + conclusion these sweeps are usually run to reach. +- The same discipline covers filters you added yourself. A `| head -20` + written to keep the output small is a truncation you must account for + when reading the result, and unlike the harness preview it leaves no + trace at all. diff --git a/.claude/skills/batch-fix/SKILL.md b/.claude/skills/batch-fix/SKILL.md index 91fdcebb9..f4702af02 100644 --- a/.claude/skills/batch-fix/SKILL.md +++ b/.claude/skills/batch-fix/SKILL.md @@ -347,7 +347,7 @@ Use the issue data (title, body, comments) cached from Step 0a to populate each agent's prompt. Pass each agent the full agent prompt (see below) with ``, -``, and `` substituted. +``, ``, and `` substituted. #### Worktree mode (`ISOLATION_MODE=worktree`) @@ -365,12 +365,23 @@ For a **multi-issue wave**: launch ALL agents in a single message block `model: "opus"`. Do NOT use `run_in_background` -- wait for all agents in the wave to complete before proceeding. -**Known limitation**: Worktree agents fork from `INTEGRATION_BRANCH` at the -moment they are spawned. Within a single multi-issue wave, agents do not -see each other's in-flight work — they only see the integration-branch tip -that existed when the wave started. The merge in Step 4b reconciles their -results mechanically. For tightly coupled issues that must build on each -other, use `--sequential` or run them as a single `/fix-issue`. +**Worktree agents fork from the base, not from the integration-branch +tip.** Each worktree starts at the commit `INTEGRATION_BRANCH` was created +from in Step 1 — `main` — not at wherever the integration branch has since +advanced to. Every wave therefore begins from the same pristine pre-batch +tree, and a Wave-3 agent cannot see what Waves 1 and 2 already merged. +Serializing same-crate issues into separate waves does not, on its own, +prevent them from colliding: both agents edit the pre-batch file. Disjoint +hunks still merge cleanly in Step 4b. Semantic coupling does not, and it +fails quietly — the later agent writes against a helper the earlier one +replaced, or re-does a fix that already landed. + +**Remedy — substitute `` into every worktree-mode +agent prompt.** The prompt's setup step then merges it, which works +because worktrees share the object store (see "Setup — Environment +Verification"). That covers the cross-wave case only. Agents within one +wave run concurrently and still cannot see each other; for tightly coupled +issues use `--sequential`, or run them as a single `/fix-issue`. #### Branch mode (`ISOLATION_MODE=branch`) @@ -429,6 +440,19 @@ improvement over worktree mode where agents fork independently from `main`. > **Branch mode**: Skip this step — results are processed inline in Step 4a. +**Merge only once every agent in the wave has returned, and only against +a clean integration worktree.** Serialisation is the safeguard here; git +is not. Two things follow: + +- `git checkout ` fails outright while another + worktree holds that branch (`fatal: '' is already used by + worktree at …`). That refusal means an agent is still running. Wait for + it and retry — nothing is broken, and there is nothing to repair. +- `git merge` refuses only when the merge would overwrite an uncommitted + change to a file it actually touches. Uncommitted work in any other + file is carried through silently and ends up inside the merge commit. + Do not treat a successful merge as evidence that the tree was idle. + For each agent result in the wave: The worktree agent returns one of: @@ -550,6 +574,27 @@ git checkout Then run `make pre-commit` — the canonical gate per "Validation gates" in `AGENTS.md` (it adds udeps, doc warnings, the lint families, the self-scan gates, and `make snapshot-anchors` on top of the cargo trio). + +**Capture it to a per-invocation log and read the verdict from the log** +— never from a reported exit status, which in this harness is the status +of the trailing command on the line, not of `make`: + +```bash +log=$(mktemp /tmp/bca-pre-commit.XXXXXX.log) +make pre-commit >"$log" 2>&1 +grep '^BCA_GATE:' "$log" +``` + +`BCA_GATE: pass (gate=pre-commit)` or +`BCA_GATE: fail (gate=pre-commit, exit=2, stage=_pc-fmt)` — exactly one +line, and it is the last thing the gate writes. Neither a `make[2]: *** +… Error 2` line nor a green-looking tail is a verdict: the gate is a +parallel DAG, so a failing stage is reported the instant it fails and +other stages keep producing successful output after it. No `BCA_GATE:` +line at all means the run never finished. A fixed path such as +`/tmp/pc.log` is shared with every concurrent agent on the host and has +already caused one agent to diagnose another's failure — hence `mktemp`. + If `make` is unavailable, fall back to: ```bash @@ -735,6 +780,37 @@ your worktree. In branch mode: the orchestrator has verified a clean repo before launching you. +**In worktree mode, run `make worktree-setup` before your first +`make pre-commit`.** A fresh worktree has neither the integration corpora +under `tests/repositories/` (24 tests fail without them: 5 corpus tests and +19 CLI tests analysing a real `DeepSpeech` source file) nor +`big-code-analysis-py/.venv` (`py-typecheck` reports ~33 mypy errors, +`py-test` dies with "Couldn't find a virtualenv"). Both are bootstrap +artifacts, not regressions in your change. The target is idempotent and a +~100 ms no-op afterwards. + +If a corpus checkout is interrupted — it is long enough to hit a command +timeout — the submodule is left with its files deleted but its HEAD already +at the recorded SHA, so **a plain `git submodule update --init` is a silent +no-op**. Re-run `make worktree-setup`; it detects that state and escalates +to `--force`. Do not conclude the corpus is fine because a re-run exited 0. + +**In worktree mode, merge the integration branch before you investigate:** + +```bash +git merge --no-edit +``` + +Your worktree forked from the commit the integration branch was created +from, not from its current tip, so without this you are reading the +pre-batch tree and cannot see fixes that earlier waves already merged. +On a fresh worktree the merge is a fast-forward. It is allowed even +though `` is checked out elsewhere: git refuses to +*check out* a branch another worktree holds, not to merge from one. + +In branch mode, skip this — your branch was created from the integration +branch's current tip in Step 4a and already contains it. + Try to activate Serena: ``` @@ -972,6 +1048,27 @@ cargo test --workspace --all-features pre-commit run --all-files ``` +**When you redirect a gate to a log file, use a path that is yours alone +— put your issue number in it.** Every agent in a wave runs on the same +host and shares `/tmp`, so an obvious fixed name such as `/tmp/pc.log` +gets written by all of them at once. That has already +produced a wrong diagnosis here: one agent paired its own exit status +with a neighbour's log, read a `fmt-check` failure out of it, and spent +the attempt on code it had never touched. + +```bash +make pre-commit > /tmp/pc-.log 2>&1 +grep '^BCA_GATE:' /tmp/pc-.log +``` + +Step 6a explains how to read that verdict line and why an exit status +reported by the harness is not one. The requirement here is narrower: the +path must be unique to you. Before believing a failure you did not +expect, confirm the log is describing your tree — a passing `make +pre-commit` run names its project root well over a hundred times, so +`grep -c "$PROJECT_ROOT" /tmp/pc-.log` returning 0 means +you are reading someone else's run. + If any check fails on code you changed, fix and retry (one attempt). If it fails again or fails on code you did not change, report as FAILED. diff --git a/.gitattributes b/.gitattributes index 8f10025ae..199c3e2eb 100644 --- a/.gitattributes +++ b/.gitattributes @@ -11,3 +11,14 @@ tree-sitter-preproc/** linguist-vendored # the byte-exact comparison. Covering all `*.py` forecloses the same trap for # any future generated-Python gate. *.py text eol=lf + +# The self-scan baseline is generated wholesale by +# `make self-scan-write-baseline-headroom`; nothing in it is +# hand-authored. A textual merge of two branches that both touched a +# baselined file yields hunks that are plausible-looking and wrong on +# *both* sides, because neither side's measurement describes the merged +# tree. `-merge` makes git refuse the text merge and leave the whole +# file conflicted, which says "regenerate" rather than "pick a side". +# Resolve with: +# make self-scan-write-baseline-headroom +.bca-baseline.toml -merge diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 01b47b9ea..ccd6bf40c 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -128,6 +128,26 @@ repos: entry: python3 -m unittest -q utils/check-excluded-manifests-test.py pass_filenames: false + # Self-tests for the worktree-setup submodule classifier. It is + # what decides when to run a destructive `git submodule update + # --force`, so a logic change there re-runs them (#1171). + - id: worktree-setup-test + name: worktree-setup-test + language: system + files: '^utils/worktree-setup(-test)?\.py$' + entry: python3 -m unittest -q utils/worktree-setup-test.py + pass_filenames: false + + # Self-tests for the pre-commit/ci verdict line. gate-status.sh is + # what stands between a red gate and a log that reads as green, so + # a logic change there re-runs them (#1172). + - id: gate-status-test + name: gate-status-test + language: system + files: '^utils/gate-status(-test)?\.sh$' + entry: bash utils/gate-status-test.sh + pass_filenames: false + # Sync-test for check-grammar-crate.py's EXTENSIONS table against # src/langs.rs (#869). Re-runs when the script, its test, or the # source-of-truth language table changes. diff --git a/AGENTS.md b/AGENTS.md index e227ca61a..ccaa507d0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -139,6 +139,12 @@ and `cargo run -p big-code-analysis-web --`. fabricated measurement tables here. Build argument lists as arrays and expand them `"${ARR[@]}"`. See [`.claude/rules/shell.md`](.claude/rules/shell.md). +- **Tool output**: a truncated result is not a result. Persisted output + shown as a `Preview (first 2KB)` fragment carries no marker at the cut, + and the orderings in routine use here — `sort | uniq -c`, `rg` over a + tree, a test run's trailing summary — put the rows that matter past it. + Read the persisted file, or aggregate in the command. See + [`.claude/rules/tool-output.md`](.claude/rules/tool-output.md). - **Code search**: `rg` (ripgrep). Never `grep` via Bash. - **File search**: `fd` (or `fdfind` on Debian/Ubuntu). Never `find` via Bash. @@ -212,6 +218,15 @@ and `cargo run -p big-code-analysis-web --`. ## Validation gates +In a checkout you have not run the gate in before, run +`make worktree-setup` first. It checks out the integration corpora and +the Python-bindings venv; without them `make pre-commit` reports 24 test +failures and ~33 mypy errors that are bootstrap artifacts rather than +regressions. It is idempotent and a ~100 ms no-op afterwards, and it +repairs an interrupted corpus checkout — a state a plain +`git submodule update --init` cannot fix, because the recorded SHA +already matches and the re-run is a silent no-op (#1171). + Before considering a change done, run `make pre-commit` from the repo root. It is the canonical entry point for the full validation gate and runs, in one parallel pass: the cargo trio (`cargo fmt --check`, @@ -242,6 +257,30 @@ Python stage is skipped with a clear "X not found" message when the corresponding tool is absent). `make ci` runs the same checks without auto-fix, mirroring CI behaviour. +**Read the outcome from the `BCA_GATE:` line, nothing else.** Both gates +end with exactly one of `BCA_GATE: pass (gate=pre-commit)` or +`BCA_GATE: fail (gate=pre-commit, exit=2, stage=_pc-fmt)` on stdout, and +no other line of either gate's output starts with that token. Do not +infer an outcome from make's `Error N` lines: the gate is a parallel +DAG, so a stage's `make[2]: *** [… _pc-fmt] Error 2` is printed the +instant it fails and is routinely followed by a hundred lines of other +stages' +*successful* output. Capture each run to its own log path — a fixed +`/tmp/pc.log` is shared with every other checkout and agent on the host +— and grep that log rather than trusting a reported exit status, which +in some tooling is the status of the trailing `echo` rather than of +make: + +```bash +log=$(mktemp /tmp/bca-pre-commit.XXXXXX.log) +make pre-commit >"$log" 2>&1 +grep '^BCA_GATE:' "$log" +``` + +No `BCA_GATE:` line at all is a third state, not a pass: the run +crashed, was killed, or was interrupted. See +[`CONTRIBUTING.md`](CONTRIBUTING.md), "Reading the verdict". + **`_native.pyi` is stubtest-gated (#673).** The hand-written PyO3 stub `big-code-analysis-py/python/big_code_analysis/_native.pyi` is no longer "kept in lockstep by hand" on trust alone: `make py-stubtest` @@ -285,6 +324,29 @@ purely procedural: do not bypass pre-commit, and refresh the baseline with `make self-scan-write-baseline-headroom` in the commit that moved the metric. A red gate on `main` traces directly to skipping this step. +The same rule governs **merges**. `.bca-baseline.toml` is marked +`-merge` in `.gitattributes`, so git leaves it wholly conflicted rather +than splicing two branches' entries together. That is deliberate: +neither side's recorded values describe the merged tree, so hand +resolution is always wrong here, not merely tedious. Regenerate with +`make self-scan-write-baseline-headroom` and stage the result. + +**Price a candidate limit at both tiers before calling it free.** +Converging a limit onto a cluster of existing values is never free while +a proportional soft tier is active — the soft tier measures *distance to +the limit*, so a limit chosen to sit exactly on a population's value +maximises soft-tier breach by construction, and none of those functions +can ever clear the band because they *are* the limit. The natural +measurement is the misleading one: `bca check --threshold =` +is applied last and absolutely, never scaled, so it has no soft tier to +report. Use `bca check --explain-threshold =`, which +reports both tiers plus how many offenders each already has in the +baseline, and weigh the *new-entry* count. This repo's `nargs 7 → 6` +was approved on a hard-tier zero and would have bought 74 permanent +baseline entries (#1143, #1169); the same trap applies to a +`[thresholds.lang.]` override (#1141). See +[Tightening a limit onto a cluster](big-code-analysis-book/src/recipes/thresholds.md#converging-onto-a-cluster). + If GNU Make 4 or any of the optional tools (`taplo`, `rumdl`, `shellcheck`, `shfmt`, `checkmake`, `actionlint`, `cargo-nextest`, `ruff`, `mypy`, `pyright`, `maturin`) are unavailable, fall back to the @@ -474,12 +536,18 @@ smaller. - **When the complexity is essential, suppress with a reason.** Some functions are irreducibly complex *and clearest left whole* — a dispatch `match`, a hand-rolled parser table, an exhaustive state - machine. For these, add an in-source marker with a one-line rationale - rather than contorting the code: - `// bca: suppress()` inside the function (per-file: - `// bca: suppress-file()`). Use canonical metric names — it - is `nexits`, **never** `exit`; an unknown identifier warns *and voids - the entire marker*. `tokens` is not suppressible. See + machine. For these, add an in-source marker rather than contorting + the code, with the rationale on the same line: + `// bca: suppress() — ` inside the function (per-file: + `// bca: suppress-file() — `). Anything after the metric + list is free text; no separator is required. **Name the metrics** if + you want to write a reason: a *bare* verb (`// bca: suppress`, no + list) takes no trailing text at all, because nothing distinguishes a + rationale from prose *about* the marker — put the reason on the line + above if the `All` scope is really what you want. + Use canonical metric names — it is `nexits`, **never** `exit`; an + unknown identifier warns and is skipped, while the recognised names in + the same marker still suppress. `tokens` is not suppressible. See [Suppression markers](big-code-analysis-book/src/commands/suppression.md) and the full recipe at [`recipes/agent-feedback.md`](big-code-analysis-book/src/recipes/agent-feedback.md). diff --git a/CHANGELOG.md b/CHANGELOG.md index 172268f9c..b3e4cae1a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,6 +26,30 @@ for historical reference. ### Added +- `bca check --explain-threshold =`: preview what a + candidate threshold would cost at **both** tiers without editing + `bca.toml` or running a gate (#1169). Reports hard-tier offenders, the + resolved soft limit and its offenders, how many of each already match + a `--baseline` entry — so the new-entry count a reviewer actually + weighs is on screen — and names a cluster when the candidate lands on + top of an existing population. Repeatable, one candidate per metric; + honours `exclude_tests`, `[check] exclude`, suppression markers, + `[thresholds.lang.]` overrides and the baseline exactly as the + run it predicts, and always exits 0 on success (1 on a tool error, + such as a candidate naming a metric this build does not gate). This + closes the gap that + `--threshold` limits are absolute and never scaled, which made the + one-command way to trial a candidate limit the one way that could not + show its soft-tier cost. +- `make worktree-setup`: one idempotent bootstrap for a fresh clone or + `git worktree` (#1171). Checks out the integration corpora under + `tests/repositories/` and the Python-bindings venv, classifying each + submodule first so it is a ~100 ms no-op once the tree is set up. It + escalates to `git submodule update --force` for the interrupted- + checkout state that a plain re-run cannot repair — a plain re-run is a + silent no-op there, because the recorded SHA already matches HEAD — + and refuses to force a submodule that also carries local + modifications. - Per-language threshold overrides in `bca.toml` (#1141). A `[thresholds.lang.]` table layers over the global `[thresholds]` per metric, keyed by the same language slugs `--language` accepts, so @@ -118,6 +142,69 @@ for historical reference. ### Changed +- **(behaviour change)** `bca check` writes its offender rows to + **stdout** instead of stderr (#1167). The rows are the command's + product, so `bca check | wc -l`, `| head`, `| rg -c` and + `bca check 2>/dev/null` now reach them; previously all four reported + an empty offender list, which reads as "this tree is clean" rather + than as an error. Everything that is commentary about the run stays on + stderr: the `--- summary ---` footer, the `--- next steps ---` + remediation block, GitHub Actions annotations, and the + `bca: skipped N …` / `bca: filtered N …` / `warning:` / `error:` + diagnostics. One exception — `--report-format` without `--output` + gives the aggregated SARIF / Checkstyle / Code Climate document + stdout, so the human rows fall back to stderr rather than corrupting + a payload that parses today; `--output ` moves the document off + stdout and the rows return to it. Exit codes (including + `--exit-codes=tiered`), the `--summary-file` digest, and the + aggregated document are unchanged. *Migration:* a pipeline reading the + rows through `2>&1` needs no change; one that captured them with + `2>file` should now use `>file`. +- An unrecognized or non-suppressible metric name inside + `bca: suppress(...)` is now reported and skipped rather than voiding + the entire marker, so `suppress(cognitive, exit)` still silences + `cognitive` (#1168, reversing the contract pinned by #896). Skipping + can only narrow a marker's coverage, so a typo still cannot widen + scope — whereas voiding left the author believing an exemption was + active when it was not. +- **Baseline schema v6.** `.bca-baseline.toml` records `start_line` only + for an entry whose `(path, qualified, metric)` identity is shared with + another — the sole case matching consults it (#1170). Elsewhere the + field re-rendered on every unrelated edit above a baselined function, + churning diffs, hiding real value changes in review, and conflicting + on every merge between branches. Entry order keeps its line-number + tiebreak: `start_line` moves from third to last in the sort key, so it + now decides only between entries sharing one identity. v2–v5 baselines + read unchanged. A baseline written by a *newer* schema now reports the + version mismatch by name instead of a bare serde field error; the + reverse direction cannot be fixed from here, so a v6 file handed to an + already-released pre-v6 `bca` still surfaces the raw error and must be + regenerated with `--write-baseline`. An entry that pins no line drops + it from every rendering: `bca exemptions` omits the `:line` suffix + (text), renders `-` in the Line column (markdown), and omits the + `line` key (JSON); `bca diff-baseline --format json` omits + `start_line`. +- `.bca-baseline.toml` is marked `-merge` in `.gitattributes` (#1170). + The file is generated wholesale, so a textual merge of two branches + produces hunks that are wrong on *both* sides; git now leaves it + conflicted as a whole and the resolution is to regenerate with + `make self-scan-write-baseline-headroom`. `-merge` rather than a + `merge=ours` driver, which would need per-clone `git config` and + silently falls back to a normal merge where unconfigured. +- `make pre-commit` and `make ci` now end with a single + machine-readable verdict line — `BCA_GATE: pass (gate=pre-commit)` or + `BCA_GATE: fail (gate=pre-commit, exit=2, stage=_pc-fmt)` — replacing + the success-only `Pre-commit checks passed` / `CI checks passed` + (#1172). Grep it anchored (`^BCA_GATE:`); absence of the line is a + third state (crash, kill, interrupt), not a pass. `stage=` is a + comma-separated list in make's report order, because `-j` stops + scheduling but lets running jobs finish and fail. Both gates' exit + statuses are unchanged. +- The 24 tests that depend on the integration corpora now fail with a + diagnostic naming the cause and the remedy — including that by-hand + recovery needs `--force` — instead of `bca`'s generic "path does not + exist" or a corpus-count mismatch that conflated an absent corpus with + a drifted one (#1171). - **Metric values move.** A ternary's condition and its two branch operands now each count as a Fitzpatrick Rule 9 unary condition in `abc.conditions`, matching what Java, Groovy, and C# already did @@ -555,6 +642,44 @@ for historical reference. ### Fixed +- A threshold written with the bare `bca diff --metric` alias (`sloc`, + `ploc`, `lloc`, `cloc`, `blank`) now *overrides* the same metric's + dotted spelling instead of adding a second, independent threshold + (#1165). Aliases are resolved where each layer is parsed — the + manifest and `--config` `[thresholds]` table, `[thresholds.soft]`, + `[thresholds.lang.]`, and `--threshold` — so the layers merge by + metric rather than by spelling, one `(function, metric)` pair emits + one offender line, and `--print-effective-config` prints the limit + that actually fires; its output now round-trips through `--config` to + an identical gate result. A single table that sets one metric under + both spellings is rejected rather than silently keeping whichever key + sorts last. +- `bca check --tier=soft=RATIO` (and its `--headroom` alias, and a + `"x"` string in `[thresholds.soft]`) scaled the lower-is-worse + `mi.*` family the wrong way (#1166). A limit there is a *floor*, so + multiplying it by the ratio lowered it: `[thresholds] "mi.original" = + 20` with `--tier=soft=0.5` resolved to a soft floor of 10, below the + hard floor it was meant to warn ahead of. The early-warning band could + never fire first, making the soft tier a silent no-op for the whole + family. The ratio now tightens each limit in its own direction — 20 + with `soft=0.9` resolves to 22.2223, rounded up so the band never + resolves below the exact quotient. The `[thresholds.soft]` + soft-looser-than-hard check, previously restricted to higher-is-worse + metrics because of this defect, now applies to `mi.*` too. +- A suppression marker carrying a rationale on the same line + (`// bca: suppress(nargs) — threaded context`) is no longer rejected + as malformed and silently inert (#1168). Anything after the metric + list is free text, with no separator required: the parentheses are the + positive signal that the comment is a marker. `AGENTS.md` and the book + prescribed writing the rationale there, which is the spelling that + voided the marker. A *bare* verb (`// bca: suppress`, no list) still + takes no trailing text and warns when it carries any — no separator + set can distinguish a rationale from prose *about* the marker, since + `-`, `:`, `//`, `#` and the dashes are exactly what someone writing + `// bca: suppress - we removed this marker, see #123` reaches for, and + reading that as a marker silences every metric on its function with no + diagnostic at all. The warning now names the way out: list the metrics + you mean, or move the reason to the line above. - Corrected the inverted doc comment on `python_apply_boolean_operator`, which described its ancestor walk as counting control constructs and stopping at lambdas when `count_specific_ancestors`'s diff --git a/CLAUDE.md b/CLAUDE.md index acc03e81d..0bc35f7bd 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -23,13 +23,34 @@ destroys other agents' in-progress work: - If you see stale worktrees, leave them alone — another agent may be using them, or the user will clean them up manually. -**Run `make py-bootstrap` once per worktree before `make pre-commit`.** -A fresh worktree inherits no `.venv`, so `py-typecheck` reports ~33 -mypy errors (`pytest` untyped without its stubs) and `py-test` dies -with "Couldn't find a virtualenv". Both are bootstrap artifacts, not -regressions. The cargo stages themselves work in a worktree — that -was #1145, fixed by the `[workspace]` tables on the six excluded -crates. +**Run `make worktree-setup` once per worktree before `make +pre-commit`.** A fresh worktree inherits neither the integration +corpora nor a Python venv, and neither absence names itself: + +- Without the corpora under `tests/repositories/`, 24 tests fail — + the 5 corpus tests, plus 19 CLI tests that analyse a real source + file from the `DeepSpeech` corpus. Since #1171 each names the cause + and the remedy; before it they read as bugs in whatever you were + changing. +- Without `big-code-analysis-py/.venv`, `py-typecheck` reports ~33 + mypy errors (`pytest` untyped without its stubs) and `py-test` dies + with "Couldn't find a virtualenv". + +Both are bootstrap artifacts, not regressions. The cargo stages +themselves work in a worktree — that was #1145, fixed by the +`[workspace]` tables on the six excluded crates. + +`make worktree-setup` is idempotent and a ~100 ms no-op once the tree +is set up, so re-running it is the cheapest way to rule the +environment out. Run it again in particular if a corpus checkout was +interrupted: that leaves the submodule with its files deleted but its +HEAD already at the recorded SHA, so **a plain `git submodule update +--init` is a silent no-op** and only `--force` repairs it. +`worktree-setup` detects that state and escalates on its own; by hand +it is `git submodule update --init --force -- `. It refuses to +force a submodule that also has local modifications — accepted `.snap` +files in `big-code-analysis-output` are exactly what that protects — +and prints the command for you to run once they are safe. ### Tool choice diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index a597f5782..aa674fd7c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -19,11 +19,17 @@ essentials; the deeper conventions live in ```bash git clone https://github.com/dekobon/big-code-analysis cd big-code-analysis -git submodule update --init --recursive +make worktree-setup cargo build --workspace cargo test --workspace --all-features ``` +`make worktree-setup` checks out the integration corpora and, if `uv` is +installed, the Python-bindings venv. It is idempotent, so re-run it any +time the environment looks wrong — including after an interrupted corpus +checkout, which leaves the submodule empty at the recorded SHA where a +plain `git submodule update --init` is a silent no-op. + MSRV is `1.94`, declared once in the root `Cargo.toml` (`[workspace.package] rust-version = "1.94"`) and inherited by every member crate. @@ -31,8 +37,13 @@ member crate. Integration snapshots live in the [`big-code-analysis-output`](https://github.com/dekobon/big-code-analysis-output) submodule under `tests/repositories/`. Initialize submodules before -running the test suite, otherwise integration tests fail with a missing -fixture. +running the test suite, otherwise 24 tests fail; each says so. + +The out-of-band benchmark harness (`make bench-*`) additionally walks +`DeepSpeech`'s own submodules, which the corpus tests exclude and +`worktree-setup` therefore does not fetch. See +[Benchmarking](docs/development/benchmarking.md) for its recursive +checkout. The two binaries shipped with the workspace are: @@ -95,6 +106,57 @@ of [`AGENTS.md`](AGENTS.md). `make ci` runs the same checks without auto-fix, mirroring the GitHub Actions behaviour. +### Reading the verdict + +Both gates end with exactly one machine-readable line on stdout: + +```text +BCA_GATE: pass (gate=pre-commit) +BCA_GATE: fail (gate=pre-commit, exit=2, stage=_pc-fmt) +``` + +Grep for a line *starting* `BCA_GATE:` — nothing else in the repository +emits one, there is exactly one per run, and it is the last thing either +gate writes. + +Do **not** read an outcome out of make's `Error N` lines. The gate is a +parallel DAG, so `make[2]: *** [… _pc-fmt] Error 2` is printed the +instant a stage fails, while other stages are still running — it is +routinely followed by several more stages' *successful* output. In one +measured run the failing stage was reported at line 58 of a 236-line +log, with 174 lines of green output after it. + +`stage=` lists every stage make reported as failing, comma-separated in +the order reported: under `-j` the first failure stops *scheduling*, but +stages already running finish and can fail too. It reads `unknown` when +the gate died before make named a stage. + +**No `BCA_GATE:` line at all is a third state, not a pass.** It means +the run never finished — it crashed, was killed, or was interrupted. On +the failure path make appends its own `make: *** [… pre-commit] Error 2` +epilogue after the verdict, because the gate exits non-zero and make +says so; the verdict line does not change either gate's exit status. + +Capture a run like this: + +```bash +log=$(mktemp /tmp/bca-pre-commit.XXXXXX.log) +make pre-commit >"$log" 2>&1 +grep '^BCA_GATE:' "$log" +``` + +Two habits that idiom encodes: + +- **Give every run its own log path.** A fixed `/tmp/pc.log` is shared + by every checkout, worktree, and tool on the host. During one batch a + concurrent run's log was read as this one's, and a `fmt-check` failure + was diagnosed against code that had never been touched. +- **Read the verdict from the log, not from a reported exit status.** + Some tooling reports the status of the last command on the line, so + `make pre-commit >log 2>&1; echo "EXIT=$?"` announces `echo`'s + success. That is a property of the caller, not of make; the log is + the thing that is true either way. + If GNU Make 4 or any optional tool is unavailable, fall back to the raw cargo trio: diff --git a/Makefile b/Makefile index 934325253..9f363f91f 100644 --- a/Makefile +++ b/Makefile @@ -88,7 +88,7 @@ find-by-ext = $(if $(FD),$(FD) --extension $(1) $(FD_EXCLUDE) $(2),find . -name NEXTEST := $(shell command -v cargo-nextest 2>/dev/null) TEST_CMD = $(if $(NEXTEST),$(NEXTEST) nextest run --workspace --all-features,cargo test --workspace --all-features --lib --bins --tests) -.PHONY: help check-tools build build-release check test test-doc chain-audit fmt fmt-check markdown-fmt markdown-lint shellcheck sh-fmt sh-fmt-check toml-fmt toml-fmt-check toml-lint makefile-check actionlint snapshot-anchors grammar-marker-sync grammar-marker-sync-test check-versions check-excluded-manifests check-excluded-manifests-test check-manpage-assets enums-check enums-codegen-drift enums-codegen-drift-test self-scan self-scan-headroom self-scan-write-baseline self-scan-write-baseline-headroom vcs lint clippy udeps insta-review insta-accept clean distclean install install-cli install-web doc doc-open doc-check doc-check-docsrs book book-serve book-pot book-po-update book-ja book-deploy all pre-commit ci release-check verify-changelog pkg-deb-local pkg-rpm-local dev-env-build dev-env-run dev-env-shell dev-env-rm py-bootstrap py-sync py-relock py-clean py-fmt py-fmt-check py-lint py-typecheck py-test py-stubtest smoke smoke-cli smoke-lib bench bench-scaling bench-walk _check-find _pc-fmt _pc-clippy _pc-test _pc-doc-check _pc-udeps _pc-shellcheck _pc-markdown-lint _pc-toml-lint _pc-makefile-check _pc-actionlint _pc-snapshot-anchors _pc-grammar-marker-sync _pc-grammar-marker-sync-test _pc-check-versions _pc-check-versions-test _pc-check-grammar-crate-test _pc-check-excluded-manifests _pc-check-excluded-manifests-test _pc-check-manpage-assets _pc-enums-check _pc-enums-codegen-drift _pc-enums-codegen-drift-test _pc-self-scan _pc-self-scan-headroom _pc-py-fmt _pc-py-typecheck _pc-py-test _pc-py-stubtest _ci-fmt-check _ci-clippy _ci-test _ci-doc-check _ci-build _ci-udeps _ci-shellcheck _ci-markdown-lint _ci-toml-lint _ci-makefile-check _ci-actionlint _ci-snapshot-anchors _ci-grammar-marker-sync _ci-grammar-marker-sync-test _ci-check-versions _ci-check-versions-test _ci-check-grammar-crate-test _ci-check-excluded-manifests _ci-check-excluded-manifests-test _ci-check-manpage-assets _ci-enums-check _ci-enums-codegen-drift _ci-enums-codegen-drift-test _ci-enums-codegen-drift-test _ci-self-scan _ci-self-scan-headroom _ci-cargo-pipeline _ci-py-fmt-check _ci-py-lint _ci-py-typecheck _ci-py-test _ci-py-stubtest +.PHONY: help check-tools worktree-setup worktree-setup-test build build-release check test test-doc chain-audit fmt fmt-check markdown-fmt markdown-lint shellcheck sh-fmt sh-fmt-check toml-fmt toml-fmt-check toml-lint makefile-check actionlint snapshot-anchors grammar-marker-sync grammar-marker-sync-test check-versions check-excluded-manifests check-excluded-manifests-test check-manpage-assets gate-status-test enums-check enums-codegen-drift enums-codegen-drift-test self-scan self-scan-headroom self-scan-write-baseline self-scan-write-baseline-headroom vcs lint clippy udeps insta-review insta-accept clean distclean install install-cli install-web doc doc-open doc-check doc-check-docsrs book book-serve book-pot book-po-update book-ja book-deploy all pre-commit ci release-check verify-changelog pkg-deb-local pkg-rpm-local dev-env-build dev-env-run dev-env-shell dev-env-rm py-bootstrap py-sync py-relock py-clean py-fmt py-fmt-check py-lint py-typecheck py-test py-stubtest smoke smoke-cli smoke-lib bench bench-scaling bench-walk _check-find _pc-all _pc-fmt _pc-clippy _pc-test _pc-doc-check _pc-udeps _pc-shellcheck _pc-markdown-lint _pc-toml-lint _pc-makefile-check _pc-actionlint _pc-snapshot-anchors _pc-grammar-marker-sync _pc-grammar-marker-sync-test _pc-check-versions _pc-check-versions-test _pc-check-grammar-crate-test _pc-check-excluded-manifests _pc-check-excluded-manifests-test _pc-check-manpage-assets _pc-worktree-setup-test _pc-gate-status-test _pc-enums-check _pc-enums-codegen-drift _pc-enums-codegen-drift-test _pc-self-scan _pc-self-scan-headroom _pc-py-fmt _pc-py-typecheck _pc-py-test _pc-py-stubtest _ci-all _ci-fmt-check _ci-clippy _ci-test _ci-doc-check _ci-build _ci-udeps _ci-shellcheck _ci-markdown-lint _ci-toml-lint _ci-makefile-check _ci-actionlint _ci-snapshot-anchors _ci-grammar-marker-sync _ci-grammar-marker-sync-test _ci-check-versions _ci-check-versions-test _ci-check-grammar-crate-test _ci-check-excluded-manifests _ci-check-excluded-manifests-test _ci-check-manpage-assets _ci-worktree-setup-test _ci-gate-status-test _ci-enums-check _ci-enums-codegen-drift _ci-enums-codegen-drift-test _ci-enums-codegen-drift-test _ci-self-scan _ci-self-scan-headroom _ci-cargo-pipeline _ci-py-fmt-check _ci-py-lint _ci-py-typecheck _ci-py-test _ci-py-stubtest # Default target help: @@ -98,6 +98,7 @@ help: @echo "" @echo "Prerequisites:" @echo " check-tools Verify required tools are present" + @echo " worktree-setup Bootstrap a fresh clone/worktree: corpora + Python venv" @echo "" @echo "Build targets:" @echo " build Build debug binaries" @@ -135,6 +136,8 @@ help: @echo " check-excluded-manifests Assert excluded crates root a workspace and grammars are =-pinned" @echo " check-excluded-manifests-test Self-tests for the check-excluded-manifests gate" @echo " check-manpage-assets Assert every bca-*.1 man page is in deb+rpm asset lists" + @echo " worktree-setup-test Self-tests for the worktree-setup submodule classifier" + @echo " gate-status-test Self-tests for the pre-commit/ci BCA_GATE verdict line" @echo " enums-check cargo clippy + cargo test on workspace-excluded enums crate" @echo " enums-codegen-drift Block enums codegen output drifting from checked-in files" @echo " enums-codegen-drift-test Self-tests for the enums-codegen-drift gate" @@ -161,7 +164,7 @@ help: @echo " smoke Run the release/wheel smoke scripts locally (smoke-cli + smoke-lib)" @echo " smoke-cli CLI smoke (scripts/smoke/cli_wheel_smoke.sh) against a debug bca" @echo " smoke-lib Library smoke (scripts/smoke/lib_wheel_smoke.py) via maturin develop" - @echo " (first-time setup: 'make py-bootstrap' — installs uv-managed venv from uv.lock)" + @echo " (first-time setup: 'make worktree-setup' — corpora + this venv; 'make py-bootstrap' for the venv alone)" @echo "" @echo "Maintenance:" @echo " clean Remove cargo build artifacts (target/) only" @@ -205,6 +208,57 @@ help: check-tools: @bash $(BASE_DIR)utils/check-tools.sh +# One-shot bootstrap for a fresh clone or a fresh `git worktree`: the +# two steps nothing else states and nothing else detects (#1171). +# +# 1. The integration corpora under tests/repositories/. Without them 24 +# tests fail — 5 corpus tests and 19 CLI tests that analyse a real +# source file from the DeepSpeech corpus. +# 2. big-code-analysis-py/.venv. Without it `make pre-commit`'s +# py-typecheck stage reports ~33 mypy errors (pytest is untyped +# without its stubs) and py-test dies with "Couldn't find a +# virtualenv". +# +# Idempotent and cheap when there is nothing to do: the corpus half +# classifies each submodule with a stat-only `git diff` (~60 ms for all +# four) and skips the ones already checked out, and `uv sync --locked` +# re-audits an up-to-date venv in ~30 ms. It does rebuild the compiled +# extension when the Rust sources have moved since the last sync — that +# is the venv being brought up to date, not waste, and it is why this +# delegates to py-bootstrap rather than short-circuiting on a `.venv` +# directory that an interrupted bootstrap may have left half-built. +# Deliberately NOT wired into pre-commit or ci — it is a developer +# bootstrap, not a gate. +# +# The venv half is a soft skip when uv is absent, matching how the other +# py-* targets behave on a barebones host; `make py-bootstrap` remains +# the hard-failing entry point for someone who asked for it by name. +worktree-setup: + @python3 $(BASE_DIR)utils/worktree-setup.py + @if command -v uv >/dev/null 2>&1; then \ + $(MAKE) --no-print-directory py-bootstrap; \ + else \ + echo "worktree-setup: uv not found; skipping the Python venv."; \ + echo " Until it exists, 'make pre-commit' reports ~33 mypy errors from"; \ + echo " py-typecheck and py-test dies with \"Couldn't find a virtualenv\"."; \ + echo " Install uv (curl -LsSf https://astral.sh/uv/install.sh | sh), then"; \ + echo " re-run 'make worktree-setup'."; \ + fi + +# Self-tests for the worktree-setup classifier. Not a gate over the +# tree's contents, but the classifier decides when to run a destructive +# `git submodule update --force`, so it is held to the same standard as +# the utils/ gates and runs in pre-commit alongside them. +worktree-setup-test: + @(cd $(BASE_DIR) && python3 -m unittest -q utils/worktree-setup-test.py) + +# Self-tests for utils/gate-status.sh, which wraps `pre-commit` and `ci` +# and is the only thing standing between a red gate and a log that reads +# as green. Held to the same standard as the utils/ gates for that +# reason, and cheap enough (~0.1s, no cargo) to run in both. +gate-status-test: + @bash $(BASE_DIR)utils/gate-status-test.sh + # --------------------------------------------------------------------------- # Build # --------------------------------------------------------------------------- @@ -902,7 +956,7 @@ lint: $(MAKE) -j --output-sync=target \ _ci-clippy \ _ci-shellcheck _ci-markdown-lint _ci-toml-lint _ci-makefile-check \ - _ci-actionlint _ci-snapshot-anchors _ci-grammar-marker-sync _ci-grammar-marker-sync-test _ci-check-versions _ci-check-versions-test _ci-check-grammar-crate-test _ci-check-excluded-manifests _ci-check-excluded-manifests-test _ci-check-manpage-assets _ci-enums-check _ci-enums-codegen-drift _ci-enums-codegen-drift-test + _ci-actionlint _ci-snapshot-anchors _ci-grammar-marker-sync _ci-grammar-marker-sync-test _ci-check-versions _ci-check-versions-test _ci-check-grammar-crate-test _ci-check-excluded-manifests _ci-check-excluded-manifests-test _ci-check-manpage-assets _ci-worktree-setup-test _ci-gate-status-test _ci-enums-check _ci-enums-codegen-drift _ci-enums-codegen-drift-test # --------------------------------------------------------------------------- # Maintenance @@ -1043,24 +1097,42 @@ book-deploy: # --------------------------------------------------------------------------- all: check test build-release +# `pre-commit` and `ci` delegate their real work to the `_pc-all` / +# `_ci-all` aggregates below and run it under utils/gate-status.sh, which +# ends the log with exactly one `BCA_GATE: pass` / `BCA_GATE: fail +# (stage=…)` line (#1172). It replaces the old prose "Pre-commit checks +# passed", which existed only on the success path — a failing run had no +# terminal verdict at all, only `make[1]: *** [… _pc-fmt] Error 2` lines +# that a parallel run emits mid-stream while other stages are still +# going, and which are therefore not a verdict. The wrapper exits with +# the gate's own status; it never converts a red run into a green exit. +# +# The aggregates exist so the wrapper has a single command to run: +# `ci` needs two sequential sub-makes, and one verdict line per gate is +# the whole point. Do not invoke `_pc-all` / `_ci-all` directly — that +# skips the verdict. pre-commit: + @bash $(BASE_DIR)utils/gate-status.sh pre-commit $(MAKE) _pc-all + +ci: + @bash $(BASE_DIR)utils/gate-status.sh ci $(MAKE) _ci-all + +_pc-all: $(MAKE) -j --output-sync=target \ _pc-test \ _pc-shellcheck _pc-markdown-lint _pc-toml-lint _pc-makefile-check \ - _pc-actionlint _pc-snapshot-anchors _pc-grammar-marker-sync _pc-grammar-marker-sync-test _pc-check-versions _pc-check-versions-test _pc-check-grammar-crate-test _pc-check-excluded-manifests _pc-check-excluded-manifests-test _pc-check-manpage-assets _pc-enums-check _pc-enums-codegen-drift _pc-enums-codegen-drift-test \ + _pc-actionlint _pc-snapshot-anchors _pc-grammar-marker-sync _pc-grammar-marker-sync-test _pc-check-versions _pc-check-versions-test _pc-check-grammar-crate-test _pc-check-excluded-manifests _pc-check-excluded-manifests-test _pc-check-manpage-assets _pc-worktree-setup-test _pc-gate-status-test _pc-enums-check _pc-enums-codegen-drift _pc-enums-codegen-drift-test \ _pc-manpages \ _pc-self-scan _pc-self-scan-headroom \ _pc-py-fmt _pc-py-typecheck _pc-py-test _pc-py-stubtest - @echo "Pre-commit checks passed" -ci: +_ci-all: $(MAKE) _ci-fmt-check $(MAKE) -j --output-sync=target \ _ci-cargo-pipeline \ _ci-shellcheck _ci-markdown-lint _ci-toml-lint _ci-makefile-check \ - _ci-actionlint _ci-snapshot-anchors _ci-grammar-marker-sync _ci-grammar-marker-sync-test _ci-check-versions _ci-check-versions-test _ci-check-grammar-crate-test _ci-check-excluded-manifests _ci-check-excluded-manifests-test _ci-check-manpage-assets _ci-enums-check _ci-enums-codegen-drift _ci-enums-codegen-drift-test \ + _ci-actionlint _ci-snapshot-anchors _ci-grammar-marker-sync _ci-grammar-marker-sync-test _ci-check-versions _ci-check-versions-test _ci-check-grammar-crate-test _ci-check-excluded-manifests _ci-check-excluded-manifests-test _ci-check-manpage-assets _ci-worktree-setup-test _ci-gate-status-test _ci-enums-check _ci-enums-codegen-drift _ci-enums-codegen-drift-test \ _ci-py-fmt-check _ci-py-lint _ci-py-typecheck _ci-py-test _ci-py-stubtest - @echo "CI checks passed" # --------------------------------------------------------------------------- # Parallel pre-commit DAG @@ -1090,6 +1162,8 @@ ci: # ├── _pc-grammar-marker-sync # ├── _pc-grammar-marker-sync-test # ├── _pc-check-versions +# ├── _pc-worktree-setup-test +# ├── _pc-gate-status-test # ├── _pc-enums-check # ├── _pc-enums-codegen-drift (chained after _pc-enums-check) # ├── _pc-py-fmt @@ -1180,6 +1254,12 @@ _pc-check-excluded-manifests-test: _pc-fmt _pc-check-manpage-assets: _pc-fmt $(MAKE) check-manpage-assets +_pc-worktree-setup-test: _pc-fmt + $(MAKE) worktree-setup-test + +_pc-gate-status-test: _pc-fmt + $(MAKE) gate-status-test + _pc-enums-check: _pc-fmt $(MAKE) enums-check @@ -1337,6 +1417,12 @@ _ci-check-excluded-manifests-test: _ci-check-manpage-assets: $(MAKE) check-manpage-assets +_ci-worktree-setup-test: + $(MAKE) worktree-setup-test + +_ci-gate-status-test: + $(MAKE) gate-status-test + _ci-enums-check: $(MAKE) enums-check diff --git a/README.ja.md b/README.ja.md index fc89fa51f..3806deabb 100644 --- a/README.ja.md +++ b/README.ja.md @@ -51,7 +51,7 @@ 問題のある関数をモデルのコンテキストへ報告します。 必要なのは `PATH` 上の `bca`([クイックスタート](#クイックスタート)参照)と数行の設定だけです。 -- **Claude Code** — `PostToolUse` フックが編集されたファイルに対して `bca check` を実行し、違反を stderr 経由でフィードバックします。 +- **Claude Code** — `PostToolUse` フックが編集されたファイルに対して `bca check` を実行し、違反をモデルにフィードバックします。 本リポジトリ自身がリファレンス実装のフック [`.claude/hooks/bca-check.sh`](.claude/hooks/bca-check.sh) をドッグフーディングしています。 - **opencode** — `tool.execute.after` プラグインが同じ役割を果たします。 リファレンスコピーは [`.opencode/plugins/bca-check.js`](.opencode/plugins/bca-check.js) にあります。 diff --git a/README.md b/README.md index f4c02ffb9..edbe83b17 100644 --- a/README.md +++ b/README.md @@ -65,7 +65,7 @@ lands. All it needs is `bca` on `PATH` (see [Quick start](#quick-start)) plus a few lines of config. - **Claude Code**: a `PostToolUse` hook runs `bca check` on the edited - file and feeds violations back through stderr. This repository + file and feeds violations back to the model. This repository dogfoods a reference hook at [`.claude/hooks/bca-check.sh`](.claude/hooks/bca-check.sh). - **opencode**: a `tool.execute.after` plugin does the same; the diff --git a/STABILITY.md b/STABILITY.md index 06d085bed..afda85a0b 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -649,6 +649,21 @@ window for in-place migration; older versions surface a "regenerate with `--write-baseline`" hint instead of silently mis-matching. Recent transitions: +- **v5 → v6** ([#1170](https://github.com/dekobon/big-code-analysis/issues/1170)): + `start_line` became optional and is written only for an entry whose + `(path, qualified, metric)` identity is shared with another — the + sole case matching consults it. Elsewhere the field re-rendered on + every edit above the function, churning diffs and conflicting on + every merge. v2–v5 baselines read unchanged: a recorded line is + still honoured exactly as before. A v6 file read by a pre-v6 bca + fails to parse (a required field is missing) rather than + mis-matching; upgrade bca or regenerate with `--write-baseline`. + The optionality reaches four rendered surfaces, so an entry with no + recorded line changes shape in each: `bca exemptions --format json` + omits the `line` key and `bca diff-baseline --format json` omits + `start_line` (omitted, never `null`), `bca exemptions --format + markdown` renders `-` in the Line column, and the text format drops + the `:line` suffix entirely. - **v3 → v4** ([#377](https://github.com/dekobon/big-code-analysis/issues/377)): entries key on the qualified symbol (`function` renamed to `qualified`) plus a `start_line` *tolerance* rather than the exact diff --git a/bca.toml b/bca.toml index 7aed8dad9..8cc4637bb 100644 --- a/bca.toml +++ b/bca.toml @@ -128,14 +128,18 @@ exclude = [ # `abc` has converged onto the shipped default (40). # # `nargs` deliberately stays at 7. #1143 proposed 7 -> 6 as a free -# ratchet on the strength of a hard-tier measurement: zero new offenders, -# because all 73 offenders at limit 5 sit at exactly 6. But the soft tier -# scales every limit by BCA_HEADROOM (0.95), so a limit of 6 puts all 73 -# of them permanently in the 95-100% band at once — they cannot ever -# clear it, because they are the limit. That is ~73 baseline entries -# bought for no hard-tier gain, which is the "reads as debt rather than -# as a decision" outcome #1143 exists to avoid. Measured 2026-08-01. -# The honest options are to stay at 7 or to do the real 6 -> 5 work. +# ratchet on the strength of a hard-tier measurement: every offender at +# limit 6 is already baselined, so `bca check --threshold nargs=6` +# reports zero. But the soft tier scales every limit by BCA_HEADROOM +# (0.95), and 74 functions sit at exactly 6, so a limit of 6 puts all of +# them 0.3 below the band at once — they cannot ever clear it, because +# they are the limit. That is 74 baseline entries bought for no +# hard-tier gain, which is the "reads as debt rather than as a decision" +# outcome #1143 exists to avoid. The honest options are to stay at 7 or +# to do the real 6 -> 5 work. +# +# Re-measure with `bca check --explain-threshold nargs=6` (#1169) rather +# than a bare `--threshold`, which has no soft tier by design. cognitive = 25 cyclomatic = 15 "halstead.effort" = 50000 diff --git a/big-code-analysis-book/src/commands/check.md b/big-code-analysis-book/src/commands/check.md index 0ffef5a71..ea189861b 100644 --- a/big-code-analysis-book/src/commands/check.md +++ b/big-code-analysis-book/src/commands/check.md @@ -40,7 +40,7 @@ threshold failures, not broken input, so none of them lets `--exit-codes=tiered` (or `[check] exit_codes = "tiered"` in `bca.toml`) splits the single violation code `2` by severity so CI can -branch on it without parsing the `[new]` / `[regr +N%]` stderr tags: +branch on it without parsing the `[new]` / `[regr +N%]` row tags: | Code | Meaning (tiered mode) | |------|-----------------------| @@ -158,6 +158,12 @@ run gates correctly. A bare family head with no single threshold scalar (`halstead`, `mi`) is ambiguous and rejected with a "did you mean" hint listing the concrete sub-metrics — pick one (e.g. `halstead.volume`). +The two spellings name one metric, so they override each other wherever +limits merge: a `[thresholds.lang.c] ploc = 100` replaces the global +`"loc.ploc"`, and `--threshold ploc=100` replaces either. Within a single +table, writing both is an error rather than a silent winner — set the +metric once, under whichever spelling you prefer. + ### Per-language limits (`[thresholds.lang.]`) {#per-language-limits} Metric distributions vary by language more than by project — the @@ -225,10 +231,19 @@ cyclomatic = 15.0 `--tier ` selects which threshold tier the gate compares against. `hard` (the default) uses the `[thresholds]` table verbatim; `soft` is an early-warning tier that fires *before* the hard -gate, flagging a function at `RATIO` of any limit. A bare `--tier` +gate, tightening every limit by `RATIO`. A bare `--tier` means `soft`; `soft` alone uses the default ratio `0.95`; `soft=0.90` pins the ratio to `0.90`; `soft=1.0` disables the blanket scale. +`RATIO` scales the *band*, not the number. For most metrics a limit is +a ceiling, so tightening it means multiplying: `cognitive = 15` with +`soft=0.9` warns at `13.5`. For the lower-is-worse `mi.*` family a +limit is a **floor** — a value below it is the violation — so the same +`0.9` *divides*: `"mi.original" = 20` warns at `20 / 0.9 = 22.2223`, +rounded up so the band never resolves below the exact quotient. +Multiplying a floor would move the warning to `18`, *under* the hard +gate, where nothing could reach it first. + A `[thresholds.soft]` table sets per-metric soft limits, each either an absolute number or a `"x"` string that scales the metric's hard limit: @@ -256,8 +271,8 @@ The soft tier resolves in a fixed order: 2. If a `[thresholds.soft]` table exists, merge its overrides on top; metrics absent from it inherit their hard limit. The blanket `RATIO` does not apply (explicit per-metric limits win). -3. Otherwise scale every limit by the soft `RATIO` (default `0.95` for - a bare `soft`; `soft=1.0` disables scaling). +3. Otherwise tighten every limit by the soft `RATIO` (default `0.95` + for a bare `soft`; `soft=1.0` disables scaling). 4. Repeated `--threshold name=value` flags apply last, absolutely. Steps 1 to 3 run **once per language**, against that language's own @@ -271,9 +286,10 @@ and 30 reports "also breaches the hard limit" while sitting inside the limit the project configured for it. The one combination that *can* invert the two tiers is an absolute -`[thresholds.soft]` value above a language's tightened hard limit — +`[thresholds.soft]` value looser than the hard limit it shadows — `[thresholds.soft] cognitive = 12` alongside -`[thresholds.lang.csharp] cognitive = 4`. That is a tool error (exit +`[thresholds.lang.csharp] cognitive = 4`, or an `mi.*` soft floor +*below* its hard floor. That is a tool error (exit `1`) naming the offending table, for the same reason a `"x"` factor above `1` is rejected at parse time: a soft tier that fires *after* the hard gate is never the intent. @@ -288,10 +304,59 @@ reports the resolved `tier` alongside the post-merge limits. See the [Local threshold gates](../recipes/local-gates.md#two-tier-thresholds) recipe for the migration tip and rationale. +## Previewing a candidate limit (`--explain-threshold`) {#explain-threshold} + +`--explain-threshold =` reports what a candidate limit +would cost — at **both** tiers — instead of gating. It is repeatable, +takes one candidate per metric, and writes nothing. + +```console +$ bca check --explain-threshold nargs=6 +nargs: candidate limit 6 + hard tier (limit 6): 60 offenders, 60 already baselined, 0 new + soft tier (limit 5.7, 0.95x): 135 offenders, 61 already baselined, 74 new + cluster: 75 of 75 soft-band offenders sit at exactly 6 — the candidate + limit itself. The soft tier measures distance to the limit, so a limit + of 6 places them inside the 5.7 band by construction and none of them + can clear it without real work. +``` + +The `new` column is the figure to weigh: it is how many baseline +entries adopting the limit would add. Here the hard tier reads as free +and the soft tier costs 74 entries, none of which any amount of tidying +can retire — which is the whole reason the flag exists. + +`--threshold nargs=6` cannot tell you this. Its limits are applied last +and **absolutely, never scaled**, so a candidate trialled that way has +no soft tier at all. Passing both flags for the same metric is +rejected rather than silently resolved. + +The soft limit is derived from the candidate exactly as a real run +would derive it: a `[thresholds.soft]` entry for the metric wins, then +the `--tier=soft=RATIO` ratio if one was given, then `0.95`. A +`[thresholds.lang.]` table that overrides the metric keeps its own +limit — a candidate *global* limit does not reach it — and the report +says so on its own line. + +Everything else matches the run being predicted: `exclude_tests`, +`[check] exclude`, in-source suppression markers, `--changed-only`, and +the baseline all apply as usual. The one difference is that +baseline-covered offenders are counted rather than dropped, which is +what makes the `already baselined` / `new` split possible. + +The preview replaces the gate, so it never fails: the exit code is `0` +unless a tool error (exit `1`) stops the run, and a one-line reminder of +that goes to stderr. It conflicts with `--write-baseline`, +`--print-effective-config`, `--report-format`, and `--output`, each of +which would produce a second, different artifact. + +See [Choosing thresholds](../recipes/thresholds.md#converging-onto-a-cluster) +for the rule this flag exists to make visible. + ## Offender output -Every offending `(function, metric)` pair prints one line to stderr in -this stable format: +Every offending `(function, metric)` pair prints one line to **stdout** +in this stable format: ```text :-: : = (limit ) @@ -307,6 +372,40 @@ src/parser.rs:42-117: parse_expression: cognitive = 31 (limit 20) Lines are sorted by path, then start line, then metric name, so output is deterministic across runs over the same tree. +### Which stream {#which-stream} + +The offender rows are the command's product, so they go to **stdout**: +`bca check | wc -l`, `| head`, `| rg -c` and `2>/dev/null` all reach +them. + +Everything the run says *about itself* goes to **stderr**: + +- the per-file `--- summary ---` footer, +- the `--- next steps ---` remediation block, +- the GitHub Actions `::error` annotations, +- the `bca: skipped N violations via [check.exclude]` and + `bca: filtered N violations via baseline` counts, +- every `warning:` and `error:` diagnostic. + +One combination inverts this. `--report-format ` without +`--output` puts the aggregated SARIF / Checkstyle / Code Climate +document on stdout, so the human rows fall back to stderr rather than +corrupting it: + +```bash +bca check --report-format sarif | jq '.runs[0].results | length' # document on stdout +bca check --report-format sarif --output report.sarif | wc -l # rows back on stdout +``` + +The `--summary-file` digest is a file, not a stream, and appears on +neither. + +Earlier releases sent the rows to stderr along with everything else, +which made `| wc -l` and `2>/dev/null` report an empty offender list — +indistinguishable from a clean tree. A pipeline that reads the rows +through `2>&1` needs no change; one that captured them with `2>file` +should now use `>file`. + ## Silencing violations with suppression markers In-source comments can silence threshold violations on individual @@ -410,7 +509,7 @@ capture them. # Listed offenders are filtered from threshold checks; a function that # gets worse than its recorded value still fails. Refresh with # `--write-baseline` when entries become stale. -version = 5 +version = 6 [provenance] tier = "hard" @@ -418,16 +517,18 @@ tier = "hard" [[entry]] path = "src/parser.rs" qualified = "Parser::parse_expression" -start_line = 42 metric = "cyclomatic" value = 22.0 ``` The `qualified` field is the function's qualified symbol (the `::`-joined chain of enclosing named containers plus the function -name); `start_line` is retained only to disambiguate a symbol shared by -several functions. With `--baseline-fuzzy-match`, each entry also -carries a `body_hash` for rename-tolerant matching. +name). An entry carries a `start_line` only when its +`(path, qualified, metric)` identity is shared with another entry, the +one case matching consults a line number; recording it elsewhere would +only rewrite the file every time an edit above the function shifted it. +With `--baseline-fuzzy-match`, each entry also carries a `body_hash` +for rename-tolerant matching. Functions already covered by an in-source suppression marker are excluded. Pass `--no-suppress` together with `--write-baseline` to @@ -485,7 +586,7 @@ adoption flow and CI integration patterns. ## Reporting without failing -`--no-fail` prints offenders to stderr but exits `0`. Useful while +`--no-fail` reports offenders as usual but exits `0`. Useful while adopting baselines without flipping CI red. Other CI tools call this behavior `--report-only` or `--soft-fail`; here the flag is spelled `--no-fail`. @@ -510,9 +611,10 @@ common CI case needs zero explicit configuration. | `--summary-file ` | Append markdown digest (per-file rollup + breakdown + top-10 offenders); `never` suppresses it | `auto` detects `GITHUB_STEP_SUMMARY` | | `--no-remediation` | Suppress the trailing `--- next steps ---` block | Block emitted on failure unless this flag is passed | -The per-violation stderr lines and the per-file rollup footer -remain unchanged when none of the above are active, so existing -CI tooling that grep-anchors on the legacy output keeps working. +The per-violation rows and the per-file rollup footer remain +unchanged in *content* when none of the above are active, so CI +tooling that grep-anchors on the legacy text keeps working — but +see [Which stream](#which-stream) for where each now lands. See the [CI integration recipe](../recipes/ci.md#actionable-failure-output) for worked examples — including a "putting it all together" GHA diff --git a/big-code-analysis-book/src/commands/suppression.md b/big-code-analysis-book/src/commands/suppression.md index a08e158b1..3b28e82ca 100644 --- a/big-code-analysis-book/src/commands/suppression.md +++ b/big-code-analysis-book/src/commands/suppression.md @@ -31,6 +31,19 @@ forms: | `bca: suppress-file` | File | Suppress every metric | | `bca: suppress-file(metric, ...)` | File | Suppress only the listed metrics | +The two metric-list forms may carry a **rationale** — free text on the +same line, after the marker itself: + +```rust +// bca: suppress(nargs) — threaded context, not a god-function +``` + +No separator is required and none is privileged: `— why`, `- why`, +`: why`, `// why`, and bare prose all read the same. The two *bare* +verbs take no trailing text at all — `bca: suppress see #123` is not a +marker, whatever punctuation opens the trailing words. See +[Rationale text](#rationale-text) for why. + A function-scope marker attaches to the innermost `FuncSpace` (see the [`FuncSpace` rustdoc](https://docs.rs/big-code-analysis/*/big_code_analysis/spaces/struct.FuncSpace.html)) whose source range contains the comment. @@ -133,7 +146,8 @@ These match the threshold names and the JSON field names emitted on - `nexits` is the canonical spelling — `bca: suppress(nexits)` silences a `nexits` threshold violation. The legacy `exit` alias was retired in #555 and is no longer accepted; spelling it `exit` is an unknown - identifier, which warns *and voids the entire marker* (see below). + identifier, which warns and is skipped — the recognized names beside + it still suppress (see below). - `tokens` is a threshold-checkable metric (and a `CodeMetrics` JSON field) but is deliberately absent from the suppression list: a marker cannot turn it off. Treat `tokens` as a hard resource cap, @@ -150,11 +164,42 @@ of the form warning: path/to/file.rs:42: unknown metric 'no_such_metric' in bca suppression marker; known metrics: abc, cognitive, ... ``` -The marker is dropped — a typo never silently widens scope to other -metrics. Unknown verbs (anything other than `suppress` / `suppress-file`) -and malformed bodies (unbalanced parentheses, trailing garbage) -produce the same shape of warning and are similarly dropped. None of -these are fatal: a typo in one file does not derail a workspace walk. +Only the offending identifier is dropped: `bca: suppress(cognitive, +exit)` still silences `cognitive`. Skipping can only ever shrink what a +marker covers, so a typo never widens scope to a metric the author did +not name — while voiding the whole marker, which is what earlier +releases did, produced the opposite hazard: a suppression its author +believed was active and the gate did not. + +Unknown verbs (anything other than `suppress` / `suppress-file`) and +bodies that parse to no directive at all (an unbalanced parenthesis, a +bare verb followed by any trailing text) produce the same shape of +warning, and those *do* void the marker — there is nothing left of it to +honour. None of these are fatal: a typo in one file does not derail a +workspace walk, and a doc comment that merely mentions the syntax does +not fail your gate. + +### Rationale text {#rationale-text} + +Text after a metric list is yours; `bca` parses past it and does not +interpret it. The asymmetry between the two forms is deliberate: + +- **After a metric list** — `bca: suppress(nargs) threaded context` — + anything goes. The parentheses are a positive signal that this comment + is a marker, and the list has already stated the intent unambiguously, + so whatever follows is prose. +- **After a bare verb** — `bca: suppress — irreducible dispatch` — there + is no trailing text. Nothing in this shape separates a rationale from + a sentence *about* the feature, and no separator can: `-`, `:`, `//`, + `#` and the dashes are exactly the punctuation someone writing + `// bca: suppress - we removed this marker, see #123` reaches for. + Accepting them silences every metric in the enclosing function on the + strength of a comment, so the whole shape warns instead. + + If you want to record a reason, name the metrics — + `bca: suppress(cognitive, cyclomatic) — reason` — which is more + precise than the `All` hammer anyway. If `All` really is what you + mean, put the reason on the line above the marker. ## Where markers may appear @@ -214,8 +259,8 @@ bca check --report-format sarif --no-fail --report-suppressed \ Offenders silenced by an in-source marker or covered by the baseline are emitted into the SARIF document with a SARIF `suppressions` entry — `kind: "inSource"` for markers, `kind: "external"` for the baseline. The -suppression never fails the gate (exit code and the human stderr stream are -unaffected); the `suppressions` entry lets downstream tooling tell +suppression never fails the gate (exit code and the human offender rows +are unaffected); the `suppressions` entry lets downstream tooling tell suppressed debt apart from active offenders. > **GitHub Code Scanning caveat.** GitHub does **not** honor the SARIF @@ -267,7 +312,7 @@ bca exemptions --paths src/ tests/** # Baseline (.bca-baseline.toml, 1 entry) - src/markdown_report.rs:88 write_language_section cognitive 29 + src/markdown_report.rs write_language_section cognitive 29 ``` The surrounding function (for function-scoped markers) gives scope @@ -348,13 +393,13 @@ there is no removal date. These shapes are reserved for future use and are **not** parsed today: -- `bca: suppress(metric, reason = "...")` — audit-trail prose alongside - the metric list, mirroring Rust's `reason = "…"` attribute argument. - `bca: suppress-next` — silence the immediately following declaration - rather than the enclosing function. - -Authors should avoid using either form today: a `reason = "..."` -argument is currently parsed as an unknown metric identifier and -discarded with a stderr warning, and `bca: suppress-next` is rejected -as an unknown verb. Both will be promoted to first-class behavior -in a future release without breaking existing markers. + rather than the enclosing function. Rejected today as an unknown verb; + it will be promoted to first-class behavior in a future release + without breaking existing markers. + +The `reason = "..."` argument this section used to reserve is no longer +planned. Write the rationale after the metric list instead — see +[Rationale text](#rationale-text) — which is the spelling authors +already reach for. Inside the parentheses, `reason = "..."` is still an +unknown metric identifier and is skipped with a stderr warning. diff --git a/big-code-analysis-book/src/recipes/agent-feedback.md b/big-code-analysis-book/src/recipes/agent-feedback.md index 07940840c..0e8ed7713 100644 --- a/big-code-analysis-book/src/recipes/agent-feedback.md +++ b/big-code-analysis-book/src/recipes/agent-feedback.md @@ -13,7 +13,7 @@ is **batch, after-edit, structured** — exactly what So this recipe ships no new binary surface. `bca check` already gives an agent everything it needs: -- A machine-parseable offender list (per-violation lines on stderr, or +- A machine-parseable offender list (per-violation rows on stdout, or `--report-format sarif | code-climate | clang-warning | msvc-warning | checkstyle` for a structured document). - A tiered exit code — `2` when offenders are present, `0` clean, `1` @@ -164,7 +164,9 @@ file_path="$(jq -r '.tool_input.file_path // empty')" # Thresholds, baseline, and excludes come from the repo-root bca.toml. # --no-summary / --no-remediation keep the feedback to the offender -# lines themselves; the guidance below tells the agent what to do. +# rows themselves; the guidance below tells the agent what to do. The +# rows are on stdout and any diagnostic on stderr, so `2>&1` captures +# whichever the run produced. status=0 report="$(bca check "$file_path" --no-summary --no-remediation 2>&1)" || status=$? @@ -282,8 +284,10 @@ export const BcaCheck = async ({ $ }) => { // / `exit_codes = "tiered"`) still report. if (res.exitCode < 2) return - // Surface the offenders to the agent by throwing. - const offenders = res.stderr.toString().trim() || res.stdout.toString().trim() + // Surface the offenders to the agent by throwing. The rows are on + // stdout; stderr is the fallback for a run whose only output is a + // diagnostic. + const offenders = res.stdout.toString().trim() || res.stderr.toString().trim() throw new Error(`bca flagged complexity in ${filePath}:\n\n${offenders}\n\n${GUIDANCE}`) }, } @@ -355,8 +359,10 @@ down. functions are irreducibly complex *and clearest left whole* — a dispatch `match`, a hand-rolled parser table, an exhaustive state machine. For these, do not contort the code: add a suppression marker - with a one-line rationale and move on. A clear function with an - honest `// bca: suppress(...)` is better than a "compliant" tangle. + with the rationale on the same line — + `// bca: suppress(cognitive) — exhaustive opcode dispatch` — and move + on. A clear function with an honest marker is better than a + "compliant" tangle. - **Keep the fix where the violation is.** The flag is scoped to the function you just edited. Fix it there, mention anything larger you noticed, and do not widen the change into a module rewrite to bring @@ -373,22 +379,29 @@ precisely (full reference: - **Per function** — place the marker in a comment inside the function body, naming the metric(s): - `// bca: suppress(cyclomatic, abc)`. A bare `// bca: suppress` - (no list) silences *every* metric for that function. + `// bca: suppress(cyclomatic, abc) — hand-rolled parser table`. A + bare `// bca: suppress` (no list) silences *every* metric for that + function. - **Whole file** — `// bca: suppress-file(halstead, nargs, nexits)` anywhere in the file; the bare `// bca: suppress-file` form silences every metric file-wide. +- **Put the rationale on the marker line, after a metric list.** + Anything after the list is free text and does not need a separator, so + the reason lives where the next reader of the flagged function will + see it. (A *bare* verb takes no trailing text at all — nothing there + distinguishes a rationale from prose about the marker, so naming the + metrics is what buys you a reason.) Write it every time: + a reviewer — human or agent — needs to tell an honest exemption from a + dodge, and `bca exemptions` lists every marker in the tree for exactly + that audit. - **Use canonical metric names.** The accepted identifiers are `abc`, `cognitive`, `cyclomatic`, `halstead`, `loc`, `mi`, `nargs`, `nexits`, `nom`, `npa`, `npm`, `wmc`. It is **`nexits`, not `exit`** - (the legacy `exit` alias was retired) — and an unknown identifier - both warns *and voids the entire marker*, silently un-suppressing - everything it listed. `tokens` is deliberately *not* suppressible; - treat it as a hard resource cap. -- **Always pair a suppression with a rationale comment** so a reviewer - — human or agent — can later tell an honest exemption from a dodge. - `bca exemptions` lists every marker in the tree for exactly this - audit. + (the legacy `exit` alias was retired). An unknown identifier warns and + is skipped; the recognized names beside it still suppress, so + `suppress(cognitive, exit)` silences `cognitive` and complains about + `exit`. `tokens` is deliberately *not* suppressible; treat it as a + hard resource cap. ## Gate at the task boundary, not per edit diff --git a/big-code-analysis-book/src/recipes/baselines.md b/big-code-analysis-book/src/recipes/baselines.md index a29cad464..a7cd0818a 100644 --- a/big-code-analysis-book/src/recipes/baselines.md +++ b/big-code-analysis-book/src/recipes/baselines.md @@ -110,6 +110,27 @@ A shrinking diff is the goal. Two `--write-baseline` runs over an unchanged tree produce byte-identical output, so spurious diffs only appear when actual offenders changed. +#### Before tightening a limit, price it at both tiers {#price-a-candidate-limit} + +Paying debt down invites tightening the limit that produced it, and the +obvious measurement — `bca check --threshold =` — +answers only half the question. A `--threshold` value is applied last +and absolutely, never scaled, so it has **no soft tier**: a candidate +that costs nothing at the hard gate can still put a whole population +permanently inside the `--tier=soft` band, and you will not see it +until you have edited `bca.toml` and run the other gate. + +```bash +bca check --explain-threshold cognitive=15 +``` + +Weigh the `new` column, not the offender count: it is how many baseline +entries the change would add, and it is what a reviewer is actually +being asked to approve. A `cluster:` line means the candidate landed on +top of an existing population, and those entries can never be retired — +see +[Tightening a limit onto a cluster](thresholds.md#converging-onto-a-cluster). + ### 5. PR-review heuristics Run `bca diff-baseline ` and read the summary instead of @@ -188,7 +209,7 @@ Tag prefixes: Tags only appear when `--baseline` is passed; without it the line format is byte-identical to the no-baseline default. CI tooling that -grep-pipes the stderr stream can suppress the trailing summary with +reads the merged streams can suppress the trailing summary with `--no-summary`. The summary footer groups violations by file, cites the single worst @@ -196,9 +217,32 @@ metric per file (max `value / limit` ratio), and sorts rows by violation count descending then path ascending. It is the fastest way to read a long offender list and spot which file to start with. +### Merging two branches + +A baseline is generated wholesale, so a *textual* merge of two branches +that both touched it is never right: each side's values were measured +against its own tree, and the merged tree matches neither. Tell git not +to try, with one line in `.gitattributes`: + +```gitattributes +.bca-baseline.toml -merge +``` + +Git then leaves the file conflicted as a whole instead of splicing the +two sides together, and the resolution is always the same — regenerate: + +```bash +bca check --write-baseline +git add .bca-baseline.toml +``` + +Unlike a `merge=ours` driver, `-merge` needs no per-clone `git config`, +so it works for everyone who clones the repository rather than only for +those who already knew to set it up. + ### 6. Retire the baseline -When `.bca-baseline.toml` contains only `version = 5` and no entries, +When `.bca-baseline.toml` contains only `version = 6` and no entries, drop the `--baseline` flag from CI and delete the file. The thresholds now stand on their own. @@ -208,7 +252,7 @@ A baseline written with `--write-baseline` (v5+) records *which gate it was written against* in a `[provenance]` table: ```toml -version = 5 +version = 6 [provenance] tier = "soft" @@ -276,6 +320,13 @@ containers plus the function name (`MyStruct::do_thing`, `start_line` is closest to the violation — and within `--baseline-line-tolerance` lines (default 50) — wins. Beyond the tolerance the violation is `[new]`. + + This is the only rule that reads a line number, so from schema v6 a + `start_line` is written **only** for entries in such a group. + Everywhere else the field is absent, and the file therefore does not + change when an edit above a baselined function moves it — which is + what keeps a real baseline change (a `value` moving) visible in a + diff instead of buried in line-number noise. 3. **Body hash (opt-in).** With `--baseline-fuzzy-match`, a violation whose qualified symbol no longer matches is matched against entries with an identical normalised body hash within the same @@ -297,8 +348,9 @@ which produce the bulk of baseline churn. ### Remediation footer When the gate finds violations, `bca check` emits a trailing -`--- next steps ---` block on stderr (and inside the -`$GITHUB_STEP_SUMMARY` digest) that names the artifact, prints a +`--- next steps ---` block on stderr — the offender rows themselves +go to stdout — and inside the +`$GITHUB_STEP_SUMMARY` digest, that names the artifact, prints a copy-paste-safe `--write-baseline` refresh invocation, and links back to this recipe. The refresh invocation mirrors the gate's resolved `--paths` / `--exclude` / `--exclude-from` / `--config` / @@ -306,7 +358,7 @@ resolved `--paths` / `--exclude` / `--exclude-from` / `--config` / can refresh the baseline without leaving the page. Suppress the block with `--no-remediation` if downstream tooling -grep-pipes the stderr stream and the trailing block confuses it. +reads the merged streams and the trailing block confuses it. ## Composition with suppression markers @@ -350,7 +402,7 @@ bca exemptions --paths src/ tests/** # Baseline (.bca-baseline.toml, 417 entries) - src/markdown_report.rs:88 write_language_section cognitive 29 + src/markdown_report.rs write_language_section cognitive 29 ... ``` diff --git a/big-code-analysis-book/src/recipes/ci.md b/big-code-analysis-book/src/recipes/ci.md index cb096938a..95d955151 100644 --- a/big-code-analysis-book/src/recipes/ci.md +++ b/big-code-analysis-book/src/recipes/ci.md @@ -297,7 +297,7 @@ A new offender not in the baseline still fails. An improved function passes silently and stays in the baseline until the next `--write-baseline` refresh. -Each surviving violation in the stderr stream is prefixed with a tag +Each surviving violation row is prefixed with a tag so a developer can tell at a glance whether they are looking at a brand-new offender or a known one that has worsened: @@ -307,13 +307,17 @@ brand-new offender or a known one that has worsened: was zero, `[regr +>9999%]` when the regression exceeds 100× the baseline, `[regr NaN]` when the current value is NaN. -After the per-violation lines the stderr stream emits a per-file -rollup footer with the format `: violations (worst: - = vs limit at L)`, sorted by -violation count descending. This is intended to be the first thing a -reader looks at: which file has the most problems, and which metric -is the loudest in that file. Pass `--no-summary` to suppress the -footer for downstream tooling that grep-pipes the stderr stream. +The violation rows go to stdout; everything below goes to stderr. +See [Which stream](../commands/check.md#which-stream) for the full +split. + +After the per-violation rows, stderr carries a per-file rollup footer +with the format `: violations (worst: = + vs limit at L)`, sorted by violation count +descending. This is intended to be the first thing a reader looks at: +which file has the most problems, and which metric is the loudest in +that file. Pass `--no-summary` to suppress the footer for tooling that +reads the merged streams. ### Actionable failure output {#actionable-failure-output} @@ -468,8 +472,8 @@ The artifact URL is derived from `$GITHUB_REPOSITORY` and runs — where there is no upload to point at — instead suggest running `bca report` to see the detailed view locally. -Suppress the block with `--no-remediation` for downstream tooling -that grep-pipes stderr. +Suppress the block with `--no-remediation` for tooling that reads +the merged streams; a plain `bca check | ...` pipeline never sees it. Refresh after focused refactors: @@ -511,8 +515,8 @@ recommended invocation is: What this gives you on a failing PR: -1. **Per-violation stderr lines** — same shape as the legacy gate, - so existing grep tooling keeps working. +1. **Per-violation rows on stdout** — same shape as the legacy + gate, so existing grep tooling keeps working. 2. **Per-file rollup footer** with `Files in this range:` (touched in the PR) listed before `Other offenders:` — the developer sees their own contributions first. diff --git a/big-code-analysis-book/src/recipes/local-gates.md b/big-code-analysis-book/src/recipes/local-gates.md index edfadd9cf..aeedd7c01 100644 --- a/big-code-analysis-book/src/recipes/local-gates.md +++ b/big-code-analysis-book/src/recipes/local-gates.md @@ -59,17 +59,19 @@ and that the broader ratchet pattern formalises. The pattern is two recipes wrapping the same checker, plus two recipes for refreshing the baseline at each tier. -| Target | Tier | Thresholds | Baseline-filtered | Use case | -| ------------------------------------- | ---- | --------------------- | ----------------- | --------------------------------------------------- | -| `self-scan` | hard | 100% of config | yes | Mirror of CI. Must stay green on every commit. | -| `self-scan-headroom` | soft | config × `HEADROOM` | yes | Early-warning band. Fires before the hard tier. | -| `self-scan-write-baseline` | hard | 100% of config | (write) | Absorb today's hard-tier offenders. | -| `self-scan-write-baseline-headroom` | soft | config × `HEADROOM` | (write) | Absorb soft-tier offenders when launching or widening the band. | +| Target | Tier | Thresholds | Baseline-filtered | Use case | +| ------------------------------------- | ---- | ----------------------- | ----------------- | --------------------------------------------------- | +| `self-scan` | hard | 100% of config | yes | Mirror of CI. Must stay green on every commit. | +| `self-scan-headroom` | soft | tightened by `HEADROOM` | yes | Early-warning band. Fires before the hard tier. | +| `self-scan-write-baseline` | hard | 100% of config | (write) | Absorb today's hard-tier offenders. | +| `self-scan-write-baseline-headroom` | soft | tightened by `HEADROOM` | (write) | Absorb soft-tier offenders when launching or widening the band. | The hard tier and the soft tier consume the **same** `[thresholds]` table and the **same** `.bca-baseline.toml`. The -only difference between them is a scalar multiplier applied to -every threshold value before `bca check` sees it. +only difference between them is the `HEADROOM` ratio applied to +every threshold value before `bca check` sees it — tightening each +limit, which for the lower-is-worse `mi.*` family means *raising* it +rather than lowering it. Write the shared baseline at the **soft** tier (`self-scan-write-baseline-headroom`). A v5 baseline records the tier @@ -98,8 +100,8 @@ against. `hard` (the default) compares against `[thresholds]` verbatim. Metrics absent from the soft table inherit their hard limit (no soft band). When a soft table is present, the blanket `RATIO` does not apply — explicit per-metric limits win over the scalar. -3. Otherwise scale every limit by the soft `RATIO` (default `0.95` for - a bare `soft`, so `--tier=soft` is never a silent no-op; +3. Otherwise tighten every limit by the soft `RATIO` (default `0.95` + for a bare `soft`, so `--tier=soft` is never a silent no-op; `soft=1.0` disables scaling). 4. Repeated `--threshold name=value` flags apply last, absolutely. @@ -129,6 +131,14 @@ the `"x"` form for the large-valued metrics (`halstead.*`, scale factor must be in `(0, 1]` — the soft tier is an early-warning band that fires *before* the hard gate, never looser than it. +That "tighter, never looser" rule is what fixes the *direction* of the +scaling, and it is not always a multiplication. A `mi.*` limit is a +floor rather than a ceiling — a value **below** it is the violation — +so `RATIO` divides instead: `"mi.original" = 20` under `--tier=soft=0.9` +resolves to `22.2223`, not `18`. The intuition that a soft band is +"90% of the limit" is right for every other metric and exactly +backwards here. + Both tiers ratchet through the **same** `.bca-baseline.toml` (no separate soft baseline file). `bca check --print-effective-config --tier=soft` prints the resolved limits — paste its `[thresholds]` @@ -378,7 +388,7 @@ violation that also breaches the hard limit. The tiered codes are opt-in; the default stays `0`/`1`/`2`, and every fail-state remains non-zero. Use them when CI needs to route "a new offender appeared" differently from "a baselined offender got worse" without parsing -the `[new]` / `[regr +N%]` stderr tags. +the `[new]` / `[regr +N%]` row tags. ### Wiring into pre-commit and CI @@ -582,7 +592,7 @@ ordering still applies: legitimately moved (someone *did* pay down debt), regenerate the baseline and review the diff. 4. **Retire when empty.** When `.bca-baseline.toml` shrinks to - just `version = 5` (the bare schema stamp with no offender entries), + just `version = 6` (the bare schema stamp with no offender entries), drop the `--baseline` flag and delete the file. The thresholds now stand on their own. diff --git a/big-code-analysis-book/src/recipes/thresholds.md b/big-code-analysis-book/src/recipes/thresholds.md index 57e6bac60..6761f89bf 100644 --- a/big-code-analysis-book/src/recipes/thresholds.md +++ b/big-code-analysis-book/src/recipes/thresholds.md @@ -341,6 +341,52 @@ function of the metrics you are already gating, so gating it too double-counts. already in the table, or because their distributions are dominated by a language idiom rather than by design quality. +## Tightening a limit onto a cluster {#converging-onto-a-cluster} + +**Converging a limit onto a cluster of existing values is never free while a proportional soft tier +is active.** Measure a candidate at *both* tiers before calling it free — the hard-tier reading is +the one you naturally take, and it is the one that misleads. + +The soft tier measures distance to the limit. A limit chosen to sit exactly on a population's value +therefore maximises soft-tier breach by construction: every function at that value is inside the +band the moment the limit lands, and no amount of tidying moves it out, because it *is* the limit. + +This project walked into it. `nargs` was at `7`, and tightening it to `6` looked free — every +offender at the tighter limit was already in the baseline, so `bca check --threshold nargs=6` +reported nothing: + +| `nargs` limit | hard offenders | soft limit (`0.95`) | soft offenders | +|---|---|---|---| +| 7 (kept) | 27 | 6.65 | 60 | +| 6 (proposed) | 60 | 5.7 | 134 | + +Seventy-four functions sit at exactly `6`. A limit of `6` puts all of them `0.3` below the soft +band at once — 74 baseline entries bought for no hard-tier gain, none of which can ever be retired +short of rewriting the signatures. The limit stayed at `7`; the honest alternative was the real +`6 → 5` work. ([#1143](https://github.com/dekobon/big-code-analysis/issues/1143), +[#1169](https://github.com/dekobon/big-code-analysis/issues/1169).) + +`bca check --explain-threshold =` measures both tiers in one walk and names the +cluster when it finds one, without touching `bca.toml`: + +```console +$ bca check --explain-threshold nargs=6 +nargs: candidate limit 6 + hard tier (limit 6): 60 offenders, 60 already baselined, 0 new + soft tier (limit 5.7, 0.95x): 135 offenders, 61 already baselined, 74 new + cluster: 75 of 75 soft-band offenders sit at exactly 6 — the candidate limit itself. … +``` + +The trap is not confined to the global table. A +[per-language override](#per-language) is another way to converge a limit onto a population, and it +has the same cliff — `--explain-threshold` reports a language keeping its own limit on a separate +line rather than folding it into the count. + +The rule generalises past the soft tier: a limit that lands on a cluster leaves the population with +no room, so the *next* tightening is the one that has to move real code. Prefer a limit in the gap +between clusters. See [Preview a candidate limit](../commands/check.md#explain-threshold) for the +flag's full contract. + ## Re-deriving for your own codebase {#re-deriving} Corpus percentiles are a starting point. Your own distribution is better evidence, and producing it diff --git a/big-code-analysis-cli/src/baseline.rs b/big-code-analysis-cli/src/baseline.rs index 36b16544e..ed271174b 100644 --- a/big-code-analysis-cli/src/baseline.rs +++ b/big-code-analysis-cli/src/baseline.rs @@ -5,15 +5,22 @@ //! express "exempt forever", baselines express "tech debt we're paying //! down". //! -//! The on-disk shape is TOML, sorted by `(path, qualified, start_line, -//! metric)` so diffs are reviewable. Each entry records `(path, -//! qualified, start_line, metric, value)` plus an optional `body_hash`; -//! matching keys on `(path, qualified, metric)` with `start_line` as a +//! The on-disk shape is TOML, sorted by `(path, qualified, metric, +//! start_line)` so diffs are reviewable. Each entry records `(path, +//! qualified, metric, value)` plus an optional `body_hash`; matching +//! keys on `(path, qualified, metric)` with `start_line` as a //! tolerance-band disambiguator (issue #377), so a function survives //! line drift from edits above it. The filter is "current value <= //! baseline value", so improvements pass silently and regressions //! still fail. //! +//! `start_line` is therefore written only where it is *consulted*: an +//! identity shared by two or more entries (issue #1170). Nothing about +//! the ordering or the identity depends on a line number, so an edit +//! above a baselined function produces no diff at all, and a merge of +//! two branches that both edited a baselined file has nothing +//! line-shaped left to conflict over. +//! //! Path keys are canonicalised relative to an *anchor*: the directory //! containing the baseline file (lexically resolved, no symlink //! following). Write and read use the same anchor, so a baseline @@ -22,12 +29,20 @@ //! [`normalize_path`] for the encoding pipeline. use std::collections::HashMap; -use std::path::{Component, Path, PathBuf}; +use std::path::{Path, PathBuf}; use serde::{Deserialize, Serialize}; use crate::thresholds::{Violation, breaches_limit}; +mod body_hash; +mod path_key; + +pub(crate) use body_hash::{bare_name, hash_body}; +use body_hash::{decode_body_hash, encode_body_hash}; +pub(crate) use path_key::anchor_for; +use path_key::{lexical_normalize, normalize_path}; + /// Schema version. Bump on breaking format changes. /// /// History: @@ -63,9 +78,19 @@ use crate::thresholds::{Violation, breaches_limit}; /// the "baseline-refresh discipline" guards against (#449). The /// table is omitted from baselines whose provenance is unknown. /// +/// - v5 → v6: `start_line` is optional and is written **only** for an +/// entry whose `(path, qualified, metric)` identity is shared with +/// another entry in the same file — the one case the matcher consults +/// it (issue #1170). A lone record under a key already matched +/// unconditionally, so the field was pure noise there: it re-rendered +/// on every unrelated edit above a baselined function, churning the +/// diff and conflicting on every merge of two branches that touched +/// the same file. v5 files (which carry it everywhere) read +/// unchanged — a present line is honoured exactly as before. +/// /// Either legacy version emits a one-time deprecation warning so the /// user knows to refresh. -pub(crate) const BASELINE_VERSION: u32 = 5; +pub(crate) const BASELINE_VERSION: u32 = 6; /// Lowest legacy version still accepted at read time. Below this we /// reject with a "regenerate" hint instead of silently mis-matching. @@ -241,7 +266,12 @@ struct SymbolKey { /// tolerance band. #[derive(Debug, Clone)] struct Record { - start_line: usize, + /// Recorded line, `None` when the entry omitted it. Only ever + /// consulted for an ambiguous group, and [`from_violations`] writes + /// it for exactly those, so a `None` here means either a + /// unique-identity entry (where it is unreachable) or a hand-edited + /// file (where the record simply drops out of the tolerance match). + start_line: Option, value: f64, /// Normalised body digest, present only for v4 entries written with /// fuzzy matching enabled. Feeds the rename-tolerant fallback. @@ -250,7 +280,8 @@ struct Record { /// One baseline entry flattened for cross-file diffing /// (`bca diff-baseline`): the `(path, qualified, metric)` identity plus -/// the recorded `value` and the human-review `start_line`. Produced by +/// the recorded `value` and, where the file recorded one, the +/// human-review `start_line`. Produced by /// [`Baseline::diff_entries`] *after* version validation, legacy path /// re-canonicalisation, and the non-finite/negative filter have run, so /// a diff observes exactly the records the matcher would. @@ -259,7 +290,7 @@ pub(crate) struct DiffEntry { pub(crate) path: String, pub(crate) qualified: String, pub(crate) metric: String, - pub(crate) start_line: usize, + pub(crate) start_line: Option, pub(crate) value: f64, } @@ -273,10 +304,15 @@ pub(crate) struct BaselineEntry { /// alias from legacy v2/v3 files, where it held only the bare name. #[serde(alias = "function")] qualified: String, - /// 1-based start line. No longer part of the identity key; retained - /// for human review, deterministic ordering, and as the tolerance - /// disambiguator for ambiguous qualified symbols. - start_line: usize, + /// 1-based start line, present only when this entry's identity is + /// ambiguous — i.e. shared with another entry, where the tolerance + /// band is what tells them apart (issue #377). For a unique + /// identity the matcher never reads it, so writing it would only + /// churn the file on every edit above the function (issue #1170). + /// Absent in v6+ unique entries; present on every v2–v5 entry, + /// which read back unchanged. + #[serde(default, skip_serializing_if = "Option::is_none")] + start_line: Option, metric: String, /// Metric value at baseline time. `current > value` still fails /// (ratchet-down). Both non-finite and negative values (the latter @@ -292,6 +328,36 @@ pub(crate) struct BaselineEntry { body_hash: Option, } +/// The identity a baseline entry is keyed on, in the one order every +/// baseline-shaped listing sorts by: `path`, `qualified`, `metric`, and +/// then `start_line` — last, and only as a tie-break within an +/// otherwise-ambiguous group. +/// +/// Stated once because four types spread over three modules carry the +/// same four fields and have to agree (issue #1170): rendered order +/// must not depend on where a function sits in its file, so an edit +/// above one entry cannot reshuffle a baseline, a `diff-baseline` +/// bucket, or an `exemptions` section. Ordering on the identity also +/// makes each identity group contiguous, which +/// [`clear_unambiguous_start_lines`] relies on. And since a v6 baseline +/// records `start_line` only for an ambiguous identity, `start_line` +/// alone no longer supplies a total order at all. +pub(crate) trait BaselineIdentity { + fn identity(&self) -> (&str, &str, &str, Option); +} + +impl BaselineIdentity for BaselineEntry { + fn identity(&self) -> (&str, &str, &str, Option) { + (&self.path, &self.qualified, &self.metric, self.start_line) + } +} + +/// The total order [`BaselineIdentity`] describes, shaped for +/// `slice::sort_by`. +pub(crate) fn cmp_identity(a: &T, b: &T) -> std::cmp::Ordering { + a.identity().cmp(&b.identity()) +} + /// Top-level baseline file. `version` is required; missing field is a /// hard error so a truncated file isn't silently treated as empty. /// @@ -384,16 +450,12 @@ impl Baseline { fuzzy: bool, ) -> Result { let file: BaselineFile = - toml::from_str(text).map_err(|e| format!("malformed baseline TOML: {e}"))?; + toml::from_str(text).map_err(|e| parse_failure_message(text, &e))?; let version = file .version .ok_or_else(|| "baseline missing version field".to_string())?; if !(LEGACY_MIN_VERSION..=BASELINE_VERSION).contains(&version) { - return Err(format!( - "baseline version {version} is not supported by this bca \ - (expected {LEGACY_MIN_VERSION}..={BASELINE_VERSION}); \ - regenerate with `bca check --write-baseline` or upgrade bca" - )); + return Err(unsupported_version_message(version)); } // v2/v3 store a bare function name and (v2 only) pre-canonical // paths. Below v3 the path form predates issue #376 and must be @@ -592,7 +654,13 @@ impl Baseline { records .iter() .filter_map(|r| { - let dist = r.start_line.abs_diff(v.start_line); + // A record with no recorded line cannot be placed, so it + // drops out of the tolerance match rather than matching + // at distance zero. Only reachable from a hand-edited + // file: the writer records a line for every member of an + // ambiguous group, and this arm is unreachable for a + // singleton. + let dist = r.start_line?.abs_diff(v.start_line); (dist <= self.tolerance).then_some((dist, r.value)) }) // Closest recorded line wins. On an exact distance tie — two @@ -688,23 +756,18 @@ pub(crate) fn from_violations( Some(BaselineEntry { path: normalize_path(&anchor, &v.path), qualified: v.function, - start_line: v.start_line, + start_line: Some(v.start_line), metric: v.metric.to_string(), value: v.value, body_hash: v.body_hash.map(encode_body_hash), }) }) .collect(); - // Sort by `qualified` ahead of `start_line` so the records that - // share a symbol (the disambiguation case) cluster together in the - // rendered file, which is what a reviewer reads. - entries.sort_by(|a, b| { - a.path - .cmp(&b.path) - .then(a.qualified.cmp(&b.qualified)) - .then(a.start_line.cmp(&b.start_line)) - .then(a.metric.cmp(&b.metric)) - }); + // Order on the identity alone (see [`BaselineIdentity`]), so the + // rendered order never depends on where a function sits in its file + // and `clear_unambiguous_start_lines` sees contiguous groups. + entries.sort_by(cmp_identity); + clear_unambiguous_start_lines(&mut entries); BaselineFile { version: Some(BASELINE_VERSION), provenance: Some(provenance), @@ -712,385 +775,75 @@ pub(crate) fn from_violations( } } -/// Render a `BaselineFile` to a TOML string with the standard comment -/// header prepended. Output is deterministic byte-for-byte for the -/// same input (TOML's f64 formatter is round-trip stable, struct field -/// order is fixed by `#[derive(Serialize)]` declaration order). -pub(crate) fn render(file: &BaselineFile) -> Result { - let body = toml::to_string(file)?; - Ok(format!("{HEADER}{body}")) -} - -/// The bare (innermost) name of a qualified symbol: the segment after -/// the last `::`, or the whole string when there is no separator. Used -/// to match a v4 violation's qualified symbol (`MyStruct::do_thing`) -/// against a legacy v2/v3 baseline entry that stored only `do_thing`, -/// and to elide a function's own name from its body hash. -pub(crate) fn bare_name(qualified: &str) -> &str { - qualified - .rsplit_once("::") - .map_or(qualified, |(_, tail)| tail) -} - -/// Encode a body hash as the lowercase, zero-padded 16-digit hex form -/// stored in the TOML. Hex (not a TOML integer) because FNV-1a fills -/// the full `u64` range and TOML integers are `i64`. -fn encode_body_hash(h: u64) -> String { - format!("{h:016x}") +/// The user-facing rejection for a baseline whose schema version this +/// build cannot read, naming the two ways out. Shared by the version +/// check and by [`parse_failure_message`], so both spell the remedy +/// identically. +fn unsupported_version_message(version: u32) -> String { + format!( + "baseline version {version} is not supported by this bca \ + (expected {LEGACY_MIN_VERSION}..={BASELINE_VERSION}); \ + regenerate with `bca check --write-baseline` or upgrade bca" + ) } -/// Decode a stored body hash. Returns `None` for any malformed digest -/// (wrong length, non-hex) so a hand-edited file degrades the entry to -/// "no fuzzy fallback" rather than aborting the whole load. -fn decode_body_hash(s: &str) -> Option { - (s.len() == 16) - .then(|| u64::from_str_radix(s, 16).ok()) - .flatten() -} - -/// FNV-1a offset basis and prime for the 64-bit variant. FNV is chosen -/// over the std `DefaultHasher` because the digest is persisted to disk -/// and must be byte-stable across bca versions, platforms, and process -/// runs — `DefaultHasher`'s algorithm and seed carry no such guarantee. -const FNV_OFFSET_BASIS: u64 = 0xcbf2_9ce4_8422_2325; -const FNV_PRIME: u64 = 0x0000_0100_0000_01b3; - -/// Hash the normalised body of the space spanning `start_line..=end_line` -/// (1-based, inclusive) within `source`, with the function's own `name` -/// elided. This is the "fuzzy" in fuzzy matching: a digest of a view of -/// the body that survives the two changes a rename-or-relocate refactor -/// makes (issue #377, rule 3) while still distinguishing genuinely -/// different code: +/// Explain a baseline that would not deserialize. /// -/// 1. **Whitespace**: `\r` is dropped, each line's internal whitespace -/// runs collapse to one space, leading/trailing whitespace is -/// trimmed, and blank lines are skipped — so reformatting, -/// re-indentation, or blank-line churn does not change the digest. -/// 2. **The function's own name**: every whole-word occurrence of `name` -/// (the declaration and any recursive self-calls) is replaced with a -/// fixed sentinel, so renaming `classify` to `categorize` leaves the -/// digest unchanged. Whole-word means bounded by non-identifier bytes, -/// so a `name` of `is` does not corrupt `is_valid`. +/// A file written by a *newer* bca can fail on shape — v6 made +/// `start_line` optional, so a v5-era build rejects a v6 file for a +/// missing field — and that failure lands before the version check ever +/// runs. Re-read just the `version` key and, when it is one this build +/// does not support, report the version mismatch instead: it is the +/// same defect, stated in terms that name the fix. Genuinely malformed +/// TOML falls through to the parser's own message. /// -/// Out-of-range lines are clamped (a `start_line` past EOF yields the -/// empty-body digest), so a malformed span never panics. -pub(crate) fn hash_body(source: &[u8], start_line: usize, end_line: usize, name: &str) -> u64 { - let normalized = normalize_body(source, start_line, end_line); - let elided = elide_identifier(&normalized, name.as_bytes()); - let mut hash = FNV_OFFSET_BASIS; - for &b in &elided { - hash ^= u64::from(b); - hash = hash.wrapping_mul(FNV_PRIME); +/// Only reached on the failure path, so a well-formed baseline is still +/// parsed exactly once. +fn parse_failure_message(text: &str, err: &toml::de::Error) -> String { + #[derive(Deserialize)] + struct VersionProbe { + version: Option, } - hash -} - -/// Build the whitespace-normalised byte view of a line span (step 1 of -/// [`hash_body`]). Lines are joined by `\n`; blank lines are dropped. -fn normalize_body(source: &[u8], start_line: usize, end_line: usize) -> Vec { - let first = start_line.saturating_sub(1); - let mut out = Vec::new(); - for line in source - .split(|&b| b == b'\n') - .skip(first) - .take(end_line.saturating_sub(first)) + if let Ok(VersionProbe { + version: Some(version), + }) = toml::from_str::(text) + && !(LEGACY_MIN_VERSION..=BASELINE_VERSION).contains(&version) { - let mut line_started = false; - let mut pending_space = false; - for &b in line { - if b == b'\r' { - continue; - } - if b.is_ascii_whitespace() { - pending_space = true; - continue; - } - if line_started && pending_space { - out.push(b' '); - } - pending_space = false; - out.push(b); - line_started = true; - } - if line_started { - out.push(b'\n'); - } + return unsupported_version_message(version); } - out + format!("malformed baseline TOML: {err}") } -/// Sentinel emitted in place of the elided function name. `\x00` does not -/// appear in normalised source bytes, so it cannot collide with real -/// content. -const ELIDED_NAME_SENTINEL: u8 = 0x00; - -/// Replace every whole-word occurrence of `name` in `haystack` with -/// [`ELIDED_NAME_SENTINEL`] (step 2 of [`hash_body`]). Whole-word means -/// neither neighbour is an identifier byte. An empty or non-identifier -/// `name` (e.g. the `` / `` sentinels) is left as-is. -fn elide_identifier(haystack: &[u8], name: &[u8]) -> Vec { - let is_ident = |b: u8| b.is_ascii_alphanumeric() || b == b'_'; - if name.is_empty() || !name.iter().copied().all(is_ident) { - return haystack.to_vec(); - } - let mut out = Vec::with_capacity(haystack.len()); - let mut i = 0; - while i < haystack.len() { - let left_ok = i == 0 || !is_ident(haystack[i - 1]); - let matches = left_ok - && haystack[i..].starts_with(name) - && haystack.get(i + name.len()).is_none_or(|&b| !is_ident(b)); - if matches { - out.push(ELIDED_NAME_SENTINEL); - i += name.len(); - } else { - out.push(haystack[i]); - i += 1; - } - } - out -} - -/// Derive the canonical anchor for a baseline file at `baseline_path`: -/// the lexically-absolute directory the file lives in. Used at both -/// write and read time so the key shape is independent of `--paths` -/// form, working directory drift, or whether the path was passed -/// relative or absolute. +/// Strip `start_line` from every entry whose `(path, qualified, metric)` +/// identity is unique in the file (issue #1170). /// -/// Lexical, not symlink-following: `bca` should not surprise users by -/// resolving `src/` through a symlinked directory. The cost is that a -/// baseline written via a symlinked invocation and read via the real -/// path (or vice-versa) does not match — but that mirrors how every -/// other tool (cargo, git) treats workdir identity. -pub(crate) fn anchor_for(baseline_path: &Path) -> PathBuf { - // `std::path::absolute` only fails when the path is empty or the - // platform cannot obtain the CWD (effectively never in a normal - // shell). Fall back to the input path so the rest of the pipeline - // degrades to pre-#376 behaviour (no canonicalisation) instead of - // dying — the worst case is the path-stickiness this fix targets, - // not a hard error. - let abs = std::path::absolute(baseline_path).unwrap_or_else(|_| baseline_path.to_path_buf()); - let mut abs = lexical_normalize(&abs); - // `pop` returns false if the path is already a root or empty; in - // that degenerate case the path itself becomes the anchor. - abs.pop(); - abs -} - -/// Lexically normalise `p` by folding `.` and `..` components without -/// touching the filesystem. POSIX-style folding: -/// - `..` after a `Normal` component pops it (`a/b/../c` → `a/c`). -/// - `..` immediately after a `RootDir` or Windows `Prefix` is a no-op -/// (`/..` → `/`, `C:\..` → `C:\`) — you cannot go above the root. -/// - `..` with no prior component to consume is preserved literally -/// (`../a` → `../a`, `a/../../b` → `../b`). This keeps identity for -/// baselines that legitimately reference a sibling of the anchor. -fn lexical_normalize(p: &Path) -> PathBuf { - let mut out = PathBuf::new(); - for c in p.components() { - match c { - Component::Prefix(_) | Component::RootDir | Component::Normal(_) => out.push(c), - Component::CurDir => {} - Component::ParentDir => match out.components().next_back() { - // Pop the previous Normal component (typical case). - Some(Component::Normal(_)) => { - out.pop(); - } - // POSIX/Windows: `..` past a root or drive prefix is - // a no-op. `/..` resolves to `/`, not `/..`. Without - // this case the normalised path would be non-canonical - // and downstream `strip_prefix` would mis-match. - Some(Component::RootDir | Component::Prefix(_)) => {} - // No prior Normal/Root/Prefix component (e.g., - // relative path starting with `..` or accumulating - // multiple leading `..`s). Preserve the `..` literally - // so a baseline that legitimately points at a sibling - // of the anchor keeps a distinct identity. - _ => out.push(c), - }, - } - } - out -} - -/// Normalize a path for use as a baseline identity key. -/// -/// 1. The path is made lexically absolute (using `anchor` as the base -/// directory when relative) and `.` / `..` components are folded. -/// 2. If the result is under `anchor`, the anchor prefix is stripped so -/// the key form is `src/foo.rs` rather than `{anchor}/src/foo.rs`. -/// Paths outside the anchor (rare; e.g. a baseline that records -/// files from a sibling crate) keep their absolute form. -/// 3. The resulting `OsStr` is fed through the byte-level -/// percent-encoder for TOML safety (backslash → forward slash, -/// non-unreserved bytes → `%XX`, Windows unpaired surrogates → -/// `%uHHHH`). The encoding is injective **except** for the deliberate -/// `\` → `/` separator fold: a Windows-style `src\foo.rs` and a Unix -/// `src/foo.rs` are *intended* to collapse onto the same key so a -/// baseline written on one platform matches on the other. Apart from -/// that single equivalence, distinct byte sequences produce distinct -/// strings. The fold is applied uniformly across every encoder branch -/// (UTF-8, raw-byte, WTF-16) so the equivalence holds for non-UTF-8 -/// paths carrying a literal `\` byte too (#704) — previously only the -/// UTF-8 fast path folded, so those keyed inconsistently. +/// [`Baseline::match_in_group`] returns a lone record's value +/// unconditionally, so for those entries the line is never read — it +/// only re-renders on every unrelated edit above the function, churning +/// the diff and giving two branches something to conflict over. Members +/// of an ambiguous group keep their line, because there the tolerance +/// band is the only thing that tells them apart. /// -/// Non-UTF-8 paths cannot be represented verbatim in a TOML string -/// (TOML mandates UTF-8). Falling back to `Path::display()` would -/// replace every invalid byte with U+FFFD and collapse distinct paths -/// onto the same key — exactly the lossy identity collision we have to -/// avoid. The per-byte encoder preserves identity by emitting `%XX` for -/// every byte not in the unreserved path set. -fn normalize_path(anchor: &Path, p: &Path) -> String { - // Resolve to an absolute, lexically-normalised PathBuf so the key - // is independent of CWD and the `--paths` form the user passed. - // An already-absolute `p` and an empty anchor both bypass the join - // (empty-anchor is the in-memory-test case; absolute is the - // `--paths "$PWD"` form). - let abs = if p.is_absolute() || anchor.as_os_str().is_empty() { - lexical_normalize(p) - } else { - lexical_normalize(&anchor.join(p)) +/// Requires `entries` sorted so that equal identities are adjacent (the +/// sort in [`from_violations`] guarantees it). +fn clear_unambiguous_start_lines(entries: &mut [BaselineEntry]) { + let same_identity = |a: &BaselineEntry, b: &BaselineEntry| { + a.path == b.path && a.qualified == b.qualified && a.metric == b.metric }; - let stripped = abs.strip_prefix(anchor).unwrap_or(&abs); - encode_os_path(stripped.as_os_str()) -} - -/// Encode an OS string into a TOML-safe key. Routes through the -/// UTF-8 fast path when possible (still per-byte percent-encoded for -/// `%` safety), falls back to the platform-specific byte/WTF-16 -/// encoders for non-UTF-8 paths. See [`normalize_path`] for the -/// injectivity guarantee. -fn encode_os_path(s: &std::ffi::OsStr) -> String { - match s.to_str() { - Some(s) => { - let mut out = String::with_capacity(s.len()); - // The `\` → `/` separator fold lives in - // `push_percent_encoded_byte` so every encoder branch (this - // UTF-8 fast path, the non-UTF-8 byte / WTF-16 fallbacks) - // folds identically — a Windows-style `src\foo.rs` and a - // Unix `src/foo.rs` produce the same baseline key on either - // platform. Previously only this branch folded, so a non-UTF-8 - // path carrying a literal `\` byte keyed differently (#704). - for b in s.bytes() { - push_percent_encoded_byte(&mut out, b); - } - out + for group in entries.chunk_by_mut(same_identity) { + if let [only] = group { + only.start_line = None; } - None => encode_non_utf8_os_str(s), - } -} - -#[cfg(unix)] -fn encode_non_utf8_os_str(s: &std::ffi::OsStr) -> String { - use std::os::unix::ffi::OsStrExt; - percent_encode_path_bytes(s.as_bytes()) -} - -#[cfg(windows)] -fn encode_non_utf8_os_str(s: &std::ffi::OsStr) -> String { - use std::os::windows::ffi::OsStrExt; - percent_encode_wtf16(s.encode_wide()) -} - -#[cfg(not(any(unix, windows)))] -fn encode_non_utf8_os_str(s: &std::ffi::OsStr) -> String { - // Exotic targets (wasm, etc.) where neither `OsStrExt` is available. - // Reuse the per-byte encoder on the lossy UTF-8 form so output is - // still TOML-safe; injectivity is best-effort here because the - // platform itself has already destroyed the original bytes via - // `to_string_lossy`. Prefix with U+FFFD so the key can never collide - // with one produced through the `to_str()` branch above. - let mut out = String::from("\u{FFFD}"); - for &b in s.to_string_lossy().as_bytes() { - push_percent_encoded_byte(&mut out, b); } - out } -/// Percent-encode the raw bytes of a non-UTF-8 path so the result is -/// (1) valid UTF-8 (required by TOML), (2) injective for distinct byte -/// sequences (required to keep baseline identities from collapsing), -/// and (3) human-recognizable for the common case where most bytes are -/// printable ASCII path characters. The unreserved set mirrors the -/// "safe for use in a filename" subset of RFC 3986 unreserved with -/// `/` added (path separator) and `%` excluded (escape introducer). -#[cfg(unix)] -fn percent_encode_path_bytes(bytes: &[u8]) -> String { - let mut out = String::with_capacity(bytes.len()); - for &b in bytes { - push_percent_encoded_byte(&mut out, b); - } - out -} - -/// Append a single byte to `out`, either verbatim (if it falls in the -/// unreserved path set) or as `%XX` (uppercase hex). The `%` byte -/// itself is not unreserved, so the output is unambiguous: every `%` -/// in the result was emitted by this function and is followed by -/// either two hex digits (from this function) or `u` followed by four -/// hex digits (from [`percent_encode_wtf16`]). -/// -/// A backslash byte (`\`, 0x5C) is folded to `/` *before* the -/// unreserved check, so a Windows-style separator keys identically to a -/// Unix one regardless of which encoder branch produced the byte. This -/// is the single fold point shared by the UTF-8, raw-byte, and WTF-16 -/// encoders — keeping them consistent was the #704 fix. The fold is -/// deliberately non-injective (`\` and `/` collapse to one key); the -/// `normalize_path` doc records that exception. -fn push_percent_encoded_byte(out: &mut String, b: u8) { - use std::fmt::Write; - let b = if b == b'\\' { b'/' } else { b }; - let is_unreserved = b.is_ascii_alphanumeric() - || matches!( - b, - b'-' | b'_' | b'.' | b'~' | b'/' | b':' | b'+' | b',' | b' ' - ); - if is_unreserved { - out.push(b as char); - } else { - // Writing to a String can only fail on allocation failure, which - // already panics in the standard library. - let _ = write!(out, "%{b:02X}"); - } -} - -/// Percent-encode a WTF-16 code-unit sequence into a TOML-safe UTF-8 -/// string. Valid scalar values are encoded as their UTF-8 bytes through -/// [`push_percent_encoded_byte`]; unpaired surrogates are emitted as -/// `%uHHHH` (uppercase 4-digit hex), a form the byte encoder never -/// produces. The result is: -/// -/// 1. **Injective**: every code unit maps to a distinct token (either -/// a sequence of `%XX` byte escapes / unreserved bytes for a paired -/// scalar, or one `%uHHHH` for an unpaired surrogate). Two distinct -/// WTF-16 sequences therefore always produce distinct strings. -/// 2. **Stable**: deterministic; no allocation order or hashing -/// influences output. -/// 3. **Human-debuggable enough**: ASCII path components survive -/// unchanged. -/// -/// Exposed at `pub(crate)` purely so the unit tests can drive it with -/// synthetic input on any platform (the production caller is -/// `#[cfg(windows)]` only). -#[cfg(any(windows, test))] -pub(crate) fn percent_encode_wtf16(units: impl IntoIterator) -> String { - use std::fmt::Write; - - let mut out = String::new(); - let mut buf = [0u8; 4]; - for r in char::decode_utf16(units) { - match r { - Ok(c) => { - for &b in c.encode_utf8(&mut buf).as_bytes() { - push_percent_encoded_byte(&mut out, b); - } - } - Err(e) => { - let _ = write!(out, "%u{:04X}", e.unpaired_surrogate()); - } - } - } - out +/// Render a `BaselineFile` to a TOML string with the standard comment +/// header prepended. Output is deterministic byte-for-byte for the +/// same input (TOML's f64 formatter is round-trip stable, struct field +/// order is fixed by `#[derive(Serialize)]` declaration order). +pub(crate) fn render(file: &BaselineFile) -> Result { + let body = toml::to_string(file)?; + Ok(format!("{HEADER}{body}")) } #[cfg(test)] diff --git a/big-code-analysis-cli/src/baseline/body_hash.rs b/big-code-analysis-cli/src/baseline/body_hash.rs new file mode 100644 index 000000000..82e2370e4 --- /dev/null +++ b/big-code-analysis-cli/src/baseline/body_hash.rs @@ -0,0 +1,141 @@ +//! Body hashing for the baseline's rename-tolerant fuzzy match (issue +//! #377): a digest of a normalised view of a function body that survives +//! reformatting and a rename of the function itself, plus the hex codec +//! that persists it in the baseline TOML. +//! +//! [`hash_body`] documents what the normalisation deliberately ignores +//! and why FNV-1a rather than the std hasher. + +/// The bare (innermost) name of a qualified symbol: the segment after +/// the last `::`, or the whole string when there is no separator. Used +/// to match a v4 violation's qualified symbol (`MyStruct::do_thing`) +/// against a legacy v2/v3 baseline entry that stored only `do_thing`, +/// and to elide a function's own name from its body hash. +pub(crate) fn bare_name(qualified: &str) -> &str { + qualified + .rsplit_once("::") + .map_or(qualified, |(_, tail)| tail) +} + +/// Encode a body hash as the lowercase, zero-padded 16-digit hex form +/// stored in the TOML. Hex (not a TOML integer) because FNV-1a fills +/// the full `u64` range and TOML integers are `i64`. +pub(super) fn encode_body_hash(h: u64) -> String { + format!("{h:016x}") +} + +/// Decode a stored body hash. Returns `None` for any malformed digest +/// (wrong length, non-hex) so a hand-edited file degrades the entry to +/// "no fuzzy fallback" rather than aborting the whole load. +pub(super) fn decode_body_hash(s: &str) -> Option { + (s.len() == 16) + .then(|| u64::from_str_radix(s, 16).ok()) + .flatten() +} + +/// FNV-1a offset basis and prime for the 64-bit variant. FNV is chosen +/// over the std `DefaultHasher` because the digest is persisted to disk +/// and must be byte-stable across bca versions, platforms, and process +/// runs — `DefaultHasher`'s algorithm and seed carry no such guarantee. +const FNV_OFFSET_BASIS: u64 = 0xcbf2_9ce4_8422_2325; +const FNV_PRIME: u64 = 0x0000_0100_0000_01b3; + +/// Hash the normalised body of the space spanning `start_line..=end_line` +/// (1-based, inclusive) within `source`, with the function's own `name` +/// elided. This is the "fuzzy" in fuzzy matching: a digest of a view of +/// the body that survives the two changes a rename-or-relocate refactor +/// makes (issue #377, rule 3) while still distinguishing genuinely +/// different code: +/// +/// 1. **Whitespace**: `\r` is dropped, each line's internal whitespace +/// runs collapse to one space, leading/trailing whitespace is +/// trimmed, and blank lines are skipped — so reformatting, +/// re-indentation, or blank-line churn does not change the digest. +/// 2. **The function's own name**: every whole-word occurrence of `name` +/// (the declaration and any recursive self-calls) is replaced with a +/// fixed sentinel, so renaming `classify` to `categorize` leaves the +/// digest unchanged. Whole-word means bounded by non-identifier bytes, +/// so a `name` of `is` does not corrupt `is_valid`. +/// +/// Out-of-range lines are clamped (a `start_line` past EOF yields the +/// empty-body digest), so a malformed span never panics. +pub(crate) fn hash_body(source: &[u8], start_line: usize, end_line: usize, name: &str) -> u64 { + let normalized = normalize_body(source, start_line, end_line); + let elided = elide_identifier(&normalized, name.as_bytes()); + let mut hash = FNV_OFFSET_BASIS; + for &b in &elided { + hash ^= u64::from(b); + hash = hash.wrapping_mul(FNV_PRIME); + } + hash +} + +/// Build the whitespace-normalised byte view of a line span (step 1 of +/// [`hash_body`]). Lines are joined by `\n`; blank lines are dropped. +fn normalize_body(source: &[u8], start_line: usize, end_line: usize) -> Vec { + let first = start_line.saturating_sub(1); + let mut out = Vec::new(); + for line in source + .split(|&b| b == b'\n') + .skip(first) + .take(end_line.saturating_sub(first)) + { + let mut line_started = false; + let mut pending_space = false; + for &b in line { + if b == b'\r' { + continue; + } + if b.is_ascii_whitespace() { + pending_space = true; + continue; + } + if line_started && pending_space { + out.push(b' '); + } + pending_space = false; + out.push(b); + line_started = true; + } + if line_started { + out.push(b'\n'); + } + } + out +} + +/// Sentinel emitted in place of the elided function name. `\x00` does not +/// appear in normalised source bytes, so it cannot collide with real +/// content. +const ELIDED_NAME_SENTINEL: u8 = 0x00; + +/// Replace every whole-word occurrence of `name` in `haystack` with +/// [`ELIDED_NAME_SENTINEL`] (step 2 of [`hash_body`]). Whole-word means +/// neither neighbour is an identifier byte. An empty or non-identifier +/// `name` (e.g. the `` / `` sentinels) is left as-is. +fn elide_identifier(haystack: &[u8], name: &[u8]) -> Vec { + let is_ident = |b: u8| b.is_ascii_alphanumeric() || b == b'_'; + if name.is_empty() || !name.iter().copied().all(is_ident) { + return haystack.to_vec(); + } + let mut out = Vec::with_capacity(haystack.len()); + let mut i = 0; + while i < haystack.len() { + let left_ok = i == 0 || !is_ident(haystack[i - 1]); + let matches = left_ok + && haystack[i..].starts_with(name) + && haystack.get(i + name.len()).is_none_or(|&b| !is_ident(b)); + if matches { + out.push(ELIDED_NAME_SENTINEL); + i += name.len(); + } else { + out.push(haystack[i]); + i += 1; + } + } + out +} + +#[cfg(test)] +#[path = "body_hash_tests.rs"] +mod tests; diff --git a/big-code-analysis-cli/src/baseline/body_hash_tests.rs b/big-code-analysis-cli/src/baseline/body_hash_tests.rs new file mode 100644 index 000000000..5152dff92 --- /dev/null +++ b/big-code-analysis-cli/src/baseline/body_hash_tests.rs @@ -0,0 +1,94 @@ +// Sibling-file unit tests for `bare_name` and the body-hash helpers, +// wired in via `#[path = "body_hash_tests.rs"] mod tests;` so the +// production `body_hash.rs` stays under the `bca check` per-file +// metric caps. Matched by the `./**/*_tests.rs` rule in `.bcaignore`, +// so the self-scan walker skips this file the same way it skips +// `./tests/`. + +use super::*; + +#[test] +fn bare_name_strips_qualifier() { + assert_eq!(bare_name("MyStruct::do_thing"), "do_thing"); + assert_eq!(bare_name("a::b::c"), "c"); + assert_eq!(bare_name("plain"), "plain"); + assert_eq!(bare_name(""), ""); +} + +#[test] +fn body_hash_ignores_indentation_blank_lines_and_run_width() { + // The normalisation trims leading/trailing whitespace, collapses + // internal whitespace *runs* to one space, drops `\r`, and skips + // blank lines — so re-indenting, reflowing blank lines, or changing + // CRLF/LF must not change the digest. (It is not insensitive to the + // presence/absence of whitespace *between* tokens, only its width.) + let original = b" let x = 1;\n return x + 1;\n"; + let reformatted = b"\nlet x = 1;\r\n\n return x + 1;\n\n"; + assert_eq!( + hash_body(original, 1, 2, ""), + hash_body(reformatted, 1, 6, "") + ); +} + +#[test] +fn body_hash_distinguishes_different_bodies() { + assert_ne!( + hash_body(b"let x = 1;", 1, 1, ""), + hash_body(b"let x = 2;", 1, 1, "") + ); +} + +#[test] +fn body_hash_respects_line_range() { + // Lines 2..=2 of a three-line body hash only the middle line. + let src = b"line one\nline two\nline three\n"; + assert_eq!(hash_body(src, 2, 2, ""), hash_body(b"line two", 1, 1, "")); +} + +#[test] +fn body_hash_out_of_range_is_empty_digest() { + // A start past EOF yields the empty-body digest (the FNV offset + // basis) rather than panicking on an out-of-bounds slice. + let src = b"only one line"; + assert_eq!(hash_body(src, 100, 200, ""), hash_body(b"", 1, 1, "")); +} + +#[test] +fn body_hash_elides_own_name_so_rename_matches() { + // The headline rule-3 property: renaming the function (declaration + // and recursive self-calls) leaves the digest unchanged, because the + // bare name is elided. + let before = b"fn classify(n: i32) -> i32 { classify(n - 1) }"; + let after = b"fn categorize(n: i32) -> i32 { categorize(n - 1) }"; + assert_eq!( + hash_body(before, 1, 1, "classify"), + hash_body(after, 1, 1, "categorize") + ); +} + +#[test] +fn body_hash_elision_is_whole_word_only() { + // Eliding `is` must not corrupt the substring inside `is_valid` — + // two bodies that differ only in an unrelated identifier sharing the + // elided prefix must still hash differently. + let a = b"fn is() { is_valid() }"; + let b = b"fn is() { is_ready() }"; + assert_ne!(hash_body(a, 1, 1, "is"), hash_body(b, 1, 1, "is")); +} + +#[test] +fn body_hash_round_trips_through_hex_codec() { + let h = hash_body(b"some body text", 1, 1, ""); + assert_eq!(decode_body_hash(&encode_body_hash(h)), Some(h)); +} + +#[test] +fn decode_body_hash_rejects_malformed() { + assert_eq!(decode_body_hash("not-hex"), None); + assert_eq!(decode_body_hash("dead"), None); // too short + assert_eq!(decode_body_hash(""), None); + assert_eq!( + decode_body_hash("0123456789abcdef"), + Some(0x0123_4567_89ab_cdef) + ); +} diff --git a/big-code-analysis-cli/src/baseline/path_key.rs b/big-code-analysis-cli/src/baseline/path_key.rs new file mode 100644 index 000000000..0016890db --- /dev/null +++ b/big-code-analysis-cli/src/baseline/path_key.rs @@ -0,0 +1,263 @@ +//! Path canonicalisation and percent-encoding for baseline identity +//! keys. +//! +//! A key is produced by resolving a path against the baseline file's +//! *anchor* ([`anchor_for`]), folding `.`/`..` lexically, stripping the +//! anchor prefix, and percent-encoding the remaining `OsStr` bytes so +//! the result is valid UTF-8 for TOML without collapsing distinct paths +//! onto one key. [`normalize_path`] documents the pipeline and its one +//! deliberate non-injectivity (the `\` → `/` separator fold). + +use std::path::{Component, Path, PathBuf}; + +/// Derive the canonical anchor for a baseline file at `baseline_path`: +/// the lexically-absolute directory the file lives in. Used at both +/// write and read time so the key shape is independent of `--paths` +/// form, working directory drift, or whether the path was passed +/// relative or absolute. +/// +/// Lexical, not symlink-following: `bca` should not surprise users by +/// resolving `src/` through a symlinked directory. The cost is that a +/// baseline written via a symlinked invocation and read via the real +/// path (or vice-versa) does not match — but that mirrors how every +/// other tool (cargo, git) treats workdir identity. +pub(crate) fn anchor_for(baseline_path: &Path) -> PathBuf { + // `std::path::absolute` only fails when the path is empty or the + // platform cannot obtain the CWD (effectively never in a normal + // shell). Fall back to the input path so the rest of the pipeline + // degrades to pre-#376 behaviour (no canonicalisation) instead of + // dying — the worst case is the path-stickiness this fix targets, + // not a hard error. + let abs = std::path::absolute(baseline_path).unwrap_or_else(|_| baseline_path.to_path_buf()); + let mut abs = lexical_normalize(&abs); + // `pop` returns false if the path is already a root or empty; in + // that degenerate case the path itself becomes the anchor. + abs.pop(); + abs +} + +/// Lexically normalise `p` by folding `.` and `..` components without +/// touching the filesystem. POSIX-style folding: +/// - `..` after a `Normal` component pops it (`a/b/../c` → `a/c`). +/// - `..` immediately after a `RootDir` or Windows `Prefix` is a no-op +/// (`/..` → `/`, `C:\..` → `C:\`) — you cannot go above the root. +/// - `..` with no prior component to consume is preserved literally +/// (`../a` → `../a`, `a/../../b` → `../b`). This keeps identity for +/// baselines that legitimately reference a sibling of the anchor. +pub(super) fn lexical_normalize(p: &Path) -> PathBuf { + let mut out = PathBuf::new(); + for c in p.components() { + match c { + Component::Prefix(_) | Component::RootDir | Component::Normal(_) => out.push(c), + Component::CurDir => {} + Component::ParentDir => match out.components().next_back() { + // Pop the previous Normal component (typical case). + Some(Component::Normal(_)) => { + out.pop(); + } + // POSIX/Windows: `..` past a root or drive prefix is + // a no-op. `/..` resolves to `/`, not `/..`. Without + // this case the normalised path would be non-canonical + // and downstream `strip_prefix` would mis-match. + Some(Component::RootDir | Component::Prefix(_)) => {} + // No prior Normal/Root/Prefix component (e.g., + // relative path starting with `..` or accumulating + // multiple leading `..`s). Preserve the `..` literally + // so a baseline that legitimately points at a sibling + // of the anchor keeps a distinct identity. + _ => out.push(c), + }, + } + } + out +} + +/// Normalize a path for use as a baseline identity key. +/// +/// 1. The path is made lexically absolute (using `anchor` as the base +/// directory when relative) and `.` / `..` components are folded. +/// 2. If the result is under `anchor`, the anchor prefix is stripped so +/// the key form is `src/foo.rs` rather than `{anchor}/src/foo.rs`. +/// Paths outside the anchor (rare; e.g. a baseline that records +/// files from a sibling crate) keep their absolute form. +/// 3. The resulting `OsStr` is fed through the byte-level +/// percent-encoder for TOML safety (backslash → forward slash, +/// non-unreserved bytes → `%XX`, Windows unpaired surrogates → +/// `%uHHHH`). The encoding is injective **except** for the deliberate +/// `\` → `/` separator fold: a Windows-style `src\foo.rs` and a Unix +/// `src/foo.rs` are *intended* to collapse onto the same key so a +/// baseline written on one platform matches on the other. Apart from +/// that single equivalence, distinct byte sequences produce distinct +/// strings. The fold is applied uniformly across every encoder branch +/// (UTF-8, raw-byte, WTF-16) so the equivalence holds for non-UTF-8 +/// paths carrying a literal `\` byte too (#704) — previously only the +/// UTF-8 fast path folded, so those keyed inconsistently. +/// +/// Non-UTF-8 paths cannot be represented verbatim in a TOML string +/// (TOML mandates UTF-8). Falling back to `Path::display()` would +/// replace every invalid byte with U+FFFD and collapse distinct paths +/// onto the same key — exactly the lossy identity collision we have to +/// avoid. The per-byte encoder preserves identity by emitting `%XX` for +/// every byte not in the unreserved path set. +pub(super) fn normalize_path(anchor: &Path, p: &Path) -> String { + // Resolve to an absolute, lexically-normalised PathBuf so the key + // is independent of CWD and the `--paths` form the user passed. + // An already-absolute `p` and an empty anchor both bypass the join + // (empty-anchor is the in-memory-test case; absolute is the + // `--paths "$PWD"` form). + let abs = if p.is_absolute() || anchor.as_os_str().is_empty() { + lexical_normalize(p) + } else { + lexical_normalize(&anchor.join(p)) + }; + let stripped = abs.strip_prefix(anchor).unwrap_or(&abs); + encode_os_path(stripped.as_os_str()) +} + +/// Encode an OS string into a TOML-safe key. Routes through the +/// UTF-8 fast path when possible (still per-byte percent-encoded for +/// `%` safety), falls back to the platform-specific byte/WTF-16 +/// encoders for non-UTF-8 paths. See [`normalize_path`] for the +/// injectivity guarantee. +fn encode_os_path(s: &std::ffi::OsStr) -> String { + match s.to_str() { + Some(s) => { + let mut out = String::with_capacity(s.len()); + // The `\` → `/` separator fold lives in + // `push_percent_encoded_byte` so every encoder branch (this + // UTF-8 fast path, the non-UTF-8 byte / WTF-16 fallbacks) + // folds identically — a Windows-style `src\foo.rs` and a + // Unix `src/foo.rs` produce the same baseline key on either + // platform. Previously only this branch folded, so a non-UTF-8 + // path carrying a literal `\` byte keyed differently (#704). + for b in s.bytes() { + push_percent_encoded_byte(&mut out, b); + } + out + } + None => encode_non_utf8_os_str(s), + } +} + +#[cfg(unix)] +fn encode_non_utf8_os_str(s: &std::ffi::OsStr) -> String { + use std::os::unix::ffi::OsStrExt; + percent_encode_path_bytes(s.as_bytes()) +} + +#[cfg(windows)] +fn encode_non_utf8_os_str(s: &std::ffi::OsStr) -> String { + use std::os::windows::ffi::OsStrExt; + percent_encode_wtf16(s.encode_wide()) +} + +#[cfg(not(any(unix, windows)))] +fn encode_non_utf8_os_str(s: &std::ffi::OsStr) -> String { + // Exotic targets (wasm, etc.) where neither `OsStrExt` is available. + // Reuse the per-byte encoder on the lossy UTF-8 form so output is + // still TOML-safe; injectivity is best-effort here because the + // platform itself has already destroyed the original bytes via + // `to_string_lossy`. Prefix with U+FFFD so the key can never collide + // with one produced through the `to_str()` branch above. + // + // This is the documented exception to `AGENTS.md`'s blanket ban on + // `to_string_lossy()` for identifier paths: the two branches that + // can recover the real bytes are `#[cfg]`-ed out on this target, so + // there is nothing lossless left to call, and the U+FFFD prefix + // confines the lossiness to a key space of its own. + let mut out = String::from("\u{FFFD}"); + for &b in s.to_string_lossy().as_bytes() { + push_percent_encoded_byte(&mut out, b); + } + out +} + +/// Percent-encode the raw bytes of a non-UTF-8 path so the result is +/// (1) valid UTF-8 (required by TOML), (2) injective for distinct byte +/// sequences (required to keep baseline identities from collapsing), +/// and (3) human-recognizable for the common case where most bytes are +/// printable ASCII path characters. The unreserved set mirrors the +/// "safe for use in a filename" subset of RFC 3986 unreserved with +/// `/` added (path separator) and `%` excluded (escape introducer). +#[cfg(unix)] +fn percent_encode_path_bytes(bytes: &[u8]) -> String { + let mut out = String::with_capacity(bytes.len()); + for &b in bytes { + push_percent_encoded_byte(&mut out, b); + } + out +} + +/// Append a single byte to `out`, either verbatim (if it falls in the +/// unreserved path set) or as `%XX` (uppercase hex). The `%` byte +/// itself is not unreserved, so the output is unambiguous: every `%` +/// in the result was emitted by this function and is followed by +/// either two hex digits (from this function) or `u` followed by four +/// hex digits (from [`percent_encode_wtf16`]). +/// +/// A backslash byte (`\`, 0x5C) is folded to `/` *before* the +/// unreserved check, so a Windows-style separator keys identically to a +/// Unix one regardless of which encoder branch produced the byte. This +/// is the single fold point shared by the UTF-8, raw-byte, and WTF-16 +/// encoders — keeping them consistent was the #704 fix. The fold is +/// deliberately non-injective (`\` and `/` collapse to one key); the +/// `normalize_path` doc records that exception. +fn push_percent_encoded_byte(out: &mut String, b: u8) { + use std::fmt::Write; + let b = if b == b'\\' { b'/' } else { b }; + let is_unreserved = b.is_ascii_alphanumeric() + || matches!( + b, + b'-' | b'_' | b'.' | b'~' | b'/' | b':' | b'+' | b',' | b' ' + ); + if is_unreserved { + out.push(b as char); + } else { + // Writing to a String can only fail on allocation failure, which + // already panics in the standard library. + let _ = write!(out, "%{b:02X}"); + } +} + +/// Percent-encode a WTF-16 code-unit sequence into a TOML-safe UTF-8 +/// string. Valid scalar values are encoded as their UTF-8 bytes through +/// [`push_percent_encoded_byte`]; unpaired surrogates are emitted as +/// `%uHHHH` (uppercase 4-digit hex), a form the byte encoder never +/// produces. The result is: +/// +/// 1. **Injective**: every code unit maps to a distinct token (either +/// a sequence of `%XX` byte escapes / unreserved bytes for a paired +/// scalar, or one `%uHHHH` for an unpaired surrogate). Two distinct +/// WTF-16 sequences therefore always produce distinct strings. +/// 2. **Stable**: deterministic; no allocation order or hashing +/// influences output. +/// 3. **Human-debuggable enough**: ASCII path components survive +/// unchanged. +/// +/// Compiled under `test` as well as `windows` purely so the unit tests +/// can drive it with synthetic input on any platform (the production +/// caller is `#[cfg(windows)]` only). +#[cfg(any(windows, test))] +fn percent_encode_wtf16(units: impl IntoIterator) -> String { + use std::fmt::Write; + + let mut out = String::new(); + let mut buf = [0u8; 4]; + for r in char::decode_utf16(units) { + match r { + Ok(c) => { + for &b in c.encode_utf8(&mut buf).as_bytes() { + push_percent_encoded_byte(&mut out, b); + } + } + Err(e) => { + let _ = write!(out, "%u{:04X}", e.unpaired_surrogate()); + } + } + } + out +} + +#[cfg(test)] +#[path = "path_key_tests.rs"] +mod tests; diff --git a/big-code-analysis-cli/src/baseline/path_key_tests.rs b/big-code-analysis-cli/src/baseline/path_key_tests.rs new file mode 100644 index 000000000..f8a309589 --- /dev/null +++ b/big-code-analysis-cli/src/baseline/path_key_tests.rs @@ -0,0 +1,312 @@ +// Sibling-file unit tests for baseline path-key canonicalisation and +// percent-encoding, wired in via `#[path = "path_key_tests.rs"] mod +// tests;` so the production `path_key.rs` stays under the `bca check` +// per-file metric caps. Matched by the `./**/*_tests.rs` rule in +// `.bcaignore`, so the self-scan walker skips this file the same way +// it skips `./tests/`. + +use super::*; + +/// Canonical empty anchor for unit tests: the violation path is keyed +/// as-passed without prepending a synthetic CWD. Real callers always +/// derive their anchor via [`anchor_for`] from the baseline file path, +/// but for the in-memory tests in this file an empty anchor preserves +/// the pre-#376 semantics of "key on the literal path string the test +/// supplied" while still exercising the new lexical normalisation. +fn test_anchor() -> &'static Path { + Path::new("") +} + +// -- anchor + lexical normalisation (issue #376) ---------------------- + +#[test] +fn lexical_normalize_folds_curdir_and_parent() { + assert_eq!(lexical_normalize(Path::new("./a/b")), Path::new("a/b")); + assert_eq!(lexical_normalize(Path::new("a/./b")), Path::new("a/b")); + assert_eq!(lexical_normalize(Path::new("a/b/../c")), Path::new("a/c")); + assert_eq!( + lexical_normalize(Path::new("a/b/c/../../d")), + Path::new("a/d") + ); +} + +#[test] +fn lexical_normalize_preserves_escaping_parents() { + // `..` past every accumulated Normal component is preserved so + // an entry that genuinely lives one level above the anchor + // (e.g., a sibling-crate analysis) still has an identity. + assert_eq!(lexical_normalize(Path::new("../a")), Path::new("../a")); + assert_eq!(lexical_normalize(Path::new("a/../../b")), Path::new("../b")); +} + +#[cfg(unix)] +#[test] +fn lexical_normalize_folds_parent_past_root() { + // POSIX: `..` immediately after a RootDir is a no-op. Before the + // fix the function preserved `..` literally, yielding `/..` — + // non-canonical and `strip_prefix` would not match a canonical + // anchor. A hand-crafted v2 entry like `path = "/../etc/passwd"` + // could exploit this to produce keys that bypass anchor + // relativisation; the fold keeps the encoder's output canonical. + assert_eq!(lexical_normalize(Path::new("/..")), Path::new("/")); + assert_eq!(lexical_normalize(Path::new("/../..")), Path::new("/")); + assert_eq!( + lexical_normalize(Path::new("/../etc/passwd")), + Path::new("/etc/passwd") + ); + // Mixing: Normal pop still works after a root-fold no-op. + assert_eq!( + lexical_normalize(Path::new("/foo/../../bar")), + Path::new("/bar") + ); +} + +#[cfg(unix)] +#[test] +fn anchor_for_strips_baseline_filename() { + // `anchor_for` is lexical-only — no filesystem access — so the + // assertion can be a pure path comparison against synthetic + // input. Pinning to a fixed prefix keeps the test independent + // of `$TMPDIR` shape across CI hosts. + assert_eq!( + anchor_for(Path::new("/tmp/bca-anchor-test/baseline.toml")), + Path::new("/tmp/bca-anchor-test"), + ); +} + +#[cfg(unix)] +#[test] +fn normalize_path_canonicalises_against_anchor() { + // Three distinct typings of the same file under one anchor must + // collapse to the same key. + let anchor = Path::new("/repo"); + let key_dot = normalize_path(anchor, Path::new("/repo/src/foo.rs")); + let key_rel = normalize_path(anchor, Path::new("src/./foo.rs")); + let key_parent = normalize_path(anchor, Path::new("src/x/../foo.rs")); + assert_eq!(key_dot, "src/foo.rs"); + assert_eq!(key_rel, "src/foo.rs"); + assert_eq!(key_parent, "src/foo.rs"); +} + +#[cfg(unix)] +#[test] +fn normalize_path_outside_anchor_uses_absolute_form() { + // A path that isn't under the anchor keeps its absolute form + // rather than degrading to `../` chains. Legitimate use case: + // a baseline at the repo root recording offenders from a + // sibling vendored crate kept outside the tree. + let key = normalize_path(Path::new("/repo"), Path::new("/elsewhere/file.rs")); + assert_eq!(key, "/elsewhere/file.rs"); +} + +// -- non-UTF-8 path identity ------------------------------------------ + +#[test] +fn normalize_path_utf8_unchanged_for_unreserved_ascii() { + // Regression guard: the common UTF-8 case (all-unreserved-ASCII + // path components) must round-trip untouched. Non-UTF-8 + // encoding shenanigans must not leak into ordinary inputs (no + // unexpected percent escapes, no extra markers). + assert_eq!( + normalize_path(test_anchor(), Path::new("src/foo.rs")), + "src/foo.rs" + ); + assert_eq!( + normalize_path(test_anchor(), Path::new("crates/a/b.rs")), + "crates/a/b.rs" + ); + // Backslashes are still normalized to forward slashes for the + // UTF-8 path so that cross-OS baselines match. + assert_eq!( + normalize_path(test_anchor(), Path::new("a\\b\\c.rs")), + "a/b/c.rs" + ); +} + +#[test] +fn normalize_path_utf8_escapes_percent() { + // `%` must be escaped in the UTF-8 fast path so it cannot collide + // with a non-UTF-8 byte's `%XX` escape. See `normalize_path_utf8_ + // non_utf8_byte_no_collision` for the actual collision check. + assert_eq!( + normalize_path(test_anchor(), Path::new("foo%FF.rs")), + "foo%25FF.rs" + ); + assert_eq!( + normalize_path(test_anchor(), Path::new("a%b%c.rs")), + "a%25b%25c.rs" + ); +} + +#[cfg(unix)] +#[test] +fn normalize_path_utf8_percent_vs_non_utf8_byte_no_collision() { + // The bug: a UTF-8 path containing the literal text `%FF` and a + // non-UTF-8 path containing the byte `0xFF` at the same position + // used to normalize to the same key (both `foo%FF.rs`), so a + // baseline written for one silently covered violations from the + // other. With `%` percent-encoded on the UTF-8 side, the keys + // diverge. + use std::ffi::OsStr; + use std::os::unix::ffi::OsStrExt; + + let utf8 = Path::new("foo%FF.rs"); + let non_utf8 = PathBuf::from(OsStr::from_bytes(b"foo\xff.rs")); + let key_utf8 = normalize_path(test_anchor(), utf8); + let key_non_utf8 = normalize_path(test_anchor(), &non_utf8); + assert_eq!(key_utf8, "foo%25FF.rs"); + assert_eq!(key_non_utf8, "foo%FF.rs"); + assert_ne!(key_utf8, key_non_utf8); +} + +#[cfg(unix)] +#[test] +fn baseline_key_preserves_non_utf8_identity() { + use std::ffi::OsStr; + use std::os::unix::ffi::OsStrExt; + + // Two distinct non-UTF-8 paths must produce two distinct + // baseline keys. The previous `display().to_string()` fallback + // collapsed both onto a sequence of U+FFFD replacement chars, + // so a baseline written from path A would silently cover + // violations from path B. + let a = PathBuf::from("src").join(OsStr::from_bytes(b"bad-\xff\xfe.rs")); + let b = PathBuf::from("src").join(OsStr::from_bytes(b"bad-\xfe\xff.rs")); + let key_a = normalize_path(test_anchor(), &a); + let key_b = normalize_path(test_anchor(), &b); + assert_ne!(key_a, key_b); + // The encoded keys are valid UTF-8 (required by TOML) and + // contain only ASCII bytes after percent-encoding. + assert!(key_a.is_ascii()); + assert!(key_b.is_ascii()); +} + +// -- WTF-16 percent-encoding (always-on, synthetic input) ------------ + +#[test] +fn wtf16_encode_pure_ascii() { + // ASCII path bytes are unreserved, so they survive unchanged. + let out = percent_encode_wtf16("src/foo.rs".encode_utf16()); + assert_eq!(out, "src/foo.rs"); +} + +#[test] +fn wtf16_encode_empty() { + assert_eq!(percent_encode_wtf16(std::iter::empty::()), ""); +} + +#[test] +fn wtf16_encode_bmp_non_ascii() { + // U+00E9 (é) is BMP; UTF-8 = 0xC3 0xA9; both bytes are + // non-unreserved and percent-encode to %C3%A9. + let out = percent_encode_wtf16("é".encode_utf16()); + assert_eq!(out, "%C3%A9"); +} + +#[test] +fn wtf16_encode_supplementary_plane() { + // U+1F600 (😀) requires a surrogate pair in WTF-16 + // (0xD83D, 0xDE00) and UTF-8-encodes as 0xF0 0x9F 0x98 0x80. + // `char::decode_utf16` pairs the surrogates back to the scalar, + // so the encoder must emit the UTF-8 byte form. + let units = [0xD83D_u16, 0xDE00_u16]; + let out = percent_encode_wtf16(units); + assert_eq!(out, "%F0%9F%98%80"); + // Sanity: the same character entered as a string round-trips + // identically through `encode_utf16`. + assert_eq!(out, percent_encode_wtf16("😀".encode_utf16())); +} + +#[test] +fn wtf16_encode_unpaired_high_surrogate() { + let out = percent_encode_wtf16([0xD83D_u16]); + assert_eq!(out, "%uD83D"); +} + +#[test] +fn wtf16_encode_unpaired_low_surrogate() { + // A lone low surrogate (no preceding high) is unpaired. + let out = percent_encode_wtf16([0xDE00_u16]); + assert_eq!(out, "%uDE00"); +} + +#[test] +fn wtf16_encode_high_followed_by_non_low_is_unpaired() { + // High surrogate followed by ASCII: the high is unpaired and + // the ASCII byte is encoded normally afterwards. + let units = [0xD83D_u16, u16::from(b'x')]; + let out = percent_encode_wtf16(units); + assert_eq!(out, "%uD83Dx"); +} + +#[test] +fn wtf16_encode_leading_low_then_pair() { + // A lone low surrogate followed by a real pair: the leading low + // must not consume the next code unit (the high of the pair). + let units = [0xDC00_u16, 0xD83D_u16, 0xDE00_u16]; + let out = percent_encode_wtf16(units); + assert_eq!(out, "%uDC00%F0%9F%98%80"); +} + +#[test] +fn wtf16_encode_distinct_unpaired_surrogates_do_not_collide() { + // The whole point of the fix: two distinct invalid WTF-16 + // sequences that `to_string_lossy()` would have collapsed onto + // a single U+FFFD must produce two distinct encoded keys. + let a = percent_encode_wtf16([0xD83D_u16]); + let b = percent_encode_wtf16([0xDE00_u16]); + assert_ne!(a, b); + // And two different lone high surrogates also separate cleanly. + let c = percent_encode_wtf16([0xD800_u16]); + let d = percent_encode_wtf16([0xDBFF_u16]); + assert_ne!(c, d); +} + +#[test] +fn wtf16_encode_marker_never_emitted_by_scalar_bytes() { + // Regression guard: the byte encoder only emits `%` followed by + // exactly two uppercase hex digits, never `%u`. Scalars cannot + // produce a string that begins with `%u` from their UTF-8 bytes + // — `u` is unreserved, so it stays as `u`, but the preceding + // `%` only appears when a non-unreserved byte is escaped (and + // is then immediately followed by two hex digits, not `u`). + // Therefore parsing `%u…` is unambiguous. + for codepoint in ['u', '%', '!', '\u{00E9}', '\u{1F600}'] { + let s = codepoint.to_string(); + let out = percent_encode_wtf16(s.encode_utf16()); + assert!(!out.contains("%u"), "scalar {codepoint:?} produced {out:?}"); + } +} + +#[cfg(windows)] +#[test] +fn baseline_key_preserves_non_utf16_identity_on_windows() { + use std::ffi::OsString; + use std::os::windows::ffi::OsStringExt; + + // Two distinct paths that differ only by an unpaired surrogate + // value would collapse to the same `to_string_lossy()` key + // (both surrogates become U+FFFD). With the WTF-16 encoder they + // stay distinct. + let a_units: [u16; 5] = [ + u16::from(b'a'), + u16::from(b'/'), + 0xD83D, + u16::from(b'.'), + u16::from(b's'), + ]; + let b_units: [u16; 5] = [ + u16::from(b'a'), + u16::from(b'/'), + 0xDE00, + u16::from(b'.'), + u16::from(b's'), + ]; + let path_a = PathBuf::from(OsString::from_wide(&a_units)); + let path_b = PathBuf::from(OsString::from_wide(&b_units)); + let key_a = normalize_path(test_anchor(), &path_a); + let key_b = normalize_path(test_anchor(), &path_b); + assert_ne!(key_a, key_b); + assert!(key_a.is_ascii()); + assert!(key_b.is_ascii()); +} diff --git a/big-code-analysis-cli/src/baseline_diff.rs b/big-code-analysis-cli/src/baseline_diff.rs index 81d87b316..9e7df4f8b 100644 --- a/big-code-analysis-cli/src/baseline_diff.rs +++ b/big-code-analysis-cli/src/baseline_diff.rs @@ -14,7 +14,8 @@ //! matcher from issue #377): a function that drifts up or down the file //! is the *same* entry, not an add+remove pair. When a single //! `(path, qualified, metric)` triple carries several records (genuinely -//! ambiguous symbols — overloads, duplicate `impl` blocks), the records +//! ambiguous symbols — overloads, duplicate `impl` blocks — the only +//! case a v6+ baseline records a `start_line` for at all), the records //! on each side are sorted by `(value, start_line)` and paired //! positionally; any surplus becomes added/removed. This is a //! best-effort heuristic for an inherently ambiguous case, deterministic @@ -26,7 +27,7 @@ use std::fmt::Write as _; use serde::Serialize; -use crate::baseline::DiffEntry; +use crate::baseline::{BaselineIdentity, DiffEntry, cmp_identity}; use crate::format_util::{MetricScalar, strip_path_prefix}; /// An entry present in exactly one of the two baselines (`added` / @@ -36,7 +37,11 @@ pub(crate) struct EntryDelta { pub(crate) path: String, pub(crate) qualified: String, pub(crate) metric: String, - pub(crate) start_line: usize, + /// Recorded line, absent whenever the baseline did not pin one — as + /// v6+ files do not for a unique identity (#1170). Omitted from the + /// JSON rather than rendered as a placeholder line number. + #[serde(skip_serializing_if = "Option::is_none")] + pub(crate) start_line: Option, pub(crate) value: f64, } @@ -50,7 +55,10 @@ pub(crate) struct ValueDelta { pub(crate) path: String, pub(crate) qualified: String, pub(crate) metric: String, - pub(crate) start_line: usize, + /// The *new* baseline's recorded line, absent when it pinned none + /// (see [`EntryDelta::start_line`]). + #[serde(skip_serializing_if = "Option::is_none")] + pub(crate) start_line: Option, pub(crate) old: f64, pub(crate) new: f64, } @@ -165,10 +173,10 @@ impl BaselineDiff { added.extend(news[paired..].iter().map(|e| entry_delta(e))); } - added.sort_by(cmp_entry_delta); - removed.sort_by(cmp_entry_delta); - worsened.sort_by(cmp_value_delta); - improved.sort_by(cmp_value_delta); + added.sort_by(cmp_identity); + removed.sort_by(cmp_identity); + worsened.sort_by(cmp_identity); + improved.sort_by(cmp_identity); Self { added, @@ -430,22 +438,16 @@ fn cmp_for_pairing(a: &DiffEntry, b: &DiffEntry) -> Ordering { .then(a.start_line.cmp(&b.start_line)) } -fn cmp_entry_delta(a: &EntryDelta, b: &EntryDelta) -> Ordering { - (&a.path, &a.qualified, &a.metric, a.start_line).cmp(&( - &b.path, - &b.qualified, - &b.metric, - b.start_line, - )) +impl BaselineIdentity for EntryDelta { + fn identity(&self) -> (&str, &str, &str, Option) { + (&self.path, &self.qualified, &self.metric, self.start_line) + } } -fn cmp_value_delta(a: &ValueDelta, b: &ValueDelta) -> Ordering { - (&a.path, &a.qualified, &a.metric, a.start_line).cmp(&( - &b.path, - &b.qualified, - &b.metric, - b.start_line, - )) +impl BaselineIdentity for ValueDelta { + fn identity(&self) -> (&str, &str, &str, Option) { + (&self.path, &self.qualified, &self.metric, self.start_line) + } } #[cfg(test)] diff --git a/big-code-analysis-cli/src/baseline_diff_tests.rs b/big-code-analysis-cli/src/baseline_diff_tests.rs index 2406c90e0..b0a13c1e3 100644 --- a/big-code-analysis-cli/src/baseline_diff_tests.rs +++ b/big-code-analysis-cli/src/baseline_diff_tests.rs @@ -9,7 +9,7 @@ fn entry(path: &str, qualified: &str, metric: &str, start_line: usize, value: f6 path: path.to_string(), qualified: qualified.to_string(), metric: metric.to_string(), - start_line, + start_line: Some(start_line), value, } } @@ -336,6 +336,39 @@ fn json_emits_all_buckets_with_summary_ignoring_filter() { assert_eq!(parsed["added"][0]["value"], 5.0); } +/// `EntryDelta` and `ValueDelta` both carry `start_line` as an +/// `Option` a v6 baseline usually leaves empty (#1170), but `entry()` +/// hard-codes `Some`, so neither `skip_serializing_if` branch is +/// otherwise exercised. Deleting either attribute would put a +/// `"start_line": null` on every entry of every diff, and nothing else +/// here would notice. +#[test] +fn json_omits_start_line_for_an_entry_that_pins_none() { + let lineless = |path: &str, value: f64| DiffEntry { + path: path.to_string(), + qualified: "x".to_string(), + metric: "cognitive".to_string(), + start_line: None, + value, + }; + let old = vec![lineless("a.rs", 10.0)]; + let new = vec![lineless("a.rs", 12.0), lineless("b.rs", 5.0)]; + let json = BaselineDiff::compute(&old, &new).render_json().unwrap(); + let parsed: serde_json::Value = serde_json::from_str(&json).unwrap(); + // `added` is an `EntryDelta`; `worsened` is a `ValueDelta`. Both + // declare the field separately, so both are asserted. + assert_eq!(parsed["added"][0]["path"], "b.rs"); + assert!( + parsed["added"][0].get("start_line").is_none(), + "EntryDelta must omit the key, not null it; got: {json}" + ); + assert_eq!(parsed["worsened"][0]["path"], "a.rs"); + assert!( + parsed["worsened"][0].get("start_line").is_none(), + "ValueDelta must omit the key, not null it; got: {json}" + ); +} + #[test] fn file_level_metric_identity_uses_file_sentinel() { let old = vec![]; diff --git a/big-code-analysis-cli/src/baseline_tests.rs b/big-code-analysis-cli/src/baseline_tests.rs index 12bbbcf0d..e3a6ab1ac 100644 --- a/big-code-analysis-cli/src/baseline_tests.rs +++ b/big-code-analysis-cli/src/baseline_tests.rs @@ -263,37 +263,50 @@ fn from_violations_skips_non_finite() { #[test] fn from_violations_deterministic_order() { // Inputs are crafted so every tiebreaker in the - // (path, qualified, start_line, metric) sort (issue #377 clusters - // same-symbol records together for review) is the deciding + // (path, qualified, metric, start_line) sort is the deciding // comparator for at least one adjacent pair in the output: // - // [0] vs [1]: same path + qualified + start_line -> metric breaks tie - // [1] vs [2]: same path + qualified, different start_line - // -> start_line breaks tie + // [0] vs [1]: same path + qualified -> metric breaks tie + // [1] vs [2]: same path + qualified + metric (an ambiguous + // identity) -> start_line breaks tie // [2] vs [3]: same path, different qualified -> qualified breaks tie // [3] vs [4]: different path -> path breaks tie + // + // The same fixture pins which entries keep a `start_line` (#1170): + // only [1] and [2], the two sharing one identity. + // + // The ambiguous pair is fed 99-before-10, against the order it must + // come back in. That is what makes the `start_line` claim above + // falsifiable: `sort_by` is stable, so an already-ascending pair + // comes back ascending whether or not the comparator ever looks at + // the line. Dropping `start_line` from `BaselineEntry::identity` + // failed none of the suite's 5054 tests while this vector was + // pre-sorted. let unsorted = vec![ v("src/z.rs", "z", 100, "cyclomatic", 5.0), v("src/a.rs", "b", 10, "cognitive", 4.0), v("src/a.rs", "a", 10, "cognitive", 3.0), - v("src/a.rs", "a", 10, "cyclomatic", 5.0), v("src/a.rs", "a", 99, "cyclomatic", 6.0), + v("src/a.rs", "a", 10, "cyclomatic", 5.0), ]; let file = from_violations(unsorted, test_anchor(), Provenance::hard()); assert_eq!(file.entries[0].path, "src/a.rs"); assert_eq!(file.entries[0].qualified, "a"); - assert_eq!(file.entries[0].start_line, 10); + // Unique identity -> no line recorded. + assert_eq!(file.entries[0].start_line, None); assert_eq!(file.entries[0].metric, "cognitive"); assert_eq!(file.entries[1].path, "src/a.rs"); assert_eq!(file.entries[1].qualified, "a"); - assert_eq!(file.entries[1].start_line, 10); + assert_eq!(file.entries[1].start_line, Some(10)); assert_eq!(file.entries[1].metric, "cyclomatic"); assert_eq!(file.entries[2].path, "src/a.rs"); assert_eq!(file.entries[2].qualified, "a"); - assert_eq!(file.entries[2].start_line, 99); + assert_eq!(file.entries[2].start_line, Some(99)); assert_eq!(file.entries[3].path, "src/a.rs"); assert_eq!(file.entries[3].qualified, "b"); + assert_eq!(file.entries[3].start_line, None); assert_eq!(file.entries[4].path, "src/z.rs"); + assert_eq!(file.entries[4].start_line, None); } #[test] @@ -312,6 +325,83 @@ fn from_violations_byte_equal_across_two_calls() { assert_eq!(a, b); } +/// The churn half of #1170, and the more valuable one: a baseline +/// written before and after an edit that shifted every function down the +/// file must be byte-identical, so a *real* baseline change (a `value` +/// moving) is the only thing a reviewer ever sees in the diff. +#[test] +fn line_drift_leaves_the_rendered_baseline_byte_identical() { + let before = vec![ + v("src/a.rs", "foo", 10, "cyclomatic", 5.0), + v("src/a.rs", "Bar::baz", 40, "cognitive", 7.0), + v("src/b.rs", "qux", 20, "cognitive", 7.0), + ]; + // Same functions, same values, each pushed 500 lines down by an + // import block or a sibling function added above. + let after: Vec = before + .iter() + .map(|v| Violation { + start_line: v.start_line + 500, + end_line: v.end_line + 500, + ..v.clone() + }) + .collect(); + let rendered_before = + render(&from_violations(before, test_anchor(), Provenance::hard())).expect("render before"); + let rendered_after = + render(&from_violations(after, test_anchor(), Provenance::hard())).expect("render after"); + assert_eq!(rendered_before, rendered_after); + // Guard against passing for the wrong reason: the file must really + // hold the entries, and really hold no line numbers. + assert!(rendered_before.contains("qualified = \"Bar::baz\"")); + assert!( + !rendered_before.contains("start_line"), + "unique identities record no line:\n{rendered_before}" + ); +} + +/// The exception that keeps the tolerance disambiguator working: when +/// two entries share one `(path, qualified, metric)` identity, a line is +/// the only thing that tells them apart, so both record one — and those +/// two entries do still move under line drift. +/// +/// `src/b.rs`'s `unique` is what pins the `path` third of that identity, +/// and it has to sit *immediately after* `src/a.rs`'s `unique` to do it: +/// the grouping is `chunk_by_mut`, so only adjacent entries are ever +/// compared, and entries are sorted by path first. A same-named function +/// in a non-adjacent file would chunk alone under a broken predicate too, +/// and prove nothing. Dropping `a.path == b.path` from `same_identity` +/// failed none of the suite's 5053 tests before this entry existed — +/// which is #1170's churn bug returning for `new` / `fmt` / `default`, +/// the commonest symbol shape in a Rust tree. +#[test] +fn ambiguous_identity_records_a_line_for_every_member() { + let file = from_violations( + vec![ + v("src/a.rs", "Trait::is_valid", 10, "cyclomatic", 5.0), + v("src/a.rs", "Trait::is_valid", 900, "cyclomatic", 6.0), + v("src/a.rs", "unique", 40, "cyclomatic", 8.0), + v("src/b.rs", "unique", 40, "cyclomatic", 8.0), + ], + test_anchor(), + Provenance::hard(), + ); + let lines: Vec<(&str, &str, Option)> = file + .entries + .iter() + .map(|e| (e.path.as_str(), e.qualified.as_str(), e.start_line)) + .collect(); + assert_eq!( + lines, + vec![ + ("src/a.rs", "Trait::is_valid", Some(10)), + ("src/a.rs", "Trait::is_valid", Some(900)), + ("src/a.rs", "unique", None), + ("src/b.rs", "unique", None), + ] + ); +} + #[test] fn path_normalized_forward_slash_on_serialize() { // Construct a Violation with a backslash path directly (so the @@ -346,13 +436,22 @@ fn entry( BaselineEntry { path: path.to_string(), qualified: qualified.to_string(), - start_line, + start_line: Some(start_line), metric: metric.to_string(), value, body_hash: None, } } +/// A hand-built entry that pins no `start_line` — the shape a v6 file +/// writes for a unique identity (#1170). +fn entry_without_line(path: &str, qualified: &str, metric: &str, value: f64) -> BaselineEntry { + BaselineEntry { + start_line: None, + ..entry(path, qualified, 0, metric, value) + } +} + #[test] fn classify_at_exact_baseline_is_covered() { let b = baseline_with(vec![entry("a", "f", 1, "cyclomatic", 5.0)]); @@ -569,7 +668,7 @@ fn classify_fuzzy_matches_renamed_function_by_body_hash() { entries: vec![BaselineEntry { path: "a".to_string(), qualified: "old_name".to_string(), - start_line: 1, + start_line: Some(1), metric: "cyclomatic".to_string(), value: 5.0, body_hash: Some(format!("{:016x}", 0xdead_beef_u64)), @@ -669,76 +768,155 @@ fn classify_recorded_round_trips_bit_exactly() { } } -// -- anchor + lexical normalisation (issue #376) ---------------------- +// -- optional start_line (issue #1170) -------------------------------- +/// The acceptance criterion for the v6 re-key: a legacy baseline that +/// records a line for every entry and its v6 rewrite that records none +/// must classify the *same set* of violations the same way. Asserting +/// the whole set, not a count — a count can hold steady while the set +/// changes. #[test] -fn lexical_normalize_folds_curdir_and_parent() { - assert_eq!(lexical_normalize(Path::new("./a/b")), Path::new("a/b")); - assert_eq!(lexical_normalize(Path::new("a/./b")), Path::new("a/b")); - assert_eq!(lexical_normalize(Path::new("a/b/../c")), Path::new("a/c")); +fn v5_and_v6_baselines_classify_an_identical_offender_set() { + use std::fmt::Write as _; + + let recorded = [ + ("src/a.rs", "foo", 10, "cyclomatic", 5.0), + ("src/a.rs", "Bar::baz", 40, "cognitive", 7.0), + ("src/b.rs", "qux", 300, "halstead.effort", 60_000.0), + ]; + // Same entries either way; only the schema stamp and the presence + // of `start_line` differ. + let mut v5 = String::from("version = 5\n"); + for (path, qualified, line, metric, value) in recorded { + let _ = write!( + v5, + "[[entry]]\npath = \"{path}\"\nqualified = \"{qualified}\"\n\ + start_line = {line}\nmetric = \"{metric}\"\nvalue = {value}\n" + ); + } + let v6 = render(&from_violations( + recorded + .iter() + .map(|&(p, q, l, m, val)| v(p, q, l, m, val)) + .collect(), + test_anchor(), + Provenance::hard(), + )) + .expect("render v6"); + assert!(!v6.contains("start_line")); + + // Probe every interesting outcome: covered at the recorded value, + // covered below it, regressed above it, drifted far past the + // tolerance, and an entry the baseline never had. + let probes = [ + v("src/a.rs", "foo", 10, "cyclomatic", 5.0), + v("src/a.rs", "foo", 4_000, "cyclomatic", 4.0), + v("src/a.rs", "Bar::baz", 40, "cognitive", 9.0), + v("src/b.rs", "qux", 1, "halstead.effort", 60_000.0), + v("src/b.rs", "never_recorded", 7, "cognitive", 3.0), + ]; + let classify_all = |text: &str| -> Vec { + let b = parse(text).expect("parse"); + probes.iter().map(|p| b.classify(p)).collect() + }; + assert_eq!(classify_all(&v5), classify_all(&v6)); + // And pin what that agreed-on set actually is, so the assertion + // cannot pass by both sides degrading to `New` together. assert_eq!( - lexical_normalize(Path::new("a/b/c/../../d")), - Path::new("a/d") + classify_all(&v6), + vec![ + Coverage::Covered { recorded: 5.0 }, + Coverage::Covered { recorded: 5.0 }, + Coverage::Regressed { recorded: 7.0 }, + Coverage::Covered { recorded: 60_000.0 }, + Coverage::New, + ] ); } +/// Two entries that differ only in `start_line` are one identity, and +/// stay one identity when the lines are gone: the group is what the +/// tolerance rule operates on, not the individual lines. #[test] -fn lexical_normalize_preserves_escaping_parents() { - // `..` past every accumulated Normal component is preserved so - // an entry that genuinely lives one level above the anchor - // (e.g., a sibling-crate analysis) still has an identity. - assert_eq!(lexical_normalize(Path::new("../a")), Path::new("../a")); - assert_eq!(lexical_normalize(Path::new("a/../../b")), Path::new("../b")); +fn entries_differing_only_in_start_line_share_one_identity() { + let b = baseline_with(vec![ + entry("a", "Trait::f", 10, "cyclomatic", 5.0), + entry("a", "Trait::f", 900, "cyclomatic", 9.0), + ]); + assert_eq!(b.by_symbol.len(), 1, "one key, two records under it"); + // With the lines present the tolerance still places each violation + // on the nearer record. + assert!(matches!( + b.classify(&v("a", "Trait::f", 12, "cyclomatic", 5.0)), + Coverage::Covered { recorded } if recorded == 5.0 + )); + assert!(matches!( + b.classify(&v("a", "Trait::f", 902, "cyclomatic", 9.0)), + Coverage::Covered { recorded } if recorded == 9.0 + )); } -#[cfg(unix)] +/// A lone record matches regardless of line, so dropping the line +/// changes nothing for it — the property that makes omitting the field +/// safe in the first place. #[test] -fn lexical_normalize_folds_parent_past_root() { - // POSIX: `..` immediately after a RootDir is a no-op. Before the - // fix the function preserved `..` literally, yielding `/..` — - // non-canonical and `strip_prefix` would not match a canonical - // anchor. A hand-crafted v2 entry like `path = "/../etc/passwd"` - // could exploit this to produce keys that bypass anchor - // relativisation; the fold keeps the encoder's output canonical. - assert_eq!(lexical_normalize(Path::new("/..")), Path::new("/")); - assert_eq!(lexical_normalize(Path::new("/../..")), Path::new("/")); - assert_eq!( - lexical_normalize(Path::new("/../etc/passwd")), - Path::new("/etc/passwd") - ); - // Mixing: Normal pop still works after a root-fold no-op. - assert_eq!( - lexical_normalize(Path::new("/foo/../../bar")), - Path::new("/bar") - ); +fn lone_record_without_a_line_matches_at_any_distance() { + let b = baseline_with(vec![entry_without_line("a", "f", "cyclomatic", 5.0)]); + assert!(matches!( + b.classify(&v("a", "f", 100_000, "cyclomatic", 5.0)), + Coverage::Covered { recorded } if recorded == 5.0 + )); } -#[cfg(unix)] +/// Only reachable by hand-editing: a record inside an *ambiguous* group +/// with no line cannot be placed, so it drops out of the tolerance match +/// rather than matching at distance zero and shadowing its sibling. #[test] -fn anchor_for_strips_baseline_filename() { - // `anchor_for` is lexical-only — no filesystem access — so the - // assertion can be a pure path comparison against synthetic - // input. Pinning to a fixed prefix keeps the test independent - // of `$TMPDIR` shape across CI hosts. - assert_eq!( - anchor_for(Path::new("/tmp/bca-anchor-test/baseline.toml")), - Path::new("/tmp/bca-anchor-test"), +fn unplaceable_record_in_an_ambiguous_group_is_skipped() { + let b = baseline_with(vec![ + entry_without_line("a", "Trait::f", "cyclomatic", 99.0), + entry("a", "Trait::f", 900, "cyclomatic", 9.0), + ]); + // The lineless record's generous 99.0 must not be what covers a + // violation sitting on the other record's line. + assert!(matches!( + b.classify(&v("a", "Trait::f", 900, "cyclomatic", 20.0)), + Coverage::Regressed { recorded } if recorded == 9.0 + )); + // And a violation nowhere near the placed record is `New`, not + // silently covered by the unplaceable one. + assert!(matches!( + b.classify(&v("a", "Trait::f", 5, "cyclomatic", 20.0)), + Coverage::New + )); +} + +/// A file from a *newer* schema can fail to deserialize before the +/// version check runs — v6 dropping a field a v5-era build requires is +/// exactly that shape. The error must still name the remedy rather than +/// surfacing a bare serde message. +#[test] +fn parse_reports_the_version_when_a_newer_schema_fails_to_deserialize() { + // `value` is required in every schema, so its absence stands in for + // "a field this build needs that version 99 no longer writes". + let err = parse( + "version = 99\n[[entry]]\npath = \"a\"\nqualified = \"f\"\nmetric = \"cyclomatic\"\n", + ) + .unwrap_err(); + assert!( + err.contains("version 99") && err.contains("--write-baseline"), + "msg: {err}" ); } -#[cfg(unix)] +/// …but a genuinely malformed file still gets the parser's own message, +/// which is the one that says *where* it broke. #[test] -fn normalize_path_canonicalises_against_anchor() { - // Three distinct typings of the same file under one anchor must - // collapse to the same key. - let anchor = Path::new("/repo"); - let key_dot = normalize_path(anchor, Path::new("/repo/src/foo.rs")); - let key_rel = normalize_path(anchor, Path::new("src/./foo.rs")); - let key_parent = normalize_path(anchor, Path::new("src/x/../foo.rs")); - assert_eq!(key_dot, "src/foo.rs"); - assert_eq!(key_rel, "src/foo.rs"); - assert_eq!(key_parent, "src/foo.rs"); +fn parse_reports_the_toml_error_when_the_version_is_supported() { + let err = parse("version = 6\n[[entry]]\npath = \"a\"\nvalue = \"oops\"\n").unwrap_err(); + assert!(err.contains("malformed baseline TOML"), "msg: {err}"); } +// -- path-key integration through `Baseline` --------------------------- #[cfg(unix)] #[test] @@ -771,229 +949,6 @@ fn from_str_defensive_anchor_normalization() { )); } -#[cfg(unix)] -#[test] -fn normalize_path_outside_anchor_uses_absolute_form() { - // A path that isn't under the anchor keeps its absolute form - // rather than degrading to `../` chains. Legitimate use case: - // a baseline at the repo root recording offenders from a - // sibling vendored crate kept outside the tree. - let key = normalize_path(Path::new("/repo"), Path::new("/elsewhere/file.rs")); - assert_eq!(key, "/elsewhere/file.rs"); -} - -// -- non-UTF-8 path identity ------------------------------------------ - -#[test] -fn normalize_path_utf8_unchanged_for_unreserved_ascii() { - // Regression guard: the common UTF-8 case (all-unreserved-ASCII - // path components) must round-trip untouched. Non-UTF-8 - // encoding shenanigans must not leak into ordinary inputs (no - // unexpected percent escapes, no extra markers). - assert_eq!( - normalize_path(test_anchor(), Path::new("src/foo.rs")), - "src/foo.rs" - ); - assert_eq!( - normalize_path(test_anchor(), Path::new("crates/a/b.rs")), - "crates/a/b.rs" - ); - // Backslashes are still normalized to forward slashes for the - // UTF-8 path so that cross-OS baselines match. - assert_eq!( - normalize_path(test_anchor(), Path::new("a\\b\\c.rs")), - "a/b/c.rs" - ); -} - -#[test] -fn normalize_path_utf8_escapes_percent() { - // `%` must be escaped in the UTF-8 fast path so it cannot collide - // with a non-UTF-8 byte's `%XX` escape. See `normalize_path_utf8_ - // non_utf8_byte_no_collision` for the actual collision check. - assert_eq!( - normalize_path(test_anchor(), Path::new("foo%FF.rs")), - "foo%25FF.rs" - ); - assert_eq!( - normalize_path(test_anchor(), Path::new("a%b%c.rs")), - "a%25b%25c.rs" - ); -} - -#[cfg(unix)] -#[test] -fn normalize_path_utf8_percent_vs_non_utf8_byte_no_collision() { - // The bug: a UTF-8 path containing the literal text `%FF` and a - // non-UTF-8 path containing the byte `0xFF` at the same position - // used to normalize to the same key (both `foo%FF.rs`), so a - // baseline written for one silently covered violations from the - // other. With `%` percent-encoded on the UTF-8 side, the keys - // diverge. - use std::ffi::OsStr; - use std::os::unix::ffi::OsStrExt; - - let utf8 = Path::new("foo%FF.rs"); - let non_utf8 = PathBuf::from(OsStr::from_bytes(b"foo\xff.rs")); - let key_utf8 = normalize_path(test_anchor(), utf8); - let key_non_utf8 = normalize_path(test_anchor(), &non_utf8); - assert_eq!(key_utf8, "foo%25FF.rs"); - assert_eq!(key_non_utf8, "foo%FF.rs"); - assert_ne!(key_utf8, key_non_utf8); -} - -#[cfg(unix)] -#[test] -fn baseline_key_preserves_non_utf8_identity() { - use std::ffi::OsStr; - use std::os::unix::ffi::OsStrExt; - - // Two distinct non-UTF-8 paths must produce two distinct - // baseline keys. The previous `display().to_string()` fallback - // collapsed both onto a sequence of U+FFFD replacement chars, - // so a baseline written from path A would silently cover - // violations from path B. - let a = PathBuf::from("src").join(OsStr::from_bytes(b"bad-\xff\xfe.rs")); - let b = PathBuf::from("src").join(OsStr::from_bytes(b"bad-\xfe\xff.rs")); - let key_a = normalize_path(test_anchor(), &a); - let key_b = normalize_path(test_anchor(), &b); - assert_ne!(key_a, key_b); - // The encoded keys are valid UTF-8 (required by TOML) and - // contain only ASCII bytes after percent-encoding. - assert!(key_a.is_ascii()); - assert!(key_b.is_ascii()); -} - -// -- WTF-16 percent-encoding (always-on, synthetic input) ------------ - -#[test] -fn wtf16_encode_pure_ascii() { - // ASCII path bytes are unreserved, so they survive unchanged. - let out = percent_encode_wtf16("src/foo.rs".encode_utf16()); - assert_eq!(out, "src/foo.rs"); -} - -#[test] -fn wtf16_encode_empty() { - assert_eq!(percent_encode_wtf16(std::iter::empty::()), ""); -} - -#[test] -fn wtf16_encode_bmp_non_ascii() { - // U+00E9 (é) is BMP; UTF-8 = 0xC3 0xA9; both bytes are - // non-unreserved and percent-encode to %C3%A9. - let out = percent_encode_wtf16("é".encode_utf16()); - assert_eq!(out, "%C3%A9"); -} - -#[test] -fn wtf16_encode_supplementary_plane() { - // U+1F600 (😀) requires a surrogate pair in WTF-16 - // (0xD83D, 0xDE00) and UTF-8-encodes as 0xF0 0x9F 0x98 0x80. - // `char::decode_utf16` pairs the surrogates back to the scalar, - // so the encoder must emit the UTF-8 byte form. - let units = [0xD83D_u16, 0xDE00_u16]; - let out = percent_encode_wtf16(units); - assert_eq!(out, "%F0%9F%98%80"); - // Sanity: the same character entered as a string round-trips - // identically through `encode_utf16`. - assert_eq!(out, percent_encode_wtf16("😀".encode_utf16())); -} - -#[test] -fn wtf16_encode_unpaired_high_surrogate() { - let out = percent_encode_wtf16([0xD83D_u16]); - assert_eq!(out, "%uD83D"); -} - -#[test] -fn wtf16_encode_unpaired_low_surrogate() { - // A lone low surrogate (no preceding high) is unpaired. - let out = percent_encode_wtf16([0xDE00_u16]); - assert_eq!(out, "%uDE00"); -} - -#[test] -fn wtf16_encode_high_followed_by_non_low_is_unpaired() { - // High surrogate followed by ASCII: the high is unpaired and - // the ASCII byte is encoded normally afterwards. - let units = [0xD83D_u16, u16::from(b'x')]; - let out = percent_encode_wtf16(units); - assert_eq!(out, "%uD83Dx"); -} - -#[test] -fn wtf16_encode_leading_low_then_pair() { - // A lone low surrogate followed by a real pair: the leading low - // must not consume the next code unit (the high of the pair). - let units = [0xDC00_u16, 0xD83D_u16, 0xDE00_u16]; - let out = percent_encode_wtf16(units); - assert_eq!(out, "%uDC00%F0%9F%98%80"); -} - -#[test] -fn wtf16_encode_distinct_unpaired_surrogates_do_not_collide() { - // The whole point of the fix: two distinct invalid WTF-16 - // sequences that `to_string_lossy()` would have collapsed onto - // a single U+FFFD must produce two distinct encoded keys. - let a = percent_encode_wtf16([0xD83D_u16]); - let b = percent_encode_wtf16([0xDE00_u16]); - assert_ne!(a, b); - // And two different lone high surrogates also separate cleanly. - let c = percent_encode_wtf16([0xD800_u16]); - let d = percent_encode_wtf16([0xDBFF_u16]); - assert_ne!(c, d); -} - -#[test] -fn wtf16_encode_marker_never_emitted_by_scalar_bytes() { - // Regression guard: the byte encoder only emits `%` followed by - // exactly two uppercase hex digits, never `%u`. Scalars cannot - // produce a string that begins with `%u` from their UTF-8 bytes - // — `u` is unreserved, so it stays as `u`, but the preceding - // `%` only appears when a non-unreserved byte is escaped (and - // is then immediately followed by two hex digits, not `u`). - // Therefore parsing `%u…` is unambiguous. - for codepoint in ['u', '%', '!', '\u{00E9}', '\u{1F600}'] { - let s = codepoint.to_string(); - let out = percent_encode_wtf16(s.encode_utf16()); - assert!(!out.contains("%u"), "scalar {codepoint:?} produced {out:?}"); - } -} - -#[cfg(windows)] -#[test] -fn baseline_key_preserves_non_utf16_identity_on_windows() { - use std::ffi::OsString; - use std::os::windows::ffi::OsStringExt; - - // Two distinct paths that differ only by an unpaired surrogate - // value would collapse to the same `to_string_lossy()` key - // (both surrogates become U+FFFD). With the WTF-16 encoder they - // stay distinct. - let a_units: [u16; 5] = [ - u16::from(b'a'), - u16::from(b'/'), - 0xD83D, - u16::from(b'.'), - u16::from(b's'), - ]; - let b_units: [u16; 5] = [ - u16::from(b'a'), - u16::from(b'/'), - 0xDE00, - u16::from(b'.'), - u16::from(b's'), - ]; - let path_a = PathBuf::from(OsString::from_wide(&a_units)); - let path_b = PathBuf::from(OsString::from_wide(&b_units)); - let key_a = normalize_path(test_anchor(), &path_a); - let key_b = normalize_path(test_anchor(), &path_b); - assert_ne!(key_a, key_b); - assert!(key_a.is_ascii()); - assert!(key_b.is_ascii()); -} - #[cfg(unix)] #[test] fn baseline_covers_distinguishes_non_utf8_paths() { @@ -1044,94 +999,6 @@ fn baseline_covers_distinguishes_non_utf8_paths() { assert!(matches!(b.classify(&violation_b), Coverage::New)); } -// -- bare_name / body-hash helpers (issue #377) ----------------------- - -#[test] -fn bare_name_strips_qualifier() { - assert_eq!(bare_name("MyStruct::do_thing"), "do_thing"); - assert_eq!(bare_name("a::b::c"), "c"); - assert_eq!(bare_name("plain"), "plain"); - assert_eq!(bare_name(""), ""); -} - -#[test] -fn body_hash_ignores_indentation_blank_lines_and_run_width() { - // The normalisation trims leading/trailing whitespace, collapses - // internal whitespace *runs* to one space, drops `\r`, and skips - // blank lines — so re-indenting, reflowing blank lines, or changing - // CRLF/LF must not change the digest. (It is not insensitive to the - // presence/absence of whitespace *between* tokens, only its width.) - let original = b" let x = 1;\n return x + 1;\n"; - let reformatted = b"\nlet x = 1;\r\n\n return x + 1;\n\n"; - assert_eq!( - hash_body(original, 1, 2, ""), - hash_body(reformatted, 1, 6, "") - ); -} - -#[test] -fn body_hash_distinguishes_different_bodies() { - assert_ne!( - hash_body(b"let x = 1;", 1, 1, ""), - hash_body(b"let x = 2;", 1, 1, "") - ); -} - -#[test] -fn body_hash_respects_line_range() { - // Lines 2..=2 of a three-line body hash only the middle line. - let src = b"line one\nline two\nline three\n"; - assert_eq!(hash_body(src, 2, 2, ""), hash_body(b"line two", 1, 1, "")); -} - -#[test] -fn body_hash_out_of_range_is_empty_digest() { - // A start past EOF yields the empty-body digest (the FNV offset - // basis) rather than panicking on an out-of-bounds slice. - let src = b"only one line"; - assert_eq!(hash_body(src, 100, 200, ""), hash_body(b"", 1, 1, "")); -} - -#[test] -fn body_hash_elides_own_name_so_rename_matches() { - // The headline rule-3 property: renaming the function (declaration - // and recursive self-calls) leaves the digest unchanged, because the - // bare name is elided. - let before = b"fn classify(n: i32) -> i32 { classify(n - 1) }"; - let after = b"fn categorize(n: i32) -> i32 { categorize(n - 1) }"; - assert_eq!( - hash_body(before, 1, 1, "classify"), - hash_body(after, 1, 1, "categorize") - ); -} - -#[test] -fn body_hash_elision_is_whole_word_only() { - // Eliding `is` must not corrupt the substring inside `is_valid` — - // two bodies that differ only in an unrelated identifier sharing the - // elided prefix must still hash differently. - let a = b"fn is() { is_valid() }"; - let b = b"fn is() { is_ready() }"; - assert_ne!(hash_body(a, 1, 1, "is"), hash_body(b, 1, 1, "is")); -} - -#[test] -fn body_hash_round_trips_through_hex_codec() { - let h = hash_body(b"some body text", 1, 1, ""); - assert_eq!(decode_body_hash(&encode_body_hash(h)), Some(h)); -} - -#[test] -fn decode_body_hash_rejects_malformed() { - assert_eq!(decode_body_hash("not-hex"), None); - assert_eq!(decode_body_hash("dead"), None); // too short - assert_eq!(decode_body_hash(""), None); - assert_eq!( - decode_body_hash("0123456789abcdef"), - Some(0x0123_4567_89ab_cdef) - ); -} - // -- provenance (issue #486) ------------------------------------------- #[test] diff --git a/big-code-analysis-cli/src/cli_args/check.rs b/big-code-analysis-cli/src/cli_args/check.rs index 443c7b62c..f144c7575 100644 --- a/big-code-analysis-cli/src/cli_args/check.rs +++ b/big-code-analysis-cli/src/cli_args/check.rs @@ -24,6 +24,45 @@ pub(crate) struct CheckArgs { /// `0` is allowed and means "no value permitted". #[clap(long = "threshold", value_parser = parse_cli_threshold)] pub(crate) thresholds: Vec<(String, f64)>, + /// Preview what a candidate `=` would cost, at both + /// tiers, instead of gating. Reports the hard-tier offender count, + /// the resolved soft limit and its offender count, and how many of + /// each already match a `--baseline` entry — so the decision-grade + /// figure (new baseline entries the change would add) is on the + /// screen. Repeatable, one candidate per metric; every other metric + /// is left out of the walk. + /// + /// This exists because `--threshold` cannot answer the question: + /// its limits are applied last and absolutely, never scaled, so a + /// candidate trialled that way has no soft tier at all and reads as + /// free when it is not. The soft band is derived from the candidate + /// itself, at the `--tier=soft=RATIO` ratio when one is given and + /// 0.95 otherwise. + /// + /// Honours `exclude_tests`, `[check] exclude`, in-source suppression + /// markers and the baseline exactly as the run it predicts, and + /// writes nothing: no gate runs, so it always exits 0 on success + /// (1 on a tool error, such as a candidate naming a metric this + /// build does not gate). Conflicts with `--write-baseline`, + /// `--print-effective-config`, `--report-format`, `--output`, and an + /// explicit `--summary-file `, each of which would produce a + /// second, different artifact. + // The candidate-limit preview landed in issue #1169; see the + // "Choosing thresholds" and "Baselines" recipes in the book. + // + // `--summary-file` is rejected by `reject_summary_file_path` rather + // than by `conflicts_with_all`: clap conflicts on the flag's + // *presence*, and the keyword forms `auto` / `never` must keep + // working — `auto` is what a GHA workflow leaves implicit, and the + // preview simply produces no step summary the way any other + // non-gating run does. + #[clap( + long = "explain-threshold", + value_name = "METRIC=LIMIT", + value_parser = parse_cli_threshold, + conflicts_with_all = ["write_baseline", "print_effective_config", "output_format", "output"], + )] + pub(crate) explain_thresholds: Vec<(String, f64)>, /// Path to a TOML config with a `[thresholds]` table; CLI /// `--threshold` flags override values read from it. // The indented example lives in `long_help`, not the `///` doc @@ -45,7 +84,7 @@ Path to a TOML config with a `[thresholds]` table. Example: CLI `--threshold` flags override values read from this file." )] pub(crate) config: Option, - /// Print offenders to stderr but exit 0 even when thresholds are + /// Report offenders as usual but exit 0 even when thresholds are /// exceeded. Useful while adopting baselines without flipping CI red. /// Default: exit 2 when any threshold is exceeded. #[clap(long = "no-fail")] @@ -59,7 +98,7 @@ CLI `--threshold` flags override values read from this file." /// Surface suppressed debt in the offender document instead of /// dropping it. Offenders silenced by an in-source `bca: suppress` /// marker or covered by the baseline are still kept out of the gate - /// (exit code and human stream are unaffected), but are emitted into + /// (exit code and offender rows are unaffected), but are emitted into /// the `--format sarif` document carrying a SARIF /// `suppressions` entry — GitHub Code Scanning renders them as /// suppressed (closed) alerts so the debt stays visible. Only the @@ -73,8 +112,11 @@ CLI `--threshold` flags override values read from this file." /// lines, MSVC warning lines). Named `--report-format` to separate /// "which CI report dialect" from the data-serialization `--format` /// the structured subcommands use. When omitted *and* - /// `--output` is also omitted, only the human-readable stderr stream - /// is emitted; the exit-code contract is unaffected. When omitted but + /// `--output` is also omitted, only the human-readable offender rows + /// are emitted; the exit-code contract is unaffected. Note that + /// passing this flag without `--output` gives the document stdout, + /// so the human rows fall back to stderr for that combination — + /// add `--output ` to keep both. When omitted but /// `--output` is given, the dialect is inferred from the output /// extension (`.sarif` → sarif, `.xml` → checkstyle); an extension /// with no unique dialect is a usage error. The old `--format` / `-O` @@ -132,8 +174,9 @@ CLI `--threshold` flags override values read from this file." pub(crate) write_baseline: Option>, /// Skip the trailing per-file rollup footer. The footer groups /// violations by file and cites the single worst-ratio metric per - /// file. Pass this when downstream tooling grep-pipes the stderr - /// stream and would be confused by the trailing summary block. + /// file. It is written to stderr, so a plain `bca check | ...` + /// pipeline never sees it; pass this when a tool reads the *merged* + /// streams and would be confused by the trailing summary block. /// Default: footer enabled. #[clap(long = "no-summary")] pub(crate) no_summary: bool, @@ -155,10 +198,10 @@ CLI `--threshold` flags override values read from this file." pub(crate) changed_only: bool, /// Emit GitHub Actions `::error file=…,line=…,title=…::msg` /// workflow commands per violation so the GHA UI renders them as - /// inline annotations on the file-diff view. Additive to the - /// human-readable stderr stream — annotations ride on top, they - /// don't replace it. Tri-state `` mirroring - /// `--color`: `auto` (default) emits annotations when + /// inline annotations on the file-diff view. Written to stderr, + /// additive to the human-readable offender rows — annotations ride + /// on top, they don't replace them. Tri-state `` + /// mirroring `--color`: `auto` (default) emits annotations when /// `$GITHUB_ACTIONS == "true"`; `always` forces them on; `never` /// suppresses them even inside a GHA step (so a workflow that runs /// `bca check` twice can annotate from only one run). A bare @@ -220,14 +263,18 @@ CLI `--threshold` flags override values read from this file." /// /// - `hard` (default) — flag a function only when a metric is at or /// over its `[thresholds]` limit. - /// - `soft` — early-warning tier: flag a function when a metric - /// reaches `RATIO` (default 0.95) of any limit, i.e. before the - /// hard gate trips. With a `[thresholds.soft]` table present, the + /// - `soft` — early-warning tier: tighten every limit by `RATIO` + /// (default 0.95) so a function is flagged before the hard gate + /// trips. With a `[thresholds.soft]` table present, the /// per-metric soft limits take precedence over the blanket ratio /// (metrics absent from it inherit their hard limit). - /// - `soft=0.90` — soft tier scaling every limit by 0.90; `soft=1.0` - /// disables the blanket scale (a soft tier driven only by an - /// explicit `[thresholds.soft]` table). + /// - `soft=0.90` — soft tier tightening every limit by 0.90; + /// `soft=1.0` disables the blanket scale (a soft tier driven only + /// by an explicit `[thresholds.soft]` table). + /// + /// `RATIO` scales the band, not the number: a ceiling comes down + /// (`cognitive = 15` warns at 13.5), while a lower-is-worse `mi.*` + /// floor goes up (`mi.original = 20` warns at 22.2223). /// /// Resolution order: `[thresholds]` (manifest + `--config`) → /// `[thresholds.soft]` or the soft ratio → absolute @@ -253,7 +300,7 @@ CLI `--threshold` flags override values read from this file." pub(crate) headroom: Option, /// Exit-code style: `default` keeps the stable /// 0/1/2 contract; `tiered` splits exit `2` by severity so CI can - /// branch without parsing the `[new]` / `[regr +N%]` stderr tags: + /// branch without parsing the `[new]` / `[regr +N%]` row tags: /// /// - `0` — clean. /// - `1` — tool error (bad config, unknown metric, unreadable path). diff --git a/big-code-analysis-cli/src/cli_args/mod.rs b/big-code-analysis-cli/src/cli_args/mod.rs index 5b236c30e..441c4cffd 100644 --- a/big-code-analysis-cli/src/cli_args/mod.rs +++ b/big-code-analysis-cli/src/cli_args/mod.rs @@ -436,6 +436,14 @@ pub(crate) enum Command { /// distinguish "metric regression" from "tool crashed". /// `--strict-exit-codes` opts into tiered codes (2-5) that split the /// violation case by severity. + /// + /// Streams: the offender rows go to stdout, so `bca check | wc -l` + /// and `bca check 2>/dev/null` see them. The summary footer, + /// remediation block, GitHub Actions annotations, and every + /// `bca:` / `warning:` / `error:` diagnostic go to stderr. The one + /// exception is `--report-format` without `--output`: the + /// aggregated document takes stdout there and the human rows fall + /// back to stderr, so a SARIF payload stays parseable. // Boxed because `CheckArgs` is by far the largest variant payload // (its many gate-tuning flags dwarf the other subcommands' args); // boxing keeps `Command` small and silences `large_enum_variant`. diff --git a/big-code-analysis-cli/src/commands.rs b/big-code-analysis-cli/src/commands.rs index d610cb7c3..b631e5edc 100644 --- a/big-code-analysis-cli/src/commands.rs +++ b/big-code-analysis-cli/src/commands.rs @@ -45,7 +45,8 @@ use crate::metric_diff::DiffSide; use crate::threshold_lang::LanguageThresholds; use crate::threshold_soft::{SoftLimit, scale_threshold}; use crate::thresholds::{ - ParsedThresholds, ThresholdSet, Violation, breaches_limit, render_violation_line, + ParsedThresholds, ThresholdSet, Violation, breaches_limit, metric_is_lower_is_worse, + render_violation_line, }; use big_code_analysis::{FuncSpace, Ops}; @@ -53,10 +54,11 @@ use crate::{ Action, AggregateItem, CheckArgs, Cli, Command, Config, CountArgs, DiffBaselineArgs, ExemptionsArgs, FindArgs, GlobalOpts, InitArgs, LineRange, ListMetricsArgs, MetricsArgs, OutputFormat, PreprocArgs, PrintConfigFormat, ReportArgs, StripCommentsArgs, StructuredArgs, - SummaryFile, Tier, TierSpec, die, die_io, group_files_by_basename, legacy_hint, load_baseline, - load_preproc_data, load_threshold_config, note, read_exclude_patterns_from, resolve_walk_files, - run_walk, run_walk_collecting, run_walk_resolved, validate_output_path, warn, write_atomic, - write_output_or_stdout, write_stdout_or_die, writeln_stdout_or_die, + SummaryFile, Tier, TierSpec, die, die_io, die_unless_broken_pipe, group_files_by_basename, + legacy_hint, load_baseline, load_preproc_data, load_threshold_config, note, + read_exclude_patterns_from, resolve_walk_files, run_walk, run_walk_collecting, + run_walk_resolved, validate_output_path, warn, write_atomic, write_output_or_stdout, + write_stdout_or_die, writeln_stdout_or_die, }; mod analyze; diff --git a/big-code-analysis-cli/src/commands/check.rs b/big-code-analysis-cli/src/commands/check.rs index 5967c0490..9519348ea 100644 --- a/big-code-analysis-cli/src/commands/check.rs +++ b/big-code-analysis-cli/src/commands/check.rs @@ -3,12 +3,15 @@ use super::*; mod effective_config; +mod explain; mod footer; mod outcome; mod remediation; mod thresholds; -pub(crate) use {effective_config::*, footer::*, outcome::*, remediation::*, thresholds::*}; +pub(crate) use { + effective_config::*, explain::*, footer::*, outcome::*, remediation::*, thresholds::*, +}; pub(crate) fn run_check( globals: GlobalOpts, @@ -40,10 +43,21 @@ pub(crate) fn run_check( // the CLI value winning in either direction. let tier = args.resolved_tier(); let tiered_exit_codes = args.resolved_exit_codes() == Some(crate::ExitCodes::Tiered); + let layers = merge_threshold_layers(&args, base_thresholds); + // `--explain-threshold` is a candidate-limit preview, not a gate + // (#1169): it splices its own limits into the same merged layers, + // walks once, and reports both tiers' cost. Branching here — after + // the manifest merge, before the gate's own resolution — is what + // makes the preview honour `exclude_tests`, `[check] exclude` and the + // baseline exactly as the run it predicts. + if !args.explain_thresholds.is_empty() { + run_explain_thresholds(globals, &args, &layers, tier, preproc); + return; + } let ResolvedThresholds { thresholds, provenance, - } = validate_and_build_thresholds(&mut args, base_thresholds, tier); + } = validate_and_build_thresholds(&mut args, layers, tier); // `--print-effective-config` is a read-only debug aid: print the // resolved configuration and exit 0 before the walk. clap already // rejects pairing with `--write-baseline` (conflicts_with), so by @@ -60,28 +74,16 @@ pub(crate) fn run_check( ); return; } - let scope = resolve_diff_scope(&args); - // Clone globals for the remediation builder: `run_check_walk` - // consumes `globals` by value (it passes through to `run_walk` - // which spawns worker threads with ownership), but - // `format_remediation_block` needs the resolved `--paths` / - // `--exclude` set to compose a copy-paste-safe refresh command. - let globals_for_remediation = globals.clone(); - let walk = run_check_walk(globals, &args, preproc, thresholds); - enforce_usable_input(&walk); - let violations = walk.violations; - - // Drop offenders from `[check.exclude]` files (#378) before *any* - // downstream consumer sees them — so `--write-baseline` never - // records the structural exemptions and the gate never fails on - // them. Applied after the empty-input guard above: exempt files are - // still walked and counted, only their violations are dropped. - let violations = apply_check_exclude( + let CollectedViolations { violations, - &args, - &globals_for_remediation.paths, - globals_for_remediation.paths_from.as_deref(), - ); + scope, + // Kept for the remediation builder: `run_check_walk` consumes + // `globals` by value (it passes through to `run_walk` which + // spawns worker threads with ownership), but + // `format_remediation_block` needs the resolved `--paths` / + // `--exclude` set to compose a copy-paste-safe refresh command. + globals: globals_for_remediation, + } = collect_check_violations(globals, &args, preproc, thresholds); // `--write-baseline ` writes there; a bare `--write-baseline` // is resolved to the manifest `baseline` by `merge_check` (#496), so @@ -98,22 +100,13 @@ pub(crate) fn run_check( return; } - let pairs = filter_by_baseline( + let pairs = classify_check_violations( violations, - args.baseline.as_deref(), - args.baseline_line_tolerance - .unwrap_or(baseline::DEFAULT_LINE_TOLERANCE), - args.baseline_fuzzy_match.unwrap_or(false), + &args, + scope.as_ref(), provenance, args.report_suppressed, ); - // Apply `--changed-only` diff-scope filtering to ALL offenders before - // splitting, so the suppressed set surfaced in the report respects the - // touched-file scope exactly as the active set does — otherwise - // `--changed-only --report-suppressed` would leak suppressed debt from - // files outside the diff. With `--report-suppressed` off this is the - // original pre-feature ordering (filter, then everything is active). - let pairs = apply_changed_only(pairs, scope.as_ref(), args.changed_only); // Split the report-only suppressed debt — in-source markers // (`v.suppressed`) plus baseline-covered offenders // (`Coverage::Covered`), present only under `--report-suppressed` — from @@ -156,6 +149,87 @@ pub(crate) fn run_check( } } +/// What the walk half of the pipeline yields: the offenders that +/// survived `[check.exclude]`, the diff scope both callers need +/// afterwards, and the `GlobalOpts` clone `run_check_walk` did not +/// consume. +struct CollectedViolations { + violations: Vec, + scope: Option, + globals: GlobalOpts, +} + +/// The gate's pre-baseline stages, in the order [`run_check`] applies +/// them: diff scope, walk, empty-input guard, `[check.exclude]`. +/// +/// Shared with the `--explain-threshold` preview (#1169) so that the +/// preview cannot describe a different run from the one it predicts. A +/// stage added here reaches both; a stage added to one call site would +/// compile in the other and silently diverge, and no test would catch +/// it because each path would still pass its own assertions. +/// +/// The `[check.exclude]` drop (#378) happens before *any* downstream +/// consumer sees the offenders — so `--write-baseline` never records +/// the structural exemptions and the gate never fails on them. It runs +/// after the empty-input guard: exempt files are still walked and +/// counted, only their violations are dropped. +fn collect_check_violations( + globals: GlobalOpts, + args: &CheckArgs, + preproc: Option>, + thresholds: Arc, +) -> CollectedViolations { + let scope = resolve_diff_scope(args); + let globals_kept = globals.clone(); + let walk = run_check_walk(globals, args, preproc, thresholds); + enforce_usable_input(&walk); + let violations = apply_check_exclude( + walk.violations, + args, + &globals_kept.paths, + globals_kept.paths_from.as_deref(), + ); + CollectedViolations { + violations, + scope, + globals: globals_kept, + } +} + +/// The gate's post-exclude stages, in the order [`run_check`] applies +/// them: baseline classification, then `--changed-only` scoping. The +/// companion to [`collect_check_violations`], and shared with the +/// preview for the same reason. +/// +/// It also owns the three `CheckArgs` defaults that feed +/// [`filter_by_baseline`], which is what stops the preview and the gate +/// resolving the same `--baseline` differently. +/// +/// `--changed-only` filters ALL offenders before the caller splits +/// them, so the suppressed set surfaced in the report respects the +/// touched-file scope exactly as the active set does — otherwise +/// `--changed-only --report-suppressed` would leak suppressed debt from +/// files outside the diff. With `--report-suppressed` off this is the +/// original pre-feature ordering (filter, then everything is active). +fn classify_check_violations( + violations: Vec, + args: &CheckArgs, + scope: Option<&diff::DiffScope>, + provenance: baseline::Provenance, + keep_covered: bool, +) -> Vec<(Violation, Option)> { + let pairs = filter_by_baseline( + violations, + args.baseline.as_deref(), + args.baseline_line_tolerance + .unwrap_or(baseline::DEFAULT_LINE_TOLERANCE), + args.baseline_fuzzy_match.unwrap_or(false), + provenance, + keep_covered, + ); + apply_changed_only(pairs, scope, args.changed_only) +} + /// What the check walk produced: the sorted violations plus the /// post-walk tally `run_check` consults before it trusts the gate /// verdict. @@ -550,6 +624,25 @@ pub(crate) fn apply_changed_only_inner( ChangedOnlyOutcome { kept, diagnostic } } +/// Render one `path:lines: name: metric = N (limit M)` row per pair, in +/// the order the caller sorted them, then flush. +/// +/// Split out from [`emit_check_results`] so the same loop serves both +/// destinations the stream contract there admits. The flush is part of +/// the helper for the same reason [`write_parts_flushed`] carries one: +/// a buffered writer that reports every `write_all` as `Ok` can still +/// fail at flush time, and on the stdout path that error decides between +/// `die` and a silent truncated report. +fn write_violation_rows_flushed( + out: &mut impl Write, + pairs: &[(Violation, Option)], +) -> std::io::Result<()> { + pairs + .iter() + .try_for_each(|(v, tag)| writeln!(out, "{}", render_violation_line(v, tag.as_ref())))?; + out.flush() +} + pub(crate) fn emit_check_results( pairs: Vec<(Violation, Option)>, suppressed: Vec<(Violation, Option)>, @@ -557,26 +650,65 @@ pub(crate) fn emit_check_results( scope: Option<&diff::DiffScope>, remediation: Option<&str>, ) { + // Stream contract (#1167) — do not move a line across it without + // reading the issue first. + // + // **stdout** carries the offender rows, and nothing else. They are + // this command's product, so the obvious ways to work with a list — + // `| wc -l`, `| head`, `| rg -c`, `2>/dev/null` — must reach them. + // They used to go to stderr, where all four silently reported an + // empty offender set: a *plausible* "this tree is clean" rather than + // an error. + // + // **stderr** carries everything that is commentary about the run: + // the per-file summary footer, the GitHub Actions annotations, the + // remediation block, and the `bca: …` / `warning:` / `error:` + // diagnostics the upstream stages emit (`skipped N violations via + // [check.exclude]`, `filtered N violations via baseline`, …). + // + // The one exception: `--report-format` without `--output` puts the + // aggregated SARIF / Checkstyle / Code Climate document on stdout, + // so the rows stay on stderr for that combination instead of + // corrupting a machine-readable payload. `--output ` moves the + // document off stdout and the rows go back to it. + let document_owns_stdout = args.output_format.is_some() && args.output.is_none(); + + if !document_owns_stdout { + // Written and flushed before the first stderr write below, so a + // terminal (or a `2>&1`-merged CI log) still shows the rows + // above the footer that summarizes them. + // + // `BrokenPipe` is exempt — `bca check | head` is routine and + // must still exit on the gate verdict, matching every other + // subcommand's stdout policy (`write_stdout_or_die`, #1132). + // Any other write failure is a real tool error: reporting a gate + // verdict whose evidence never reached the consumer is the + // silent-success shape this whole contract exists to avoid. + let mut stdout = BufWriter::new(std::io::stdout().lock()); + let written = write_violation_rows_flushed(&mut stdout, &pairs); + die_unless_broken_pipe(written, "writing check offenders"); + } + // BrokenPipe on stderr (e.g. when piped to `head`) is the only // realistic write failure here; swallow it rather than die so the // exit-code contract is honored. // - // `stderr` is unbuffered — one `write(2)` per violation line — so the - // lock is wrapped in a `BufWriter`. What makes that safe is the + // `stderr` is unbuffered — one `write(2)` per line — so the lock is + // wrapped in a `BufWriter`. What makes that safe is the // explicit `drop` at the end of this function, and it is about // *ordering*, not about `process::exit`: `BufWriter::drop` does // flush (it only discards the error), and this buffer is dropped // here, long before `run_check` ever reaches `process::exit`. // Deleting that drop lets the `eprintln!` diagnostic and the stdout - // document that follow overtake the violation report — which is what - // `check_violations_are_flushed_before_later_stderr_writes` catches. + // document that follow overtake the summary footer — which is what + // `check_stderr_block_is_flushed_before_later_stderr_writes` catches. // // The one thing buffering does cost: a panic between the writes // below and that drop loses the entire report, because // `BufWriter::drop` skips the flush when `self.panicked`. let mut stderr = BufWriter::new(std::io::stderr().lock()); - for (v, tag) in &pairs { - let _ = writeln!(stderr, "{}", render_violation_line(v, tag.as_ref())); + if document_owns_stdout { + let _ = write_violation_rows_flushed(&mut stderr, &pairs); } if !args.no_summary && !pairs.is_empty() { let _ = write_summary_footer(&mut stderr, &pairs, scope); @@ -622,28 +754,42 @@ pub(crate) fn emit_check_results( ); } - // Emit the aggregated CI/IDE document if requested. Empty input - // produces a well-formed but offender-free document, which CI - // consumers can ingest unchanged on clean runs. The exit-code - // contract is unaffected by this branch. - if let Some(fmt) = args.output_format { - let offenders: Vec<_> = pairs - .into_iter() - .map(|(v, _)| violation_to_offender(v)) - .collect(); - // Only the SARIF format can represent suppression, so route active + - // suppressed offenders through the suppression-aware writer there. For - // every other format (and the default no-suppressed case) fall back to - // the plain dump so output is byte-for-byte unchanged. - let written = if !suppressed.is_empty() - && matches!(fmt, check_format::AggregatedFormat::Sarif) - { - check_format::dump_sarif_with_suppressed(&offenders, suppressed, args.output.as_deref()) - } else { - fmt.dump(&offenders, args.output.as_deref()) - }; - written.unwrap_or_else(|e| die(format_args!("failed to write {}: {e}", fmt.name()))); - } + emit_aggregated_document(pairs, suppressed, args); +} + +/// Emit the aggregated CI/IDE document (`--report-format`, or a dialect +/// inferred from `--output`) — the machine-readable counterpart to the +/// human rows [`emit_check_results`] writes, and the other half of that +/// function's stream contract: it lands in `--output ` when given +/// and on stdout otherwise. +/// +/// A no-op when neither flag is in effect. Empty input still produces a +/// well-formed but offender-free document, which CI consumers can ingest +/// unchanged on clean runs, and a successful write never perturbs the +/// exit-code contract. +fn emit_aggregated_document( + pairs: Vec<(Violation, Option)>, + suppressed: Vec<(Violation, Option)>, + args: &CheckArgs, +) { + let Some(fmt) = args.output_format else { + return; + }; + let offenders: Vec<_> = pairs + .into_iter() + .map(|(v, _)| violation_to_offender(v)) + .collect(); + // Only the SARIF format can represent suppression, so route active + + // suppressed offenders through the suppression-aware writer there. For + // every other format (and the default no-suppressed case) fall back to + // the plain dump so output is byte-for-byte unchanged. + let written = if !suppressed.is_empty() && matches!(fmt, check_format::AggregatedFormat::Sarif) + { + check_format::dump_sarif_with_suppressed(&offenders, suppressed, args.output.as_deref()) + } else { + fmt.dump(&offenders, args.output.as_deref()) + }; + written.unwrap_or_else(|e| die(format_args!("failed to write {}: {e}", fmt.name()))); } /// Decide whether GitHub Actions `::error` annotations should be diff --git a/big-code-analysis-cli/src/commands/check/explain.rs b/big-code-analysis-cli/src/commands/check/explain.rs new file mode 100644 index 000000000..e324f6116 --- /dev/null +++ b/big-code-analysis-cli/src/commands/check/explain.rs @@ -0,0 +1,508 @@ +//! `bca check --explain-threshold =`: what a candidate +//! limit would cost, at **both** tiers, without editing anything (#1169). +//! +//! Tightening a limit onto a cluster of existing values is free at the +//! hard tier and expensive at the soft one, and the hard-tier reading is +//! the one a person naturally takes. `bca check --threshold nargs=6` over +//! this repository reported zero offenders — every one was already +//! baselined — while the same limit written into `bca.toml` puts 73 +//! functions permanently inside the `0.95` soft band, because they sit at +//! exactly 6 and the band starts at 5.7. They cannot clear it; they *are* +//! the limit. +//! +//! `--threshold` cannot be made to show this. Its values are applied last +//! and absolutely, never scaled, so the one-command way to trial a +//! candidate limit is by contract the one way that has no soft tier to +//! report. Hence a separate surface. +//! +//! # Why the counts match a real run +//! +//! Everything downstream of the walk is the gate's own code, in the +//! gate's own order, because both run it through the same two helpers: +//! `collect_check_violations` (walk, empty-input guard, +//! [`apply_check_exclude`]) and `classify_check_violations` +//! ([`filter_by_baseline`], [`apply_changed_only`]). A stage added to +//! the gate therefore reaches the preview too, rather than compiling +//! here while quietly predicting a different run. In-source suppression +//! markers are honoured inside the walk exactly as they always are. The +//! only deliberate difference is that baseline-covered offenders are +//! *kept* (tagged `Covered`) instead of dropped, because the split +//! between "already baselined" and "new" is the number a reviewer +//! weighs — 134 soft offenders of which 61 are baselined means 73 new +//! entries. +//! +//! That split is safe to read off a single walk because +//! [`Baseline::classify`](crate::baseline::Baseline::classify) is a +//! function of the offender's path, symbol, metric, and *value* — never +//! of the limit it breached. One walk at the soft limit therefore yields +//! coverage tags that are correct for both tiers. + +use super::super::*; +use super::*; + +/// Minimum number of soft-band offenders before a shared value is called +/// a cluster. Below ten, "they all sit at N" describes a handful rather +/// than a population, and the offender counts printed above already say +/// everything a reviewer needs. +const CLUSTER_MIN_FUNCTIONS: usize = 10; + +/// Minimum share of the soft band a single value must account for to be +/// called a cluster. A simple majority is the point at which converging +/// the limit onto that value moves most of the band at once — which is +/// the cost this report exists to surface. +const CLUSTER_MIN_SHARE: f64 = 0.5; + +/// Run the candidate-limit preview. This *replaces* the gate: nothing +/// here consults or produces an exit code, so the invocation exits 0 +/// unless a tool error kills it first. +pub(crate) fn run_explain_thresholds( + globals: GlobalOpts, + args: &CheckArgs, + layers: &ParsedThresholds, + tier: TierSpec, + preproc: Option>, +) { + reject_summary_file_path(args); + let candidates = candidate_limits(args); + // The soft band is derived from the candidate, so the tier the user + // asked to *gate* at does not apply — the report covers both tiers + // regardless. A `--tier=soft=R` still pins the ratio; the `hard` + // default leaves it at `DEFAULT_SOFT_HEADROOM`, which is what + // `make self-scan-headroom` and every other proportional soft gate + // use. + let ratio = tier.ratio(); + let resolved = Arc::new(build_candidate_gate(layers, &candidates, ratio)); + + let CollectedViolations { + violations, scope, .. + } = collect_check_violations(globals, args, preproc, Arc::clone(&resolved)); + let pairs = classify_check_violations( + violations, + args, + scope.as_ref(), + // The provenance the *user's* tier would stamp, not the soft tier + // this walk ran at: the walk is soft only to collect a superset, + // and warning that the baseline may under-cover would report an + // artifact of that implementation choice as a fact about the run + // being previewed. + resolve_provenance(tier, !layers.soft.is_empty()), + // Keep baseline-covered offenders instead of dropping them: the + // split between "already baselined" and "new" is the number the + // report exists to show. + true, + ); + + let context = CandidateGate { + layers, + resolved: &resolved, + ratio, + soft_table_applies: soft_table_applies(layers, &candidates), + }; + let report: Vec = candidates + .iter() + .map(|(metric, candidate)| context.explain(metric, *candidate, &pairs)) + .collect(); + write_report(&report); +} + +/// The candidate configuration one preview run shares across every metric +/// it explains: the merged manifest layers the candidates were spliced +/// into, the gate resolved from them, and the proportional soft ratio in +/// effect. Grouped because all three answer one question — "what would +/// this limit resolve to, and where" — and every per-metric lookup below +/// needs the same three. +struct CandidateGate<'a> { + layers: &'a ParsedThresholds, + resolved: &'a LanguageThresholds, + ratio: Option, + /// Whether any *explained* metric carries a `[thresholds.soft]` + /// entry. [`resolve_tier`]'s soft branch is all-or-nothing per + /// table: one such entry switches the whole table into merge mode, + /// and every other explained metric then inherits its hard limit + /// with no ratio applied at all. + soft_table_applies: bool, +} + +impl CandidateGate<'_> { + /// Tally one candidate limit over the offenders the shared walk + /// produced. + fn explain( + &self, + metric: &str, + candidate: f64, + pairs: &[(Violation, Option)], + ) -> CandidateOutcome { + // Resolving under the caller's spelling is also the proof that it + // matches `Violation::metric`: the gate yields each entry's + // registry name, so a hit means the two strings are the same one + // and the filter below cannot silently select nothing. + let global_soft = resolved_limit(self.resolved.global(), metric).unwrap_or_else(|| { + die(format_args!( + "--explain-threshold {metric}: candidate limit did not resolve to a gated metric" + )) + }); + let mut outcome = CandidateOutcome { + metric: metric.to_owned(), + candidate, + hard: TierTally::new(candidate), + soft: TierTally::new(global_soft), + soft_derivation: self.soft_derivation(metric), + band_values: Vec::new(), + language_overrides: self.language_overrides(metric, candidate, global_soft), + }; + // `v.suppressed` is only ever set under `--report-suppressed`, + // which keeps marker-silenced offenders in the stream for the + // SARIF document. `run_check` partitions them away from the gate; + // dropping them here is the same partition, so the preview counts + // what the gate would count under the user's own flags rather + // than what one report format happens to surface. + for (v, coverage) in pairs + .iter() + .filter(|(v, _)| v.metric == outcome.metric && !v.suppressed) + { + // Every record here already breaches its language's soft + // limit — that is what the walk gated on. The hard tier is the + // subset that also breaches that language's candidate ceiling, + // which the walk stamped on the violation. + outcome.soft.record(coverage.as_ref()); + let hard_breach = v + .hard_limit + .is_some_and(|ceiling| breaches_limit(v.value, ceiling, v.lower_is_worse)); + if hard_breach { + outcome.hard.record(coverage.as_ref()); + } else { + outcome.band_values.push(v.value); + } + } + outcome + } + + /// Where this metric's soft limit came from. A `[thresholds.soft]` + /// entry overrides the proportional ratio, so the two are exclusive + /// — and a soft table naming *some other* explained metric suppresses + /// the ratio for this one without giving it a band of its own. + fn soft_derivation(&self, metric: &str) -> SoftDerivation { + if self.layers.soft.contains_key(metric) { + SoftDerivation::Table + } else if self.soft_table_applies { + SoftDerivation::Inherited + } else { + SoftDerivation::Ratio(self.ratio.unwrap_or(DEFAULT_SOFT_HEADROOM)) + } + } + + /// Languages gating this metric at something other than the candidate. + // Both comparands are threshold *limits*, not measurements: `hard` is + // a config value verbatim and `soft` is `scale_threshold` applied to + // one, so two limits derived the same way from the same base are + // bit-identical and an exact comparison is the contract. Mirrors the + // module-level allow in `crate::threshold_lang`. + #[allow(clippy::float_cmp)] + fn language_overrides( + &self, + metric: &str, + candidate: f64, + global_soft: f64, + ) -> Vec { + self.resolved + .languages() + .filter_map(|(slug, set)| { + let soft = resolved_limit(set, metric)?; + let hard = self + .layers + .lang + .get(slug) + .and_then(|overrides| overrides.get(metric)) + .copied() + .unwrap_or(candidate); + // A language table that overrides only *other* metrics + // resolves this one exactly as the global set does; the + // candidate applies there in full, so it earns no line. + (hard != candidate || soft != global_soft).then_some(LanguageOverride { + slug, + hard, + soft, + }) + }) + .collect() + } +} + +/// One tier's share of the answer. +struct TierTally { + limit: f64, + total: usize, + /// Offenders that matched a `--baseline` entry, whether or not the + /// value worsened against it. The complement (`total - baselined`) is + /// the number of *new* entries a baseline refresh at this limit would + /// add, which is the figure a reviewer weighs. + baselined: usize, +} + +impl TierTally { + fn new(limit: f64) -> Self { + Self { + limit, + total: 0, + baselined: 0, + } + } + + fn record(&mut self, coverage: Option<&Coverage>) { + self.total += 1; + if matches!( + coverage, + Some(Coverage::Covered { .. } | Coverage::Regressed { .. }) + ) { + self.baselined += 1; + } + } + + fn new_entries(&self) -> usize { + self.total - self.baselined + } +} + +/// The whole answer for one candidate limit. +struct CandidateOutcome { + metric: String, + candidate: f64, + hard: TierTally, + soft: TierTally, + /// How the soft limit was derived, for the report line. + soft_derivation: SoftDerivation, + /// Values of the offenders that breach the soft band but *not* the + /// candidate ceiling — the population the candidate would newly place + /// in the early-warning band. Cluster detection runs over these. + band_values: Vec, + /// Languages whose `[thresholds.lang.]` table keeps its own + /// limits for this metric. Their files are counted against those + /// numbers, not the candidate. + language_overrides: Vec, +} + +impl CandidateOutcome { + /// The single value most of the soft band sits on, when there is one. + fn cluster(&self) -> Option<(f64, usize)> { + let band = self.band_values.len(); + if band < CLUSTER_MIN_FUNCTIONS { + return None; + } + let mut sorted = self.band_values.clone(); + sorted.sort_by(f64::total_cmp); + let (value, run) = sorted + .chunk_by(|a, b| a.total_cmp(b).is_eq()) + .map(|run| (run[0], run.len())) + .max_by_key(|(_, len)| *len)?; + #[allow(clippy::cast_precision_loss)] + let share = run as f64 / band as f64; + (share >= CLUSTER_MIN_SHARE).then_some((value, run)) + } +} + +/// How a candidate's soft limit was derived, for the report line. +enum SoftDerivation { + /// A `[thresholds.soft]` entry names this metric, so the soft limit + /// is that table's value and no ratio was applied to it. + Table, + /// A `[thresholds.soft]` table is in force but names only *other* + /// explained metrics. `resolve_tier` merges such a table onto the + /// hard limits rather than scaling them, so this metric keeps its + /// hard limit verbatim and has no soft band at all — reporting a + /// ratio here would name an arithmetic step that never ran. + Inherited, + /// No soft table applies, so the soft limit is the candidate scaled + /// by the proportional ratio in effect. + Ratio(f64), +} + +impl std::fmt::Display for SoftDerivation { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Table => f.write_str("[thresholds.soft]"), + Self::Inherited => f.write_str( + "no soft band; [thresholds.soft] names other metrics, so the hard limit stands", + ), + Self::Ratio(ratio) => write!(f, "{}x", MetricScalar(*ratio)), + } + } +} + +/// One language gating this metric at something other than the +/// candidate. Named fields rather than a positional `(slug, hard, +/// soft)`: the two limits are same-typed, and transposing them in a +/// tuple compiles and prints a plausible-but-wrong note line. +struct LanguageOverride { + slug: &'static str, + hard: f64, + soft: f64, +} + +/// Reject an explicit `--summary-file ` alongside the preview. +/// +/// `--explain-threshold` returns before `emit_check_results`, so the +/// markdown digest is never appended — the user names a destination and +/// finds it untouched, which is the silent no-op the structurally +/// identical `--output` is already a hard usage error for. +/// +/// Not expressible in the flag's `conflicts_with_all`, which fires on +/// the *presence* of `--summary-file` and would take the keyword forms +/// with it. `auto` — the default, and what a GHA workflow leaves +/// implicit — must keep working: it names no destination of its own, so +/// producing no step summary is the same thing every other non-gating +/// run does. Only an explicit path is a promise this command breaks. +fn reject_summary_file_path(args: &CheckArgs) { + if let Some(SummaryFile::Path(path)) = &args.summary_file { + die(format_args!( + "--explain-threshold cannot be used with --summary-file {}: the \ + preview replaces the gate, so there are no results to digest \ + into that file", + path.display() + )); + } +} + +/// Resolve `--explain-threshold` into canonical `metric -> limit` pairs, +/// rejecting the two ways the request can contradict itself. +fn candidate_limits(args: &CheckArgs) -> BTreeMap { + let mut candidates: BTreeMap = BTreeMap::new(); + for (name, limit) in canonical_cli_thresholds("--explain-threshold", &args.explain_thresholds) { + // Two candidate limits for one metric cannot both be previewed in + // one walk, and silently keeping the last would answer a question + // the user did not ask. + if let Some(previous) = candidates.insert(name.clone(), limit) { + die(format_args!( + "--explain-threshold {name}={} conflicts with --explain-threshold \ + {name}={}: preview one candidate limit per metric per run", + MetricScalar(limit), + MetricScalar(previous), + )); + } + } + // A `--threshold` override is absolute and never scaled, so it has no + // soft tier — the exact gap this flag exists to close. Letting one sit + // alongside a candidate for the same metric would make the reported + // soft limit depend on which layer won, so reject the pairing. + for (name, _) in canonical_cli_thresholds("--threshold", &args.thresholds) { + if candidates.contains_key(&name) { + die(format_args!( + "--threshold {name}=… and --explain-threshold {name}=… name the same \ + metric; a --threshold limit is absolute and has no soft tier to \ + preview, so pass only --explain-threshold" + )); + } + } + candidates +} + +/// Whether any `[thresholds.soft]` entry survives narrowing to the +/// explained metrics. +/// +/// `build_candidate_gate` applies exactly that narrowing before +/// `resolve_tier` sees the table, and `resolve_tier`'s soft branch is +/// all-or-nothing per table — so this is the emptiness test that decides +/// which of its two soft behaviours the preview actually ran. +fn soft_table_applies(layers: &ParsedThresholds, candidates: &BTreeMap) -> bool { + candidates + .keys() + .any(|metric| layers.soft.contains_key(metric)) +} + +/// The limit `set` resolved for `metric`, or `None` when it gates no such +/// metric. +fn resolved_limit(set: &ThresholdSet, metric: &str) -> Option { + set.iter() + .find_map(|(name, limit)| (name == metric).then_some(limit)) +} + +/// Render the preview to stdout. The report is this invocation's product +/// — per the `emit_check_results` stream contract that means stdout, so +/// `| rg`, `| head`, and `2>/dev/null` all reach it. +fn write_report(report: &[CandidateOutcome]) { + // Said once, on stderr: the preview never gates, so a bare exit 0 must + // not read as "this tree is clean at the candidate limit". + eprintln!("bca: --explain-threshold is a preview; no gate ran and the exit code is always 0"); + let mut stdout = BufWriter::new(std::io::stdout().lock()); + let written = report + .iter() + .try_for_each(|outcome| write_outcome(&mut stdout, outcome)) + .and_then(|()| stdout.flush()); + die_unless_broken_pipe(written, "writing threshold preview"); +} + +fn write_outcome(out: &mut impl Write, outcome: &CandidateOutcome) -> std::io::Result<()> { + let CandidateOutcome { + metric, + candidate, + hard, + soft, + soft_derivation, + .. + } = outcome; + writeln!( + out, + "{metric}: candidate limit {}", + MetricScalar(*candidate) + )?; + writeln!(out, " hard tier {}", tier_line(hard, None))?; + writeln!( + out, + " soft tier {}", + tier_line(soft, Some(soft_derivation)) + )?; + for over in &outcome.language_overrides { + writeln!( + out, + " note: [thresholds.lang.{}] keeps {metric} at {} (soft {}); its \ + files are counted against that, not the candidate", + over.slug, + MetricScalar(over.hard), + MetricScalar(over.soft), + )?; + } + match outcome.cluster() { + Some((value, count)) => write_cluster(out, outcome, value, count), + None => Ok(()), + } +} + +/// One tier's row. `derivation` names where a limit that was *derived* +/// came from; the hard tier passes `None`, because the candidate is the +/// number the user typed. +fn tier_line(tally: &TierTally, derivation: Option<&SoftDerivation>) -> String { + let derivation = derivation.map_or_else(String::new, |d| format!(", {d}")); + format!( + "(limit {}{derivation}): {} offenders, {} already baselined, {} new", + MetricScalar(tally.limit), + tally.total, + tally.baselined, + tally.new_entries(), + ) +} + +/// The finding the whole command exists for: a candidate limit that lands +/// on top of a population puts that population in the soft band by +/// construction, because the soft tier measures distance to the limit. +fn write_cluster( + out: &mut impl Write, + outcome: &CandidateOutcome, + value: f64, + count: usize, +) -> std::io::Result<()> { + let band = outcome.band_values.len(); + let converged = if value.total_cmp(&outcome.candidate).is_eq() { + " — the candidate limit itself" + } else { + "" + }; + writeln!( + out, + " cluster: {count} of {band} soft-band offenders sit at exactly {}{converged}. \ + The soft tier measures distance to the limit, so a limit of {} places them \ + inside the {} band by construction and none of them can clear it without \ + real work.", + MetricScalar(value), + MetricScalar(outcome.candidate), + MetricScalar(outcome.soft.limit), + ) +} diff --git a/big-code-analysis-cli/src/commands/check/thresholds.rs b/big-code-analysis-cli/src/commands/check/thresholds.rs index e91db805a..7e938b812 100644 --- a/big-code-analysis-cli/src/commands/check/thresholds.rs +++ b/big-code-analysis-cli/src/commands/check/thresholds.rs @@ -98,32 +98,23 @@ pub(crate) fn resolve_check_output_format(args: &mut CheckArgs) { args.output_format = Some(fmt); } -/// Validate `--output` / `--output-format` pairing, then resolve the -/// effective threshold sets per the documented resolution order -/// (#373/#374/#375/#380): the manifest `[thresholds]` base, the -/// `--config` file merged on top (keys win on collision), the -/// per-language `[thresholds.lang.]` overrides layered per metric -/// (#1141), the tier resolution (hard verbatim, or soft via -/// `[thresholds.soft]` / `--headroom`), and finally the absolute -/// `--threshold` CLI overrides. Dies if no thresholds were configured. -/// The result is wrapped in `Arc` so it can be cloned into each walker -/// worker's `Config`. -pub(crate) fn validate_and_build_thresholds( - args: &mut CheckArgs, +/// Merge the two file-sourced threshold layers into one set of +/// unresolved tables: the manifest `[thresholds]` table (empty when no +/// `bca.toml` was discovered) with `--config` layered on top, its keys +/// winning on collision so every existing recipe is preserved. The hard, +/// soft, and per-language layers all merge the same way — per-language +/// nested one level deeper, so a `--config` override of one metric for +/// one language leaves that language's other limits alone. +/// +/// Split out of [`validate_and_build_thresholds`] so the +/// `--explain-threshold` preview (`super::explain`) starts from exactly +/// the same merged tables the gate does rather than re-deriving them: a +/// preview whose configuration differs from the run it predicts is worse +/// than none. +pub(crate) fn merge_threshold_layers( + args: &CheckArgs, base_thresholds: ParsedThresholds, - tier: TierSpec, -) -> ResolvedThresholds { - // Resolve the `--output` / `--format` pairing before the walk so a - // misconfigured invocation fails fast instead of after a full parse. - resolve_check_output_format(args); - - // Layer 1: the manifest `[thresholds]` table (empty when no - // `bca.toml` was discovered). Layer 2: `--config` merges on top, - // its keys winning on collision, preserving every existing recipe. - // The hard, soft, and per-language layers all merge the same way — - // per-language nested one level deeper, so a `--config` override of - // one metric for one language leaves that language's other limits - // alone. +) -> ParsedThresholds { let ParsedThresholds { mut hard, mut soft, @@ -137,6 +128,28 @@ pub(crate) fn validate_and_build_thresholds( lang.entry(slug).or_default().extend(overrides); } } + ParsedThresholds { hard, soft, lang } +} + +/// Validate `--output` / `--output-format` pairing, then resolve the +/// effective threshold sets per the documented resolution order +/// (#373/#374/#375/#380): the `merged` manifest + `--config` base from +/// [`merge_threshold_layers`], the per-language +/// `[thresholds.lang.]` overrides layered per metric (#1141), the +/// tier resolution (hard verbatim, or soft via `[thresholds.soft]` / +/// `--headroom`), and finally the absolute `--threshold` CLI overrides. +/// Dies if no thresholds were configured. The result is wrapped in `Arc` +/// so it can be cloned into each walker worker's `Config`. +pub(crate) fn validate_and_build_thresholds( + args: &mut CheckArgs, + merged: ParsedThresholds, + tier: TierSpec, +) -> ResolvedThresholds { + // Resolve the `--output` / `--format` pairing before the walk so a + // misconfigured invocation fails fast instead of after a full parse. + resolve_check_output_format(args); + + let ParsedThresholds { hard, soft, lang } = merged; // The soft ratio (the `RATIO` in `--tier=soft=RATIO`) was already // validated to `(0, 1]` by `TierSpec::from_str` at parse time, so a @@ -147,7 +160,8 @@ pub(crate) fn validate_and_build_thresholds( // branch the tier resolver takes. let soft_table_present = !soft.is_empty(); - let layers = SharedLayers::new(&args.thresholds, tier, &soft, &lang, &hard); + let cli = canonical_cli_thresholds("--threshold", &args.thresholds); + let layers = SharedLayers::new(&cli, tier, &soft, &lang, &hard); let thresholds = layers.resolve_all(&hard, &lang); if thresholds.is_empty() { die( @@ -171,9 +185,105 @@ pub(crate) fn validate_and_build_thresholds( } } +/// Canonicalise the `--threshold` metric names so the CLI layer merges +/// with the manifest, `--config`, and `[thresholds.lang.]` layers +/// by *metric* rather than by spelling (#1165) — `--threshold ploc=100` +/// must override a manifest `"loc.ploc"`, not gate it a second time. +/// +/// Deliberately **not** done inside `parse_cli_threshold`, which is the +/// clap `value_parser` for `--threshold`. That parser owns the +/// `metric=limit` *syntax*; resolving a name against the metric registry +/// is a semantic check, and the sibling semantic check — +/// `--threshold not_a_metric=1`, rejected by +/// [`ThresholdSet::build_tiered`] with the did-you-mean list — already +/// lives here. Split across the two layers, two adjacent typo classes +/// would report through different surfaces: an ambiguous family head +/// wrapped as clap's `invalid value '…' for '--threshold '`, +/// an unknown metric as the plain `error:` form. +/// +/// Note this is *not* an exit-code argument. `exit_clap_error` +/// (#561/#594) already remaps clap's usage exit 2 to `EXIT_TOOL_ERROR` +/// precisely so `bca check`'s exit 2 stays reserved for "thresholds +/// exceeded", so either placement exits 1. +/// +/// Repeating one metric stays last-wins, as it already is for two +/// occurrences of the same spelling; only the enclosing tables reject a +/// metric named twice. +/// +/// `flag` is the spelling to blame in the diagnostic. It is a parameter +/// rather than a literal because `--explain-threshold` (#1169) routes +/// its own values through here, and a hardcoded `--threshold` named a +/// flag that invocation never passed. +pub(crate) fn canonical_cli_thresholds(flag: &str, raw: &[(String, f64)]) -> Vec<(String, f64)> { + raw.iter() + .map(|(name, limit)| { + let canonical = crate::metric_alias::normalize_for_check(name) + .unwrap_or_else(|e| die(format_args!("{flag}: {e}"))); + (canonical.into_owned(), *limit) + }) + .collect() +} + /// The per-metric override tables, keyed by canonical language slug. type LanguageOverrides = BTreeMap<&'static str, BTreeMap>; +/// Resolve the gate a `--explain-threshold` preview walks with (#1169). +/// +/// `candidates` replaces the global `[thresholds]` hard limits for +/// exactly the metrics being explained, and every other metric is dropped +/// — the preview reports only what it was asked about, and narrowing the +/// set narrows the metric families the walk computes with it (#1113). +/// +/// Resolved at the **soft** tier on purpose. The soft band is the more +/// permissive of the two (it fires *before* the hard gate), so one walk +/// against it collects a superset already containing every hard-tier +/// offender, and each emitted [`Violation`] carries both numbers: +/// `limit` is that language's resolved soft limit and `hard_limit` its +/// candidate ceiling. Counting the hard tier is then a `breaches_limit` +/// filter over the records in hand rather than a second parse of the +/// tree. +/// +/// Per-language `[thresholds.lang.]` overrides of an explained +/// metric are kept as they are: a candidate *global* limit does not +/// change what a language that overrode the metric gates at, and each +/// language's soft band is derived from its own limit by +/// [`SharedLayers::resolve_one`]. That is what keeps the counts exact on +/// the trees #1141 exists for. +pub(crate) fn build_candidate_gate( + layers: &ParsedThresholds, + candidates: &BTreeMap, + ratio: Option, +) -> LanguageThresholds { + let soft = retain_explained(&layers.soft, candidates); + let lang: LanguageOverrides = layers + .lang + .iter() + .map(|(slug, overrides)| (*slug, retain_explained(overrides, candidates))) + .filter(|(_, kept)| !kept.is_empty()) + .collect(); + // No `--threshold` layer: an absolute CLI override is never scaled, + // so letting one through would hand the preview a metric with no soft + // tier to report. `run_explain_thresholds` rejects the overlap up + // front instead. + let shared = SharedLayers::new(&[], TierSpec::Soft(ratio), &soft, &lang, candidates); + shared.resolve_all(candidates, &lang) +} + +/// Drop every entry of one threshold layer whose metric is not being +/// explained. Shared by the `[thresholds.soft]` table and each +/// `[thresholds.lang.]` table so both are narrowed by the same +/// rule. +fn retain_explained( + table: &BTreeMap, + candidates: &BTreeMap, +) -> BTreeMap { + table + .iter() + .filter(|(name, _)| candidates.contains_key(name.as_str())) + .map(|(name, value)| (name.clone(), *value)) + .collect() +} + /// The threshold layers every table in one run shares, separated from /// the per-table hard limits they are applied to. /// @@ -337,8 +447,12 @@ pub(crate) fn resolve_tier( ); } let mut out = hard; - for limit in out.values_mut() { - *limit = scale_threshold(*limit, ratio); + for (name, limit) in &mut out { + // Keys arrive canonical (#1165), which is what makes this + // direction lookup correct by construction rather than by luck: + // scaling a lower-is-worse floor the higher-is-worse way inverts + // the whole tier (#1166). + *limit = scale_threshold(*limit, ratio, metric_is_lower_is_worse(name)); } out } diff --git a/big-code-analysis-cli/src/commands/exemptions.rs b/big-code-analysis-cli/src/commands/exemptions.rs index b78b6a722..556e08c4e 100644 --- a/big-code-analysis-cli/src/commands/exemptions.rs +++ b/big-code-analysis-cli/src/commands/exemptions.rs @@ -138,13 +138,10 @@ pub(crate) fn resolve_baseline_section(args: &ExemptionsArgs) -> BaselineSection .into_iter() .map(BaselineRow::from) .collect(); - entries.sort_by(|a, b| { - (a.path.as_str(), a.start_line, a.metric.as_str()).cmp(&( - b.path.as_str(), - b.start_line, - b.metric.as_str(), - )) - }); + // Order on the full identity (see [`BaselineIdentity`]). + // `diff_entries` walks a `HashMap`, so the input order is arbitrary + // and every displayed row needs a total order to be reproducible. + entries.sort_by(baseline::cmp_identity); BaselineSection { path: path.display().to_string(), entries, diff --git a/big-code-analysis-cli/src/commands/init.rs b/big-code-analysis-cli/src/commands/init.rs index b0f45bcbd..3c0c134a2 100644 --- a/big-code-analysis-cli/src/commands/init.rs +++ b/big-code-analysis-cli/src/commands/init.rs @@ -145,6 +145,7 @@ pub(crate) fn scaffold_baseline( tuning: crate::WalkTuningArgs::default(), preproc: crate::PreprocConsumeArgs::default(), thresholds: Vec::new(), + explain_thresholds: Vec::new(), config: Some(manifest_path.to_path_buf()), no_fail: false, no_suppress: false, diff --git a/big-code-analysis-cli/src/commands_tests.rs b/big-code-analysis-cli/src/commands_tests.rs index b6f7848d2..68124b012 100644 --- a/big-code-analysis-cli/src/commands_tests.rs +++ b/big-code-analysis-cli/src/commands_tests.rs @@ -330,6 +330,7 @@ fn base_check_args() -> CheckArgs { tuning: crate::WalkTuningArgs::default(), preproc: crate::PreprocConsumeArgs::default(), thresholds: Vec::new(), + explain_thresholds: Vec::new(), config: None, no_fail: false, no_suppress: false, @@ -711,7 +712,42 @@ fn effective_config_reflects_resolved_threshold_set() { #[test] #[allow(clippy::float_cmp)] // The exact rounded output is the contract under test. fn scale_threshold_trims_float_artifact_to_six_sig_figs() { - assert_eq!(scale_threshold(7.0, 0.95), 6.65); + assert_eq!(scale_threshold(7.0, 0.95, false), 6.65); +} + +/// The direction of the scaling is the whole of #1166, so pin both +/// arithmetics in one test: a higher-is-worse limit is a ceiling and +/// must come *down*; a lower-is-worse `mi.*` limit is a floor and must +/// go *up*. Asserted together so a future "simplify" cannot invert both +/// and stay green — which is exactly what a single-direction test would +/// have allowed. +#[test] +#[allow(clippy::float_cmp)] // The exact resolved limits are the contract. +fn scale_threshold_raises_a_floor_and_lowers_a_ceiling() { + // Ceiling: 15 * 0.9. Exact in decimal, and the float product is + // exact too, so this is a bit-exact expectation. + assert_eq!(scale_threshold(15.0, 0.9, false), 13.5); + // Floor: 20 / 0.9 = 22.222…, above the hard floor it derives from. + // Rounded up at 6 significant figures (see `scale_threshold`). + assert_eq!(scale_threshold(20.0, 0.9, true), 22.2223); + // The bug this pins: the floor must never land at or below its own + // hard limit, which `20 * 0.9 == 18` did. + assert!(scale_threshold(20.0, 0.9, true) > 20.0); +} + +/// The floor's last significant figure rounds **up**, never down: a +/// floor rounded down is a band that fires marginally late. Pins the +/// direction with a quotient whose 6-sig-fig nearest-rounding would go +/// the other way (`21.05263…` → `21.0526` to nearest, `21.0527` up). +#[test] +#[allow(clippy::float_cmp)] // The rounding direction is the contract. +fn scale_threshold_floor_rounds_up_not_to_nearest() { + let resolved = scale_threshold(20.0, 0.95, true); + assert_eq!(resolved, 21.0527); + assert!( + resolved >= 20.0 / 0.95, + "a floor must not resolve below the exact quotient; got {resolved}" + ); } /// Exact products must pass through untouched: `50_000 * 0.95` is @@ -721,7 +757,7 @@ fn scale_threshold_trims_float_artifact_to_six_sig_figs() { #[test] #[allow(clippy::float_cmp)] // The exact rounded output is the contract under test. fn scale_threshold_preserves_large_exact_products() { - assert_eq!(scale_threshold(50_000.0, 0.95), 47_500.0); + assert_eq!(scale_threshold(50_000.0, 0.95, false), 47_500.0); } /// `ratio == 1.0` is the documented no-op: every limit must survive @@ -730,8 +766,20 @@ fn scale_threshold_preserves_large_exact_products() { #[test] #[allow(clippy::float_cmp)] // ratio == 1.0 must be a bit-exact identity. fn scale_threshold_ratio_one_is_identity() { - for &limit in &[0.0, 7.0, 15.0, 300.0, 50_000.0] { - assert_eq!(scale_threshold(limit, 1.0), limit); + // `8.3` is the case a grid-exact list cannot see: its double sits a + // hair above the decimal, so `8.3 * 1e5` is `830000.0000000001` and + // a bare `ceil` in the floor direction promotes it to `8.30001` — + // a soft floor strictly *above* the hard one on a run documented as + // parity. Every other limit here is already on the sig-fig grid, so + // it passes whether or not the snap exists. + for &limit in &[0.0, 7.0, 8.3, 15.0, 300.0, 50_000.0] { + for lower_is_worse in [false, true] { + assert_eq!( + scale_threshold(limit, 1.0, lower_is_worse), + limit, + "limit {limit} lower_is_worse {lower_is_worse}", + ); + } } } @@ -741,7 +789,10 @@ fn scale_threshold_ratio_one_is_identity() { #[test] #[allow(clippy::float_cmp)] // A zero limit must stay bit-exact zero. fn scale_threshold_zero_limit_stays_zero() { - assert_eq!(scale_threshold(0.0, 0.5), 0.0); + assert_eq!(scale_threshold(0.0, 0.5, false), 0.0); + // `0 / 0.5` is also zero, so the floor direction hits the same + // short-circuit rather than the `log10(0) == -inf` maths. + assert_eq!(scale_threshold(0.0, 0.5, true), 0.0); } /// A subnormal-range limit must not poison the result with `NaN`: the @@ -751,8 +802,10 @@ fn scale_threshold_zero_limit_stays_zero() { /// later be rejected by `ThresholdSet::build` with a confusing error. #[test] fn scale_threshold_subnormal_limit_stays_finite() { - let scaled = scale_threshold(1e-320, 0.5); - assert!(scaled.is_finite(), "expected finite, got {scaled}"); + for lower_is_worse in [false, true] { + let scaled = scale_threshold(1e-320, 0.5, lower_is_worse); + assert!(scaled.is_finite(), "expected finite, got {scaled}"); + } } /// Minimal `CheckArgs` carrying only the `[check.exclude]` inputs under diff --git a/big-code-analysis-cli/src/exemptions.rs b/big-code-analysis-cli/src/exemptions.rs index eb931b24a..1264e802f 100644 --- a/big-code-analysis-cli/src/exemptions.rs +++ b/big-code-analysis-cli/src/exemptions.rs @@ -57,7 +57,16 @@ pub(crate) struct BaselineRow { pub(crate) qualified: String, pub(crate) metric: String, pub(crate) value: f64, - pub(crate) start_line: usize, + /// Line recorded by the baseline, absent for a v6+ entry whose + /// identity is unique (#1170) — such an entry pins no line, and the + /// renderers say so rather than inventing one. + pub(crate) start_line: Option, +} + +impl crate::baseline::BaselineIdentity for BaselineRow { + fn identity(&self) -> (&str, &str, &str, Option) { + (&self.path, &self.qualified, &self.metric, self.start_line) + } } impl From for BaselineRow { @@ -134,10 +143,14 @@ impl ExemptionsReport { if open_section(&mut out, &header, " (none)\n", section.entries.is_empty()) { for e in §ion.entries { let path = strip_path_prefix(&e.path, strip_prefix); + // The `:line` suffix is dropped rather than filled + // with a placeholder when the entry pins no line, so + // the field never reads as a line number that is not + // there (#1170). + let line = e.start_line.map_or_else(String::new, |l| format!(":{l}")); let _ = writeln!( out, - " {path}:{} {} {} {}", - e.start_line, + " {path}{line} {} {} {}", e.qualified, e.metric, MetricScalar(e.value), @@ -200,7 +213,10 @@ impl ExemptionsReport { out, "| {} | {} | {} | {} | {} |", escape_gfm_cell(path), - e.start_line, + // `-` for an entry that pins no line (#1170); a + // table column cannot simply omit its cell. + e.start_line + .map_or_else(|| "-".to_owned(), |l| l.to_string()), escape_gfm_cell(&e.qualified), escape_gfm_cell(&e.metric), MetricScalar(e.value), @@ -385,7 +401,9 @@ struct JsonMarker<'a> { #[derive(Serialize)] struct JsonBaseline<'a> { path: &'a str, - line: usize, + /// Omitted, not nulled, when the entry pins no line (#1170). + #[serde(skip_serializing_if = "Option::is_none")] + line: Option, qualified: &'a str, metric: &'a str, value: f64, diff --git a/big-code-analysis-cli/src/exemptions_tests.rs b/big-code-analysis-cli/src/exemptions_tests.rs index f39df4a9d..6549dc3c3 100644 --- a/big-code-analysis-cli/src/exemptions_tests.rs +++ b/big-code-analysis-cli/src/exemptions_tests.rs @@ -75,7 +75,7 @@ fn sample_report() -> ExemptionsReport { qualified: "write_language_section".to_owned(), metric: "cognitive".to_owned(), value: 29.0, - start_line: 88, + start_line: Some(88), }], }), } @@ -185,7 +185,7 @@ fn markdown_escapes_pipe_in_path_symbol_and_function_cells() { qualified: "Mod::sym|bol".to_owned(), metric: "cognitive".to_owned(), value: 12.0, - start_line: 7, + start_line: Some(7), }], }), }; @@ -217,6 +217,66 @@ fn markdown_escapes_pipe_in_path_symbol_and_function_cells() { ); } +/// A v6 baseline records `start_line` only for an entry whose identity +/// is shared, so most entries render through a `None` branch that no +/// other fixture in this file reaches — `sample_report` pins `Some(88)` +/// and the escaping fixture `Some(7)` (#1170). +/// +/// All three formats are asserted together because each drops the line +/// a different way — omit the suffix, render `-`, omit the key — and +/// each is a separate branch. A `unwrap_or(0)` in the text renderer, a +/// blank markdown cell, or a `"line": null` in the JSON are all changes +/// the rest of the suite is blind to. +#[test] +fn a_lineless_baseline_entry_drops_the_line_in_every_format() { + let report = ExemptionsReport { + markers: None, + excludes: None, + baseline: Some(BaselineSection { + path: ".bca-baseline.toml".to_owned(), + entries: vec![BaselineRow { + path: "src/lonely.rs".to_owned(), + qualified: "only_one".to_owned(), + metric: "cognitive".to_owned(), + value: 21.0, + start_line: None, + }], + }), + }; + + // Text: the `:line` suffix is absent, not filled with a placeholder + // that would read as a real line number. + let text = report.render(OutputFormat::Text, "").expect("tty render"); + assert!( + text.contains(" src/lonely.rs only_one cognitive 21\n"), + "got: {text}" + ); + assert!( + !text.contains("src/lonely.rs:"), + "no colon-line suffix may survive; got: {text}" + ); + + // Markdown: a table row cannot omit a cell, so the column reads `-`. + let md = report + .render(OutputFormat::Markdown, "") + .expect("markdown render"); + assert!( + md.contains("| src/lonely.rs | - | only_one | cognitive | 21 |"), + "got: {md}" + ); + + // JSON: the key is omitted rather than nulled, so a consumer reads + // "no line recorded" as absence. + let json = report.render(OutputFormat::Json, "").expect("json render"); + let v: Value = serde_json::from_str(&json).expect("valid JSON"); + let entry = &v["suppressions"]["baseline"][0]; + assert_eq!(entry["qualified"], "only_one"); + assert!( + entry.get("line").is_none(), + "the line key must be omitted, not nulled; got: {json}" + ); +} + #[test] fn json_nests_three_sections_under_suppressions_envelope() { let out = sample_report() diff --git a/big-code-analysis-cli/src/threshold_lang.rs b/big-code-analysis-cli/src/threshold_lang.rs index 538070061..85f7a87ba 100644 --- a/big-code-analysis-cli/src/threshold_lang.rs +++ b/big-code-analysis-cli/src/threshold_lang.rs @@ -26,7 +26,7 @@ use std::sync::Arc; use big_code_analysis::{LANG, Metric}; -use crate::thresholds::{ThresholdSet, threshold_scalar}; +use crate::thresholds::{ThresholdSet, insert_canonical_limit, threshold_scalar}; /// Reserved key inside `[thresholds]` that introduces the per-language /// override sub-tables (`[thresholds.lang.]`). The nesting under a @@ -97,7 +97,8 @@ fn parse_one_language_table( [thresholds.lang.{slug}] without one" )); } - limits.insert(name.clone(), threshold_scalar(&context, name, value)?); + let limit = threshold_scalar(&context, name, value)?; + insert_canonical_limit(&mut limits, &context, name, limit)?; } Ok(limits) } diff --git a/big-code-analysis-cli/src/threshold_soft.rs b/big-code-analysis-cli/src/threshold_soft.rs index b6223c6f3..e318b5ae8 100644 --- a/big-code-analysis-cli/src/threshold_soft.rs +++ b/big-code-analysis-cli/src/threshold_soft.rs @@ -35,9 +35,12 @@ pub(crate) enum SoftLimit { impl SoftLimit { /// Resolve to a concrete limit. `Absolute` ignores `hard`; `Scale` - /// multiplies the metric's hard limit, erroring when no hard limit - /// exists for the metric to scale (a scale factor relative to - /// nothing is meaningless). + /// tightens the metric's hard limit by its factor, erroring when no + /// hard limit exists for the metric to scale (a scale factor + /// relative to nothing is meaningless). + /// + /// `name` must already be canonical (#1165), because the scaling + /// direction is looked up from it — see [`scale_threshold`]. pub(crate) fn resolve(self, name: &str, hard: Option) -> Result { match self { Self::Absolute(value) => Ok(value), @@ -49,7 +52,11 @@ impl SoftLimit { absolute soft limit or add a hard limit first" ) })?; - Ok(scale_threshold(base, factor)) + Ok(scale_threshold( + base, + factor, + crate::thresholds::metric_is_lower_is_worse(name), + )) } } } @@ -80,15 +87,45 @@ pub(crate) fn is_valid_scale_ratio(ratio: f64) -> bool { 0.0 < ratio && ratio <= 1.0 } -/// Scale a threshold `limit` by `ratio`, rounding to +/// Tighten a threshold `limit` by `ratio`, rounding to /// [`HEADROOM_SIG_FIGS`] significant figures. `ratio` is assumed already /// validated (see [`is_valid_scale_ratio`]) to lie in `(0, 1]`. Shared /// by the `--headroom` scalar path and the `[thresholds.soft]` /// scale-relative form so both round identically. -pub(crate) fn scale_threshold(limit: f64, ratio: f64) -> f64 { - let scaled = limit * ratio; +/// +/// `ratio` scales the *band*, not the number: it always makes the soft +/// tier stricter than the hard gate, and which arithmetic does that +/// depends on the metric's direction (#1166). A higher-is-worse limit is +/// a ceiling, so tightening it means lowering it — multiply. A +/// lower-is-worse `mi.*` limit is a *floor*, so tightening it means +/// **raising** it — divide. Multiplying a floor lowers it, which put the +/// early-warning band *below* the hard gate and made `--tier=soft` a +/// silent no-op for the whole `mi.*` family. +/// +/// The rounding is likewise direction-aware, and asymmetric on purpose. +/// A ceiling rounds to nearest: `limit * ratio` has an exact decimal +/// value that the float product misses by an ulp (`7 * 0.95` is +/// `6.6499999999999995`), so nearest-rounding *recovers* the true +/// product, and the resulting output parity is a contract (#373). A +/// floor's `limit / ratio` generally has no exact decimal at all +/// (`20 / 0.9` repeats), so the last figure is a real choice — and it +/// goes **up**, because a floor rounded down is a band that fires +/// marginally late, the same defect in miniature — but only when there +/// is a real remainder to round, see [`FLOOR_GRID_SNAP_ULPS`]. Rounding +/// up also keeps the resolved floor at or above the exact quotient, so +/// it can never land under the hard floor it is derived from and trip +/// [`ThresholdSet::build_tiered`](crate::thresholds::ThresholdSet::build_tiered)'s +/// soft-looser-than-hard guard. +pub(crate) fn scale_threshold(limit: f64, ratio: f64, lower_is_worse: bool) -> f64 { + let scaled = if lower_is_worse { + limit / ratio + } else { + limit * ratio + }; // `log10(0)` is `-inf`; short-circuit the degenerate inputs so the - // magnitude maths below only sees finite, non-zero values. + // magnitude maths below only sees finite, non-zero values. A zero + // `ratio` is rejected upstream, but were one to arrive the division + // yields an infinity that lands here rather than panicking. if scaled == 0.0 || !scaled.is_finite() { return scaled; } @@ -108,7 +145,37 @@ pub(crate) fn scale_threshold(limit: f64, ratio: f64) -> f64 { if !factor.is_finite() { return scaled; } - (scaled * factor).round() / factor + let ticks = scaled * factor; + if lower_is_worse { + ceil_off_grid(ticks) / factor + } else { + ticks.round() / factor + } +} + +/// How far from an exact sig-fig grid position a scaled floor may sit +/// and still count as being *on* it, in ULPs of the value itself. +/// +/// `limit / ratio` with `ratio == 1.0` reproduces `limit`, but the +/// double for a limit like `8.3` is a hair above the decimal it prints +/// as, so `8.3 * 1e5` is `830000.0000000001` — one ULP over the grid. +/// A bare `ceil` promotes that to the next whole tick and resolves the +/// soft floor to `8.30001`, which makes the documented `ratio == 1.0` +/// identity false and gates the soft tier above the hard one. Four ULPs +/// clears that single-ULP case with room to spare and is orders of +/// magnitude below the smallest genuine remainder a real ratio leaves +/// (`20 / 0.9` is `0.22` of a tick short). +const FLOOR_GRID_SNAP_ULPS: f64 = 4.0; + +/// `ticks.ceil()`, except that a value within [`FLOOR_GRID_SNAP_ULPS`] +/// of a grid position is already on it and rounds to it instead. +fn ceil_off_grid(ticks: f64) -> f64 { + let nearest = ticks.round(); + if (ticks - nearest).abs() <= FLOOR_GRID_SNAP_ULPS * f64::EPSILON * ticks.abs() { + nearest + } else { + ticks.ceil() + } } /// Parse one `[thresholds.soft]` value: a number (absolute) or a diff --git a/big-code-analysis-cli/src/thresholds.rs b/big-code-analysis-cli/src/thresholds.rs index a0b05536c..b93658243 100644 --- a/big-code-analysis-cli/src/thresholds.rs +++ b/big-code-analysis-cli/src/thresholds.rs @@ -255,6 +255,12 @@ pub(crate) fn parse_fail_above(s: &str) -> Result { /// Parse a single `--threshold metric=limit` token. Only one `=` is /// allowed, both sides must be non-empty, and `limit` must parse as a /// finite, non-negative `f64`. +/// +/// Syntax only: the metric name is passed through verbatim, and is +/// resolved against the registry — canonicalised, then checked for +/// existence — one layer down, where the manifest and `--config` names +/// are resolved too. See `canonical_cli_thresholds` for why the two +/// halves are not split across the clap boundary (#1165). pub(crate) fn parse_cli_threshold(s: &str) -> Result<(String, f64), String> { let (name, limit) = s .split_once('=') @@ -324,27 +330,82 @@ pub(crate) fn split_thresholds_table( let mut out = ParsedThresholds::default(); for (key, value) in raw { match key.as_str() { - SOFT_SUBTABLE_KEY => { - let table = value.as_table().ok_or_else(|| { - "[thresholds.soft] must be a table of `metric = ` entries" - .to_string() - })?; - for (name, sub) in table { - out.soft.insert(name.clone(), parse_soft_value(name, sub)?); - } - } + SOFT_SUBTABLE_KEY => out.soft = parse_soft_table(value)?, crate::threshold_lang::LANG_SUBTABLE_KEY => { out.lang = crate::threshold_lang::parse_language_tables(value)?; } _ => { - out.hard - .insert(key.clone(), threshold_scalar("[thresholds]", key, value)?); + let limit = threshold_scalar("[thresholds]", key, value)?; + insert_canonical_limit(&mut out.hard, "[thresholds]", key, limit)?; } } } Ok(out) } +/// Parse the `[thresholds.soft]` sub-table into its unresolved +/// [`SoftLimit`] entries, keyed by canonical metric id. +/// +/// Mirrors [`crate::threshold_lang::parse_language_tables`], the other +/// reserved sub-table of `[thresholds]`, so both nested layers are read +/// by a named parser rather than one inline loop and one delegation. +fn parse_soft_table(value: &toml::Value) -> Result, String> { + let table = value.as_table().ok_or_else(|| { + "[thresholds.soft] must be a table of `metric = ` entries".to_string() + })?; + let context = format!("[thresholds.{SOFT_SUBTABLE_KEY}]"); + let mut out = BTreeMap::new(); + for (name, sub) in table { + let limit = parse_soft_value(name, sub)?; + insert_canonical_limit(&mut out, &context, name, limit)?; + } + Ok(out) +} + +/// Insert `value` under `name`'s canonical metric id, so every threshold +/// map is keyed by the dotted registry id from the parse boundary onward +/// (#1165). +/// +/// The bare `bca diff --metric` spelling of a `loc` sub-metric is an +/// alias for the dotted threshold id (`sloc` == `loc.sloc`, #514). +/// Canonicalising where the maps are *built* — rather than where they +/// are consumed — is what makes the manifest, `--config`, +/// `[thresholds.lang.]`, and `--threshold` layers merge by *metric* +/// instead of by spelling. Keyed by the raw spelling, a merge kept both +/// and gated the same extractor twice: two offender lines for one +/// `(function, metric)` pair, and a `--print-effective-config` that +/// printed one of the two limits while the other fired. Every consumer +/// downstream — [`resolve_tier`](crate::commands::check::resolve_tier), +/// [`ThresholdSet::build_tiered`], `--print-effective-config` — may +/// therefore assume canonical keys. +/// +/// An ambiguous family head (`halstead`, `mi`) has no single threshold +/// scalar and is rejected here, as is a table naming one metric under two +/// spellings: silently keeping whichever key sorts last is the same +/// surprise this function exists to remove. +pub(crate) fn insert_canonical_limit( + into: &mut BTreeMap, + table: &str, + name: &str, + value: V, +) -> Result<(), String> { + let canonical = crate::metric_alias::normalize_for_check(name) + .map_err(|e| format!("{table} {e}"))? + .into_owned(); + if into.contains_key(&canonical) { + // Names both spellings rather than quoting the earlier key, + // which would read as `"loc.ploc": "loc.ploc" is already set` + // whenever the dotted form is the one written second. + return Err(format!( + "{table} {name:?}: {canonical:?} is already set in this table under its other \ + spelling; a bare `loc` sub-metric name and its dotted id (`ploc`, `loc.ploc`) \ + are one metric, so set it once" + )); + } + into.insert(canonical, value); + Ok(()) +} + /// Parse a hard-tier scalar limit. Accepts TOML integers and floats; /// `i64 -> f64` is exact for the small limits metrics carry in practice. /// `table` names the enclosing table for the error message, so a @@ -621,8 +682,13 @@ struct ResolvedThreshold { /// Maintainability Index family). Thin alias for the library catalog's /// [`big_code_analysis::metric_catalog::lower_is_worse`] — the single /// source of truth shared with the offender wording and Code Climate -/// severity inversion. -fn metric_is_lower_is_worse(name: &str) -> bool { +/// severity inversion, the soft-tier scaling direction (#1166), and the +/// per-function gate. +/// +/// `name` must be canonical — every layer that can introduce a threshold +/// key resolves the bare `diff --metric` alias spelling at its own parse +/// boundary (#1165), so an alias never reaches this lookup. +pub(crate) fn metric_is_lower_is_worse(name: &str) -> bool { big_code_analysis::metric_catalog::lower_is_worse(name) } @@ -671,13 +737,15 @@ impl ThresholdSet { ) -> Result { let mut entries = Vec::with_capacity(raw.len()); for (name, limit) in raw { - // Accept the bare `diff --metric` spelling as an alias for the - // dotted threshold id (issue #514): `sloc` -> `loc.sloc`. An - // ambiguous family head (`halstead`, `mi`, which have no single - // threshold scalar) is rejected here with a "did you mean" - // hint rather than guessing a sub-metric. - let canonical = crate::metric_alias::normalize_for_check(name)?; - let extractor = lookup_extractor(&canonical).ok_or_else(|| { + // Names arrive canonical: every layer that can introduce one + // — the manifest and `--config` tables, the per-language + // tables, the `--threshold` flags — resolves the bare + // `diff --metric` alias spelling at its own parse boundary + // (#1165, via `insert_canonical_limit`). Re-normalising here + // is what let a merge key two spellings of one metric and + // gate it twice, so this layer now takes canonical keys as an + // invariant rather than restoring it after the fact. + let extractor = lookup_extractor(name).ok_or_else(|| { let known = known_metric_names(); format!( "unknown threshold metric {name:?}{}; known metrics: {}", @@ -697,17 +765,10 @@ impl ThresholdSet { // absolute form, which per-language hard overrides make easy // to hit by accident (#1141). // - // Higher-is-worse metrics only. The lower-is-worse `mi.*` - // family is a *floor*, so tightening it means raising it — - // but `resolve_tier`'s blanket ratio multiplies every limit, - // which lowers an `mi.*` floor and therefore already emits a - // soft tier looser than its hard one for any pre-existing - // `[thresholds] mi.original = N` plus `--tier=soft`. That is - // a separate defect in the scaling direction (#1166); - // rejecting it here would fail those runs rather than fix - // them. Drop the guard once #1166 lands. - if !lower_is_worse - && let Some(hard) = hard_limit + // Both directions, since #1166: a lower-is-worse `mi.*` + // limit is a floor, so "looser" means *below* the hard floor + // — which is exactly what `breaches_limit` tests for it. + if let Some(hard) = hard_limit && breaches_limit(*limit, hard, lower_is_worse) { return Err(format!( diff --git a/big-code-analysis-cli/src/thresholds_tests.rs b/big-code-analysis-cli/src/thresholds_tests.rs index 88a478bee..2466a507b 100644 --- a/big-code-analysis-cli/src/thresholds_tests.rs +++ b/big-code-analysis-cli/src/thresholds_tests.rs @@ -111,26 +111,91 @@ fn build_accepts_zero_limit() { ThresholdSet::build(&raw).expect("zero limit is valid"); } +/// Parse a `[thresholds]` table the way both the manifest and +/// `--config` do, so alias/duplicate assertions exercise the real parse +/// boundary rather than a hand-built map. +fn split(toml_src: &str) -> Result { + let cfg: ThresholdConfig = toml::from_str(toml_src).expect("parses as TOML"); + split_thresholds_table(&cfg.thresholds) +} + /// Issue #514: the bare `bca diff --metric` spelling of a `loc` /// sub-metric is accepted as an alias and resolves to the dotted /// extractor, so a name copy-pasted from a `diff` run gates correctly. -#[test] -fn build_accepts_bare_loc_submetric_alias() { - let mut raw = BTreeMap::new(); - raw.insert("sloc".to_string(), 100.0); - let set = ThresholdSet::build(&raw).expect("bare loc alias resolves"); +/// +/// Issue #1165 moved the resolution from `ThresholdSet::build_tiered` to +/// the parse boundary, which is why this asserts on the parsed key and +/// not only on the built set: the key is what every later layer merges +/// against. +#[test] +fn parse_canonicalises_bare_loc_submetric_alias() { + let parsed = split("[thresholds]\nsloc = 100\n").expect("bare loc alias resolves"); + assert_eq!(parsed.hard.keys().collect::>(), ["loc.sloc"]); + let set = ThresholdSet::build(&parsed.hard).expect("canonical key builds"); let resolved: Vec<(&str, f64)> = set.iter().collect(); assert_eq!(resolved, [("loc.sloc", 100.0)]); } +/// The soft tier canonicalises too (#1165). Keyed by the raw spelling, a +/// `[thresholds.soft] sloc = "0.9x"` would find no hard limit named +/// `sloc` to scale and die with "no hard [thresholds] limit exists". +#[test] +fn parse_canonicalises_soft_subtable_alias() { + let parsed = split("[thresholds]\n\"loc.sloc\" = 100\n[thresholds.soft]\nsloc = \"0.9x\"\n") + .expect("bare loc alias resolves in the soft table"); + assert_eq!(parsed.soft.get("loc.sloc"), Some(&SoftLimit::Scale(0.9))); + assert_eq!( + parsed + .soft + .get("loc.sloc") + .expect("soft entry") + .resolve("loc.sloc", parsed.hard.get("loc.sloc").copied()), + Ok(90.0), + "the canonical soft key must find the canonical hard limit to scale", + ); +} + +/// Issue #1165: a table that names one metric under two spellings is +/// rejected. Before the fix the merge kept both and gated the same +/// extractor twice; silently keeping whichever key sorts last would be +/// the same surprise in a quieter form. +/// +/// Both key orders are covered because the table is a `BTreeMap`: which +/// spelling is *seen* second is decided by sort order, not by the order +/// the user typed. `loc.ploc` sorts before `ploc`, but `blank` sorts +/// before `loc.blank` — so the dotted id is the reported one in the +/// second case, and the message has to read correctly either way. +#[test] +fn parse_rejects_one_metric_under_two_spellings() { + for (src, canonical) in [ + ("[thresholds]\n\"loc.ploc\" = 6\nploc = 100\n", "loc.ploc"), + // `blank` sorts before `loc.blank`, so here the *dotted* id is + // the key reported — the direction whose message would otherwise + // read `"loc.blank": "loc.blank" is already set`, naming one key + // twice and the other not at all. + ( + "[thresholds]\nblank = 6\n\"loc.blank\" = 100\n", + "loc.blank", + ), + ] { + let err = split(src).expect_err("two spellings of one metric"); + assert!(err.contains("[thresholds]"), "{err}"); + assert!(err.contains(canonical), "{err}"); + assert!( + err.contains("are one metric, so set it once"), + "the message explains the alias relation in both directions: {err}" + ); + } +} + /// Issue #514: a bare family head with no single threshold scalar is /// ambiguous and rejected with the concrete candidates, not silently -/// mapped to one sub-metric. +/// mapped to one sub-metric. Since #1165 the rejection happens at the +/// parse boundary, attributed to the table it came from. #[test] -fn build_rejects_ambiguous_family_head() { - let mut raw = BTreeMap::new(); - raw.insert("halstead".to_string(), 1.0); - let err = ThresholdSet::build(&raw).expect_err("ambiguous head"); +fn parse_rejects_ambiguous_family_head() { + let err = split("[thresholds]\nhalstead = 1\n").expect_err("ambiguous head"); + assert!(err.contains("[thresholds]"), "{err}"); assert!(err.contains("ambiguous"), "{err}"); assert!(err.contains("halstead.volume"), "{err}"); } diff --git a/big-code-analysis-cli/tests/check/check_baseline.rs b/big-code-analysis-cli/tests/check/check_baseline.rs index 52c803950..221aa5543 100644 --- a/big-code-analysis-cli/tests/check/check_baseline.rs +++ b/big-code-analysis-cli/tests/check/check_baseline.rs @@ -192,8 +192,8 @@ fn regressed_function_fails_even_when_baselined() { ]) .assert() .code(2) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic = 7")); + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic = 7")); } #[test] @@ -230,7 +230,7 @@ fn new_offender_fails_even_with_baseline() { ]) .assert() .code(2) - .stderr(predicate::str::contains("classify")); + .stdout(predicate::str::contains("classify")); } // -- Ratchet semantics ---------------------------------------------------- @@ -322,7 +322,7 @@ fn moved_function_still_covered_after_line_drift() { ]) .assert() .success() - .stderr(predicate::str::contains("[new]").not()) + .stdout(predicate::str::contains("[new]").not()) // Prove the violation was actually classified and filtered — // a clean exit alone could mask a parse that found nothing. .stderr(predicate::str::contains("filtered 1 violations")); @@ -480,7 +480,7 @@ fn baseline_line_tolerance_flag_is_honored_end_to_end() { ]) .assert() .code(2) - .stderr(predicate::str::contains("[new]")); + .stdout(predicate::str::contains("[new]")); } #[test] @@ -529,7 +529,7 @@ fn fuzzy_match_covers_renamed_function() { ]) .assert() .code(2) - .stderr(predicate::str::contains("[new]")); + .stdout(predicate::str::contains("[new]")); // With fuzzy: the body hash matches, so it stays covered. cli(dir.path()) @@ -587,7 +587,7 @@ pub fn classify(n: i32) -> &'static str { .stderr(predicate::str::contains("wrote 0 baseline entries")); let content = fs::read_to_string(&baseline).unwrap(); - assert!(content.contains("version = 5")); + assert!(content.contains("version = 6")); assert!(content.contains("tier = \"hard\"")); assert!(!content.contains("[[entry]]")); } @@ -822,7 +822,7 @@ fn no_fail_overrides_baseline_fail() { ]) .assert() .success() - .stderr(predicate::str::contains("classify")); + .stdout(predicate::str::contains("classify")); } #[test] @@ -857,8 +857,8 @@ fn stale_baseline_entries_do_not_cover_unrelated_violations() { // does not cover it. A regression that treated stale entries // as wildcards would flip this to exit 0. .code(2) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic = 5")); + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic = 5")); } // -- Determinism & UX ----------------------------------------------------- @@ -1055,7 +1055,7 @@ fn clean_tree_write_baseline_produces_empty_versioned_file() { .stderr(predicate::str::contains("wrote 0 baseline entries")); let content = fs::read_to_string(&baseline).unwrap(); - assert!(content.contains("version = 5")); + assert!(content.contains("version = 6")); // No `--tier=soft`, so the write is stamped at the hard tier (#486). assert!(content.contains("tier = \"hard\"")); assert!(!content.contains("headroom")); @@ -1100,7 +1100,7 @@ fn regressed_violation_carries_tag_prefix() { .assert() .code(2) // (7-5)/5*100 = 40, rounded → +40% - .stderr(predicate::str::contains("[regr +40%] ")); + .stdout(predicate::str::contains("[regr +40%] ")); } #[test] @@ -1137,13 +1137,13 @@ fn new_violation_carries_new_tag() { ]) .assert() .code(2) - .stderr(predicate::str::contains("[new] ")); + .stdout(predicate::str::contains("[new] ")); } #[test] fn no_baseline_emits_unprefixed_lines() { // Backward-compatibility invariant: without --baseline the - // stderr line format is byte-identical to today. No `[new]` / + // offender line format is byte-identical to today. No `[new]` / // `[regr` prefix may appear on the violation line. let dir = TempDir::new().unwrap(); let src = write_fixture(&dir, "branchy.rs", BRANCHY_RUST); @@ -1164,9 +1164,9 @@ fn no_baseline_emits_unprefixed_lines() { .code(2) // The violation line must contain the function name and metric, // and must NOT carry a bracket tag prefix. - .stderr(predicate::str::contains("classify: cyclomatic =")) - .stderr(predicate::str::contains("[new]").not()) - .stderr(predicate::str::contains("[regr").not()); + .stdout(predicate::str::contains("classify: cyclomatic =")) + .stdout(predicate::str::contains("[new]").not()) + .stdout(predicate::str::contains("[regr").not()); } // -- Path canonicalisation (issue #376) ----------------------------------- diff --git a/big-code-analysis-cli/tests/check/check_exclude.rs b/big-code-analysis-cli/tests/check/check_exclude.rs index 9d744aac4..2e5b1dbc4 100644 --- a/big-code-analysis-cli/tests/check/check_exclude.rs +++ b/big-code-analysis-cli/tests/check/check_exclude.rs @@ -63,8 +63,8 @@ fn check_exclude_flag_drops_matching_offenders_only() { ]) .assert() .code(2) - .stderr(predicate::str::contains("kept_offender")) - .stderr(predicate::str::contains("excluded_offender").not()) + .stdout(predicate::str::contains("kept_offender")) + .stdout(predicate::str::contains("excluded_offender").not()) .stderr(predicate::str::contains( "skipped 1 violations via [check.exclude]", )); @@ -92,8 +92,8 @@ fn check_exclude_bare_relative_glob_drops_matching_offenders() { ]) .assert() .code(2) - .stderr(predicate::str::contains("kept_offender")) - .stderr(predicate::str::contains("excluded_offender").not()) + .stdout(predicate::str::contains("kept_offender")) + .stdout(predicate::str::contains("excluded_offender").not()) .stderr(predicate::str::contains( "skipped 1 violations via [check.exclude]", )); @@ -125,7 +125,7 @@ fn check_exclude_covering_sole_offender_exits_zero() { .stderr(predicate::str::contains( "skipped 1 violations via [check.exclude]", )) - .stderr(predicate::str::contains("excluded_offender").not()); + .stdout(predicate::str::contains("excluded_offender").not()); } /// `--check-exclude-from` reads `.gitignore`-style globs from a file; @@ -148,8 +148,8 @@ fn check_exclude_from_file_drops_matching_offenders() { ]) .assert() .code(2) - .stderr(predicate::str::contains("kept_offender")) - .stderr(predicate::str::contains("excluded_offender").not()); + .stdout(predicate::str::contains("kept_offender")) + .stdout(predicate::str::contains("excluded_offender").not()); } /// Acceptance: `--write-baseline` does NOT record entries for @@ -236,8 +236,8 @@ fn manifest_check_exclude_table_drops_offenders() { .arg("check") .assert() .code(2) - .stderr(predicate::str::contains("kept_offender")) - .stderr(predicate::str::contains("excluded_offender").not()); + .stdout(predicate::str::contains("kept_offender")) + .stdout(predicate::str::contains("excluded_offender").not()); } /// As a negative filter key (#539), an explicit `--check-exclude` UNIONs @@ -261,8 +261,14 @@ fn cli_check_exclude_unions_with_manifest_table() { .args(["check", "--check-exclude", "**/kept.rs"]) .assert() .success() - .stderr(predicate::str::contains("excluded_offender").not()) - .stderr(predicate::str::contains("kept_offender").not()); + // Absence alone would pass against a run that produced no + // offenders at all, so pair it with the positive diagnostic + // proving both violations existed and were exempted. + .stderr(predicate::str::contains( + "skipped 2 violations via [check.exclude]", + )) + .stdout(predicate::str::contains("excluded_offender").not()) + .stdout(predicate::str::contains("kept_offender").not()); } /// `--no-config` is the escape hatch that ignores the manifest entirely: @@ -293,8 +299,8 @@ fn no_config_drops_manifest_check_exclude_from_union() { ]) .assert() .code(2) - .stderr(predicate::str::contains("excluded_offender")) - .stderr(predicate::str::contains("kept_offender").not()); + .stdout(predicate::str::contains("excluded_offender")) + .stdout(predicate::str::contains("kept_offender").not()); } /// `--print-effective-config` surfaces the resolved `check_exclude` @@ -381,8 +387,8 @@ fn check_exclude_manifest_glob_applies_from_subdir() { .arg("check") .assert() .code(2) - .stderr(predicate::str::contains("keep_offender")) - .stderr(predicate::str::contains("vendor_offender").not()) + .stdout(predicate::str::contains("keep_offender")) + .stdout(predicate::str::contains("vendor_offender").not()) .stderr(predicate::str::contains( "skipped 1 violations via [check.exclude]", )); @@ -429,8 +435,8 @@ fn check_exclude_anchors_paths_from_seeds() { ]) .assert() .code(2) - .stderr(predicate::str::contains("kept_offender")) - .stderr(predicate::str::contains("excluded_offender").not()) + .stdout(predicate::str::contains("kept_offender")) + .stdout(predicate::str::contains("excluded_offender").not()) .stderr(predicate::str::contains( "skipped 1 violations via [check.exclude]", )); diff --git a/big-code-analysis-cli/tests/check/check_exit_codes.rs b/big-code-analysis-cli/tests/check/check_exit_codes.rs index 9700394a5..8187c2981 100644 --- a/big-code-analysis-cli/tests/check/check_exit_codes.rs +++ b/big-code-analysis-cli/tests/check/check_exit_codes.rs @@ -210,7 +210,7 @@ fn strict_regression_only_exits_three() { ]) .assert() .code(3) - .stderr(predicate::str::contains("[regr")); + .stdout(predicate::str::contains("[regr")); } #[test] @@ -296,6 +296,84 @@ fn strict_soft_encroachment_exits_two_not_five() { .code(2); } +/// The soft band of a lower-is-worse `mi.*` floor sits *above* the hard +/// floor, so a function between the two is an encroachment (exit 2) and +/// not a hard breach (exit 5) — the same tier split the higher-is-worse +/// tests above pin, in the opposite direction (#1166). +/// +/// Before the fix the ratio *lowered* the floor (60 * 0.5 = 30), which +/// this fixture clears easily, so the soft tier reported nothing at all +/// and the run exited 0. The hard-tier control run is what makes the +/// exit 2 meaningful: it proves the fixture is inside the hard floor and +/// the offender came from the soft band alone. +#[test] +fn strict_mi_between_the_two_floors_is_an_encroachment_not_a_hard_breach() { + let dir = TempDir::new().unwrap(); + // `classify` scores mi.original ≈ 105, comfortably inside the band + // between the hard floor (60) and the soft one (60 / 0.5 = 120). + let src = write_branchy(&dir, 6); + let config = dir.path().join("thresholds.toml"); + fs::write(&config, "[thresholds]\n\"mi.original\" = 60\n").unwrap(); + let config = config.to_str().unwrap(); + + // Hard tier: the floor is 60 and nothing is under it. + cli(dir.path()) + .args([ + "check", + "--paths", + &src, + "--config", + config, + "--exit-codes=tiered", + ]) + .assert() + .success(); + + cli(dir.path()) + .args([ + "check", + "--paths", + &src, + "--config", + config, + "--tier=soft=0.5", + "--exit-codes=tiered", + ]) + .assert() + .code(2) + .stdout( + predicate::str::is_match(r"classify: mi\.original = \d+\.\d+ \(limit 120\)").unwrap(), + ); +} + +/// The escalation to exit 5 still works in the lower-is-worse direction: +/// a floor the function is genuinely under is a hard breach even though +/// the soft floor it also trips sits above it. +/// +/// Companion to the test above — together they pin that raising the +/// floor moved the *soft band*, not the hard gate. +#[test] +fn strict_mi_under_the_hard_floor_exits_five() { + let dir = TempDir::new().unwrap(); + // mi.original ≈ 105, under a hard floor of 110. + let src = write_branchy(&dir, 6); + let config = dir.path().join("thresholds.toml"); + fs::write(&config, "[thresholds]\n\"mi.original\" = 110\n").unwrap(); + + cli(dir.path()) + .args([ + "check", + "--paths", + &src, + "--config", + config.to_str().unwrap(), + "--tier=soft=0.5", + "--exit-codes=tiered", + ]) + .assert() + .code(5); +} + /// A `[thresholds.soft]` absolute limit with no `[thresholds]` /// counterpart gives the metric a soft band and no hard ceiling, so no /// value can escalate it to a hard breach however far it overshoots. diff --git a/big-code-analysis-cli/tests/check/check_explain_threshold.rs b/big-code-analysis-cli/tests/check/check_explain_threshold.rs new file mode 100644 index 000000000..e87dfa26e --- /dev/null +++ b/big-code-analysis-cli/tests/check/check_explain_threshold.rs @@ -0,0 +1,1068 @@ +//! Integration tests for `bca check --explain-threshold` (issue #1169). +//! +//! The feature's whole value is that its counts match the gate run it +//! predicts, so most of what follows is that equality asserted against a +//! *real* `bca check` at the same limit, once per configuration layer +//! that could make the two diverge: `[check] exclude`, `--exclude-tests`, +//! in-source suppression markers, and a baseline. +//! +//! Every fixture is sized so the two tiers disagree and both are +//! non-zero. A preview whose filter selected nothing would otherwise +//! agree with a gate run that also selected nothing, and the whole suite +//! would pass vacuously. + +use std::fmt::Write as _; +use std::fs; +use std::path::Path; + +use assert_cmd::Command; +use predicates::prelude::*; +use tempfile::TempDir; + +use crate::common; + +fn cli(dir: &Path) -> Command { + common::cli_in(dir) +} + +/// One Rust function taking `arity` parameters, named `f{id}`. +fn nargs_fn(id: usize, arity: usize) -> String { + let params: Vec = (0..arity).map(|p| format!("a{p}: u8")).collect(); + format!("pub fn f{id}({}) -> u8 {{ {arity} }}\n", params.join(", ")) +} + +/// A Rust source file whose `nargs` distribution is exactly `spec`, read +/// as `(arity, how many functions at it)`. +fn nargs_source(spec: &[(usize, usize)]) -> String { + let mut out = String::new(); + let mut id = 0; + for (arity, count) in spec { + for _ in 0..*count { + out.push_str(&nargs_fn(id, *arity)); + id += 1; + } + } + out +} + +/// The fixture every equality test below shares: 20 functions at exactly +/// 4 parameters, 6 at 5, and 3 at 6. +/// +/// Against a candidate `nargs = 4` that resolves to 9 hard-tier offenders +/// (`> 4`) and 29 soft-tier ones (`> 3.8`), so the two tiers cannot be +/// confused for one another, and the 20-function soft band sits on one +/// value — the shape the whole issue is about. +const CANDIDATE_SPEC: &[(usize, usize)] = &[(4, 20), (5, 6), (6, 3)]; +const CANDIDATE_METRIC: &str = "nargs"; +const CANDIDATE_LIMIT: &str = "4"; +const EXPECTED_HARD: usize = 9; +const EXPECTED_SOFT: usize = 29; + +/// One tier's row, parsed back out of the preview's stdout. +#[derive(Debug, PartialEq, Eq)] +struct Tier { + limit: String, + total: usize, + baselined: usize, + new: usize, +} + +/// The preview report for one metric. +#[derive(Debug)] +struct Preview { + hard: Tier, + soft: Tier, + cluster: Option, +} + +/// Parse ` tier (limit [, ]): N offenders, M +/// already baselined, K new`. +/// +/// Splitting on `"): "` rather than on every comma is deliberate: the +/// soft row's limit segment carries the derivation after a comma, so a +/// naive split would shear it and silently mis-assign the counts. +fn parse_tier(line: &str) -> Tier { + let (limit, counts) = line + .split_once("): ") + .expect("tier row has a limit segment"); + let limit = limit + .split_once("(limit ") + .expect("tier row names its limit") + .1 + .to_string(); + let mut fields = counts.split(", ").map(|field| { + field + .split_whitespace() + .next() + .expect("count field is non-empty") + .parse::() + .expect("count field is a number") + }); + Tier { + limit, + total: fields.next().expect("offender count"), + baselined: fields.next().expect("baselined count"), + new: fields.next().expect("new count"), + } +} + +fn parse_preview(stdout: &str, metric: &str) -> Preview { + let header = format!("{metric}: candidate limit "); + let start = stdout + .lines() + .position(|line| line.starts_with(&header)) + .unwrap_or_else(|| panic!("preview reports {metric}; got:\n{stdout}")); + // The block runs to the next unindented header, so a multi-metric + // report cannot leak one metric's rows into another's. + let block: Vec<&str> = stdout + .lines() + .skip(start + 1) + .take_while(|line| line.starts_with(" ")) + .collect(); + let row = |prefix: &str| { + let line = block + .iter() + .find(|line| line.trim_start().starts_with(prefix)) + .unwrap_or_else(|| panic!("preview reports the {prefix} row; got:\n{stdout}")); + parse_tier(line) + }; + Preview { + hard: row("hard tier"), + soft: row("soft tier"), + cluster: block + .iter() + .find(|line| line.trim_start().starts_with("cluster: ")) + .map(|line| (*line).trim().to_string()), + } +} + +/// Offender rows for `metric` in a real `bca check` stdout stream. +fn gate_rows(stdout: &str, metric: &str) -> usize { + let needle = format!(": {metric} = "); + stdout.lines().filter(|l| l.contains(&needle)).count() +} + +fn stdout_of(cmd: &mut Command) -> String { + let out = cmd.assert().success().get_output().stdout.clone(); + String::from_utf8(out).expect("utf8 stdout") +} + +/// Run the preview and the two real gate runs it claims to predict, and +/// assert the three agree. +/// +/// `extra` is appended to all three invocations, so whatever +/// configuration layer a caller is exercising applies identically to the +/// preview and to the gate. +fn assert_preview_matches_gate(dir: &Path, extra: &[&str]) -> Preview { + // The gate has no way to express a candidate limit and a proportional + // soft tier at once: `--threshold` is absolute and never scaled, so + // the soft side has to come from a config file. That asymmetry is the + // bug this feature exists for. + fs::write( + dir.join("candidate.toml"), + format!("[thresholds]\n{CANDIDATE_METRIC} = {CANDIDATE_LIMIT}\n"), + ) + .expect("write candidate config"); + + let preview = parse_preview( + &stdout_of( + cli(dir) + .args(["check", "--explain-threshold"]) + .arg(format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}")) + .args(extra), + ), + CANDIDATE_METRIC, + ); + + let gate = |tier: &str| { + gate_rows( + &stdout_of( + cli(dir) + .args([ + "check", + "--config", + "candidate.toml", + "--no-fail", + "--no-summary", + "--no-remediation", + tier, + ]) + .args(extra), + ), + CANDIDATE_METRIC, + ) + }; + + // `new`, not `total`: a real gate run drops baseline-covered + // offenders, which the preview keeps in order to report the split. + // The fixtures below never regress against their baseline, so the + // gate's surviving rows are exactly the preview's new ones. + assert_eq!( + preview.hard.new, + gate("--tier=hard"), + "hard-tier preview must match the gate at the same limit" + ); + assert_eq!( + preview.soft.new, + gate("--tier=soft=0.95"), + "soft-tier preview must match the gate at the same limit" + ); + preview +} + +fn tree(files: &[(&str, String)]) -> TempDir { + let dir = TempDir::new().expect("tempdir"); + for (name, body) in files { + fs::write(dir.path().join(name), body).expect("write fixture"); + } + dir +} + +fn candidate_tree() -> TempDir { + tree(&[("lib.rs", nargs_source(CANDIDATE_SPEC))]) +} + +#[test] +fn preview_matches_the_gate_on_a_plain_tree() { + let dir = candidate_tree(); + let preview = assert_preview_matches_gate(dir.path(), &["--paths", "lib.rs"]); + // Pin the absolute counts too, so a helper that compared two equally + // broken numbers would still fail here. + assert_eq!(preview.hard.total, EXPECTED_HARD); + assert_eq!(preview.soft.total, EXPECTED_SOFT); +} + +#[test] +fn preview_matches_the_gate_under_check_exclude() { + // The excluded file offends at every arity in the fixture, so a + // preview that ignored `[check] exclude` would report visibly larger + // numbers rather than the same ones by luck. + let dir = tree(&[ + ("lib.rs", nargs_source(CANDIDATE_SPEC)), + ("generated.rs", nargs_source(CANDIDATE_SPEC)), + ]); + fs::write( + dir.path().join("bca.toml"), + "paths = [\".\"]\n[check]\nexclude = [\"./generated.rs\"]\n", + ) + .expect("write manifest"); + + let preview = assert_preview_matches_gate(dir.path(), &[]); + assert_eq!(preview.hard.total, EXPECTED_HARD); + assert_eq!(preview.soft.total, EXPECTED_SOFT); + + // Seed check: without the exemption the same tree doubles, so the + // equality above is the exclusion being honoured and not an empty + // filter agreeing with an empty filter. + let unexcluded = parse_preview( + &stdout_of( + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + ".", + "--explain-threshold", + ]) + .arg(format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}")), + ), + CANDIDATE_METRIC, + ); + assert_eq!(unexcluded.hard.total, EXPECTED_HARD * 2); +} + +#[test] +fn preview_matches_the_gate_under_exclude_tests() { + // The `#[cfg(test)]` module offends at every arity; `--exclude-tests` + // prunes the whole subtree before any metric is computed. + let mut body = nargs_source(CANDIDATE_SPEC); + body.push_str("#[cfg(test)]\nmod tests {\n"); + body.push_str(&nargs_source(CANDIDATE_SPEC)); + body.push_str("}\n"); + let dir = tree(&[("lib.rs", body)]); + + let preview = + assert_preview_matches_gate(dir.path(), &["--paths", "lib.rs", "--exclude-tests"]); + assert_eq!(preview.hard.total, EXPECTED_HARD); + + // Seed check: the same tree without the flag carries both copies. + let with_tests = parse_preview( + &stdout_of( + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + ]) + .arg(format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}")), + ), + CANDIDATE_METRIC, + ); + assert_eq!(with_tests.hard.total, EXPECTED_HARD * 2); +} + +#[test] +fn preview_matches_the_gate_under_suppression_markers() { + // Suppress every 6-parameter function: three hard-tier offenders + // disappear from both the preview and the gate. + let mut body = nargs_source(&[(4, 20), (5, 6)]); + for id in 26..29 { + write!( + body, + "pub fn f{id}(a0: u8, a1: u8, a2: u8, a3: u8, a4: u8, a5: u8) -> u8 {{\n \ + // bca: suppress(nargs) -- fixture: marker must drop this offender\n 6\n}}\n" + ) + .expect("write to String"); + } + let dir = tree(&[("lib.rs", body)]); + + let preview = assert_preview_matches_gate(dir.path(), &["--paths", "lib.rs"]); + assert_eq!( + preview.hard.total, + EXPECTED_HARD - 3, + "the three marked functions must not be counted" + ); + assert_eq!(preview.soft.total, EXPECTED_SOFT - 3); + + // Seed check: `--no-suppress` un-silences them, so the marker is what + // moved the number. + let raw = parse_preview( + &stdout_of( + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--no-suppress", + "--explain-threshold", + ]) + .arg(format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}")), + ), + CANDIDATE_METRIC, + ); + assert_eq!(raw.hard.total, EXPECTED_HARD); + + // `--report-suppressed` keeps marker-silenced offenders in the stream + // for the SARIF document, but the gate still excludes them — so the + // preview must too, or it would price a candidate against offenders + // no run would ever fail on. + let reported = parse_preview( + &stdout_of( + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--report-suppressed", + "--explain-threshold", + ]) + .arg(format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}")), + ), + CANDIDATE_METRIC, + ); + assert_eq!(reported.hard.total, EXPECTED_HARD - 3); +} + +#[test] +fn preview_splits_baselined_debt_from_new_entries() { + let dir = candidate_tree(); + // A baseline written at the candidate's *hard* limit covers the 9 + // hard offenders and none of the 20 soft-band ones — the exact + // asymmetry the report exists to surface. + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--threshold", + "nargs=4", + "--write-baseline", + "base.toml", + ]) + .assert() + .success(); + + let preview = assert_preview_matches_gate( + dir.path(), + &["--paths", "lib.rs", "--baseline", "base.toml"], + ); + + assert_eq!(preview.hard.total, EXPECTED_HARD); + assert_eq!(preview.hard.baselined, EXPECTED_HARD); + assert_eq!(preview.hard.new, 0, "the hard tier reads as free"); + assert_eq!(preview.soft.total, EXPECTED_SOFT); + assert_eq!(preview.soft.baselined, EXPECTED_HARD); + assert_eq!( + preview.soft.new, + EXPECTED_SOFT - EXPECTED_HARD, + "and the soft tier is where the candidate's real cost lands" + ); +} + +/// A baselined offender whose value has *worsened* still has a baseline +/// entry, so it is counted as baselined rather than as a new one — a +/// refresh would update its record, not add one. +#[test] +fn a_regressed_offender_counts_as_baselined_not_new() { + let dir = candidate_tree(); + fs::write( + dir.path().join("base.toml"), + // `f26` is the first six-parameter function and each fixture + // function occupies one line, so it starts at line 27. The + // recorded 5 is below its real 6, which is what makes it a + // regression rather than covered debt. + "version = 5\n\n[provenance]\ntier = \"hard\"\n\n[[entry]]\n\ + path = \"lib.rs\"\nqualified = \"f26\"\nstart_line = 27\n\ + metric = \"nargs\"\nvalue = 5.0\n", + ) + .expect("write baseline"); + + let preview = parse_preview( + &stdout_of( + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--baseline", + "base.toml", + "--explain-threshold", + ]) + .arg(format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}")), + ), + CANDIDATE_METRIC, + ); + assert_eq!(preview.hard.total, EXPECTED_HARD); + assert_eq!( + preview.hard.baselined, 1, + "f26 has an entry (recorded 5, now 6) and is therefore not a new entry" + ); + assert_eq!(preview.hard.new, EXPECTED_HARD - 1); +} + +/// The v6 schema omits `start_line` for a unique identity (#1170). +/// `--explain-threshold` resolves its baselined/new split through the +/// same matcher as the gate, so dropping the line must not move a single +/// offender between the two columns — asserted against the byte-for-byte +/// numbers the v5 sibling test above pins. +#[test] +fn a_regressed_offender_counts_as_baselined_without_a_recorded_line() { + let dir = candidate_tree(); + fs::write( + dir.path().join("base.toml"), + // Identical to the v5 fixture above except for the schema stamp + // and the absent `start_line`: `f26` is a unique identity, so a + // v6 `--write-baseline` records no line for it. + "version = 6\n\n[provenance]\ntier = \"hard\"\n\n[[entry]]\n\ + path = \"lib.rs\"\nqualified = \"f26\"\n\ + metric = \"nargs\"\nvalue = 5.0\n", + ) + .expect("write baseline"); + + let preview = parse_preview( + &stdout_of( + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--baseline", + "base.toml", + "--explain-threshold", + ]) + .arg(format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}")), + ), + CANDIDATE_METRIC, + ); + assert_eq!(preview.hard.total, EXPECTED_HARD); + assert_eq!( + preview.hard.baselined, 1, + "f26 still matches its entry with no line recorded" + ); + assert_eq!(preview.hard.new, EXPECTED_HARD - 1); +} + +/// The soft tier is derived direction-aware (#1166): a lower-is-worse +/// `mi.*` limit is a floor, so tightening it raises it. +#[test] +fn soft_limit_is_derived_by_metric_direction() { + let dir = candidate_tree(); + let stdout = stdout_of(cli(dir.path()).args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "mi.original=20", + "--explain-threshold", + "nargs=4", + ])); + let floor = parse_preview(&stdout, "mi.original"); + let ceiling = parse_preview(&stdout, "nargs"); + assert_eq!(floor.hard.limit, "20"); + assert_eq!( + floor.soft.limit, "21.0527, 0.95x", + "a lower-is-worse floor is tightened by raising it" + ); + assert_eq!(ceiling.hard.limit, "4"); + assert_eq!( + ceiling.soft.limit, "3.8, 0.95x", + "a higher-is-worse ceiling is tightened by lowering it" + ); +} + +/// Direction decides which *offenders* land in the hard tier, not only +/// how the soft limit is derived — and that is a second production +/// expression, `breaches_limit(v.value, ceiling, v.lower_is_worse)`. +/// +/// The test above cannot reach it: `mi.original=20` is a floor that every +/// one-line fixture function (all measure ≈147-149) clears, so its +/// population is empty at both tiers and the partition never runs. +/// Hard-coding `false` there — the pre-#1166 "higher is worse" assumption +/// — failed none of the suite's 5053 tests. +/// +/// A floor of `147.5` splits the fixture's three measured values +/// (146.9456 ×3, 147.7445 ×6, 148.655 ×20): three sit below it and +/// breach, all 29 sit below the derived `155.264` soft floor. Read with +/// the direction inverted the hard tier would instead collect the 26 +/// *above* 147.5, so the two readings share no count. +/// +/// Both metrics are explained in one run, and both carry a *non-empty* +/// population — which is the second thing this pins. Every other +/// multi-metric test here pairs a real population with an empty one, so +/// `explain`'s per-metric filter (`v.metric == outcome.metric`) could be +/// deleted outright without failing any of the suite's 5054 tests. With +/// two live populations each metric would then tally the union: 12 and +/// 58 everywhere, against the four distinct counts below. +#[test] +fn a_lower_is_worse_metric_partitions_offenders_by_direction() { + let dir = candidate_tree(); + let candidate = format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}"); + let stdout = stdout_of(cli(dir.path()).args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "mi.original=147.5", + "--explain-threshold", + &candidate, + ])); + let floor = parse_preview(&stdout, "mi.original"); + assert_eq!( + floor.hard.total, 3, + "only the three functions *below* the floor breach it; \ + reading it as a ceiling would report 26. stdout:\n{stdout}" + ); + assert_eq!(floor.hard.new, 3, "none of them is baselined"); + assert_eq!( + floor.soft.total, 29, + "the whole population sits under the raised soft floor; stdout:\n{stdout}" + ); + + // The higher-is-worse metric keeps its own counts in the same run. + let ceiling = parse_preview(&stdout, CANDIDATE_METRIC); + assert_eq!(ceiling.hard.total, EXPECTED_HARD); + assert_eq!(ceiling.soft.total, EXPECTED_SOFT); +} + +/// The candidate limit the two soft-derivation tests below share. It +/// sits above every arity in the fixture, so the *default* 0.95 band is +/// empty and any offender reported at the soft tier can only come from +/// a soft limit the run derived some other way. +const LOOSE_LIMIT: &str = "nargs=8"; + +/// `--tier=soft=` pins the ratio the preview scales its soft limit +/// by. Without it the report falls back to `DEFAULT_SOFT_HEADROOM`, +/// which is the only path the rest of this suite exercises. +/// +/// Both the derived limit and the offender count move with the ratio, +/// so a preview that ignored the flag could not pass this by luck. +#[test] +fn an_explicit_soft_ratio_derives_the_soft_limit() { + let dir = candidate_tree(); + let preview_with = |tier: &[&str]| { + parse_preview( + &stdout_of( + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + LOOSE_LIMIT, + ]) + .args(tier), + ), + CANDIDATE_METRIC, + ) + }; + + let explicit = preview_with(&["--tier=soft=0.5"]); + assert_eq!( + explicit.soft.limit, "4, 0.5x", + "the supplied ratio scales the candidate, not 0.95" + ); + assert_eq!( + explicit.soft.total, 9, + "the six 5-parameter and three 6-parameter functions breach a soft limit of 4" + ); + assert_eq!(explicit.hard.total, 0, "and none of them breach 8"); + + // Seed check: the same tree at the default tier derives 7.6, which + // nothing in the fixture reaches — so the ratio is what moved both + // the limit and the count. + let default = preview_with(&[]); + assert_eq!(default.soft.limit, "7.6, 0.95x"); + assert_eq!(default.soft.total, 0); +} + +/// A `[thresholds.soft]` entry for the explained metric supplies the +/// soft limit outright. The report must attribute it to that table +/// rather than to a ratio it never applied. +#[test] +fn a_soft_table_entry_is_named_as_the_soft_limits_source() { + let dir = candidate_tree(); + fs::write( + dir.path().join("bca.toml"), + "paths = [\"lib.rs\"]\n[thresholds.soft]\nnargs = 4\n", + ) + .expect("write manifest"); + + let preview = parse_preview( + &stdout_of(cli(dir.path()).args(["check", "--explain-threshold", LOOSE_LIMIT])), + CANDIDATE_METRIC, + ); + assert_eq!( + preview.soft.limit, "4, [thresholds.soft]", + "the table's own value, attributed to the table" + ); + assert_eq!(preview.soft.total, 9); + assert_eq!(preview.hard.total, 0); + + // Seed check: drop the manifest and the same candidate derives its + // soft limit from the ratio instead, so the attribution above is the + // table being consulted and not a constant that happens to fit. + let ratio_derived = parse_preview( + &stdout_of(cli(dir.path()).args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + LOOSE_LIMIT, + ])), + CANDIDATE_METRIC, + ); + assert_eq!(ratio_derived.soft.limit, "7.6, 0.95x"); +} + +/// The third way a soft limit is arrived at, and the one the report +/// used to misattribute. `resolve_tier`'s soft branch is all-or-nothing +/// per table: a single `[thresholds.soft]` entry switches the whole +/// table into merge mode, so an explained metric the table does *not* +/// name keeps its hard limit with no ratio applied to it at all. +/// +/// The counts were always right; the annotation was not. `limit 8, +/// 0.95x` says 8 × 0.95, which is 7.6 — a number that appears nowhere +/// in the run, on the one command whose entire purpose is to be trusted +/// for a threshold decision. +#[test] +fn a_metric_the_soft_table_omits_reports_no_soft_band() { + // A candidate the fixture straddles at *both* tiers: 3 functions + // exceed 5, and 9 exceed the 4.75 a ratio would derive. A candidate + // above the fixture's whole range would compare 0 against 0 and the + // contrast below would hold vacuously. + const TIGHT_LIMIT: &str = "nargs=5"; + const INHERITED_OFFENDERS: usize = 3; + const RATIO_OFFENDERS: usize = 9; + + let dir = candidate_tree(); + fs::write( + dir.path().join("bca.toml"), + "paths = [\"lib.rs\"]\n[thresholds.soft]\ncognitive = 4\n", + ) + .expect("write manifest"); + + let stdout = stdout_of(cli(dir.path()).args([ + "check", + "--explain-threshold", + TIGHT_LIMIT, + "--explain-threshold", + "cognitive=10", + ])); + + // The metric the table names still reports the table. + assert_eq!( + parse_preview(&stdout, "cognitive").soft.limit, + "4, [thresholds.soft]", + ); + + // The metric it omits inherits the candidate verbatim, and says so. + let nargs = parse_preview(&stdout, CANDIDATE_METRIC); + assert_eq!( + nargs.soft.limit, + "5, no soft band; [thresholds.soft] names other metrics, \ + so the hard limit stands", + ); + // Same limit at both tiers means the same offenders at both — the + // fact the annotation has to stay consistent with. + assert_eq!(nargs.hard.total, INHERITED_OFFENDERS); + assert_eq!(nargs.soft.total, INHERITED_OFFENDERS); + + // Seed check: drop the soft table and the identical candidate does + // get a ratio-derived band, over a strictly larger population. The + // reported line therefore tracks a real difference rather than a + // constant that happens to fit. + let ratio_derived = parse_preview( + &stdout_of(cli(dir.path()).args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + TIGHT_LIMIT, + ])), + CANDIDATE_METRIC, + ); + assert_eq!(ratio_derived.soft.limit, "4.75, 0.95x"); + assert_eq!(ratio_derived.soft.total, RATIO_OFFENDERS); +} + +#[test] +fn cluster_fires_when_the_soft_band_sits_on_the_candidate_limit() { + let dir = candidate_tree(); + let preview = parse_preview( + &stdout_of(cli(dir.path()).args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "nargs=4", + ])), + CANDIDATE_METRIC, + ); + let cluster = preview + .cluster + .expect("a converged limit reports a cluster"); + assert!( + cluster.contains("20 of 20 soft-band offenders sit at exactly 4"), + "cluster line names the population and the value: {cluster}" + ); + assert!( + cluster.contains("the candidate limit itself"), + "and says the limit converged onto it: {cluster}" + ); +} + +#[test] +fn cluster_stays_quiet_when_the_soft_band_is_dispersed() { + // `loc.ploc` counts one physical line per statement here, so each + // file's value is its line count. Fifteen files spread evenly over + // the five values in the (95, 100] band leave no value above the + // majority share, while the band is comfortably over the + // ten-function floor — so this exercises the *share* rule, not the + // size one. + let mut files = Vec::new(); + for ploc in 96..=100_usize { + for copy in 0..3 { + let mut body = String::new(); + for i in 0..ploc { + writeln!(body, "fn f{i}() {{ }}").expect("write to String"); + } + files.push((format!("f{ploc}_{copy}.rs"), body)); + } + } + let borrowed: Vec<(&str, String)> = files + .iter() + .map(|(name, body)| (name.as_str(), body.clone())) + .collect(); + let dir = tree(&borrowed); + + let preview = parse_preview( + &stdout_of(cli(dir.path()).args([ + "check", + "--no-config", + "--paths", + ".", + "--explain-threshold", + "loc.ploc=100", + ])), + "loc.ploc", + ); + assert_eq!( + preview.soft.total, 15, + "all fifteen files fall inside the (95, 100] soft band" + ); + assert_eq!(preview.hard.total, 0, "and none of them breach 100"); + assert_eq!( + preview.cluster, None, + "no single value holds a majority of a dispersed band" + ); +} + +/// The preview is a report, not a gate: it exits 0 even over a tree the +/// candidate limit would fail, and says so. +#[test] +fn preview_exits_zero_over_an_offending_tree() { + let dir = candidate_tree(); + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "nargs=4", + ]) + .assert() + .success() + .stderr(predicate::str::contains( + "no gate ran and the exit code is always 0", + )); +} + +/// Stream contract (#1167): the report is this invocation's product, so +/// it belongs on stdout and the diagnostics stay on stderr. +#[test] +fn report_goes_to_stdout_and_diagnostics_to_stderr() { + let dir = candidate_tree(); + let assert = cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "nargs=4", + ]) + .assert() + .success(); + let out = assert.get_output(); + let stdout = String::from_utf8(out.stdout.clone()).expect("utf8 stdout"); + let stderr = String::from_utf8(out.stderr.clone()).expect("utf8 stderr"); + + assert!(stdout.contains("nargs: candidate limit 4"), "{stdout}"); + assert!(stdout.contains("hard tier (limit 4)"), "{stdout}"); + assert!(stdout.contains("cluster: "), "{stdout}"); + assert!( + !stderr.contains("candidate limit"), + "the report must not be duplicated onto stderr: {stderr}" + ); + assert!( + stderr.contains("bca: --explain-threshold is a preview"), + "the preview caveat is a diagnostic: {stderr}" + ); +} + +/// A per-language override of the explained metric keeps its own limit, +/// and the report says so rather than reporting a number that would not +/// apply there. +#[test] +fn per_language_override_of_the_explained_metric_is_reported() { + let dir = tree(&[ + ("lib.rs", nargs_source(CANDIDATE_SPEC)), + ( + "wide.c", + "int wide(int a, int b, int c, int d, int e) { return a; }\n".to_string(), + ), + ]); + fs::write( + dir.path().join("bca.toml"), + "paths = [\".\"]\n[thresholds.lang.c]\nnargs = 8\n", + ) + .expect("write manifest"); + + let stdout = stdout_of(cli(dir.path()).args(["check", "--explain-threshold", "nargs=4"])); + let preview = parse_preview(&stdout, CANDIDATE_METRIC); + assert!( + stdout.contains("[thresholds.lang.c] keeps nargs at 8 (soft 7.6)"), + "the report names the language keeping its own limit: {stdout}" + ); + // The C function has five parameters and so would offend at the + // candidate, but its language gates at 8 — the count must exclude it. + assert_eq!(preview.soft.total, EXPECTED_SOFT); +} + +/// Two ways the request can contradict itself, both rejected rather than +/// silently resolved. +#[test] +fn contradictory_candidates_are_rejected() { + let dir = candidate_tree(); + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "nargs=4", + "--explain-threshold", + "nargs=5", + ]) + .assert() + .code(1) + .stderr(predicate::str::contains( + "preview one candidate limit per metric per run", + )); + + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "nargs=4", + "--threshold", + "nargs=6", + ]) + .assert() + .code(1) + .stderr(predicate::str::contains( + "a --threshold limit is absolute and has no soft tier to preview", + )); +} + +/// Every flag the preview refuses to share a run with, and the reason +/// each is refused: the preview returns before `emit_check_results`, so +/// a second artifact the user asked for would silently never be +/// written. Without this the whole `conflicts_with_all` list is +/// unguarded — dropping `"output"` from it would let +/// `--explain-threshold X --output f.sarif` run the preview and produce +/// no document, with nothing failing. +#[test] +fn flags_producing_a_second_artifact_are_rejected() { + let dir = candidate_tree(); + let candidate = format!("{CANDIDATE_METRIC}={CANDIDATE_LIMIT}"); + let out = dir.path().join("artifact.out"); + let out = out.to_str().expect("utf-8 tempdir path"); + for extra in [ + "--write-baseline", + "--print-effective-config=json", + "--report-format=checkstyle", + &format!("--output={out}"), + ] { + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + &candidate, + extra, + ]) + .assert() + .code(1) + // clap names both sides; asserting only the exit code would + // pass on any unrelated usage error. + .stderr(predicate::str::contains("--explain-threshold")) + .stderr(predicate::str::contains( + extra.split('=').next().unwrap_or(extra), + )); + assert!( + !std::path::Path::new(out).exists(), + "a rejected run must not have written {out}", + ); + } + + // `--summary-file ` is the same silent no-op but cannot ride + // on `conflicts_with_all`, which fires on the flag's presence and + // would take the keyword forms with it. + let summary = dir.path().join("summary.md"); + let summary = summary.to_str().expect("utf-8 tempdir path"); + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + &candidate, + "--summary-file", + summary, + ]) + .assert() + .code(1) + .stderr(predicate::str::contains( + "--explain-threshold cannot be used with --summary-file", + )); + assert!( + !std::path::Path::new(summary).exists(), + "a rejected run must not have written {summary}", + ); + + // …while `auto` keeps working: it names no destination of its own, + // so the preview producing no step summary is what every other + // non-gating run does. Rejecting it would break any CI invocation + // that passes the keyword explicitly. + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + &candidate, + "--summary-file", + "auto", + ]) + .assert() + .code(0); +} + +/// The two typo classes `--threshold` rejects, routed through the same +/// `normalize_for_check` + `build_tiered` pair — and each message must +/// name the flag the user actually passed. `canonical_cli_thresholds` +/// hardcoded `--threshold` for all three of its call sites, so an +/// ambiguous candidate blamed a flag that was never on the command line. +#[test] +fn an_unusable_candidate_metric_is_rejected_naming_this_flag() { + let dir = candidate_tree(); + // A family head with no single scalar: rejected before the walk, by + // the name-resolution layer that owns the flag label. + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "halstead=5", + ]) + .assert() + .code(1) + .stderr(predicate::str::contains( + "--explain-threshold: ambiguous metric \"halstead\"", + )) + .stderr(predicate::str::contains("halstead.effort")) + // The flag the user did not pass must not be blamed. + .stderr(predicate::str::contains("--threshold:").not()); + + // An unknown name: rejected by the threshold builder, with the + // did-you-mean vocabulary. + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + "lib.rs", + "--explain-threshold", + "not_a_metric=1", + ]) + .assert() + .code(1) + .stderr(predicate::str::contains( + "unknown threshold metric \"not_a_metric\"", + )); +} diff --git a/big-code-analysis-cli/tests/check/check_stream_contract.rs b/big-code-analysis-cli/tests/check/check_stream_contract.rs new file mode 100644 index 000000000..f3bf6e81a --- /dev/null +++ b/big-code-analysis-cli/tests/check/check_stream_contract.rs @@ -0,0 +1,393 @@ +//! `bca check`'s stdout/stderr split (issue #1167). +//! +//! The offender rows are the command's *product* and belong on stdout, +//! where `| wc -l`, `| head`, `| rg -c` and `2>/dev/null` can reach +//! them. Everything else the run says about itself — the summary footer, +//! the `skipped` / `filtered` counts, warnings, the remediation block — +//! is commentary and belongs on stderr. Before #1167 the rows went to +//! stderr too, so all four of those idioms reported an empty offender +//! set: a *plausible* "this tree is clean" rather than an error. +//! +//! Every test here asserts **both** halves. A one-sided assertion would +//! stay green against a later change that swept everything back onto one +//! stream, which is the regression this file exists to prevent. + +use std::fs; +use std::path::Path; + +use assert_cmd::Command; +use tempfile::TempDir; + +use crate::common; + +/// `classify` measures `cyclomatic = 5`, so any of these fixtures +/// offends at `cyclomatic=1`. The function name is parameterised so a +/// two-file corpus can attribute each row to its file by name alone. +fn branchy(name: &str) -> String { + format!( + "pub fn {name}(n: i32) -> i32 {{ + if n < 0 {{ return -1; }} + if n == 0 {{ return 0; }} + if n < 10 {{ return 1; }} + if n < 100 {{ return 2; }} + 3 +}} +" + ) +} + +/// A hermetic two-file corpus: `a.rs` holds `a_offender`, `b.rs` holds +/// `b_offender`. Both offend at `cyclomatic=1`, so a run over the +/// directory produces exactly two offender rows. +fn corpus() -> TempDir { + let dir = TempDir::new().expect("tempdir"); + fs::write(dir.path().join("a.rs"), branchy("a_offender")).expect("write a.rs"); + fs::write(dir.path().join("b.rs"), branchy("b_offender")).expect("write b.rs"); + dir +} + +fn cli(dir: &Path) -> Command { + common::cli_in(dir) +} + +/// A `bca check` over `dir` at `cyclomatic=1`, plus `extra` flags. +fn check(dir: &TempDir, extra: &[&str]) -> (String, String, Option) { + let out = cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + dir.path().to_str().expect("utf8 dir"), + "--threshold", + "cyclomatic=1", + ]) + .args(extra) + .output() + .expect("bca runs"); + ( + String::from_utf8(out.stdout).expect("utf8 stdout"), + String::from_utf8(out.stderr).expect("utf8 stderr"), + out.status.code(), + ) +} + +/// The offender rows in `text`, selected by the trailing `(limit N)` +/// that only a row carries. +/// +/// The selector matters more than it looks. The summary footer renders +/// `a.rs: 1 violation (worst: cyclomatic = 5 vs limit 1 at L1)`, so the +/// obvious `contains(": cyclomatic = ")` matches the *footer* and every +/// "no rows on stderr" assertion below would fail for the wrong reason — +/// or, with the polarity flipped, pass for the wrong one. ` (limit ` is +/// the one substring the footer does not contain. +fn rows(text: &str) -> Vec<&str> { + text.lines() + .filter(|line| line.contains(" (limit ")) + .collect() +} + +/// Every line with content, matched or not. Paired with [`rows`] this +/// turns "the rows are here" into "the rows are here *and nothing else +/// is*" — a claim no filtered count can make on its own. +fn non_blank_lines(text: &str) -> usize { + text.lines().filter(|line| !line.trim().is_empty()).count() +} + +/// Both halves of the contract for the default invocation: the rows are +/// on stdout, and stderr carries the footer without a single row. +#[test] +fn plain_run_puts_rows_on_stdout_and_the_footer_on_stderr() { + let dir = corpus(); + let (stdout, stderr, code) = check(&dir, &[]); + + let stdout_rows = rows(&stdout); + assert_eq!( + stdout_rows.len(), + 2, + "one row per offending function, on stdout; stdout was:\n{stdout}" + ); + // The "and nothing else" half of the contract, which every other + // assertion in this file is blind to: they all read `rows(&stdout)`, + // and a line the selector does not match is simply absent from that + // subset rather than a failure. Measured — appending one commentary + // line to the stdout writer failed none of the suite's 5053 tests. + // Counting *unfiltered* lines is what makes a stray line observable. + assert_eq!( + non_blank_lines(&stdout), + stdout_rows.len(), + "stdout carries the offender rows and nothing else; stdout was:\n{stdout}" + ); + assert!( + stdout_rows.iter().any(|r| r.contains("a_offender")) + && stdout_rows.iter().any(|r| r.contains("b_offender")), + "both offenders must be named on stdout; stdout was:\n{stdout}" + ); + assert!( + rows(&stderr).is_empty(), + "no offender row may reach stderr; stderr was:\n{stderr}" + ); + assert!( + stderr.contains("--- summary ---"), + "the footer is commentary and stays on stderr; stderr was:\n{stderr}" + ); + assert_eq!(code, Some(2), "the gate verdict is unchanged"); +} + +/// `--baseline`: the covered offender is dropped and the count that says +/// so stays on stderr, while the surviving row is on stdout. +#[test] +fn baseline_filtered_run_keeps_its_diagnostic_on_stderr() { + let dir = corpus(); + let baseline = dir.path().join("baseline.toml"); + + // Record only `a.rs`, so the later two-file run has exactly one + // covered offender and one uncovered one. + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + dir.path().join("a.rs").to_str().expect("utf8 path"), + "--threshold", + "cyclomatic=1", + "--write-baseline", + baseline.to_str().expect("utf8 path"), + ]) + .assert() + .success(); + + let (stdout, stderr, code) = + check(&dir, &["--baseline", baseline.to_str().expect("utf8 path")]); + + let stdout_rows = rows(&stdout); + assert_eq!( + stdout_rows.len(), + 1, + "only the uncovered offender survives; stdout was:\n{stdout}" + ); + assert!( + stdout_rows[0].contains("b_offender"), + "the surviving row is `b.rs`'s; stdout was:\n{stdout}" + ); + assert!( + rows(&stderr).is_empty(), + "no offender row may reach stderr; stderr was:\n{stderr}" + ); + assert!( + stderr.contains("bca: filtered 1 violations via baseline"), + "the filter count is a diagnostic and stays on stderr; stderr was:\n{stderr}" + ); + assert_eq!(code, Some(2), "the gate verdict is unchanged"); +} + +/// `[check.exclude]`: same shape, different diagnostic. +#[test] +fn check_exclude_run_keeps_its_diagnostic_on_stderr() { + let dir = corpus(); + let (stdout, stderr, code) = check(&dir, &["--check-exclude", "**/a.rs"]); + + let stdout_rows = rows(&stdout); + assert_eq!( + stdout_rows.len(), + 1, + "only the unexcluded offender survives; stdout was:\n{stdout}" + ); + assert!( + stdout_rows[0].contains("b_offender"), + "the surviving row is `b.rs`'s; stdout was:\n{stdout}" + ); + assert!( + rows(&stderr).is_empty(), + "no offender row may reach stderr; stderr was:\n{stderr}" + ); + assert!( + stderr.contains("bca: skipped 1 violations via [check.exclude]"), + "the skip count is a diagnostic and stays on stderr; stderr was:\n{stderr}" + ); + assert_eq!(code, Some(2), "the gate verdict is unchanged"); +} + +/// #1167 is a stream change and nothing else. Anyone scripting on `$?` +/// must see no difference, in the default contract and the tiered one. +/// +/// The tiered leg is the discriminating one: it exercises the +/// `--exit-codes=tiered` mapping (`NewOnly` → 2, `RegressionOnly` → 3), +/// which shares no code with the default collapse-to-2. +#[test] +fn exit_codes_are_unchanged_by_the_stream_split() { + let dir = corpus(); + let baseline = dir.path().join("baseline.toml"); + + assert_eq!( + check(&dir, &["--threshold", "cyclomatic=100"]).2, + Some(0), + "a clean run exits 0" + ); + assert_eq!(check(&dir, &[]).2, Some(2), "a breach exits 2"); + assert_eq!( + check(&dir, &["--no-fail"]).2, + Some(0), + "--no-fail forces 0 while still reporting" + ); + assert_eq!( + check(&dir, &["--exit-codes=tiered"]).2, + Some(2), + "unbaselined offenders are NewOnly, which is 2 in either contract" + ); + + // Baseline both offenders, then worsen one so the tiered contract + // has a category of its own to report. + cli(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + dir.path().to_str().expect("utf8 dir"), + "--threshold", + "cyclomatic=1", + "--write-baseline", + baseline.to_str().expect("utf8 path"), + ]) + .assert() + .success(); + fs::write( + dir.path().join("a.rs"), + branchy("a_offender").replace(" 3\n", " if n < 1000 { return 3; }\n 4\n"), + ) + .expect("worsen a.rs"); + + let (stdout, stderr, code) = check( + &dir, + &[ + "--exit-codes=tiered", + "--baseline", + baseline.to_str().expect("utf8 path"), + ], + ); + assert_eq!( + code, + Some(3), + "a lone baseline regression is RegressionOnly; stdout:\n{stdout}stderr:\n{stderr}" + ); + assert_eq!( + rows(&stdout).len(), + 1, + "the regressed row is on stdout; stdout was:\n{stdout}" + ); + assert!( + rows(&stderr).is_empty(), + "no offender row may reach stderr; stderr was:\n{stderr}" + ); +} + +/// `--summary-file` writes a separate artifact. It is not a stream, so +/// #1167 must not move it — and its digest must stay out of *both* +/// streams. +#[test] +fn summary_file_digest_is_a_separate_artifact() { + let dir = corpus(); + let digest = dir.path().join("summary.md"); + + let (stdout, stderr, code) = check( + &dir, + &["--summary-file", digest.to_str().expect("utf8 path")], + ); + assert_eq!(code, Some(2)); + + let written = fs::read_to_string(&digest).expect("digest written"); + assert!( + written.contains("bca-step-summary"), + "the digest keeps its replace markers; got:\n{written}" + ); + assert!( + written.contains("a_offender") && written.contains("b_offender"), + "the digest still names every offender; got:\n{written}" + ); + assert_eq!( + rows(&stdout).len(), + 2, + "the rows still go to stdout; stdout was:\n{stdout}" + ); + assert!( + rows(&stderr).is_empty(), + "no offender row may reach stderr; stderr was:\n{stderr}" + ); + for (name, stream) in [("stdout", &stdout), ("stderr", &stderr)] { + assert!( + !stream.contains("bca-step-summary"), + "the digest is a file, not a stream; {name} was:\n{stream}" + ); + } +} + +/// The one exception to the split, and the reason it exists: with +/// `--report-format` and no `--output`, the aggregated document owns +/// stdout. Mixing human rows into it would corrupt a SARIF payload, so +/// they stay on stderr for that combination alone. +#[test] +fn report_format_on_stdout_keeps_the_document_parseable() { + let dir = corpus(); + let (stdout, stderr, code) = check(&dir, &["--report-format", "sarif"]); + assert_eq!(code, Some(2)); + + let doc: serde_json::Value = + serde_json::from_str(&stdout).expect("stdout is an uncorrupted SARIF document"); + assert_eq!( + doc["runs"][0]["results"] + .as_array() + .expect("SARIF results array") + .len(), + 2, + "the document still carries both offenders" + ); + assert!( + rows(&stdout).is_empty(), + "no human row may be interleaved into the document; stdout was:\n{stdout}" + ); + assert_eq!( + rows(&stderr).len(), + 2, + "the human rows fall back to stderr here; stderr was:\n{stderr}" + ); +} + +/// `--output ` moves the document off stdout, so the rows go back +/// to it. Without this the exception above would read as "any +/// `--report-format` keeps the old behaviour", which is not the rule. +#[test] +fn report_format_with_output_file_puts_rows_back_on_stdout() { + let dir = corpus(); + let out = dir.path().join("report.sarif"); + + let (stdout, stderr, code) = check( + &dir, + &[ + "--report-format", + "sarif", + "--output", + out.to_str().expect("utf8 path"), + ], + ); + assert_eq!(code, Some(2)); + + let doc: serde_json::Value = + serde_json::from_str(&fs::read_to_string(&out).expect("document written")) + .expect("the file is a SARIF document"); + assert_eq!( + doc["runs"][0]["results"] + .as_array() + .expect("SARIF results array") + .len(), + 2, + "the document is unaffected by the stream split" + ); + assert_eq!( + rows(&stdout).len(), + 2, + "stdout is free, so the rows take it; stdout was:\n{stdout}" + ); + assert!( + rows(&stderr).is_empty(), + "no offender row may reach stderr; stderr was:\n{stderr}" + ); +} diff --git a/big-code-analysis-cli/tests/check/check_suppression.rs b/big-code-analysis-cli/tests/check/check_suppression.rs index 761adbc24..19c24d8b5 100644 --- a/big-code-analysis-cli/tests/check/check_suppression.rs +++ b/big-code-analysis-cli/tests/check/check_suppression.rs @@ -85,6 +85,7 @@ fn suppression_marker_silences_violation_by_default() { .args(["check", "--paths", &path, "--threshold", "cyclomatic=1"]) .assert() .success() + .stdout(predicate::str::is_empty()) .stderr(predicate::str::is_empty()); } @@ -106,8 +107,8 @@ fn no_suppress_flag_re_enables_violation() { ]) .assert() .code(2) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic")); + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic")); } #[test] @@ -122,6 +123,7 @@ fn lizard_compat_marker_silences_violation() { .args(["check", "--paths", &path, "--threshold", "cyclomatic=1"]) .assert() .success() + .stdout(predicate::str::is_empty()) .stderr(predicate::str::is_empty()); } @@ -137,6 +139,7 @@ fn file_scoped_marker_silences_nested_function_violation() { .args(["check", "--paths", &path, "--threshold", "cyclomatic=1"]) .assert() .success() + .stdout(predicate::str::is_empty()) .stderr(predicate::str::is_empty()); } @@ -170,7 +173,7 @@ fn legacy_allow_marker_does_not_suppress() { // surfaces clearly: // 1. exit code 2 — the violation is reported, the marker did not // suppress it; - // 2. stderr names the offender and metric — the violation line + // 2. stdout names the offender and metric — the violation line // exists and is intelligible; // 3. stderr names the bad verb — the user gets a diagnostic // pointing them at the rename, not a silent drop. @@ -181,17 +184,16 @@ fn legacy_allow_marker_does_not_suppress() { .args(["check", "--paths", &path, "--threshold", "cyclomatic=1"]) .assert() .code(2) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic")) .stderr(predicate::str::contains( "unknown bca directive verb 'allow'", )); } -/// Regression fixture for #896. The marker lists one valid metric -/// (`cyclomatic`) and one unknown one (`bogusmetric`). The contract is -/// that an unknown identifier voids the *entire* marker — so the -/// otherwise-valid `cyclomatic` entry must NOT suppress either. +/// Fixture for the mixed known/unknown metric list. The marker names +/// one recognized metric (`cyclomatic`) beside one that does not exist +/// (`bogusmetric`). const UNKNOWN_METRIC_RUST: &str = r#" pub fn classify(n: i32) -> &'static str { // bca: suppress(cyclomatic, bogusmetric) @@ -205,40 +207,222 @@ pub fn classify(n: i32) -> &'static str { } "#; +/// Fixture whose marker names *only* an unrecognized metric, so nothing +/// in it can suppress. Separates "skip the bad name" from "void the +/// marker": under both contracts the mixed fixture above differs, but +/// this one must behave identically — the violation still fires. +const ONLY_UNKNOWN_METRIC_RUST: &str = r#" +pub fn classify(n: i32) -> &'static str { + // bca: suppress(bogusmetric) + if n < 0 { + "neg" + } else if n == 0 { + "zero" + } else { + "pos" + } +} +"#; + #[test] -fn unknown_metric_voids_entire_marker() { - // Void-on-unknown regression (#896): a `bca: suppress(...)` marker - // whose list contains an unknown metric must warn to stderr AND - // void the *whole* marker — the valid `cyclomatic` sibling does not - // get to suppress on its own. This is the unknown-*metric* twin of - // `legacy_allow_marker_does_not_suppress` (which covers an unknown - // *verb*); the two take distinct `SuppressionError` variants and - // render distinct stderr strings, so each needs its own end-to-end - // pin through the binary. +fn unknown_metric_is_skipped_and_the_rest_of_the_list_suppresses() { + // Issue #1168 replaced the void-on-unknown contract (#896) with + // skip-and-report: an unrecognized name costs its own name and + // nothing else. `exit`-for-`nexits` is a typo `AGENTS.md` documents + // people making, and voiding the marker wholesale turned it into a + // suppression the author believed was active while the gate + // disagreed — the exact failure #1168 is about. // - // Three things must all be true; we pin each one independently so a - // regression in any single half surfaces clearly: - // 1. exit code 2 — the violation is reported, not silenced (the - // most dangerous regression would treat the unknown metric as - // suppress-all and swallow the violation); - // 2. stderr names the offender and metric — the violation line - // exists and is intelligible; - // 3. stderr carries the unknown-metric diagnostic — the user gets - // a typo pointer, not a silent drop. + // Skipping cannot widen scope: the honored set only ever shrinks. + // Three things must all be true, pinned independently: + // 1. exit code 0 — `cyclomatic` is genuinely suppressed; + // 2. stdout carries no offender row for it (#1167 put offender + // rows on stdout, diagnostics on stderr); + // 3. stderr still carries the unknown-metric diagnostic, so the + // typo is not silent. let dir = TempDir::new().unwrap(); let path = write_fixture(&dir, "branchy.rs", UNKNOWN_METRIC_RUST); + cli(dir.path()) + .args(["check", "--paths", &path, "--threshold", "cyclomatic=1"]) + .assert() + .code(0) + .stdout(predicate::str::contains("classify").not()) + .stderr(predicate::str::contains( + "unknown metric 'bogusmetric' in bca suppression marker", + )); +} + +#[test] +fn a_marker_naming_only_unknown_metrics_suppresses_nothing() { + // The other half of the #1168 contract: skipping every name in the + // list leaves an empty one, which silences nothing. A regression + // that treated an unusable list as a bare `suppress` (all metrics) + // would swallow the violation — the most dangerous direction, and + // the one the void-on-unknown rule was written to prevent. This + // still fires the violation, so that protection survives the + // relaxation. + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "branchy.rs", ONLY_UNKNOWN_METRIC_RUST); + cli(dir.path()) .args(["check", "--paths", &path, "--threshold", "cyclomatic=1"]) .assert() .code(2) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic")) .stderr(predicate::str::contains( "unknown metric 'bogusmetric' in bca suppression marker", )); } +/// The issue #1168 reproducer: two identical over-parameterised +/// functions, one marker bare and one carrying the rationale +/// `AGENTS.md` asks contributors to write. +const RATIONALE_RUST: &str = r" +pub fn many_bare(a: u8, b: u8, c: u8, d: u8, e: u8, f: u8, g: u8, h: u8) -> u8 { + // bca: suppress(nargs) + a + b + c + d + e + f + g + h +} + +pub fn many_prose(a: u8, b: u8, c: u8, d: u8, e: u8, f: u8, g: u8, h: u8) -> u8 { + // bca: suppress(nargs) — threaded context, not a god-function + a + b + c + d + e + f + g + h +} +"; + +#[test] +fn a_trailing_rationale_suppresses_exactly_like_a_bare_marker() { + // The #1168 reproducer end-to-end. Before the fix the second + // function's marker was rejected as malformed, so `many_prose` was + // reported while `many_bare` was not — a suppression its author had + // every reason to believe was active. + // + // Both halves are asserted: the run is clean (neither function is + // reported) *and* no `warning:` reaches stderr, since a marker that + // works must not also complain about itself. + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "wide.rs", RATIONALE_RUST); + + cli(dir.path()) + .args(["check", "--paths", &path, "--threshold", "nargs=5"]) + .assert() + .code(0) + .stdout(predicate::str::contains("many_prose").not()) + .stdout(predicate::str::contains("many_bare").not()) + .stderr(predicate::str::contains("warning:").not()); + + // Exit 0 plus an empty stdout is also what a tree with no violation + // at all produces, so the assertions above cannot tell "suppressed" + // from "never offended". `--no-suppress` supplies the positive + // control: with markers ignored, both functions must be reported. + cli(dir.path()) + .args([ + "check", + "--paths", + &path, + "--threshold", + "nargs=5", + "--no-suppress", + ]) + .assert() + .code(2) + .stdout(predicate::str::contains("many_bare")) + .stdout(predicate::str::contains("many_prose")); + + // …and the marker silences only the metric it names. Gating a + // second metric the list omits keeps both functions on the report, + // which is what separates `SuppressionScope::Some([nargs])` from a + // scope-widening `All` — the two are indistinguishable while + // `nargs` is the only metric under test. + cli(dir.path()) + .args([ + "check", + "--paths", + &path, + "--threshold", + "nargs=5", + "--threshold", + "cyclomatic=0", + ]) + .assert() + .code(2) + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("many_bare")) + .stdout(predicate::str::contains("many_prose")) + .stdout(predicate::str::contains("nargs").not()); +} + +#[test] +fn a_bare_verb_with_trailing_text_warns_without_failing_the_run() { + // The genuine-malformation path. A bare verb followed by words is + // not a marker — but it must warn rather than fail, or a doc comment + // or test fixture that merely mentions the syntax would break the + // gate. + // + // The exit code is 0 because the fixture has no violation at the + // threshold under test: the marker is inert *and* the malformation + // is not itself fatal. + let dir = TempDir::new().unwrap(); + let path = write_fixture( + &dir, + "prose.rs", + "pub fn tiny(a: u8) -> u8 {\n // bca: suppress markers are honoured here\n a\n}\n", + ); + + cli(dir.path()) + .args(["check", "--paths", &path, "--threshold", "nargs=5"]) + .assert() + .code(0) + .stderr(predicate::str::contains( + "malformed bca suppression marker body", + )); +} + +/// Two over-parameterised functions whose in-body comments are prose +/// *about* a marker, written with the punctuation a rationale would use. +const PROSE_ABOUT_A_MARKER_RUST: &str = r" +pub fn dashed(a: u8, b: u8, c: u8, d: u8, e: u8, f: u8, g: u8, h: u8) -> u8 { + // bca: suppress - we removed this marker, see #123 + a + b + c + d + e + f + g + h +} + +pub fn colonned(a: u8, b: u8, c: u8, d: u8, e: u8, f: u8, g: u8, h: u8) -> u8 { + // bca: suppress: not applicable to this function + a + b + c + d + e + f + g + h +} +"; + +#[test] +fn prose_about_a_bare_marker_neither_suppresses_nor_passes_silently() { + // #1168 briefly let a bare verb carry a rationale when it opened + // with `-`, `:`, `//`, `#`, or an em/en dash — the same punctuation + // an author uses to write *about* a marker. Both functions below + // then reported no violation at all, with nothing on stderr: the + // most expensive failure this tool has, since it looks exactly like + // compliant code. + // + // Both halves matter. The violations must fire (stdout, exit 2, per + // #1167's stream split), and the warning must reach stderr so the + // author of a genuinely-intended marker learns it did nothing. + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "prose_marker.rs", PROSE_ABOUT_A_MARKER_RUST); + + cli(dir.path()) + .args(["check", "--paths", &path, "--threshold", "nargs=5"]) + .assert() + .code(2) + .stdout(predicate::str::contains("dashed")) + .stdout(predicate::str::contains("colonned")) + .stderr(predicate::str::contains( + "malformed bca suppression marker body", + )) + // The warning is the only route out of this shape, so it has to + // say where to go: name the metrics, or move the reason up. + .stderr(predicate::str::contains("`bca: suppress()`")) + .stderr(predicate::str::contains("line above")); +} + #[test] fn unsuppressed_metric_still_violates() { // Per-metric scoping: `bca: suppress(cyclomatic)` leaves other diff --git a/big-code-analysis-cli/tests/check/check_thresholds.rs b/big-code-analysis-cli/tests/check/check_thresholds.rs index 4aa762a2a..b93733eca 100644 --- a/big-code-analysis-cli/tests/check/check_thresholds.rs +++ b/big-code-analysis-cli/tests/check/check_thresholds.rs @@ -51,7 +51,7 @@ fn check_clean_exits_zero_with_no_offenders() { } #[test] -fn check_violation_exits_two_with_stable_stderr() { +fn check_violation_exits_two_with_stable_stdout() { fixtures::cli_shared() .args([ "check", @@ -64,28 +64,28 @@ fn check_violation_exits_two_with_stable_stderr() { .code(2) // The classify function exceeds cyclomatic=1; the offender line // must mention the file, function name, metric, and limit in the - // documented format. - .stderr(predicate::str::contains(fixtures::branchy_rs())) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic")) - .stderr(predicate::str::contains("(limit 1)")); -} - -/// The violation report is buffered before it reaches stderr (#1115), so -/// it must be drained before anything else writes to either stream — -/// otherwise the very run that has offenders reports them out of order, -/// or (once `run_check` reaches `process::exit`, which runs no -/// destructors) not at all. + // documented format — on stdout, per the #1167 stream contract. + .stdout(predicate::str::contains(fixtures::branchy_rs())) + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("(limit 1)")); +} + +/// The stderr commentary block — summary footer, GHA annotations, +/// remediation — is buffered before it reaches fd 2 (#1115), so it must +/// be drained before anything else writes to either stream; otherwise +/// the very run that has offenders reports them out of order, or (once +/// `run_check` reaches `process::exit`, which runs no destructors) not +/// at all. /// /// `--summary-file` pointed at a path under a missing directory makes /// `write_step_summary` fail, which `emit_check_results` reports with a /// bare `eprintln!` — an unbuffered write straight to fd 2, emitted after /// the buffered block. Pinning the two in order is what makes the flush -/// observable: delete both the explicit `stderr.flush()` and the -/// `drop(stderr)` that precede the step-summary call and the diagnostic -/// overtakes the offenders, failing here. +/// observable: delete the `drop(stderr)` that precedes the step-summary +/// call and the diagnostic overtakes the footer, failing here. #[test] -fn check_violations_are_flushed_before_later_stderr_writes() { +fn check_stderr_block_is_flushed_before_later_stderr_writes() { let dir = TempDir::new().unwrap(); let path = write_fixture(&dir, "branchy.rs", BRANCHY_RUST); let unwritable = dir.path().join("absent").join("summary.md"); @@ -107,16 +107,16 @@ fn check_violations_are_flushed_before_later_stderr_writes() { .clone(); let stderr = String::from_utf8(output).expect("utf8 stderr"); - let offender = stderr - .find("classify") - .unwrap_or_else(|| panic!("offender line missing from stderr:\n{stderr}")); + let footer = stderr + .find("--- summary ---") + .unwrap_or_else(|| panic!("summary footer missing from stderr:\n{stderr}")); let diagnostic = stderr .find("failed to append step summary") .unwrap_or_else(|| panic!("step-summary diagnostic missing from stderr:\n{stderr}")); assert!( - offender < diagnostic, - "the buffered offender report must be flushed before the unbuffered \ + footer < diagnostic, + "the buffered stderr block must be flushed before the unbuffered \ step-summary diagnostic; got:\n{stderr}" ); } @@ -238,8 +238,8 @@ fn check_no_fail_keeps_exit_zero_but_still_reports() { ]) .assert() .success() - .stderr(predicate::str::contains("cyclomatic")) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("(limit 1)")); } #[test] @@ -389,8 +389,8 @@ fn check_reads_thresholds_from_toml_config() { ]) .assert() .code(2) - .stderr(predicate::str::contains("cyclomatic")) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("(limit 1)")); } #[test] @@ -417,6 +417,145 @@ fn check_cli_threshold_overrides_config() { .stderr(predicate::str::is_empty()); } +/// Issue #1165: a `--threshold` written with the bare `diff --metric` +/// alias *overrides* a config limit for the same metric instead of +/// adding a second, independent one. +/// +/// The config's `loc.sloc = 1` gates every fixture; the CLI's +/// `sloc=1000` is the same extractor and must replace it, so the run is +/// clean. Before the fix the two spellings were unrelated map keys and +/// the tight config limit still fired. +#[test] +fn check_cli_alias_threshold_overrides_the_dotted_config_limit() { + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "branchy.rs", BRANCHY_RUST); + let cfg_path = dir.path().join("thresholds.toml"); + fs::write(&cfg_path, "[thresholds]\n\"loc.sloc\" = 1\n").unwrap(); + + cli(dir.path()) + .args([ + "check", + "--paths", + &path, + "--config", + cfg_path.to_str().unwrap(), + "--threshold", + "sloc=1000", + ]) + .assert() + .success() + .stderr(predicate::str::is_empty()); +} + +/// The mirror of the test above, and the regression it guards: a config +/// written with the *alias* must still merge with a `--threshold` in the +/// *dotted* form. Canonicalising only the manifest keys would fix the +/// reported bug and break this one. +#[test] +fn check_dotted_cli_threshold_overrides_an_alias_config_limit() { + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "branchy.rs", BRANCHY_RUST); + let cfg_path = dir.path().join("thresholds.toml"); + fs::write(&cfg_path, "[thresholds]\nsloc = 1\n").unwrap(); + + cli(dir.path()) + .args([ + "check", + "--paths", + &path, + "--config", + cfg_path.to_str().unwrap(), + "--threshold", + "loc.sloc=1000", + ]) + .assert() + .success() + .stderr(predicate::str::is_empty()); +} + +/// Issue #1165: one metric, one offender line. With `loc.sloc = 1` and +/// `sloc = 2` both configured, the pre-fix gate resolved two independent +/// thresholds onto the same extractor and emitted the file's `loc.sloc` +/// breach twice. +/// +/// The chosen resolution is to reject the table rather than pick a +/// winner: a config that sets one metric under two spellings has no +/// defensible interpretation, and silently keeping whichever key sorts +/// last is the same surprise in a quieter form. +#[test] +fn check_rejects_one_metric_configured_under_two_spellings() { + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "branchy.rs", BRANCHY_RUST); + let cfg_path = dir.path().join("thresholds.toml"); + fs::write(&cfg_path, "[thresholds]\n\"loc.sloc\" = 1\nsloc = 2\n").unwrap(); + + cli(dir.path()) + .args([ + "check", + "--paths", + &path, + "--config", + cfg_path.to_str().unwrap(), + ]) + .assert() + // Exit 1 (tool error), not 2: a misconfigured table must stay + // distinguishable from a metric regression. + .code(1) + .stderr(predicate::str::contains( + "\"loc.sloc\" is already set in this table under its other spelling", + )); +} + +/// Issue #1165: the two neighbouring `--threshold` typo classes — a +/// metric that does not exist, and a family head that has no single +/// threshold scalar — report through the same surface at the same exit +/// code. +/// +/// #1165 canonicalises the CLI layer at its consumption site rather than +/// inside `parse_cli_threshold`, which *is* the clap `value_parser`. +/// Putting it in the parser would fire the ambiguity diagnostic at +/// argument-parse time, wrapped as clap's `invalid value '…' for +/// '--threshold '`, while `not_a_metric=1` — resolved +/// against the same registry, one layer down — kept the plain `error:` +/// form. Two wordings for two adjacent typos, and only one of them +/// carrying the did-you-mean list the other surface owns. +/// +/// The exit code is asserted for both, but note it is *not* what this +/// placement buys: `exit_clap_error` (#561/#594) already remaps clap's +/// exit 2 to `EXIT_TOOL_ERROR`, so a value-parser rejection would exit 1 +/// too. The shared surface is the property that actually moves. +#[test] +fn check_ambiguous_and_unknown_cli_metrics_share_one_error_surface() { + for (spec, expected) in [ + ("halstead=5", "ambiguous metric \"halstead\""), + ("not_a_metric=1", "unknown threshold metric"), + ] { + let assert = fixtures::cli_shared() + .args([ + "check", + "--paths", + fixtures::trivial_rs(), + "--threshold", + spec, + ]) + .assert() + // Exit 1 (tool error), never 2: CI must keep telling a + // misconfigured gate apart from a metric regression. + .code(1); + let stderr = String::from_utf8(assert.get_output().stderr.clone()).expect("utf8 stderr"); + assert!(stderr.contains(expected), "{spec}: {stderr}"); + assert!( + stderr.contains("did you mean") || stderr.contains("halstead.volume"), + "{spec}: both surfaces list the candidates: {stderr}" + ); + assert!( + !stderr.contains("invalid value"), + "{spec}: a clap value-parser wrapper means the two typo classes \ + now report through different surfaces: {stderr}" + ); + } +} + #[test] fn check_emits_one_line_per_metric_per_function() { // Two thresholds tight enough that the same function violates both. @@ -435,12 +574,12 @@ fn check_emits_one_line_per_metric_per_function() { .assert() .code(2); let output = assert.get_output(); - let stderr = String::from_utf8_lossy(&output.stderr); - let cyclomatic_lines = stderr + let stdout = String::from_utf8_lossy(&output.stdout); + let cyclomatic_lines = stdout .lines() .filter(|l| l.contains("classify") && l.contains("cyclomatic")) .count(); - let cognitive_lines = stderr + let cognitive_lines = stdout .lines() .filter(|l| l.contains("classify") && l.contains("cognitive")) .count(); @@ -451,7 +590,7 @@ fn check_emits_one_line_per_metric_per_function() { assert!( cyclomatic_lines == 1 && cognitive_lines == 1, "expected exactly one line per (function, metric) for classify; \ - got cyclomatic={cyclomatic_lines}, cognitive={cognitive_lines}; stderr was:\n{stderr}", + got cyclomatic={cyclomatic_lines}, cognitive={cognitive_lines}; stdout was:\n{stdout}", ); } @@ -483,15 +622,15 @@ fn check_uses_file_sentinel_for_top_level_space() { ]) .assert() .code(2); - let stderr = String::from_utf8_lossy(&assert.get_output().stderr); - let file_lines: Vec<&str> = stderr + let stdout = String::from_utf8_lossy(&assert.get_output().stdout); + let file_lines: Vec<&str> = stdout .lines() .filter(|l| l.contains("") && l.contains("loc.sloc")) .collect(); assert_eq!( file_lines.len(), 1, - "expected exactly one file-level violation line; stderr was:\n{stderr}", + "expected exactly one file-level violation line; stdout was:\n{stdout}", ); // The file path appears once as the location prefix; the function // slot is the sentinel, not the path. @@ -817,12 +956,12 @@ pub fn outer() -> i32 { .args(["check", "--paths", &path, "--threshold", "cyclomatic=1"]) .assert() .code(2); - let stderr = String::from_utf8_lossy(&assert.get_output().stderr); + let stdout = String::from_utf8_lossy(&assert.get_output().stdout); // The inner function is a child FuncSpace of `outer`; if the // recursion doesn't descend, we'd miss it entirely. assert!( - stderr.contains("inner"), - "expected nested function to be reported; stderr was:\n{stderr}", + stdout.contains("inner"), + "expected nested function to be reported; stdout was:\n{stdout}", ); } @@ -1263,9 +1402,9 @@ fn check_headroom_scales_config_limit_into_offender() { ]) .assert() .code(2) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic")) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("(limit 1)")); } /// The deprecated `--headroom ` alias now promotes the gate to @@ -1293,8 +1432,8 @@ fn check_headroom_alias_promotes_to_soft_tier() { .stderr(predicate::str::contains( "`--headroom ` is deprecated; use `--tier=soft=`", )) - .stderr(predicate::str::contains("cyclomatic")) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("(limit 1)")); } /// `--tier=soft --headroom 1.0` is the documented no-op. The limit is @@ -1475,6 +1614,127 @@ fn check_headroom_print_effective_config_shows_scaled_values_and_ratio() { .stdout(predicate::str::contains("tier = \"soft\"")); } +/// The soft ratio scales the *band*, not the number: it must always make +/// the soft tier stricter than the hard gate, whichever way the metric +/// points (#1166). +/// +/// A higher-is-worse limit is a ceiling, so 0.9 lowers it. A +/// lower-is-worse `mi.*` limit is a *floor*, so 0.9 must **raise** it — +/// `20 / 0.9`, not `20 * 0.9`. Multiplying put the early-warning floor +/// at 18, below the hard floor of 20, which no value could reach before +/// the hard gate: `--tier=soft` was a silent no-op for the whole `mi.*` +/// family. +/// +/// Both directions are asserted in one test on purpose. Each is a +/// one-character change away from the other, so a single-direction test +/// would let a future "simplify" invert both and stay green. +/// +/// All three entry points to the ratio share `scale_threshold` and are +/// exercised here, because fixing only the blanket path would leave a +/// `[thresholds.soft]` `"0.9x"` string inverted with nothing to catch it. +#[test] +fn check_soft_ratio_raises_an_mi_floor_and_lowers_a_cognitive_ceiling() { + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "branchy.rs", BRANCHY_RUST); + let hard = write_fixture( + &dir, + "hard.toml", + "[thresholds]\n\"mi.original\" = 20\ncognitive = 15\n", + ); + // The `"x"` form of the same 0.9, resolved through + // `SoftLimit::Scale` rather than the blanket-ratio loop. + let scaled = write_fixture( + &dir, + "scaled.toml", + "[thresholds]\n\"mi.original\" = 20\ncognitive = 15\n\ + [thresholds.soft]\n\"mi.original\" = \"0.9x\"\ncognitive = \"0.9x\"\n", + ); + + let resolved = |cfg: &str, tier_args: &[&str]| -> toml::Table { + let mut args = vec![ + "check", + "--paths", + &path, + "--config", + cfg, + "--print-effective-config", + "toml", + ]; + args.extend_from_slice(tier_args); + let assert = cli(dir.path()).args(args).assert().success(); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let parsed: toml::Table = toml::from_str(&stdout).expect("effective config is valid TOML"); + parsed["thresholds"] + .as_table() + .expect("[thresholds] is a table") + .clone() + }; + + for (label, cfg, tier_args) in [ + ("--tier=soft=0.9", hard.as_str(), &["--tier=soft=0.9"][..]), + // `--headroom` folds into the same `TierSpec::Soft(Some(0.9))`, + // and `make self-scan-headroom` is the motivating consumer. + ("--headroom 0.9", hard.as_str(), &["--headroom", "0.9"][..]), + ( + "[thresholds.soft] \"0.9x\"", + scaled.as_str(), + &["--tier=soft"][..], + ), + ] { + let thresholds = resolved(cfg, tier_args); + // 15 * 0.9 is exact in binary, so the ceiling is pinned exactly. + assert_eq!( + thresholds["cognitive"].as_float(), + Some(13.5), + "{label}: a higher-is-worse ceiling must come down" + ); + // 20 / 0.9 repeats, so the resolved floor is the quotient rounded + // *up* at 6 significant figures. The epsilon is 1e-9 rather than + // anything looser on purpose: the two wrong answers this test + // exists to reject — multiplying (18) and rounding to nearest + // (22.2222) — miss by 4.2 and 1e-4 respectively. + let floor = thresholds["mi.original"] + .as_float() + .expect("mi.original is a float"); + assert!( + (floor - 22.2223).abs() < 1e-9, + "{label}: expected an mi.original floor of 22.2223, got {floor}" + ); + assert!( + floor > 20.0, + "{label}: a lower-is-worse floor must rise above its hard limit, got {floor}" + ); + } +} + +/// A `[thresholds.soft]` limit looser than its own hard limit is +/// rejected for the lower-is-worse `mi.*` family too. +/// +/// #1141 added that guard for higher-is-worse metrics only, because the +/// inverted scaling of #1166 meant every `[thresholds] mi.original = N` +/// plus `--tier=soft` would have failed the run rather than been fixed +/// by it. With the scaling corrected the exclusion is gone, so an +/// `mi.*` soft *floor* below the hard floor — a band that can never fire +/// first — is the usage error it always was. +#[test] +fn check_soft_mi_floor_below_the_hard_floor_is_rejected() { + let dir = TempDir::new().unwrap(); + let path = write_fixture(&dir, "branchy.rs", BRANCHY_RUST); + let cfg = write_fixture( + &dir, + "thresholds.toml", + "[thresholds]\n\"mi.original\" = 20\n[thresholds.soft]\n\"mi.original\" = 10\n", + ); + + cli(dir.path()) + .args(["check", "--paths", &path, "--config", &cfg, "--tier=soft"]) + .assert() + .code(1) + .stderr(predicate::str::contains( + "[thresholds.soft] \"mi.original\": soft limit 10 is looser than the hard limit 20", + )); +} + // ─── [thresholds.soft] per-metric soft tier (#375) ──────────────────── // // `classify` in BRANCHY_RUST has cyclomatic == 5. A `[thresholds.soft]` @@ -1505,7 +1765,7 @@ fn check_soft_table_absolute_override_trips_at_soft_tier() { .args(["check", "--paths", &path, "--config", &cfg, "--tier=soft"]) .assert() .code(2) - .stderr(predicate::str::contains("cyclomatic = 5 (limit 3)")); + .stdout(predicate::str::contains("cyclomatic = 5 (limit 3)")); } /// `"NNx"` scale syntax resolves against the metric's hard limit: @@ -1525,7 +1785,7 @@ fn check_soft_table_scale_relative_resolves_against_hard() { .args(["check", "--paths", &path, "--config", &cfg, "--tier=soft"]) .assert() .code(2) - .stderr(predicate::str::contains("cyclomatic = 5 (limit 4)")); + .stdout(predicate::str::contains("cyclomatic = 5 (limit 4)")); } /// A metric absent from `[thresholds.soft]` inherits its hard limit at @@ -1886,7 +2146,7 @@ fn check_tokens_threshold_fires_under_narrowed_metric_selection() { ]) .assert() .code(2) - .stderr(predicate::str::is_match(r"classify: tokens = [1-9]\d* \(limit 5\)").unwrap()); + .stdout(predicate::str::is_match(r"classify: tokens = [1-9]\d* \(limit 5\)").unwrap()); } /// Narrowing must not change a single reported value (#1113). @@ -1921,13 +2181,13 @@ fn check_mi_value_is_identical_whether_or_not_the_walk_is_narrowed() { .assert() .code(2) .get_output() - .stderr + .stdout .clone(); - let stderr = String::from_utf8(out).expect("utf8 stderr"); - stderr + let stdout = String::from_utf8(out).expect("utf8 stdout"); + stdout .lines() .find(|l| l.contains("mi.original =")) - .unwrap_or_else(|| panic!("no mi.original offender in:\n{stderr}")) + .unwrap_or_else(|| panic!("no mi.original offender in:\n{stdout}")) .to_owned() }; @@ -1986,9 +2246,9 @@ fn check_multi_metric_config_gates_on_every_named_metric() { .assert() .code(2) .get_output() - .stderr + .stdout .clone(); - let stderr = String::from_utf8(out).expect("utf8 stderr"); + let stdout = String::from_utf8(out).expect("utf8 stdout"); for metric in [ "cyclomatic", @@ -1998,8 +2258,8 @@ fn check_multi_metric_config_gates_on_every_named_metric() { "halstead.difficulty", ] { assert!( - stderr.contains(&format!("classify: {metric} = ")), - "no {metric} offender in:\n{stderr}" + stdout.contains(&format!("classify: {metric} = ")), + "no {metric} offender in:\n{stdout}" ); } } diff --git a/big-code-analysis-cli/tests/check/main.rs b/big-code-analysis-cli/tests/check/main.rs index 99e67eec4..66584ed1b 100644 --- a/big-code-analysis-cli/tests/check/main.rs +++ b/big-code-analysis-cli/tests/check/main.rs @@ -17,7 +17,9 @@ mod action_enforcement; mod check_baseline; mod check_exclude; mod check_exit_codes; +mod check_explain_threshold; mod check_report_suppressed_scope; +mod check_stream_contract; mod check_suppression; mod check_thresholds; mod exemptions; diff --git a/big-code-analysis-cli/tests/check_lang_thresholds.rs b/big-code-analysis-cli/tests/check_lang_thresholds.rs index 1b80614b7..a8d90424f 100644 --- a/big-code-analysis-cli/tests/check_lang_thresholds.rs +++ b/big-code-analysis-cli/tests/check_lang_thresholds.rs @@ -21,15 +21,17 @@ fn cli(dir: &Path) -> Command { common::cli_in(dir) } -/// The offender lines from a `bca check` stderr stream, isolated from -/// the summary and remediation blocks. +/// The offender rows from a `bca check` stdout stream (#1167 put them +/// there; the summary and remediation blocks stay on stderr). /// -/// Filtering matters here: the remediation footer echoes the resolved -/// `--paths` list, so a bare `stderr.contains("branchy.c")` reads as an -/// offender even when C was gated clean — the precise false pass these -/// tests exist to rule out. -fn offenders(stderr: &str) -> Vec<&str> { - stderr +/// The ` (limit ` filter is kept rather than trusting the stream split: +/// it is the shape assertion these tests actually depend on, and it +/// still rules out the false pass they exist for — the remediation +/// footer echoes the resolved `--paths` list, so a bare +/// `contains("branchy.c")` reads as an offender even when C was gated +/// clean. +fn offenders(stdout: &str) -> Vec<&str> { + stdout .lines() .filter(|line| line.contains(" (limit ")) .collect() @@ -137,8 +139,8 @@ fn override_applies_to_its_language_only() { ); let assert = cli(dir.path()).arg("check").assert().code(2); - let stderr = String::from_utf8(assert.get_output().stderr.clone()).expect("utf8 stderr"); - let offenders = offenders(&stderr); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let offenders = offenders(&stdout); assert_eq!(offenders.len(), 1, "exactly one offender: {offenders:?}"); assert!( offenders[0].contains("branchy.rs") && offenders[0].ends_with("cyclomatic = 7 (limit 5)"), @@ -161,8 +163,8 @@ fn unoverridden_metric_inherits_the_global_limit() { ); let assert = cli(dir.path()).arg("check").assert().code(2); - let stderr = String::from_utf8(assert.get_output().stderr.clone()).expect("utf8 stderr"); - let offenders = offenders(&stderr); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let offenders = offenders(&stdout); assert_eq!(offenders.len(), 1, "exactly one offender: {offenders:?}"); assert!( offenders[0].ends_with("classify: cognitive = 6 (limit 4)"), @@ -188,8 +190,8 @@ fn corrective_overrides_at_both_ends_of_the_spread() { ); let assert = cli(dir.path()).arg("check").assert().code(2); - let stderr = String::from_utf8(assert.get_output().stderr.clone()).expect("utf8 stderr"); - let offenders = offenders(&stderr); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let offenders = offenders(&stdout); assert_eq!(offenders.len(), 1, "exactly one offender: {offenders:?}"); assert!( offenders[0].ends_with("Sample::Classify: cognitive = 6 (limit 4)"), @@ -222,8 +224,8 @@ fn a_metric_only_a_language_table_gates_is_still_computed() { ); let assert = cli(dir.path()).arg("check").assert().code(2); - let stderr = String::from_utf8(assert.get_output().stderr.clone()).expect("utf8 stderr"); - let offenders = offenders(&stderr); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let offenders = offenders(&stdout); assert_eq!(offenders.len(), 1, "exactly one offender: {offenders:?}"); assert!( offenders[0].ends_with("Sample: nom = 4 (limit 3)"), @@ -293,13 +295,14 @@ fn language_without_an_override_uses_the_global_table() { fs::write(dir.path().join("unknown.zzz"), "nothing parses this\n").expect("write fixture"); let assert = cli(dir.path()).arg("check").assert().code(2); - let stderr = String::from_utf8(assert.get_output().stderr.clone()).expect("utf8 stderr"); - let offenders = offenders(&stderr); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let offenders = offenders(&stdout); assert_eq!(offenders.len(), 1, "exactly one offender: {offenders:?}"); assert!( offenders[0].ends_with("Sample::Classify: cyclomatic = 7 (limit 5)"), "C# falls through to the global limit: {offenders:?}" ); + let stderr = String::from_utf8(assert.get_output().stderr.clone()).expect("utf8 stderr"); assert!( stderr.contains("skipping explicitly-named file with unrecognized language"), "an unrecognised file language never reaches the gate at all: {stderr}" @@ -323,7 +326,7 @@ fn cli_threshold_outranks_a_language_override() { .args(["check", "--threshold", "cyclomatic=6"]) .assert() .code(2) - .stderr(predicate::str::contains( + .stdout(predicate::str::contains( "classify: cyclomatic = 7 (limit 6)", )); } @@ -417,7 +420,7 @@ fn soft_tier_derives_from_the_language_hard_limit() { .args(["check", "--tier=soft=0.5", "--exit-codes=tiered"]) .assert() .code(2) - .stderr(predicate::str::contains( + .stdout(predicate::str::contains( "classify: cognitive = 6 (limit 5)", )); } @@ -439,7 +442,7 @@ fn soft_tier_still_escalates_past_the_language_hard_limit() { .args(["check", "--tier=soft=0.5", "--exit-codes=tiered"]) .assert() .code(5) - .stderr(predicate::str::contains("wide: cognitive = 12 (limit 5)")); + .stdout(predicate::str::contains("wide: cognitive = 12 (limit 5)")); } /// At the soft tier each table is resolved against its *own* hard @@ -661,8 +664,115 @@ fn a_language_only_manifest_warns_that_nothing_else_is_gated() { stderr.contains("no global [thresholds] table: only c is gated"), "the run must say the rest of the tree is ungated: {stderr}" ); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); assert!( - offenders(&stderr).is_empty(), - "nothing breaches a limit of 100: {stderr}" + offenders(&stdout).is_empty(), + "nothing breaches a limit of 100: {stdout}" + ); +} + +/// Issue #1165: a per-language table written with the bare +/// `diff --metric` alias overrides the global dotted limit for the same +/// metric, rather than adding a second threshold on the same extractor. +/// +/// This is the issue's own reproducer. Both fixtures measure +/// `loc.ploc = 9` at file scope; C raises the limit to 100 and must +/// therefore pass, while Rust keeps the global 6 and must fail. Before +/// the fix `ploc` and `loc.ploc` were unrelated map keys, so C's table +/// resolved *both* and the tighter global limit still fired — with +/// `--print-effective-config` showing the 100 that did not. +#[test] +fn a_language_alias_override_replaces_the_global_dotted_limit() { + let dir = polyglot_tree( + "paths = [\"branchy.c\", \"branchy.rs\"]\n\ + [thresholds]\n\ + \"loc.ploc\" = 6\n\ + [thresholds.lang.c]\n\ + ploc = 100\n", + ); + + let assert = cli(dir.path()).arg("check").assert().code(2); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let offenders = offenders(&stdout); + assert_eq!(offenders.len(), 1, "exactly one offender: {offenders:?}"); + assert!( + offenders[0].contains("branchy.rs") && offenders[0].ends_with("loc.ploc = 9 (limit 6)"), + "C is raised to 100 and passes; Rust keeps the global 6: {offenders:?}" + ); + + // The printed configuration must name the limit that fired — one + // entry per metric, under the canonical id. + let assert = cli(dir.path()) + .args(["check", "--print-effective-config", "toml"]) + .assert() + .success(); + let stdout = String::from_utf8(assert.get_output().stdout.clone()).expect("utf8 stdout"); + let parsed: toml::Table = toml::from_str(&stdout).expect("effective config is valid TOML"); + let c = parsed["thresholds"]["lang"]["c"] + .as_table() + .expect("[thresholds.lang.c] is a table"); + assert_eq!( + c.keys().collect::>(), + ["loc.ploc"], + "one entry, canonically spelled: {c:?}" + ); + assert_eq!(c["loc.ploc"].as_float(), Some(100.0)); +} + +/// Issue #1165: `--print-effective-config` piped back through +/// `--config` reproduces the *gate result*, not merely the printed +/// table — same offender lines, same exit code. +/// +/// This is the property the printed config exists to provide, and it is +/// strictly stronger than comparing the two printed tables against each +/// other: a duplicated extractor collapses into one key on the way +/// *out* (the printer is keyed by the registry's canonical id), so a +/// print-versus-reprint comparison agrees with itself while the gate +/// disagrees with both. Before the fix this run reported two offenders +/// and the round trip reported one. +#[test] +fn the_printed_config_round_trips_to_the_same_gate_result() { + let dir = polyglot_tree( + "paths = [\"branchy.c\", \"branchy.rs\"]\n\ + [thresholds]\n\ + \"loc.ploc\" = 6\n\ + [thresholds.lang.c]\n\ + ploc = 100\n", + ); + + // Both runs name the fixtures the same way, so the offender lines + // are comparable verbatim rather than through a path fixup: the + // manifest's own `paths` resolve against the manifest directory and + // come back absolute. + let paths = ["--paths", "branchy.c", "--paths", "branchy.rs"]; + + let direct = cli(dir.path()).arg("check").args(paths).assert().code(2); + let direct = String::from_utf8(direct.get_output().stdout.clone()).expect("utf8 stdout"); + + let printed = cli(dir.path()) + .args(["check", "--print-effective-config", "toml"]) + .assert() + .success(); + let printed = String::from_utf8(printed.get_output().stdout.clone()).expect("utf8 stdout"); + let echoed = dir.path().join("effective.toml"); + fs::write(&echoed, &printed).expect("write effective config"); + + let round_tripped = cli(dir.path()) + .args(["check", "--no-config"]) + .args(paths) + .args(["--config", echoed.to_str().expect("utf8 path")]) + .assert() + // Same exit code, asserted before the offender comparison: two + // empty offender lists would otherwise compare equal. + .code(2); + let round_tripped = + String::from_utf8(round_tripped.get_output().stdout.clone()).expect("utf8 stdout"); + + let direct = offenders(&direct); + assert_eq!(direct.len(), 1, "one offender before the round trip"); + assert_eq!( + direct, + offenders(&round_tripped), + "the printed config must gate identically to the config it was printed from" ); } diff --git a/big-code-analysis-cli/tests/cli_ux/cli_smoke.rs b/big-code-analysis-cli/tests/cli_ux/cli_smoke.rs index cfcbcce61..de202e5da 100644 --- a/big-code-analysis-cli/tests/cli_ux/cli_smoke.rs +++ b/big-code-analysis-cli/tests/cli_ux/cli_smoke.rs @@ -15,16 +15,11 @@ fn cli() -> Command { common::bca_command() } +/// A small fixture file known to the repo, resolved relative to the workspace +/// root so the path is valid regardless of the test runner's CWD. The shared +/// helper makes a missing integration corpus name itself (#1171). fn fixture_path() -> String { - let manifest = env!("CARGO_MANIFEST_DIR"); - let workspace = std::path::Path::new(manifest) - .parent() - .expect("manifest dir has parent"); - workspace - .join("tests/repositories/DeepSpeech/stats.py") - .to_str() - .expect("path is utf-8") - .to_string() + common::corpus_fixture_path() } #[test] diff --git a/big-code-analysis-cli/tests/cli_ux/corpus_fixture.rs b/big-code-analysis-cli/tests/cli_ux/corpus_fixture.rs new file mode 100644 index 000000000..ec606c8a0 --- /dev/null +++ b/big-code-analysis-cli/tests/cli_ux/corpus_fixture.rs @@ -0,0 +1,96 @@ +//! Unit tests for the integration-corpus guard in `tests/common`. +//! +//! Nineteen tests in this crate analyse a real source file from the +//! `DeepSpeech` corpus, which is a git submodule and therefore absent +//! from a fresh clone or `git worktree`. Before #1171 each of them +//! failed with `bca`'s generic `error: path does not exist: …`, which +//! reads as a bug in whatever the author was changing. +//! +//! The guard is tested against synthetic trees rather than the real +//! corpus: the corpus is present in every tree where this suite runs, +//! so a test that waited for its absence would never execute (see +//! `.claude/rules/testing.md`). + +use std::fs; + +use crate::common::corpus_checkout_hint; + +/// Build a synthetic workspace root whose `tests/repositories/Corpus` +/// directory holds `entries`, and return it with its `TempDir` guard. +fn workspace_with(entries: &[&str]) -> (tempfile::TempDir, std::path::PathBuf) { + let dir = tempfile::tempdir().expect("create tempdir"); + let corpus = dir.path().join("tests/repositories/Corpus"); + fs::create_dir_all(&corpus).expect("create corpus dir"); + for entry in entries { + fs::write(corpus.join(entry), "content\n").expect("write corpus entry"); + } + let root = dir.path().to_path_buf(); + (dir, root) +} + +#[test] +fn present_fixture_produces_no_hint() { + let (_guard, root) = workspace_with(&["stats.py"]); + assert_eq!(corpus_checkout_hint(&root, "Corpus", "stats.py"), None); +} + +#[test] +fn empty_corpus_reads_as_not_checked_out() { + let (_guard, root) = workspace_with(&[]); + let hint = corpus_checkout_hint(&root, "Corpus", "stats.py") + .expect("an empty corpus directory must produce a hint"); + assert!( + hint.contains("integration corpus not checked out"), + "hint must name the cause, got: {hint}" + ); + assert!( + hint.contains("make worktree-setup"), + "hint must name the remedy, got: {hint}" + ); + assert!( + hint.contains("--force"), + "hint must say the by-hand recovery needs --force, got: {hint}" + ); +} + +#[test] +fn a_corpus_holding_only_dot_git_still_reads_as_not_checked_out() { + // The interrupted-checkout shape: git writes the submodule's `.git` + // file before any content, so its presence alone is not evidence + // that the corpus is usable. + let (_guard, root) = workspace_with(&[".git"]); + let hint = corpus_checkout_hint(&root, "Corpus", "stats.py") + .expect("a corpus holding only .git must produce a hint"); + assert!( + hint.contains("integration corpus not checked out"), + "a lone .git must not read as partial content, got: {hint}" + ); +} + +#[test] +fn a_populated_corpus_missing_the_fixture_reads_as_partial() { + // The other #1171 tell: "a corpus directory containing only + // snapshots/". Some content arrived, the wanted file did not. + let (_guard, root) = workspace_with(&[".git", "README.rst"]); + let hint = corpus_checkout_hint(&root, "Corpus", "stats.py") + .expect("a corpus missing the fixture must produce a hint"); + assert!( + hint.contains("integration corpus partially checked out"), + "partial content must be distinguished from an absent corpus, got: {hint}" + ); + assert!( + hint.contains("--force"), + "the partial case is the one --force exists for, got: {hint}" + ); +} + +#[test] +fn an_absent_corpus_directory_reads_as_not_checked_out() { + let dir = tempfile::tempdir().expect("create tempdir"); + let hint = corpus_checkout_hint(dir.path(), "Corpus", "stats.py") + .expect("a missing corpus directory must produce a hint"); + assert!( + hint.contains("integration corpus not checked out"), + "hint must name the cause, got: {hint}" + ); +} diff --git a/big-code-analysis-cli/tests/cli_ux/init.rs b/big-code-analysis-cli/tests/cli_ux/init.rs index e2b6787d5..e29eb2eee 100644 --- a/big-code-analysis-cli/tests/cli_ux/init.rs +++ b/big-code-analysis-cli/tests/cli_ux/init.rs @@ -180,7 +180,7 @@ fn init_no_baseline_writes_empty_placeholder() { // Empty placeholder still carries the version key the loader // requires; without it `--baseline` would reject the file. assert!( - baseline.contains("version = 5"), + baseline.contains("version = 6"), "empty baseline placeholder missing version key: {baseline}" ); // No actual entries. diff --git a/big-code-analysis-cli/tests/cli_ux/main.rs b/big-code-analysis-cli/tests/cli_ux/main.rs index 550b654db..3336e1f35 100644 --- a/big-code-analysis-cli/tests/cli_ux/main.rs +++ b/big-code-analysis-cli/tests/cli_ux/main.rs @@ -9,6 +9,7 @@ mod common; mod cli_smoke; +mod corpus_fixture; mod deprecated_aliases; mod flag_scoping; mod help_text; diff --git a/big-code-analysis-cli/tests/cli_ux/manifest.rs b/big-code-analysis-cli/tests/cli_ux/manifest.rs index c20db8867..6144f4850 100644 --- a/big-code-analysis-cli/tests/cli_ux/manifest.rs +++ b/big-code-analysis-cli/tests/cli_ux/manifest.rs @@ -78,9 +78,9 @@ fn bare_check_uses_manifest_paths_and_thresholds() { .arg("check") .assert() .code(2) - .stderr(predicate::str::contains("classify")) - .stderr(predicate::str::contains("cyclomatic")) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("classify")) + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("(limit 1)")); } /// A loose manifest limit produces a clean run — proving the manifest @@ -94,6 +94,7 @@ fn bare_check_clean_under_loose_manifest_threshold() { .arg("check") .assert() .success() + .stdout(predicate::str::is_empty()) .stderr(predicate::str::is_empty()); } @@ -125,7 +126,7 @@ fn cli_threshold_overrides_manifest() { .args(["check", "--threshold", "cyclomatic=1"]) .assert() .code(2) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("(limit 1)")); } /// `--paths` overrides the manifest `paths`. Pointed at an empty @@ -161,7 +162,7 @@ fn config_file_merges_over_manifest_thresholds() { .arg(&cfg) .assert() .code(2) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("(limit 1)")); } /// The manifest `headroom` key scales the manifest `[thresholds]` at the @@ -178,11 +179,11 @@ fn manifest_headroom_scales_thresholds_at_soft_tier() { .args(["check", "--tier=soft"]) .assert() .code(2) - .stderr(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("cyclomatic")) // Pin the *scaled* limit: 100 × 0.01 = 1. A bare "cyclomatic" // match would pass even if the scale were wrong (e.g. headroom // ignored and the limit left at 100, or applied twice). - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("(limit 1)")); } /// The manifest `baseline` key is honored: a baseline that already @@ -284,7 +285,7 @@ fn manifest_baseline_fuzzy_match_is_honored() { .arg("check") .assert() .success() - .stderr(predicate::str::contains("[new]").not()) + .stdout(predicate::str::contains("[new]").not()) // Confirm the renamed function was actually covered via the // fuzzy fallback, not silently dropped by an empty parse. .stderr(predicate::str::contains("filtered 1 violations")); @@ -331,7 +332,7 @@ int f(double x) { .arg("check") .assert() .code(2) - .stderr(predicate::str::contains("[new]")); + .stdout(predicate::str::contains("[new]")); } /// Manifest discovery climbs from the working directory to the repo @@ -351,7 +352,7 @@ fn discovery_climbs_from_subdirectory() { .code(2) // `paths = ["."]` resolved against the manifest dir, so the // root `branchy.rs` is analysed even though cwd is two levels down. - .stderr(predicate::str::contains("classify")); + .stdout(predicate::str::contains("classify")); } /// Unrecognized top-level keys (forthcoming features such as @@ -450,8 +451,8 @@ fn manifest_cyclomatic_count_try_false_flips_gate_without_warning() { .arg("check") .assert() .code(2) - .stderr(predicate::str::contains("cyclomatic")) - .stderr(predicate::str::contains("(limit 2)")); + .stdout(predicate::str::contains("cyclomatic")) + .stdout(predicate::str::contains("(limit 2)")); // Key false → `?` is linear → cyclomatic 1 ≤ 2 → clean, no warning. let linear = try_fixture( @@ -462,6 +463,7 @@ fn manifest_cyclomatic_count_try_false_flips_gate_without_warning() { .arg("check") .assert() .success() + .stdout(predicate::str::is_empty()) .stderr(predicate::str::is_empty()); } @@ -491,7 +493,7 @@ fn manifest_soft_threshold_subtable_applies_only_at_soft_tier() { .args(["check", "--tier=soft"]) .assert() .code(2) - .stderr(predicate::str::contains("(limit 1)")); + .stdout(predicate::str::contains("(limit 1)")); } /// `bca init` must NOT consume an existing manifest: it scaffolds @@ -672,5 +674,5 @@ fn cli_cyclomatic_count_try_overrides_manifest_both_directions() { .args(["check", "--cyclomatic-count-try=true"]) .assert() .code(2) - .stderr(predicate::str::contains("cyclomatic")); + .stdout(predicate::str::contains("cyclomatic")); } diff --git a/big-code-analysis-cli/tests/common/mod.rs b/big-code-analysis-cli/tests/common/mod.rs index fec1889a3..7550cbf8a 100644 --- a/big-code-analysis-cli/tests/common/mod.rs +++ b/big-code-analysis-cli/tests/common/mod.rs @@ -29,6 +29,125 @@ pub mod fixtures; #[allow(dead_code)] pub mod validators; +/// Workspace-relative root of the integration corpora. Every entry +/// under it is a git submodule, so all of it is absent from a fresh +/// clone or a fresh `git worktree` until it is checked out. +const CORPUS_ROOT: &str = "tests/repositories"; + +/// The corpus holding [`FIXTURE_FILE`]. +const FIXTURE_CORPUS: &str = "DeepSpeech"; + +/// A small real-source file, relative to [`FIXTURE_CORPUS`]. Nineteen +/// tests in this crate analyse it, which is what makes its absence +/// worth a named diagnostic. +const FIXTURE_FILE: &str = "stats.py"; + +/// Absolute path to the workspace root, derived from this crate's +/// manifest directory rather than the process cwd (which the tests +/// move around). +fn workspace_root() -> std::path::PathBuf { + std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .parent() + .expect("manifest dir has parent") + .to_path_buf() +} + +/// Absolute path to the shared real-source fixture, panicking with a +/// message that names its own cause when the corpus is not checked out. +/// +/// Without the submodule these nineteen tests failed with `bca`'s +/// generic `error: path does not exist: …`, which reads as a bug in +/// whatever the author was changing rather than as missing setup — the +/// papercut #1171 is about. +#[allow(dead_code)] +pub fn corpus_fixture_path() -> String { + let root = workspace_root(); + if let Some(hint) = corpus_checkout_hint(&root, FIXTURE_CORPUS, FIXTURE_FILE) { + panic!("{hint}"); + } + root.join(CORPUS_ROOT) + .join(FIXTURE_CORPUS) + .join(FIXTURE_FILE) + .into_os_string() + .into_string() + .expect("fixture path is utf-8") +} + +/// The tail of [`corpus_fixture_path`] below the corpus root — +/// `DeepSpeech/stats.py` — in the platform's own spelling. +/// +/// Built by joining the same components [`corpus_fixture_path`] joins, +/// never from a literal. `bca` renders a report row with whatever +/// separator the walker was handed, and `--strip-prefix` is a plain +/// `str::strip_prefix`, so a test that spells this `"DeepSpeech/stats.py"` +/// asserts a unix path shape. That passed here and failed the +/// `windows-latest` leg the moment #1171 replaced a single +/// `join("tests/repositories/DeepSpeech/stats.py")` — whose embedded +/// slashes survive on Windows — with per-component joins. +#[allow(dead_code)] +pub fn corpus_fixture_suffix() -> String { + std::path::Path::new(FIXTURE_CORPUS) + .join(FIXTURE_FILE) + .into_os_string() + .into_string() + .expect("fixture suffix is utf-8") +} + +/// The `--strip-prefix` value that reduces [`corpus_fixture_path`] to +/// exactly [`corpus_fixture_suffix`]. +/// +/// Derived by removing the one from the other rather than by rebuilding +/// the parent, so `prefix + suffix == corpus_fixture_path()` holds by +/// construction on every platform — the property both `report +/// --strip-prefix` tests depend on. +#[allow(dead_code)] +pub fn corpus_fixture_strip_prefix() -> String { + let full = corpus_fixture_path(); + let suffix = corpus_fixture_suffix(); + full.strip_suffix(&suffix) + .unwrap_or_else(|| { + panic!("fixture path {full:?} does not end with its corpus-relative suffix {suffix:?}") + }) + .to_owned() +} + +/// `Some(diagnostic)` when `file` is missing from the `corpus` +/// submodule under `workspace_root`; `None` when it is present. +/// +/// Split from [`corpus_fixture_path`] so the diagnostic is testable +/// against a synthetic tree. The real corpus is checked out in any tree +/// where this suite runs, so a test that waited for its absence would +/// never execute — see `.claude/rules/testing.md`. +#[allow(dead_code)] +pub fn corpus_checkout_hint(workspace_root: &Path, corpus: &str, file: &str) -> Option { + let corpus_dir = workspace_root.join(CORPUS_ROOT).join(corpus); + if corpus_dir.join(file).exists() { + return None; + } + // A submodule git has started checking out always has its `.git` + // file, so ignore that entry when deciding whether any content + // landed. Everything else distinguishes "never initialized" from + // "initialized and then interrupted", and only the second needs the + // `--force`. + let has_content = std::fs::read_dir(&corpus_dir).is_ok_and(|mut entries| { + entries.any(|entry| entry.is_ok_and(|entry| entry.file_name() != ".git")) + }); + let state = if has_content { + "partially checked out" + } else { + "not checked out" + }; + Some(format!( + "integration corpus {state}: {} is missing. Run `make worktree-setup` \ + from the repository root. By hand it is `git submodule update --init \ + --force -- {CORPUS_ROOT}/{corpus}`, and the `--force` is \ + load-bearing: after an interrupted checkout the submodule HEAD \ + already matches the recorded SHA, so a plain re-run is a silent \ + no-op.", + corpus_dir.join(file).display(), + )) +} + /// Scrub CI-side env vars that `bca check` auto-detects from a /// freshly-built `Command`. On a GitHub Actions runner the parent /// process exports `GITHUB_STEP_SUMMARY` pointing to the runner's diff --git a/big-code-analysis-cli/tests/discovery/explicit_path_excludes.rs b/big-code-analysis-cli/tests/discovery/explicit_path_excludes.rs index ed6d04873..ef4e3fc77 100644 --- a/big-code-analysis-cli/tests/discovery/explicit_path_excludes.rs +++ b/big-code-analysis-cli/tests/discovery/explicit_path_excludes.rs @@ -80,7 +80,7 @@ fn explicit_path_overrides_manifest_exclude_and_warns_naming_the_glob() { .args(["check", "skipme/a.rs", "--no-summary", "--no-remediation"]) .assert() .code(2) - .stderr(predicate::str::contains("skipme_offender")) + .stdout(predicate::str::contains("skipme_offender")) .stderr(predicate::str::contains( "bca: warning: skipme/a.rs matches an exclude pattern (./skipme/**) \ but was named explicitly; analyzing anyway", @@ -102,8 +102,8 @@ fn same_manifest_exclude_still_drops_the_file_on_a_directory_walk() { .args(["check", "--no-summary", "--no-remediation"]) .assert() .code(2) - .stderr(predicate::str::contains("kept_offender")) - .stderr(predicate::str::contains("skipme_offender").not()) + .stdout(predicate::str::contains("kept_offender")) + .stdout(predicate::str::contains("skipme_offender").not()) // No override happened, so no warning — the diagnostic must not // fire for files the walk selected. .stderr(predicate::str::contains("named explicitly").not()); @@ -170,7 +170,7 @@ fn explicit_path_overrides_exclude_from_file_and_warns() { ]) .assert() .code(2) - .stderr(predicate::str::contains("skipme_offender")) + .stdout(predicate::str::contains("skipme_offender")) .stderr(predicate::str::contains( "matches an exclude pattern (./skipme/**)", )); @@ -237,7 +237,7 @@ fn override_warning_survives_a_mixed_case_extension() { .code(2) // Reported as an offender, so the file really was analyzed and // the unannounced override was a live one. - .stderr(predicate::str::contains("upper_offender")) + .stdout(predicate::str::contains("upper_offender")) .stderr(predicate::str::contains( "bca: warning: skipme/B.RS matches an exclude pattern (./skipme/**) \ but was named explicitly; analyzing anyway", @@ -258,7 +258,7 @@ fn explicit_path_does_not_override_check_exclude() { .args(["check", "skipme/a.rs", "--no-summary", "--no-remediation"]) .assert() .success() - .stderr(predicate::str::contains("skipme_offender").not()) + .stdout(predicate::str::contains("skipme_offender").not()) // The skip line proves the offender existed and was dropped; // without it a clean exit could mean the fixture stopped // offending. @@ -288,7 +288,7 @@ fn absolute_explicit_path_does_not_override_check_exclude() { ]) .assert() .success() - .stderr(predicate::str::contains("skipme_offender").not()) + .stdout(predicate::str::contains("skipme_offender").not()) .stderr(predicate::str::contains( "skipped 1 violations via [check.exclude]", )); @@ -314,7 +314,7 @@ fn explicit_path_does_not_override_include() { ]) .assert() .code(1) - .stderr(predicate::str::contains("skipme_offender").not()) + .stdout(predicate::str::contains("skipme_offender").not()) .stderr(predicate::str::contains("no input files matched")); } @@ -337,5 +337,5 @@ fn explicit_path_is_analyzed_when_include_admits_it() { ]) .assert() .code(2) - .stderr(predicate::str::contains("skipme_offender")); + .stdout(predicate::str::contains("skipme_offender")); } diff --git a/big-code-analysis-cli/tests/discovery/read_failures.rs b/big-code-analysis-cli/tests/discovery/read_failures.rs index cf2c38266..372f9b3f1 100644 --- a/big-code-analysis-cli/tests/discovery/read_failures.rs +++ b/big-code-analysis-cli/tests/discovery/read_failures.rs @@ -1146,4 +1146,171 @@ mod unix { "a closed consumer pipe must be silent; stderr: {stderr}" ); } + + /// A Rust corpus whose `bca check --threshold cyclomatic=1` offender + /// report is comfortably larger than [`PIPE_BUFFER_BYTES`]. Each + /// function branches once, so every one of them is an offender and + /// contributes one row. + fn many_offenders(dir: &TempDir) -> String { + use std::fmt::Write as _; + + let mut body = String::new(); + for i in 0..600 { + writeln!( + body, + "pub fn offender_number_{i}(n: i32) -> i32 {{ if n > {i} {{ 1 }} else {{ 0 }} }}" + ) + .expect("writing to a String is infallible"); + } + for name in ["one.rs", "two.rs", "three.rs", "four.rs"] { + write_fixture(dir, name, &body); + } + dir.path().to_str().expect("utf8 dir").to_owned() + } + + /// Run `bca check` over `source` with the child's stdout pointed at + /// `stdout`. Split from [`run_with_stdout`] because `check`'s + /// arguments are not the shared `--no-config --paths` shape: it needs + /// a threshold, and its clean-run control cannot assert exit 0. + fn run_check_with_stdout(dir: &TempDir, source: &str, stdout: Stdio) -> Output { + common::std_bca_command_in(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + source, + "--threshold", + "cyclomatic=1", + "--no-summary", + "--no-remediation", + ]) + .stdout(stdout) + .stderr(Stdio::piped()) + .spawn() + .expect("spawn bca") + .wait_with_output() + .expect("wait for bca") + } + + /// #1167 moved `check`'s offender rows onto stdout, which puts them + /// on the failure path #1132 hardened. An unwritable stdout must + /// therefore behave like every other subcommand's: `EXIT_TOOL_ERROR` + /// with the CLI's own `error:` diagnostic, never a panic and never + /// the gate's own exit 2 — a gate verdict whose evidence never + /// reached the consumer is exactly the silent-success shape the + /// stream change exists to remove. + #[test] + fn check_exits_one_when_stdout_cannot_be_written() { + let dir = TempDir::new().expect("tempdir"); + let source = write_fixture( + &dir, + "branchy.rs", + "pub fn f(n: i32) -> i32 { if n > 0 { 1 } else { 0 } }\n", + ); + let Some(full) = dev_full() else { + eprintln!("skipping: no write-failing /dev/full on this platform"); + return; + }; + + let failed = run_check_with_stdout(&dir, &source, full.into()); + let stderr = String::from_utf8_lossy(&failed.stderr).into_owned(); + assert_eq!( + failed.status.code(), + Some(1), + "`bca check` must exit 1 on an unwritable stdout; stderr: {stderr}" + ); + assert!( + stderr.contains("error: writing check offenders"), + "the diagnostic must name the emission that failed; stderr: {stderr}" + ); + assert!( + stderr.contains("No space left on device"), + "the diagnostic must name the I/O failure; stderr: {stderr}" + ); + assert!( + !stderr.contains("panicked at"), + "`bca check` must not panic on an unwritable stdout; stderr: {stderr}" + ); + + // Control: the same invocation against a writable stdout reaches + // the gate and reports the violation (exit 2), so the exit-1 + // above came from the write and not from a rejected flag set or + // an unusable fixture. + let ok = run_check_with_stdout(&dir, &source, Stdio::null()); + assert_eq!( + ok.status.code(), + Some(2), + "the fixture must reach the gate with a writable stdout; stderr: {}", + String::from_utf8_lossy(&ok.stderr) + ); + } + + /// The other half of #1167's stdout policy: `bca check | head -1` is + /// routine. `BrokenPipe` must stay exempt from the `die` above, and + /// the run must still report the *gate* verdict (exit 2) rather than + /// a tool error — otherwise CI piping the offender list through + /// `head` cannot tell a threshold breach from a broken build. + /// + /// The corpus size is load-bearing exactly as it is for `vcs` and + /// `dump` above: a report that fits in the pipe buffer is accepted + /// whole before the reader can close, the child never meets `EPIPE`, + /// and the test passes against any implementation at all. + #[test] + fn check_reports_the_gate_verdict_when_its_consumer_closes_the_pipe() { + use std::io::{BufRead, BufReader}; + + let dir = TempDir::new().expect("tempdir"); + let source = many_offenders(&dir); + + let control = run_check_with_stdout(&dir, &source, Stdio::piped()); + assert!( + control.stdout.len() > PIPE_BUFFER_BYTES, + "the offender report must outgrow the pipe buffer for the close \ + to be observable at all; got {} bytes", + control.stdout.len(), + ); + + let mut child = common::std_bca_command_in(dir.path()) + .args([ + "check", + "--no-config", + "--paths", + &source, + "--threshold", + "cyclomatic=1", + "--no-summary", + "--no-remediation", + ]) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .expect("spawn bca"); + + // Read one line, then close the read end — the `head -1` shape. + let stdout = child.stdout.take().expect("piped stdout"); + let mut reader = BufReader::new(stdout); + let mut first = String::new(); + reader.read_line(&mut first).expect("read first line"); + assert!( + first.contains(": cyclomatic = "), + "the first line should be an offender row, got {first:?}" + ); + drop(reader); + + let output = child.wait_with_output().expect("wait for bca"); + let stderr = String::from_utf8_lossy(&output.stderr).into_owned(); + assert_eq!( + output.status.code(), + Some(2), + "a closed consumer pipe must not change the gate verdict; stderr: {stderr}" + ); + assert!( + !stderr.contains("panicked at"), + "a closed consumer pipe must not panic; stderr: {stderr}" + ); + assert!( + stderr.is_empty(), + "a closed consumer pipe must be silent; stderr: {stderr}" + ); + } } diff --git a/big-code-analysis-cli/tests/discovery/walk_channel_completeness.rs b/big-code-analysis-cli/tests/discovery/walk_channel_completeness.rs index 393a8f51c..ee9f651a7 100644 --- a/big-code-analysis-cli/tests/discovery/walk_channel_completeness.rs +++ b/big-code-analysis-cli/tests/discovery/walk_channel_completeness.rs @@ -92,13 +92,11 @@ fn check_reports_the_same_violations_serially_and_in_parallel() { "--no-summary", ]; - let (_, serial) = run_at_jobs(&dir, "1", &args); - let (_, parallel) = run_at_jobs(&dir, "16", &args); + let (serial, _) = run_at_jobs(&dir, "1", &args); + let (parallel, _) = run_at_jobs(&dir, "16", &args); // 120 files, one offending function each: a dropped send shows up - // as a short count before the line comparison even runs. Count only - // offender lines — stderr also carries the trailing remediation - // block, which `--no-summary` does not suppress. + // as a short count before the line comparison even runs. let offenders = |text: &str| { text.lines() .filter(|l| l.contains(": cyclomatic = ")) @@ -109,6 +107,16 @@ fn check_reports_the_same_violations_serially_and_in_parallel() { 120, "fixture drifted; expected one offender per file:\n{serial}" ); + // The filtered count above cannot see a line the filter skips, and + // the serial-vs-parallel comparison cancels one that appears on both + // sides — so a stray non-offender line on the product stream is + // invisible to both. Since #1167 gave stdout to the offender rows + // alone, the unfiltered count is what pins that. + assert_eq!( + serial.lines().filter(|l| !l.trim().is_empty()).count(), + 120, + "stdout carries the offender rows and nothing else:\n{serial}" + ); assert_eq!( sorted_lines(&serial), sorted_lines(¶llel), diff --git a/big-code-analysis-cli/tests/output/html_report.rs b/big-code-analysis-cli/tests/output/html_report.rs index 6e8790e0a..16db63241 100644 --- a/big-code-analysis-cli/tests/output/html_report.rs +++ b/big-code-analysis-cli/tests/output/html_report.rs @@ -15,16 +15,11 @@ fn cli() -> Command { common::bca_command() } +/// A small fixture file known to the repo, resolved relative to the workspace +/// root so the path is valid regardless of the test runner's CWD. The shared +/// helper makes a missing integration corpus name itself (#1171). fn fixture_path() -> String { - let manifest = env!("CARGO_MANIFEST_DIR"); - let workspace = std::path::Path::new(manifest) - .parent() - .expect("manifest dir has parent"); - workspace - .join("tests/repositories/DeepSpeech/stats.py") - .to_str() - .expect("path is utf-8") - .to_string() + common::corpus_fixture_path() } #[test] @@ -153,21 +148,17 @@ fn report_html_is_deterministic_across_runs() { #[test] fn report_html_strip_prefix_removes_path_prefix() { let fp = fixture_path(); - let prefix = { - let idx = fp - .find("DeepSpeech/") - .expect("fixture contains DeepSpeech/"); - &fp[..idx] - }; + let prefix = common::corpus_fixture_strip_prefix(); + let suffix = common::corpus_fixture_suffix(); let output = cli() - .args(["report", "--paths", &fp, "html", "--strip-prefix", prefix]) + .args(["report", "--paths", &fp, "html", "--strip-prefix", &prefix]) .output() .expect("invocation"); assert!(output.status.success()); let body = String::from_utf8(output.stdout).expect("utf-8"); assert!( - body.contains("DeepSpeech/stats.py"), - "stripped path should appear in HTML report" + body.contains(&suffix), + "stripped path {suffix:?} should appear in HTML report" ); // `--strip-prefix` rewrites the per-file table paths only. The provenance // footer (issue #680) deliberately records the literal seed path the user diff --git a/big-code-analysis-cli/tests/output/markdown_format.rs b/big-code-analysis-cli/tests/output/markdown_format.rs index a66a53d52..e162a4396 100644 --- a/big-code-analysis-cli/tests/output/markdown_format.rs +++ b/big-code-analysis-cli/tests/output/markdown_format.rs @@ -10,17 +10,10 @@ fn cli() -> Command { } /// A small fixture file known to the repo, resolved relative to the workspace -/// root so the path is valid regardless of the test runner's CWD. +/// root so the path is valid regardless of the test runner's CWD. The shared +/// helper makes a missing integration corpus name itself (#1171). fn fixture_path() -> String { - let manifest = env!("CARGO_MANIFEST_DIR"); // .../big-code-analysis-cli - let workspace = std::path::Path::new(manifest) - .parent() - .expect("manifest dir has parent"); - workspace - .join("tests/repositories/DeepSpeech/stats.py") - .to_str() - .expect("path is utf-8") - .to_string() + common::corpus_fixture_path() } #[test] @@ -283,12 +276,8 @@ fn report_renders_nonzero_tokens_for_real_file() { #[test] fn report_strip_prefix_removes_path_prefix() { let fp = fixture_path(); - let prefix = { - let idx = fp - .find("DeepSpeech/") - .expect("fixture contains DeepSpeech/"); - &fp[..idx] - }; + let prefix = common::corpus_fixture_strip_prefix(); + let suffix = common::corpus_fixture_suffix(); let output = cli() .args([ "report", @@ -296,15 +285,15 @@ fn report_strip_prefix_removes_path_prefix() { &fp, "markdown", "--strip-prefix", - prefix, + &prefix, ]) .output() .unwrap(); assert!(output.status.success()); let stdout = String::from_utf8_lossy(&output.stdout); assert!( - stdout.contains("DeepSpeech/stats.py"), - "stripped path should appear in report" + stdout.contains(&suffix), + "stripped path {suffix:?} should appear in report" ); // `--strip-prefix` rewrites the per-file table paths only. The provenance // footer (issue #680) deliberately records the literal seed path the user diff --git a/man/bca-check.1 b/man/bca-check.1 index 825d09918..1bb147003 100644 --- a/man/bca-check.1 +++ b/man/bca-check.1 @@ -4,14 +4,23 @@ .SH NAME bca\-check \- Check per\-function metrics against thresholds. Exits 2 when any threshold is exceeded; reserve exit 1 for tool errors so CI can distinguish "metric regression" from "tool crashed". `\-\-strict\-exit\-codes` opts into tiered codes (2\-5) that split the violation case by severity .SH SYNOPSIS -\fBbca\-check\fR [\fB\-p\fR|\fB\-\-paths\fR] [\fB\-I\fR|\fB\-\-include\fR] [\fB\-X\fR|\fB\-\-exclude\fR] [\fB\-l\fR|\fB\-\-language\fR] [\fB\-\-no\-skip\-generated\fR] [\fB\-\-paths\-from\fR] [\fB\-\-exclude\-from\fR] [\fB\-\-no\-ignore\fR] [\fB\-\-no\-config\fR] [\fB\-j\fR|\fB\-\-jobs\fR] [\fB\-\-exclude\-tests\fR] [\fB\-\-cyclomatic\-count\-try\fR] [\fB\-\-preproc\-data\fR] [\fB\-\-threshold\fR] [\fB\-\-config\fR] [\fB\-\-no\-fail\fR] [\fB\-\-no\-suppress\fR] [\fB\-\-report\-suppressed\fR] [\fB\-\-report\-format\fR] [\fB\-o\fR|\fB\-\-output\fR] [\fB\-\-baseline\fR] [\fB\-\-write\-baseline\fR] [\fB\-\-no\-summary\fR] [\fB\-\-since\fR] [\fB\-\-changed\-only\fR] [\fB\-\-github\-annotations\fR] [\fB\-\-summary\-file\fR] [\fB\-\-no\-remediation\fR] [\fB\-\-print\-effective\-config\fR] [\fB\-\-tier\fR] [\fB\-\-exit\-codes\fR] [\fB\-\-baseline\-line\-tolerance\fR] [\fB\-\-baseline\-fuzzy\-match\fR] [\fB\-\-check\-exclude\fR] [\fB\-\-check\-exclude\-from\fR] [\fB\-h\fR|\fB\-\-help\fR] [\fIPATHS\fR] +\fBbca\-check\fR [\fB\-p\fR|\fB\-\-paths\fR] [\fB\-I\fR|\fB\-\-include\fR] [\fB\-X\fR|\fB\-\-exclude\fR] [\fB\-l\fR|\fB\-\-language\fR] [\fB\-\-no\-skip\-generated\fR] [\fB\-\-paths\-from\fR] [\fB\-\-exclude\-from\fR] [\fB\-\-no\-ignore\fR] [\fB\-\-no\-config\fR] [\fB\-j\fR|\fB\-\-jobs\fR] [\fB\-\-exclude\-tests\fR] [\fB\-\-cyclomatic\-count\-try\fR] [\fB\-\-preproc\-data\fR] [\fB\-\-threshold\fR] [\fB\-\-explain\-threshold\fR] [\fB\-\-config\fR] [\fB\-\-no\-fail\fR] [\fB\-\-no\-suppress\fR] [\fB\-\-report\-suppressed\fR] [\fB\-\-report\-format\fR] [\fB\-o\fR|\fB\-\-output\fR] [\fB\-\-baseline\fR] [\fB\-\-write\-baseline\fR] [\fB\-\-no\-summary\fR] [\fB\-\-since\fR] [\fB\-\-changed\-only\fR] [\fB\-\-github\-annotations\fR] [\fB\-\-summary\-file\fR] [\fB\-\-no\-remediation\fR] [\fB\-\-print\-effective\-config\fR] [\fB\-\-tier\fR] [\fB\-\-exit\-codes\fR] [\fB\-\-baseline\-line\-tolerance\fR] [\fB\-\-baseline\-fuzzy\-match\fR] [\fB\-\-check\-exclude\fR] [\fB\-\-check\-exclude\-from\fR] [\fB\-h\fR|\fB\-\-help\fR] [\fIPATHS\fR] .SH DESCRIPTION -Check per\-function metrics against thresholds. Exits 2 when any threshold is exceeded; reserve exit 1 for tool errors so CI can distinguish "metric regression" from "tool crashed". `\-\-strict\-exit\-codes` opts into tiered codes (2\-5) that split the violation case by severity +Check per\-function metrics against thresholds. Exits 2 when any threshold is exceeded; reserve exit 1 for tool errors so CI can distinguish "metric regression" from "tool crashed". `\-\-strict\-exit\-codes` opts into tiered codes (2\-5) that split the violation case by severity. +.PP +Streams: the offender rows go to stdout, so `bca check | wc \-l` and `bca check 2>/dev/null` see them. The summary footer, remediation block, GitHub Actions annotations, and every `bca:` / `warning:` / `error:` diagnostic go to stderr. The one exception is `\-\-report\-format` without `\-\-output`: the aggregated document takes stdout there and the human rows fall back to stderr, so a SARIF payload stays parseable. .SH OPTIONS .TP \fB\-\-threshold\fR \fI\fR Threshold expressed as `=`. Repeatable. Metric names match `bca list\-metrics`; sub\-metrics use a dotted form (e.g. `loc.lloc`, `halstead.volume`). The bare `bca diff \-\-metric` spelling of a `loc` sub\-metric is accepted as an alias (`sloc` == `loc.sloc`); a bare family head with no single scalar (`halstead`, `mi`) is rejected with a "did you mean" hint. CLI flags override values from `\-\-config`. Limits must be finite and non\-negative; `0` is allowed and means "no value permitted" .TP +\fB\-\-explain\-threshold\fR \fI\fR +Preview what a candidate `=` would cost, at both tiers, instead of gating. Reports the hard\-tier offender count, the resolved soft limit and its offender count, and how many of each already match a `\-\-baseline` entry — so the decision\-grade figure (new baseline entries the change would add) is on the screen. Repeatable, one candidate per metric; every other metric is left out of the walk. + +This exists because `\-\-threshold` cannot answer the question: its limits are applied last and absolutely, never scaled, so a candidate trialled that way has no soft tier at all and reads as free when it is not. The soft band is derived from the candidate itself, at the `\-\-tier=soft=RATIO` ratio when one is given and 0.95 otherwise. + +Honours `exclude_tests`, `[check] exclude`, in\-source suppression markers and the baseline exactly as the run it predicts, and writes nothing: no gate runs, so it always exits 0 on success (1 on a tool error, such as a candidate naming a metric this build does not gate). Conflicts with `\-\-write\-baseline`, `\-\-print\-effective\-config`, `\-\-report\-format`, `\-\-output`, and an explicit `\-\-summary\-file `, each of which would produce a second, different artifact. +.TP \fB\-\-config\fR \fI\fR Path to a TOML config with a `[thresholds]` table. Example: @@ -22,16 +31,16 @@ Path to a TOML config with a `[thresholds]` table. Example: CLI `\-\-threshold` flags override values read from this file. .TP \fB\-\-no\-fail\fR -Print offenders to stderr but exit 0 even when thresholds are exceeded. Useful while adopting baselines without flipping CI red. Default: exit 2 when any threshold is exceeded +Report offenders as usual but exit 0 even when thresholds are exceeded. Useful while adopting baselines without flipping CI red. Default: exit 2 when any threshold is exceeded .TP \fB\-\-no\-suppress\fR Ignore in\-source suppression markers (`bca: suppress`, `#lizard forgives`, etc.). Every threshold violation is reported regardless of comment\-based silencers. CI auditors pass this to see the raw, un\-silenced offender list .TP \fB\-\-report\-suppressed\fR -Surface suppressed debt in the offender document instead of dropping it. Offenders silenced by an in\-source `bca: suppress` marker or covered by the baseline are still kept out of the gate (exit code and human stream are unaffected), but are emitted into the `\-\-format sarif` document carrying a SARIF `suppressions` entry — GitHub Code Scanning renders them as suppressed (closed) alerts so the debt stays visible. Only the SARIF format represents suppression; other formats ignore the flag. Mutually exclusive with `\-\-no\-suppress` (which un\-silences markers) and `\-\-write\-baseline` +Surface suppressed debt in the offender document instead of dropping it. Offenders silenced by an in\-source `bca: suppress` marker or covered by the baseline are still kept out of the gate (exit code and offender rows are unaffected), but are emitted into the `\-\-format sarif` document carrying a SARIF `suppressions` entry — GitHub Code Scanning renders them as suppressed (closed) alerts so the debt stays visible. Only the SARIF format represents suppression; other formats ignore the flag. Mutually exclusive with `\-\-no\-suppress` (which un\-silences markers) and `\-\-write\-baseline` .TP \fB\-\-report\-format\fR \fI\fR -CI/IDE *report dialect* for offender records (Checkstyle 4.3 XML, SARIF 2.1.0 JSON, GitLab Code Climate JSON, clang/GCC warning lines, MSVC warning lines). Named `\-\-report\-format` to separate "which CI report dialect" from the data\-serialization `\-\-format` the structured subcommands use. When omitted *and* `\-\-output` is also omitted, only the human\-readable stderr stream is emitted; the exit\-code contract is unaffected. When omitted but `\-\-output` is given, the dialect is inferred from the output extension (`.sarif` → sarif, `.xml` → checkstyle); an extension with no unique dialect is a usage error. The old `\-\-format` / `\-O` / `\-\-output\-format` spellings stay hidden aliases for one release cycle and are slated for removal in the next major +CI/IDE *report dialect* for offender records (Checkstyle 4.3 XML, SARIF 2.1.0 JSON, GitLab Code Climate JSON, clang/GCC warning lines, MSVC warning lines). Named `\-\-report\-format` to separate "which CI report dialect" from the data\-serialization `\-\-format` the structured subcommands use. When omitted *and* `\-\-output` is also omitted, only the human\-readable offender rows are emitted; the exit\-code contract is unaffected. Note that passing this flag without `\-\-output` gives the document stdout, so the human rows fall back to stderr for that combination — add `\-\-output ` to keep both. When omitted but `\-\-output` is given, the dialect is inferred from the output extension (`.sarif` → sarif, `.xml` → checkstyle); an extension with no unique dialect is a usage error. The old `\-\-format` / `\-O` / `\-\-output\-format` spellings stay hidden aliases for one release cycle and are slated for removal in the next major .br .br @@ -59,7 +68,7 @@ Filter known offenders listed in this TOML baseline. A baselined function whose Walk the tree and write the current offender set to a baseline file instead of failing. The resulting file pins today\*(Aqs metric values as the baseline; subsequent `\-\-baseline ` runs ratchet down from there. Takes an optional path: `\-\-write\-baseline ` writes there; a bare `\-\-write\-baseline` (no value) writes to the `[check] baseline` key from the auto\-discovered `bca.toml` manifest (errors if no manifest baseline is set). Conflicts with `\-\-baseline`, `\-\-report\-format`, `\-\-output`, `\-\-since`, and `\-\-changed\-only` — diff\-scope filtering would write a *partial* baseline that the next non\-`\-\-changed\-only` run would treat as a complete snapshot, silently masking every offender outside the diff scope .TP \fB\-\-no\-summary\fR -Skip the trailing per\-file rollup footer. The footer groups violations by file and cites the single worst\-ratio metric per file. Pass this when downstream tooling grep\-pipes the stderr stream and would be confused by the trailing summary block. Default: footer enabled +Skip the trailing per\-file rollup footer. The footer groups violations by file and cites the single worst\-ratio metric per file. It is written to stderr, so a plain `bca check | ...` pipeline never sees it; pass this when a tool reads the *merged* streams and would be confused by the trailing summary block. Default: footer enabled .TP \fB\-\-since\fR \fI\fR Git ref to diff `HEAD` against. The set of files reported by `git diff \-\-name\-only ...HEAD` is surfaced first in the summary footer under "Files in this range:", so a reader scanning a CI log sees their own contributions before the legacy offender list. Defaults to auto\-detection from `BCA_DIFF_BASE`, `GITHUB_BASE_REF` (PR runs), or `GITHUB_EVENT_BEFORE` (push runs), in that precedence @@ -68,7 +77,7 @@ Git ref to diff `HEAD` against. The set of files reported by `git diff \-\-name\ Drop violations from files outside the `\-\-since`/auto\-detected touched set entirely (terser CI output for PR gates). Requires a resolvable diff base, either via `\-\-since` or one of the auto\-detected env vars; failing to resolve is fatal so a misconfigured CI does not silently turn the gate into a no\-op .TP \fB\-\-github\-annotations\fR[=\fI\fR] [default: auto] -Emit GitHub Actions `::error file=…,line=…,title=…::msg` workflow commands per violation so the GHA UI renders them as inline annotations on the file\-diff view. Additive to the human\-readable stderr stream — annotations ride on top, they don\*(Aqt replace it. Tri\-state `` mirroring `\-\-color`: `auto` (default) emits annotations when `$GITHUB_ACTIONS == "true"`; `always` forces them on; `never` suppresses them even inside a GHA step (so a workflow that runs `bca check` twice can annotate from only one run). A bare `\-\-github\-annotations` means `always`. Capped at 10 per metric (GitHub Actions surfaces at most 10 errors per step in the UI); overflow rolls up to one `::error::N more violations not shown` line per affected metric so the count is still visible +Emit GitHub Actions `::error file=…,line=…,title=…::msg` workflow commands per violation so the GHA UI renders them as inline annotations on the file\-diff view. Written to stderr, additive to the human\-readable offender rows — annotations ride on top, they don\*(Aqt replace them. Tri\-state `` mirroring `\-\-color`: `auto` (default) emits annotations when `$GITHUB_ACTIONS == "true"`; `always` forces them on; `never` suppresses them even inside a GHA step (so a workflow that runs `bca check` twice can annotate from only one run). A bare `\-\-github\-annotations` means `always`. Capped at 10 per metric (GitHub Actions surfaces at most 10 errors per step in the UI); overflow rolls up to one `::error::N more violations not shown` line per affected metric so the count is still visible .br .br @@ -104,12 +113,14 @@ json \fB\-\-tier\fR[=\fI\fR] [default: hard] Which threshold tier to gate against. Accepts `hard`, `soft`, or `soft=`: -\- `hard` (default) — flag a function only when a metric is at or over its `[thresholds]` limit. \- `soft` — early\-warning tier: flag a function when a metric reaches `RATIO` (default 0.95) of any limit, i.e. before the hard gate trips. With a `[thresholds.soft]` table present, the per\-metric soft limits take precedence over the blanket ratio (metrics absent from it inherit their hard limit). \- `soft=0.90` — soft tier scaling every limit by 0.90; `soft=1.0` disables the blanket scale (a soft tier driven only by an explicit `[thresholds.soft]` table). +\- `hard` (default) — flag a function only when a metric is at or over its `[thresholds]` limit. \- `soft` — early\-warning tier: tighten every limit by `RATIO` (default 0.95) so a function is flagged before the hard gate trips. With a `[thresholds.soft]` table present, the per\-metric soft limits take precedence over the blanket ratio (metrics absent from it inherit their hard limit). \- `soft=0.90` — soft tier tightening every limit by 0.90; `soft=1.0` disables the blanket scale (a soft tier driven only by an explicit `[thresholds.soft]` table). + +`RATIO` scales the band, not the number: a ceiling comes down (`cognitive = 15` warns at 13.5), while a lower\-is\-worse `mi.*` floor goes up (`mi.original = 20` warns at 22.2223). Resolution order: `[thresholds]` (manifest + `\-\-config`) → `[thresholds.soft]` or the soft ratio → absolute `\-\-threshold name=value` overrides (applied last, never scaled). Both tiers ratchet through the same `\-\-baseline`. `RATIO` must lie in `(0, 1]`; an out\-of\-range value is a usage error. .TP \fB\-\-exit\-codes\fR[=\fI\fR] -Exit\-code style: `default` keeps the stable 0/1/2 contract; `tiered` splits exit `2` by severity so CI can branch without parsing the `[new]` / `[regr +N%]` stderr tags: +Exit\-code style: `default` keeps the stable 0/1/2 contract; `tiered` splits exit `2` by severity so CI can branch without parsing the `[new]` / `[regr +N%]` row tags: \- `0` — clean. \- `1` — tool error (bad config, unknown metric, unreadable path). \- `2` — new offenders only (no baseline entry matched). \- `3` — regressions only (a baselined offender worsened). \- `4` — both new offenders and regressions. \- `5` — at least one `\-\-tier=soft` violation also breaches the hard limit (more urgent than soft\-band encroachment). Only emitted at the soft tier; at the hard tier every violation is a hard breach by definition, so the 2/3/4 split is used instead. diff --git a/src/spaces/compute.rs b/src/spaces/compute.rs index b4cec455d..3d2a7faec 100644 --- a/src/spaces/compute.rs +++ b/src/spaces/compute.rs @@ -379,10 +379,13 @@ fn open_func_space<'a, T: ParserTrait>( /// first-by-source-order won regardless of which body actually contained /// the comment (issue #289). /// -/// A malformed marker is logged and dropped (no scope attached) rather -/// than aborting the walk: a typo in one file must not derail a -/// workspace-wide pass, and dropping is the conservative choice — a typo -/// should not accidentally silence anything. +/// Every complaint the parse produced is logged, and whatever directive +/// it still yielded is applied: an unusable metric name costs its own +/// name and nothing else, while a body that parses to no directive at all +/// is logged and dropped (issue #1168). The walk never aborts — a typo in +/// one file must not derail a workspace-wide pass — and dropping stays +/// the conservative choice, since a marker can only ever lose coverage +/// this way, never gain it. fn apply_comment_suppression( state_stack: &mut Vec, node: &Node, @@ -391,15 +394,19 @@ fn apply_comment_suppression( is_comment: bool, ) { if is_comment && let Some(text) = node.utf8_text(code) { - match parse_suppression_marker(text) { - Ok(Some(s)) => apply_suppression(state_stack, &s), - Ok(None) => {} - Err(e) => { - // The `+ 1` converts tree-sitter's 0-based rows to the - // 1-based line numbers `FuncSpace::start_line` and the - // rest of this module report. - eprintln!("warning: {}:{}: {e}", diagnostic_path, node.start_row() + 1); - } + let scan = parse_suppression_marker(text); + for diagnostic in &scan.diagnostics { + // The `+ 1` converts tree-sitter's 0-based rows to the + // 1-based line numbers `FuncSpace::start_line` and the + // rest of this module report. + eprintln!( + "warning: {}:{}: {diagnostic}", + diagnostic_path, + node.start_row() + 1 + ); + } + if let Some(suppression) = &scan.suppression { + apply_suppression(state_stack, suppression); } } } diff --git a/src/suppression.rs b/src/suppression.rs index 2095e2173..0f9ec9a6f 100644 --- a/src/suppression.rs +++ b/src/suppression.rs @@ -11,6 +11,15 @@ //! metrics for the enclosing function. //! - `bca: suppress-file` — suppress all metrics for the entire file. //! - `bca: suppress-file(halstead)` — suppress listed metrics file-wide. +//! +//! A marker that names a metric list may carry a trailing rationale on +//! the same line (`bca: suppress(nargs) — threaded context, not a +//! god-function`), with no separator required: the parentheses are +//! the positive signal that distinguish a marker from prose. A *bare* +//! verb takes no trailing text at all — with nothing to anchor the +//! intent, no separator distinguishes a rationale from a sentence +//! *about* the marker, and reading the latter as a marker would +//! silence every metric in the enclosing function. //! - **Lizard compatibility markers** are recognized verbatim so //! existing Lizard-instrumented codebases migrate without rewrites: //! - `#lizard forgives` ≡ `bca: suppress`. @@ -25,6 +34,7 @@ use std::collections::BTreeSet; use std::fmt; +use std::sync::OnceLock; use serde::{Deserialize, Serialize}; @@ -192,27 +202,127 @@ pub(crate) struct Suppression { pub(crate) source: SuppressionSource, } -/// Error returned when a marker is recognized as a `bca:` directive but -/// the body is malformed (unknown verb, malformed list, unknown metric -/// identifier). Lizard-style markers never error: anything that does +/// What scanning one comment for a suppression marker produced. +/// +/// The two fields are independent, and that is the point (issue #1168): +/// a marker can be *partly* usable — `bca: suppress(cognitive, exit)` +/// silences `cognitive` and reports `exit` — where the previous +/// `Result` shape forced every flaw to void the whole marker. The +/// governing rule is that a comment recognisable as a `bca: suppress` +/// marker never silently does nothing: it either suppresses what it +/// names or produces a diagnostic, and often both. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct MarkerScan { + /// The directive to apply, when the comment carried a usable one. + /// `None` for an ordinary comment and for a body that could not be + /// parsed at all. + pub(crate) suppression: Option, + /// Everything wrong with the marker, in source order. Empty for the + /// dominant case. Callers on the threshold path render these as + /// `warning:` lines; the read-only audit walk ignores them. + pub(crate) diagnostics: Vec, +} + +impl MarkerScan { + /// The comment carries no marker — the common case. + fn not_a_marker() -> Self { + Self { + suppression: None, + diagnostics: Vec::new(), + } + } + + /// The comment opens a `bca:` directive that could not be parsed + /// into one at all. Nothing is suppressed and `error` is reported. + fn rejected(error: SuppressionError) -> Self { + Self { + suppression: None, + diagnostics: vec![error], + } + } + + /// A marker with nothing to complain about. + fn directive(suppression: Suppression) -> Self { + Self { + suppression: Some(suppression), + diagnostics: Vec::new(), + } + } + + /// A usable marker that still drew complaints — the partly-usable + /// case issue #1168 exists for. `suppression` covers what parsed; + /// `diagnostics` names what did not. + fn partial(suppression: Suppression, diagnostics: Vec) -> Self { + Self { + suppression: Some(suppression), + diagnostics, + } + } +} + +/// A flaw in a marker that is recognizably a `bca:` directive: an +/// unknown verb, an unparseable body, or a metric name that cannot be +/// honoured. Lizard-style markers never produce one: anything that does /// not match the exact `#lizard forgives` / `#lizard forgive global` /// shapes simply parses as "not a marker". +/// +/// The first two void the marker; a bad metric name only drops that one +/// name from the list (issue #1168). #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) enum SuppressionError { /// `bca:` directive used an unrecognized verb (anything other than /// `suppress` / `suppress-file`). UnknownVerb(String), /// `bca: suppress(...)` listed an identifier that is not a known - /// metric name. + /// metric name. Reported and skipped; the recognized names beside it + /// still suppress. UnknownMetric(String), /// `bca: suppress(...)` named a real metric that has no configurable /// threshold and therefore cannot be suppressed (currently only /// `tokens`). Distinct from [`Self::UnknownMetric`] so the author /// learns the name parsed but is simply not silenceable. NonSuppressibleMetric(String), - /// `bca: suppress(...)` body could not be tokenized (e.g. unbalanced - /// parentheses, stray characters). + /// `bca: suppress(...)` body could not be tokenized (e.g. an + /// unbalanced parenthesis, or a bare verb followed by any trailing + /// text). MalformedBody(String), + /// More distinct unusable names than [`MAX_MARKER_DIAGNOSTICS`], so + /// the tail was elided. Carries the number dropped, because a silent + /// truncation would understate how wrong the marker is. + ElidedDiagnostics(usize), +} + +/// Cap on the diagnostics one suppression marker may emit. +/// +/// Names are deduplicated before the cap applies, so reaching it takes a +/// marker with eight *distinct* unusable names — well past any real +/// typo, and into the territory of `bca: suppress(a,b,c,…)` in a +/// third-party tree. Each diagnostic renders the full suppressible-metric +/// hint (~130 characters), so without a cap a large enough comment turns +/// one marker into megabytes of stderr. +const MAX_MARKER_DIAGNOSTICS: usize = 8; + +/// The suppressible-metric vocabulary, rendered once. +/// +/// Built lazily and cached: [`SuppressionError::UnknownMetric`]'s +/// `Display` is invoked once per diagnostic, and rebuilding, allocating +/// and sorting this list on every one made a marker's cost quadratic in +/// its own length. +fn suppressible_metric_hint() -> &'static str { + static HINT: OnceLock = OnceLock::new(); + HINT.get_or_init(|| { + // `Metric::suppressible()` is the single source of truth for the + // suppressible vocabulary — it already excludes the + // non-suppressible `tokens` — so the hint is never re-derived + // from `Metric::NAMES` with a hardcoded filter. It iterates + // declaration order; we sort so the hint stays alphabetised and + // thus stable across releases. + let mut names: Vec = Metric::suppressible() + .map(|metric| metric.to_string()) + .collect(); + names.sort_unstable(); + names.join(", ") + }) } impl fmt::Display for SuppressionError { @@ -226,28 +336,35 @@ impl fmt::Display for SuppressionError { "unknown bca directive verb '{v}'; expected `suppress` or `suppress-file`" ), Self::UnknownMetric(m) => { - // The hint lists the suppressible metrics, derived from - // `Metric::suppressible()` (the single source of truth for - // the suppressible vocabulary — it already excludes the - // non-suppressible `tokens`) rather than re-deriving from - // `Metric::NAMES` with a hardcoded filter. `suppressible()` - // iterates declaration order; we sort so the hint stays - // alphabetised and thus stable across releases. - let mut names: Vec = Metric::suppressible() - .map(|metric| metric.to_string()) - .collect(); - names.sort_unstable(); - let known = names.join(", "); write!( f, - "unknown metric '{m}' in bca suppression marker; known metrics: {known}" + "unknown metric '{m}' in bca suppression marker; known metrics: {}", + suppressible_metric_hint() ) } Self::NonSuppressibleMetric(m) => { write!(f, "metric '{m}' has no threshold and cannot be suppressed") } + Self::ElidedDiagnostics(n) => write!( + f, + "… and {n} more unusable metric name(s) in this bca suppression marker" + ), Self::MalformedBody(body) => { - write!(f, "malformed bca suppression marker body '{body}'") + // This warning is the only thing standing between the + // author and a marker that silently does nothing, so it + // names both the accepted shapes and the two ways out of + // the commonest mistake — a reason written after a bare + // verb, which no separator can distinguish from prose + // about the marker. + write!( + f, + "malformed bca suppression marker body '{body}'; expected \ + `bca: suppress` / `bca: suppress-file` with nothing after \ + the verb, or `bca: suppress()`, which may carry a \ + rationale (`bca: suppress(cognitive, cyclomatic) — \ + reason`); to keep a reason here, name the metrics or move \ + the reason to the line above" + ) } } } @@ -256,12 +373,13 @@ impl fmt::Display for SuppressionError { impl std::error::Error for SuppressionError {} /// Parse a single comment's text and try to extract a suppression -/// directive. Returns: +/// directive, returning both the directive (if any) and every complaint +/// about it — see [`MarkerScan`]. /// -/// - `Ok(None)` when the comment carries no marker (the common case). -/// - `Ok(Some(s))` when a marker was successfully parsed. -/// - `Err(e)` only for *native* markers whose body is malformed — -/// Lizard-style markers never error. +/// A comment that is not a marker yields an empty scan; a *native* +/// marker that is recognizable but flawed yields at least one +/// diagnostic. Lizard-style markers never produce diagnostics: anything +/// off-shape simply is not a marker. /// /// The input is the raw comment text **including** the comment-syntax /// delimiters (e.g. `// bca: suppress`, `# bca: suppress`, `/* bca: suppress */`). @@ -270,13 +388,13 @@ impl std::error::Error for SuppressionError {} /// `/`, `*`, `!`, `#`, `;`, `-`, and ASCII whitespace. The `!` entry /// covers Rust inner doc comments (`//!`, `/*!`); the `;` and `-` /// entries cover Lisp / SQL / Lua line-comment shapes. -pub(crate) fn parse_marker(comment_text: &str) -> Result, SuppressionError> { +pub(crate) fn parse_marker(comment_text: &str) -> MarkerScan { // Fast-bail: this function runs on every comment node. Most // comments are license headers, doc comments, or TODO notes that // contain neither sigil. `str::contains` is SIMD-accelerated and // avoids the trim/strip chain below for the dominant case. if !comment_text.contains("bca:") && !comment_text.contains("lizard") { - return Ok(None); + return MarkerScan::not_a_marker(); } // Strip a `/*` opener and a `*/` closer if present so we don't @@ -330,8 +448,8 @@ pub(crate) fn parse_marker(comment_text: &str) -> Result, Su no_opener }; - if let Some(s) = parse_lizard(lizard_candidate) { - return Ok(Some(s)); + if let Some(suppression) = parse_lizard(lizard_candidate) { + return MarkerScan::directive(suppression); } // For native parsing, strip the same `#` opener so `# bca: suppress` @@ -377,21 +495,22 @@ fn parse_lizard(trimmed: &str) -> Option { None } -fn parse_native(body: &str) -> Result, SuppressionError> { +fn parse_native(body: &str) -> MarkerScan { // The native dialect is `bca:` followed by a verb (`suppress` or - // `suppress-file`), optionally followed by `(metric, metric, ...)`. + // `suppress-file`), optionally followed by `(metric, metric, ...)`, + // optionally followed by a free-text rationale. let Some(rest) = body.strip_prefix("bca:") else { - return Ok(None); + return MarkerScan::not_a_marker(); }; let rest = rest.trim_start(); if rest.is_empty() { // A bare `bca:` with nothing after it isn't useful; treat as // not-a-marker rather than an error so the user can write // documentation that mentions the namespace without firing. - return Ok(None); + return MarkerScan::not_a_marker(); } - let malformed = || SuppressionError::MalformedBody(body.to_owned()); + let malformed = || MarkerScan::rejected(SuppressionError::MalformedBody(body.to_owned())); // Split into verb + parenthesised body. We accept whitespace // between the verb and `(`. The verb is the longest prefix of @@ -400,43 +519,63 @@ fn parse_native(body: &str) -> Result, SuppressionError> { .find(|c: char| !(c.is_ascii_alphabetic() || c == '-')) .unwrap_or(rest.len()); let (verb, after_verb) = rest.split_at(verb_end); - if verb.is_empty() { - return Err(malformed()); - } let kind = match verb { "suppress" => SuppressionKind::Function, "suppress-file" => SuppressionKind::File, - other => return Err(SuppressionError::UnknownVerb(other.to_owned())), + "" => return malformed(), + other => return MarkerScan::rejected(SuppressionError::UnknownVerb(other.to_owned())), }; let after_verb = after_verb.trim_start(); - let scope = if after_verb.is_empty() { - SuppressionScope::All - } else if let Some(rest) = after_verb.strip_prefix('(') { - let close = rest.find(')').ok_or_else(malformed)?; - let (inside, trailing) = rest.split_at(close); - // After the `)` only whitespace (and `*/` already trimmed by - // caller) is allowed. Anything else is a malformed marker: - // reject so `bca: suppress(loc) garbage` doesn't silently succeed. - if !trailing[1..].trim().is_empty() { - return Err(malformed()); - } - parse_metric_list(inside)? + let (scope, diagnostics) = if after_verb.is_empty() { + (SuppressionScope::All, Vec::new()) + } else if let Some(list) = after_verb.strip_prefix('(') { + let Some(close) = list.find(')') else { + return malformed(); + }; + // Everything past the `)` is the author's rationale (issue + // #1168). The metric list already makes the intent unambiguous, + // so no separator is required and none is privileged: `— why`, + // `- why`, `: why`, `// why`, and bare prose all read the same. + // Rejecting them made `AGENTS.md`'s own "suppress with a reason" + // instruction produce a marker that silently did nothing. + let (metrics, diagnostics) = parse_metric_list(&list[..close]); + (SuppressionScope::Some(metrics), diagnostics) } else { - // Trailing text after the verb that isn't `(...)`: reject. - return Err(malformed()); + // A bare verb followed by anything at all. Unlike the post-`)` + // case there is no positive signal here separating a rationale + // from prose that merely mentions the marker: the punctuation + // people reach for when writing *about* one (`-`, `:`, `//`, + // `#`, an em dash) is the same punctuation they would open a + // rationale with. Accepting either silences every metric on the + // enclosing function on the strength of a sentence, so the whole + // shape stays malformed and the author is told to name the + // metrics instead. + return malformed(); }; - Ok(Some(Suppression { - kind, - scope, - source: SuppressionSource::Native, - })) + MarkerScan::partial( + Suppression { + kind, + scope, + source: SuppressionSource::Native, + }, + diagnostics, + ) } -fn parse_metric_list(inside: &str) -> Result { +fn parse_metric_list(inside: &str) -> (BTreeSet, Vec) { let mut set = BTreeSet::new(); + let mut diagnostics = Vec::new(); + // A marker is free to repeat a name, and each unusable one costs a + // diagnostic carrying the full metric hint — so report each distinct + // name once and stop after `MAX_MARKER_DIAGNOSTICS` of them. Both + // guards bound the output by the marker's *vocabulary* rather than + // its length, which is what keeps an adversarial comment in an + // untrusted tree from flooding the log. + let mut reported: BTreeSet<&str> = BTreeSet::new(); + let mut unusable = 0_usize; for token in inside.split(',') { let name = token.trim(); if name.is_empty() { @@ -449,18 +588,37 @@ fn parse_metric_list(inside: &str) -> Result // Parse through the canonical `Metric` vocabulary (the same one // selection uses) so suppression and selection never drift. A // typo surfaces the offending token via `ParseMetricError` - // (#554). `tokens` parses fine but has no threshold, so reject - // it with a distinct, actionable error rather than silently - // accepting a no-op suppression. - let metric: Metric = name - .parse() - .map_err(|_| SuppressionError::UnknownMetric(name.to_owned()))?; - if metric == Metric::Tokens { - return Err(SuppressionError::NonSuppressibleMetric(name.to_owned())); + // (#554). `tokens` parses fine but has no threshold, so it gets + // a distinct, actionable diagnostic rather than silently + // registering a no-op suppression. + // + // A name we cannot honour is *skipped and reported*, not fatal + // to the whole list (issue #1168): `suppress(cognitive, exit)` + // still silences `cognitive`, because voiding the marker + // wholesale turned one mistyped name — `exit` for `nexits` is + // the documented one — into a suppression the author believed + // was active. Skipping can only ever narrow what a marker + // silences, so a typo cannot widen scope. + let unusable_name = match name.parse::() { + Ok(Metric::Tokens) => SuppressionError::NonSuppressibleMetric(name.to_owned()), + Ok(metric) => { + set.insert(metric); + continue; + } + Err(_) => SuppressionError::UnknownMetric(name.to_owned()), + }; + if reported.insert(name) { + unusable += 1; + if diagnostics.len() < MAX_MARKER_DIAGNOSTICS { + diagnostics.push(unusable_name); + } } - set.insert(metric); } - Ok(SuppressionScope::Some(set)) + let elided = unusable.saturating_sub(MAX_MARKER_DIAGNOSTICS); + if elided > 0 { + diagnostics.push(SuppressionError::ElidedDiagnostics(elided)); + } + (set, diagnostics) } /// Whether an audited suppression marker applies to its enclosing @@ -635,9 +793,7 @@ fn marker_at( if !T::Checker::is_comment(node) { return None; } - let Ok(Some(suppression)) = parse_marker(node.utf8_text(code)?) else { - return None; - }; + let suppression = parse_marker(node.utf8_text(code)?).suppression?; let function = match suppression.kind { SuppressionKind::Function => enclosing.map(str::to_owned), SuppressionKind::File => None, @@ -655,9 +811,65 @@ fn marker_at( mod tests { use super::*; + /// The directive `text` parses to, asserting it carried one and drew + /// no complaint. Use where the subject is what a *clean* marker + /// means; anything expecting a diagnostic should read + /// [`scan_diagnostics`] instead so the complaint is asserted, not + /// discarded. + #[track_caller] + fn marker(text: &str) -> Suppression { + let scan = parse_marker(text); + assert!( + scan.diagnostics.is_empty(), + "expected a clean parse of {text:?}; got {:?}", + scan.diagnostics, + ); + scan.suppression + .unwrap_or_else(|| panic!("expected {text:?} to parse as a marker")) + } + + /// Every complaint `text` drew. + fn scan_diagnostics(text: &str) -> Vec { + parse_marker(text).diagnostics + } + + /// Whether `text` is no marker at all: no directive *and* no + /// complaint. Both halves matter — a comment that merely mentions + /// the syntax must stay silent, not warn at every reader. + fn is_not_a_marker(text: &str) -> bool { + let scan = parse_marker(text); + scan.suppression.is_none() && scan.diagnostics.is_empty() + } + + /// The single complaint `text` drew, when exactly one is expected. + #[track_caller] + fn sole_diagnostic(text: &str) -> SuppressionError { + let mut diagnostics = scan_diagnostics(text); + assert_eq!( + diagnostics.len(), + 1, + "expected exactly one diagnostic for {text:?}; got {diagnostics:?}", + ); + diagnostics.remove(0) + } + + /// The complaint `text` drew, asserting it also voided the marker + /// outright — the shape reserved for a body that parses to no + /// directive at all. + #[track_caller] + fn voiding_diagnostic(text: &str) -> SuppressionError { + let scan = parse_marker(text); + assert!( + scan.suppression.is_none(), + "expected {text:?} to yield no directive; got {:?}", + scan.suppression, + ); + sole_diagnostic(text) + } + #[test] fn native_bare_suppress_covers_all_for_function() { - let s = parse_marker("// bca: suppress").unwrap().unwrap(); + let s = marker("// bca: suppress"); assert_eq!(s.kind, SuppressionKind::Function); assert_eq!(s.source, SuppressionSource::Native); assert!(matches!(s.scope, SuppressionScope::All)); @@ -665,9 +877,7 @@ mod tests { #[test] fn native_suppress_with_metric_list() { - let s = parse_marker("// bca: suppress(cyclomatic, cognitive)") - .unwrap() - .unwrap(); + let s = marker("// bca: suppress(cyclomatic, cognitive)"); assert_eq!(s.kind, SuppressionKind::Function); let SuppressionScope::Some(metrics) = s.scope else { panic!("expected Some(...)"); @@ -678,37 +888,56 @@ mod tests { } #[test] - fn native_mixed_valid_and_unknown_metric_voids_whole_marker() { - // The void-on-typo contract: a marker listing one valid metric - // beside an unknown one must reject the ENTIRE list, not silently - // honor the valid part. Otherwise a misspelled metric would still - // suppress the correctly-spelled one beside it — the most - // dangerous failure mode, since it widens scope on a typo. Every - // other test feeds a marker whose only metric is unknown, where - // "void whole marker" and "skip the unknown token" are - // indistinguishable; only a mixed list separates them. Swapping - // the `?` in `parse_metric_list` for `continue` makes this parse - // to `Some({Cyclomatic})` and trips the assertion (#948). - let err = parse_marker("// bca: suppress(cyclomatic, no_such_metric)").unwrap_err(); + fn native_mixed_valid_and_unknown_metric_keeps_the_valid_half() { + // Issue #1168 reversed the pre-existing void-on-typo contract + // (#948, #896). A misspelled name now costs its own name and + // nothing else: `exit`-for-`nexits` is a mistake `AGENTS.md` + // itself documents people making, and voiding the marker + // wholesale turned it into a suppression the author believed was + // active while the gate disagreed. + // + // Skipping cannot widen scope — the set only ever loses entries + // — which is what made the old contract's stated danger ("a typo + // silences something the author did not name") unreachable here. + // Every other test feeds a marker whose only metric is unknown, + // where "void the marker" and "skip the token" are + // indistinguishable; only a mixed list separates them. + let scan = parse_marker("// bca: suppress(cyclomatic, no_such_metric)"); + let Some(Suppression { + scope: SuppressionScope::Some(metrics), + .. + }) = &scan.suppression + else { + panic!( + "expected an explicit metric set; got {:?}", + scan.suppression + ); + }; + assert_eq!( + metrics.iter().copied().collect::>(), + vec![Metric::Cyclomatic], + "the recognized half of the list must still suppress", + ); assert!( - matches!(&err, SuppressionError::UnknownMetric(name) if name == "no_such_metric"), - "mixed valid+unknown marker must error on the unknown token, \ - voiding the whole list; got {err:?}", + matches!( + scan.diagnostics.as_slice(), + [SuppressionError::UnknownMetric(name)] if name == "no_such_metric", + ), + "the unrecognized half must still be reported; got {:?}", + scan.diagnostics, ); } #[test] fn native_suppress_file_bare() { - let s = parse_marker("# bca: suppress-file").unwrap().unwrap(); + let s = marker("# bca: suppress-file"); assert_eq!(s.kind, SuppressionKind::File); assert!(matches!(s.scope, SuppressionScope::All)); } #[test] fn native_suppress_file_with_metric_list() { - let s = parse_marker("/* bca: suppress-file(halstead, loc) */") - .unwrap() - .unwrap(); + let s = marker("/* bca: suppress-file(halstead, loc) */"); assert_eq!(s.kind, SuppressionKind::File); let SuppressionScope::Some(metrics) = s.scope else { panic!("expected Some(...)"); @@ -719,7 +948,7 @@ mod tests { #[test] fn native_unknown_metric_errors() { - let err = parse_marker("// bca: suppress(no_such_metric)").unwrap_err(); + let err = sole_diagnostic("// bca: suppress(no_such_metric)"); assert!(matches!(err, SuppressionError::UnknownMetric(_))); // The error must mention what was unknown so authors can // diagnose typos without reading our source. This is the #554 @@ -755,7 +984,7 @@ mod tests { // `tokens` parses as a real `Metric` but has no threshold, so a // marker naming it is rejected with a distinct, actionable error // rather than silently accepted as a no-op suppression. - let err = parse_marker("// bca: suppress(tokens)").unwrap_err(); + let err = sole_diagnostic("// bca: suppress(tokens)"); assert!( matches!(&err, SuppressionError::NonSuppressibleMetric(m) if m == "tokens"), "expected NonSuppressibleMetric(\"tokens\"); got: {err:?}", @@ -770,7 +999,7 @@ mod tests { #[test] fn native_unknown_verb_errors() { - let err = parse_marker("// bca: disable").unwrap_err(); + let err = voiding_diagnostic("// bca: disable"); assert!(matches!(err, SuppressionError::UnknownVerb(_))); // The error message must guide the author toward the correct // verbs without making them grep our source. Anchor each verb @@ -798,50 +1027,277 @@ mod tests { /// in shipped source; this test catches that. #[test] fn legacy_allow_verb_is_unknown() { - let err = parse_marker("// bca: allow").unwrap_err(); + let err = voiding_diagnostic("// bca: allow"); assert!(matches!(err, SuppressionError::UnknownVerb(v) if v == "allow")); - let err = parse_marker("// bca: allow-file").unwrap_err(); + let err = voiding_diagnostic("// bca: allow-file"); assert!(matches!(err, SuppressionError::UnknownVerb(v) if v == "allow-file")); - let err = parse_marker("// bca: allow(cyclomatic)").unwrap_err(); + let err = voiding_diagnostic("// bca: allow(cyclomatic)"); assert!(matches!(err, SuppressionError::UnknownVerb(v) if v == "allow")); } #[test] fn native_malformed_body_errors() { - // Unbalanced paren. + // Unbalanced paren: there is no metric list to honour and no way + // to tell where one would have ended, so the marker is void. assert!(matches!( - parse_marker("// bca: suppress(cyclomatic").unwrap_err(), + voiding_diagnostic("// bca: suppress(cyclomatic"), SuppressionError::MalformedBody(_) )); - // Trailing garbage after the metric list. + // Bare verb followed by a word. With no metric list to anchor + // the intent, `// bca: suppress markers are honoured here` is + // prose about the feature, and reading it as a marker would + // silence every metric in the enclosing function. assert!(matches!( - parse_marker("// bca: suppress(cyclomatic) junk").unwrap_err(), - SuppressionError::MalformedBody(_) - )); - // Verb followed by something other than `(...)`. - assert!(matches!( - parse_marker("// bca: suppress garbage").unwrap_err(), + voiding_diagnostic("// bca: suppress garbage"), SuppressionError::MalformedBody(_) )); } + #[test] + fn malformed_body_message_names_the_accepted_shapes() { + // This warning is now the only signal an author gets that the + // reason they wrote after a bare verb left the marker inert, so + // it must name the shapes that parse *and* both ways out: name + // the metrics, or move the reason off the marker line. + let rendered = voiding_diagnostic("// bca: suppress - see #123").to_string(); + assert!( + rendered.contains("bca: suppress - see #123"), + "message must echo the offending body; got: {rendered}", + ); + assert!( + rendered.contains("`bca: suppress()`"), + "message must name the metric-list shape; got: {rendered}", + ); + assert!( + rendered.contains("rationale"), + "message must point at the rationale form; got: {rendered}", + ); + assert!( + rendered.contains("name the metrics"), + "message must tell the author to name the metrics; got: {rendered}", + ); + assert!( + rendered.contains("line above"), + "message must offer the move-the-reason-up escape; got: {rendered}", + ); + } + #[test] fn native_bare_colon_is_not_a_marker() { // `bca:` with nothing after it is not a marker; we want to // allow documentation comments to mention the namespace. - assert!(parse_marker("// bca:").unwrap().is_none()); + let scan = parse_marker("// bca:"); + assert_eq!(scan.suppression, None); + assert!(scan.diagnostics.is_empty()); } #[test] fn empty_metric_list_is_noop_not_error() { - let s = parse_marker("// bca: suppress()").unwrap().unwrap(); + let s = marker("// bca: suppress()"); assert!(s.scope.is_empty()); assert!(!s.scope.covers(Metric::Cyclomatic)); } + #[test] + fn trailing_rationale_after_metric_list_is_accepted() { + // The issue #1168 reproducer, at the parse boundary: the + // spelling `AGENTS.md` asks for — a metric list plus the reason + // the function is exempt — used to be rejected wholesale, so the + // author's suppression silently did nothing. + // + // No separator is privileged and none is required: after `)` the + // author has already said what they mean, so anything following + // is prose. + for text in [ + "// bca: suppress(nargs) \u{2014} threaded context, not a god-function", + "// bca: suppress(nargs) \u{2013} threaded context", + "// bca: suppress(nargs) - threaded context", + "// bca: suppress(nargs): threaded context", + "// bca: suppress(nargs) // threaded context", + "// bca: suppress(nargs) threaded context", + "/* bca: suppress(nargs) \u{2014} threaded context */", + ] { + let s = marker(text); + assert_eq!(s.kind, SuppressionKind::Function, "for {text:?}"); + assert!( + matches!(&s.scope, SuppressionScope::Some(m) + if m.iter().copied().eq([Metric::Nargs])), + "rationale must not disturb the metric list; {text:?} gave {:?}", + s.scope, + ); + } + } + + #[test] + fn a_bare_verb_takes_no_trailing_text_whatever_the_separator() { + // #1168 briefly accepted a rationale after a bare verb when it + // opened with `-`, `:`, `//`, `#`, or an em/en dash. Those are + // exactly the characters people reach for when writing *about* a + // marker, so ordinary comments silenced every metric on their + // function with no diagnostic at all. There is no positive + // signal in this shape to separate the two readings — the + // parentheses of the list form are what supply one — so every + // row below is malformed, including the paths and prose that + // never were rationales. + for text in [ + "// bca: suppress \u{2014} irreducible dispatch", + "// bca: suppress \u{2013} irreducible dispatch", + "// bca: suppress - we removed this marker, see #123", + "// bca: suppress: not applicable to this function", + "// bca: suppress // generated shim", + "// bca: suppress /some/path", + "// bca: suppress markers are honoured here", + "# bca: suppress-file # generated", + "// bca: suppress-file generated file", + ] { + let scan = parse_marker(text); + assert_eq!( + scan.suppression, None, + "a bare verb plus trailing text must not suppress; {text:?}", + ); + assert!( + matches!( + scan.diagnostics.as_slice(), + [SuppressionError::MalformedBody(_)] + ), + "{text:?} must warn that the marker is inert; got {:?}", + scan.diagnostics, + ); + } + // Positive control: a parser that rejected everything would pass + // the loop above. The verb alone still suppresses, and so does a + // metric list carrying the rationale that replaces this shape. + assert!(matches!( + marker("// bca: suppress").scope, + SuppressionScope::All + )); + assert!(matches!( + marker("// bca: suppress(nargs) \u{2014} threaded context").scope, + SuppressionScope::Some(_) + )); + } + + #[test] + fn unusable_names_are_deduplicated_and_capped_per_marker() { + // One diagnostic per *distinct* unusable name, not per token. + // Each renders the full suppressible-metric hint, so an + // unbounded marker in an untrusted tree is a log flood rather + // than a typo report. + let repeated = ["nope"; 500].join(","); + let scan = parse_marker(&format!("// bca: suppress({repeated})")); + assert_eq!( + scan.diagnostics, + vec![SuppressionError::UnknownMetric("nope".to_owned())], + "500 copies of one name must cost exactly one diagnostic", + ); + + // Distinct names past the cap are elided, but the tail is + // *counted*: a silent truncation would understate the marker. + let overflow = 5; + let distinct: Vec = (0..MAX_MARKER_DIAGNOSTICS + overflow) + .map(|i| format!("nope{i}")) + .collect(); + let scan = parse_marker(&format!("// bca: suppress({})", distinct.join(","))); + assert_eq!( + scan.diagnostics.len(), + MAX_MARKER_DIAGNOSTICS + 1, + "expected {MAX_MARKER_DIAGNOSTICS} names plus one tail; got {:?}", + scan.diagnostics, + ); + assert_eq!( + scan.diagnostics.last(), + Some(&SuppressionError::ElidedDiagnostics(overflow)), + "the elided count must survive the cap; got {:?}", + scan.diagnostics, + ); + // Render it. The variant assertion above holds even if the tail + // formats as an empty string, and this is the one diagnostic a + // reader only ever meets in the pathological case the cap exists + // for — so the count has to reach the page, not just the struct. + let tail = scan + .diagnostics + .last() + .expect("the cap always appends a tail") + .to_string(); + assert!( + tail.contains(&overflow.to_string()), + "the rendered tail must name how many were elided; got {tail:?}", + ); + assert!( + tail.contains("more unusable metric name"), + "the rendered tail must say what was elided; got {tail:?}", + ); + // The cap never touches the metrics the marker really names. + let scan = parse_marker(&format!( + "// bca: suppress(cognitive,{})", + distinct.join(",") + )); + let Some(Suppression { + scope: SuppressionScope::Some(metrics), + .. + }) = &scan.suppression + else { + panic!( + "expected an explicit metric set; got {:?}", + scan.suppression + ); + }; + assert!( + metrics.contains(&Metric::Cognitive), + "capping diagnostics must not narrow the suppression; got {metrics:?}", + ); + } + + #[test] + fn rationale_may_contain_parentheses_and_marker_syntax() { + // The metric list ends at the first `)`, so a rationale is free + // to contain further parens, and a second `suppress(` inside it + // is prose rather than a nested directive: one comment carries + // at most one marker. + let s = marker("// bca: suppress(nargs) — mirrors suppress(abc) in do_thing(x)"); + assert!( + matches!(&s.scope, SuppressionScope::Some(m) + if m.iter().copied().eq([Metric::Nargs])), + "got {:?}", + s.scope, + ); + } + + #[test] + fn rationale_survives_a_flawed_metric_list() { + // The two #1168 halves compose: a rationale is accepted *and* + // the recognized metrics still suppress while the rest is + // reported. Neither relaxation is allowed to swallow the other. + let scan = parse_marker("// bca: suppress(cognitive, exit) — hand-rolled state machine"); + assert!( + matches!(&scan.suppression, Some(s) + if matches!(&s.scope, SuppressionScope::Some(m) + if m.iter().copied().eq([Metric::Cognitive]))), + "got {:?}", + scan.suppression, + ); + assert!( + matches!( + scan.diagnostics.as_slice(), + [SuppressionError::UnknownMetric(name)] if name == "exit", + ), + "`exit` is the documented `nexits` typo and must still be \ + reported; got {:?}", + scan.diagnostics, + ); + } + + #[test] + fn whitespace_only_rationale_is_not_a_diagnostic() { + // Trailing whitespace after the list — a stray tab before the + // newline, say — is not a rationale and must not read as one. + let s = marker("// bca: suppress(nargs) \t "); + assert!(matches!(&s.scope, SuppressionScope::Some(m) if m.len() == 1)); + } + #[test] fn lizard_function_marker() { - let s = parse_marker("// #lizard forgives").unwrap().unwrap(); + let s = marker("// #lizard forgives"); assert_eq!(s.kind, SuppressionKind::Function); assert_eq!(s.source, SuppressionSource::Lizard); assert!(matches!(s.scope, SuppressionScope::All)); @@ -849,7 +1305,7 @@ mod tests { #[test] fn lizard_file_marker() { - let s = parse_marker("# #lizard forgive global").unwrap().unwrap(); + let s = marker("# #lizard forgive global"); assert_eq!(s.kind, SuppressionKind::File); assert_eq!(s.source, SuppressionSource::Lizard); } @@ -859,13 +1315,13 @@ mod tests { // Per the issue's narrow compat surface: `#lizard skip` is not // a recognized Lizard directive, so we treat it as no marker // rather than erroring or silently suppressing. - assert!(parse_marker("// #lizard skip").unwrap().is_none()); + assert!(is_not_a_marker("// #lizard skip")); } #[test] fn plain_comment_is_not_a_marker() { - assert!(parse_marker("// just a comment").unwrap().is_none()); - assert!(parse_marker("/* TODO: fix later */").unwrap().is_none()); + assert!(is_not_a_marker("// just a comment")); + assert!(is_not_a_marker("/* TODO: fix later */")); } /// Locks the fast-bail contract in `parse_marker`: comments that @@ -877,24 +1333,12 @@ mod tests { #[test] fn fast_bail_skips_sigil_free_comments() { // Long, sigil-free comments that should never trigger. - assert!( - parse_marker("// Copyright (c) 2026 Some Corp.") - .unwrap() - .is_none() - ); - assert!( - parse_marker("/* SPDX-License-Identifier: MIT */") - .unwrap() - .is_none() - ); + assert!(is_not_a_marker("// Copyright (c) 2026 Some Corp.")); + assert!(is_not_a_marker("/* SPDX-License-Identifier: MIT */")); // Substring-mention-but-not-a-marker: contains "lizard" in // prose but is not a Lizard directive. Slow path must still // return Ok(None). - assert!( - parse_marker("// authors: jane lizard, john doe") - .unwrap() - .is_none() - ); + assert!(is_not_a_marker("// authors: jane lizard, john doe")); } /// Locks the case sensitivity of both dialects: `Bca:` and @@ -904,13 +1348,13 @@ mod tests { #[test] fn marker_grammar_is_case_sensitive() { // Uppercase B in `Bca:` is not a native marker. - assert!(parse_marker("// Bca: suppress").unwrap().is_none()); - assert!(parse_marker("/* BCA: suppress */").unwrap().is_none()); + assert!(is_not_a_marker("// Bca: suppress")); + assert!(is_not_a_marker("/* BCA: suppress */")); // Uppercase L in `#Lizard` is not a Lizard marker. The // fast-bail rejects it (no lowercase "lizard" substring) and // the slow path would also reject it via `strip_prefix("lizard")`. - assert!(parse_marker("# #Lizard forgives").unwrap().is_none()); - assert!(parse_marker("// #Lizard forgives").unwrap().is_none()); + assert!(is_not_a_marker("# #Lizard forgives")); + assert!(is_not_a_marker("// #Lizard forgives")); } #[test] @@ -1017,11 +1461,11 @@ mod tests { // Without `!` in the leading-strip set the marker prefix `bca:` // would not match. Both line- and block-comment variants must // round-trip the same way. - let line = parse_marker("//! bca: suppress").unwrap().unwrap(); + let line = marker("//! bca: suppress"); assert_eq!(line.kind, SuppressionKind::Function); assert!(matches!(line.scope, SuppressionScope::All)); - let block = parse_marker("/*! bca: suppress */").unwrap().unwrap(); + let block = marker("/*! bca: suppress */"); assert_eq!(block.kind, SuppressionKind::Function); assert!(matches!(block.scope, SuppressionScope::All)); } @@ -1162,19 +1606,22 @@ mod tests { assert!(rust_markers("fn f() {}\n").is_empty()); } - /// A comment the parser rejects contributes nothing to the audit, - /// and does not stop the walk from collecting the valid markers + /// A comment that yields no directive contributes nothing to the + /// audit, and does not stop the walk from collecting the markers /// around it. /// /// Two rejections reach [`marker_at`] and both must be silent here. - /// `parse_marker` answers `Ok(None)` for an ordinary comment that - /// simply is not a marker, and `Err` for one that *looks* like a - /// marker but is malformed — an unknown metric name voids the whole - /// list (see `native_mixed_valid_and_unknown_metric_voids_whole_marker`). - /// The audit is a read-only listing of what *is* a marker; the - /// threshold walk is the surface that warns on malformed bodies, so - /// dropping them without a diagnostic is the contract, not an - /// oversight. + /// `parse_marker` yields no directive for an ordinary comment that + /// simply is not a marker, and none for a `bca:` body it cannot + /// parse at all. The audit is a read-only listing of what *is* a + /// marker; the threshold walk is the surface that warns, so dropping + /// these without a diagnostic is the contract, not an oversight. + /// + /// A merely *flawed* metric list is a third case and is deliberately + /// not dropped: since #1168 it yields the directive its recognized + /// names describe, so the audit lists it — an author reading the + /// exemptions report needs to see the suppression that is actually + /// in force. /// /// Without this, every comment the collector's tests feed it parses /// successfully, and the reject arm is never taken. @@ -1182,8 +1629,8 @@ mod tests { fn collector_skips_comments_that_are_not_valid_markers() { let src = "// an ordinary comment\n\ fn f() {\n\ - \x20 // bca: suppress(cyclomatic, no_such_metric)\n\ \x20 // bca: suppress garbage\n\ + \x20 // bca: disable(cognitive)\n\ \x20 // bca: suppress(cognitive)\n\ }\n"; let markers = rust_markers(src); @@ -1202,4 +1649,75 @@ mod tests { "a rejected marker must not be collected with a fallback scope" ); } + + /// A marker carrying a rationale still attaches when the comment is + /// the last thing in the file, with no trailing newline. + /// + /// Per `.claude/rules/testing.md`, both the `check_metrics` shim and + /// the integration suites append a newline to every fixture, so "a + /// node ending at EOF" is unreachable from them — + /// [`crate::test_support::space_verbatim`] analyses the bytes as + /// given. The rationale is what makes this worth pinning: it is the + /// part of the marker adjacent to the missing newline, so a future + /// parser that indexed past the `)` unconditionally would fail here + /// and nowhere else. + #[test] + fn rationale_marker_at_eof_without_trailing_newline() { + let space = crate::test_support::space_verbatim( + crate::LANG::Rust, + b"fn f(a: u8, b: u8) -> u8 { a + b }\n\ + // bca: suppress-file(nargs) \xe2\x80\x94 two is plenty", + crate::MetricsOptions::default(), + ); + assert!( + space.suppressed.covers(Metric::Nargs), + "file-scoped marker at EOF must attach; got {:?}", + space.suppressed, + ); + } + + /// CRLF line endings leave a `\r` inside the comment token in most + /// grammars, so it lands in the rationale rather than in the metric + /// list. Pinned because the pre-#1168 parser reached the same answer + /// for the opposite reason: it trimmed the `\r` off a body that had + /// nothing after the `)` at all. + #[test] + fn rationale_marker_survives_crlf_line_endings() { + let space = crate::test_support::space_verbatim( + crate::LANG::Rust, + "fn f(a: u8, b: u8) -> u8 {\r\n\ + // bca: suppress(nargs) \u{2014} two is plenty\r\n\ + a + b\r\n}\r\n" + .as_bytes(), + crate::MetricsOptions::default(), + ); + let f = space + .spaces + .iter() + .find(|s| s.name.as_deref() == Some("f")) + .expect("function space f"); + assert!( + f.suppressed.covers(Metric::Nargs), + "CRLF marker must attach; got {:?}", + f.suppressed, + ); + } + + #[test] + fn collector_lists_a_marker_whose_list_was_partly_unusable() { + // The audit reports the suppression that is *in force*. Since + // #1168 that is the recognized half of a flawed list, so the + // marker must appear — with `cognitive` only, not with a + // defaulted `All` scope, which would misreport it as silencing + // everything. + let src = "fn f() {\n // bca: suppress(cognitive, exit) — state machine\n}\n"; + let markers = rust_markers(src); + assert_eq!(markers.len(), 1, "got {markers:?}"); + assert!( + matches!(&markers[0].scope, SuppressionScope::Some(m) + if m.iter().copied().eq([Metric::Cognitive])), + "got {:?}", + markers[0].scope, + ); + } } diff --git a/tests/README.md b/tests/README.md index 5db51be25..a0be2436f 100644 --- a/tests/README.md +++ b/tests/README.md @@ -11,13 +11,19 @@ to add one, start here. ## Layout at a glance +Since [#1124](https://github.com/dekobon/big-code-analysis/issues/1124) +the test files are grouped into directory targets rather than one crate +root per file: each `tests//main.rs` is a single test binary that +declares the former `tests/*.rs` files as `mod`s. + ```text tests/ ├── README.md (this file) ├── common/ shared harness used by every integration test │ ├── mod.rs snapshot driver + per-corpus comparators │ ├── fixtures.rs small constructors for OffenderRecord -│ └── validators.rs SARIF + Checkstyle structural validators +│ ├── validators.rs SARIF + Checkstyle structural validators +│ └── vcs_fixture.rs in-process git repositories for the VCS tests ├── fixtures/ vendored external schemas (SARIF, Checkstyle) │ └── README.md refresh procedure + provenance ├── repositories/ integration corpora (4 git submodules: 3 upstream projects + 1 fixture+snapshot store) @@ -28,15 +34,13 @@ tests/ │ ├── csharp/ hand-written synthetic .cs fixtures (C# corpus) │ ├── php/ hand-written synthetic .php fixtures (PHP corpus) │ └── snapshots/ accepted YAML snapshots for ALL five corpora -├── checkstyle_test.rs output-format test: Checkstyle XML schema -├── csv_test.rs output-format test: CSV writer -├── sarif_test.rs output-format test: SARIF JSON schema -├── serde_test.rs corpus test: Rust / serde -├── deepspeech_test.rs corpus test: C++ / DeepSpeech -├── pdf_js_test.rs corpus test: JS / pdf.js -├── csharp_test.rs corpus test: C# / synthetic fixtures -├── php_test.rs corpus test: PHP / synthetic fixtures -└── cyclomatic_cross_language_parity.rs cross-language: 6 languages × 4 control shapes +├── api/ public API: AST seam, parser reuse, derives, book examples +├── corpus/ corpus tests: serde, DeepSpeech, pdf.js, C#, PHP, iRules +├── grammars/ grammar-specific: C, mozcpp, alterator string flattening +├── output_formats/ output-format tests: CSV, SARIF, Checkstyle +│ └── snapshots/ accepted insta snapshots owned by this target +├── parity/ cross-language and cross-parser parity suites +└── vcs/ VCS metrics: rank, trend, cache, bus factor, per-function ``` ## Test categories @@ -59,7 +63,7 @@ below; the canonical reference language for each parser family is: - Curly-brace scripting → `javascript_*` or `csharp_*` - Shell / dynamic → `python_*` or `bash_*` -### 2. Output-format tests (`tests/_test.rs`) +### 2. Output-format tests (`tests/output_formats/`) Single-purpose: assert that the writers in `src/output/` emit schema-conformant documents. @@ -71,7 +75,7 @@ schema-conformant documents. | `csv_test.rs` | CSV writer round-trip + header stability | | `serde_test.rs` | (despite the name) **not** an output test — it is the Rust corpus test, named after the `serde-rs/serde` upstream project. See corpus tests below. | -### 3. Corpus tests (`tests/_test.rs`) +### 3. Corpus tests (`tests/corpus/`) Each runs the parser → metric pipeline over every file in one `tests/repositories//` corpus, then diffs the result against @@ -103,14 +107,16 @@ Floats are rounded to 3 decimal places before comparison (machine portability) and the `name` field is redacted to `[filepath]` (path portability). -### 4. Cross-language parity (`cyclomatic_cross_language_parity.rs`) +### 4. Cross-language parity (`tests/parity/`) -The only test that pins behaviour *across* languages. Asserts that -four control-flow shapes (`switch_with_default`, `switch_without_default`, +Where behaviour is pinned *across* languages and parsers rather than +within one. `cyclomatic_cross_language_parity.rs` is representative: it +asserts that four control-flow shapes +(`switch_with_default`, `switch_without_default`, `if_else_if_else_chain`, `single_if_no_else`) produce the same cyclomatic-sum delta in Bash, C++, Java, JavaScript, Python, and Rust. A bug fixed in one language module that drifts another silently is -exactly what this catches. +exactly what these catch. ## The corpus divergence @@ -260,3 +266,40 @@ Mirror `tests/output_formats/sarif_test.rs`: vendor the schema under `tests/fixtures/`, document provenance in `tests/fixtures/README.md`, and validate every emitted document via `tests/common/validators.rs`. Keep the validator hermetic — no network access. + +## Moving a test file + +Moving a file that owns `insta` snapshots renames every one of them. +`insta` builds the snapshot file name from the **whole** module path of +the asserting file — `::` replaced by `__` — followed by the snapshot +name. It is not the last component. A test in +`tests/output_formats/csv_test.rs` writes +`output_formats__csv_test__.snap`; one in `src/metrics/abc.rs`'s +`mod tests` writes `big_code_analysis__metrics__abc__tests__.snap`. +The `snapshots/` directory moves as well, since insta resolves it +relative to the asserting file. + +The resulting failure does not mention the move. insta reports +`snapshot assertion for '' failed`, prints the entire value as +new, and writes a `.snap.new` under the new name — the same output a +genuine behaviour change produces. The name in the source is correct and +the reviewed values are still on disk under the old name, so nothing +points at the rename. #1124 hit this merging the root suite's test +binaries into directory targets: `tests/csv_test.rs` became +`tests/output_formats/csv_test.rs`, and five snapshots had to be renamed, +`tests/snapshots/csv_test__csv_cpp_widget.snap` to +`tests/output_formats/snapshots/output_formats__csv_test__csv_cpp_widget.snap`. + +Rename the files; do not re-accept them: + +```bash +git mv tests/snapshots/csv_test__csv_cpp_widget.snap \ + tests/output_formats/snapshots/output_formats__csv_test__csv_cpp_widget.snap +``` + +`cargo insta test --accept` also turns the suite green, but it records +whatever production emits today rather than the value that was reviewed — +the hazard the snapshot-anchor policy exists to prevent — and it leaves +the old file behind as an orphan. The `source:` line in the snapshot +header is metadata and is not compared, so a stale one fails nothing; +`cargo insta test --force-update-snapshots` refreshes it. diff --git a/tests/api/suppression_test.rs b/tests/api/suppression_test.rs index 63259fad1..79f1e6226 100644 --- a/tests/api/suppression_test.rs +++ b/tests/api/suppression_test.rs @@ -256,12 +256,12 @@ fn populated_scope_serializes_with_metrics_list() { #[test] fn unknown_metric_in_marker_has_no_effect() { - // Per the issue's "unknown identifiers must error so typos do not - // silently widen scope" requirement. At the library boundary this - // surfaces as a stderr warning (no propagated error type), so the - // observable behaviour from an integration test is "the marker is - // discarded": the enclosing function's scope stays empty. The - // actual `SuppressionError::UnknownMetric(_)` variant is exercised + // Typos must not silently widen scope. At the library boundary an + // unrecognized name surfaces as a stderr warning (no propagated + // error type), so the observable behaviour from an integration test + // is that it contributes nothing: this marker names only unknown + // metrics, so the enclosing function's scope stays empty. The + // `SuppressionError::UnknownMetric(_)` variant itself is exercised // by the unit test `native_unknown_metric_errors` in // `src/suppression.rs`. let src = r#" @@ -278,6 +278,50 @@ def fine(): ); } +#[test] +fn unknown_metric_beside_a_known_one_keeps_the_known_one() { + // Since #1168 an unrecognized name costs its own name only. The + // sibling test above cannot see the difference — its list has + // nothing left once the bad name is dropped — so this is the case + // that separates "skip the name" from "void the marker" at the + // library boundary, where `analyze` is the only surface a consumer + // has. + let src = r#" +def fine(x): + # bca: suppress(cyclomatic, no_such_metric) + if x: + return 1 + return 0 +"#; + let space = analyze_lang(src, "fixture.py"); + let fine = find_function(&space, "fine").expect("function fine should exist"); + assert!( + fine.suppressed.covers(Metric::Cyclomatic), + "the recognized half of the list must still suppress; got {:?}", + fine.suppressed, + ); +} + +#[test] +fn suppress_file_marker_accepts_a_trailing_rationale() { + // `suppress-file` takes a rationale on the same terms as the + // function-scoped verb (#1168). Asserted through `analyze` rather + // than the parser so the whole path — comment node, walk, file-level + // scope — is covered for the file-scoped half too. + let src = r#" +# bca: suppress-file(loc, halstead) — generated protocol tables + +def fine(): + return 1 +"#; + let space = analyze_lang(src, "fixture.py"); + assert!( + space.suppressed.covers(Metric::Loc) && space.suppressed.covers(Metric::Halstead), + "file-scoped marker with a rationale must attach both metrics; got {:?}", + space.suppressed, + ); +} + #[test] fn unknown_verb_in_marker_has_no_effect() { // Parallel to `unknown_metric_in_marker_has_no_effect`, but diff --git a/tests/common/mod.rs b/tests/common/mod.rs index df0df5528..7824905c0 100644 --- a/tests/common/mod.rs +++ b/tests/common/mod.rs @@ -194,11 +194,26 @@ pub fn compare_rca_output_with_files_under( let corpus_root = source_root.join(repo_name); let paths = resolve_corpus_files(&corpus_root, &include, &exclude); + // Zero files is a different failure from the wrong number of files, + // and conflating them costs a debugging cycle: on a fresh clone or + // `git worktree` the corpus submodule is simply not checked out + // (#938, #1171), which has nothing to do with the expected count the + // rest of the message talks about. + assert!( + !paths.is_empty(), + "integration corpus not checked out: {} resolved 0 source files. \ + Run `make worktree-setup` from the repository root. By hand it is \ + `git submodule update --init --force`, and the `--force` is \ + load-bearing: after an interrupted checkout the submodule HEAD \ + already matches the recorded SHA, so a plain re-run is a silent \ + no-op.", + corpus_root.display(), + ); + // A corpus that resolves to *fewer* files than expected makes the // runner skip the missing files' snapshot assertions while still // returning `Ok(())`, so the test passes having verified less than it - // claims. Zero files is the degenerate case (#938): an uninitialized - // submodule leaves the directory empty and every assertion is skipped. + // claims. // // The corpora are submodules pinned to a fixed SHA, so the resolved // count is deterministic for a given checkout. Asserting it exactly @@ -209,11 +224,10 @@ pub fn compare_rca_output_with_files_under( assert_eq!( paths.len(), expected_files, - "unexpected corpus file count under {}. If it resolved 0 files the \ - integration corpus is empty or missing — initialize the submodules \ - with `git submodule update --init --recursive`. Otherwise the \ - corpus or the include/exclude globs changed; update the expected \ - count alongside the snapshots.", + "unexpected corpus file count under {}. The corpus or the \ + include/exclude globs changed; update the expected count \ + alongside the snapshots. If the corpus is only partially \ + checked out, `make worktree-setup` repairs it.", corpus_root.display(), ); diff --git a/tests/output_formats/main.rs b/tests/output_formats/main.rs index 99b082b73..1266d977c 100644 --- a/tests/output_formats/main.rs +++ b/tests/output_formats/main.rs @@ -3,12 +3,14 @@ //! //! Grouped into one binary by #1124 — see `tests/api/main.rs`. The //! `insta` snapshots these modules own moved with them, from -//! `tests/snapshots/` to `tests/output_formats/snapshots/`: insta -//! resolves the snapshot directory from the asserting file's own -//! location. Their names are unchanged, because insta keys the -//! `__.snap` prefix on the *last* component of -//! `module_path!()` — still `csv_test` / `sarif_test` now that they are -//! modules rather than crate roots. +//! `tests/snapshots/` to `tests/output_formats/snapshots/`, because +//! insta resolves the snapshot directory from the asserting file's own +//! location. They were also renamed: insta keys the +//! `__.snap` prefix on the *whole* of `module_path!()`, +//! not its last component, so `csv_test__csv_cpp_widget` became +//! `output_formats__csv_test__csv_cpp_widget` once these files became +//! modules of this driver rather than crate roots. See "Moving a test +//! file" in `tests/README.md`. #[path = "../common/mod.rs"] mod common; diff --git a/utils/gate-status-test.sh b/utils/gate-status-test.sh new file mode 100755 index 000000000..670e74fff --- /dev/null +++ b/utils/gate-status-test.sh @@ -0,0 +1,129 @@ +#!/usr/bin/env bash +# Self-tests for utils/gate-status.sh. +# +# The assertion that matters most is `preserves a non-zero exit status`: +# a wrapper that reports `fail` but exits 0 would let a red branch look +# validated, which is strictly worse than the ambiguity #1172 set out to +# remove. The stage-extraction cases feed the script a canned GNU make +# transcript rather than a real gate run, so they stay fast and do not +# depend on which stage happens to be breakable today. +# +# These messages quote both verdict spellings. That is safe because the +# published contract is `^BCA_GATE:` and every message here is prefixed, +# so a failing run of this test cannot plant a second verdict in the +# gate's own log. + +set -uo pipefail + +SCRIPT_DIR=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd) +GATE_STATUS="$SCRIPT_DIR/gate-status.sh" + +failures=0 +out=$(mktemp) +err=$(mktemp) +trap 'rm -f -- "$out" "$err"' EXIT + +fail() { + printf 'FAIL: %s\n' "$1" >&2 + failures=$((failures + 1)) +} + +expect_eq() { + # expect_eq + if [ "$2" != "$3" ]; then + fail "$1: expected [$2], got [$3]" + fi +} + +run_gate() { + # run_gate ; leaves stdout in $out, stderr in + # $err, and the wrapper's status in $rc. + "$GATE_STATUS" "$@" >"$out" 2>"$err" + rc=$? +} + +# A canned transcript of what GNU make writes to stderr when stages fail +# under `-j`: the leaf target a stage delegates to, the stage itself, a +# second stage that was already running, a duplicate report, the +# aggregate this script is pointed at, and the outer epilogue. +# +# Held in a variable rather than inlined in the heredoc so the forwarded +# copy can be compared against it verbatim. A line *count* is not the +# `intact` the assertion below claims: a filter inserted into the +# forwarding path (`| sed …`, an awk annotator) rewrites content without +# changing the number of newlines, and that is exactly the edit the +# `tee`-not-awk choice in gate-status.sh exists to prevent. +read -r -d '' MAKE_TRANSCRIPT <<'TRANSCRIPT' +make[3]: *** [Makefile:200: fmt-check] Error 1 +make[2]: *** [Makefile:1171: _pc-fmt] Error 2 +make[2]: *** Waiting for unfinished jobs.... +make[2]: *** [Makefile:1191: _pc-markdown-lint] Error 2 +make[2]: *** [Makefile:1171: _pc-fmt] Error 2 +make[1]: *** [Makefile:1093: _pc-all] Error 2 +make: *** [Makefile:1092: pre-commit] Error 2 +TRANSCRIPT +export MAKE_TRANSCRIPT + +emit_make_transcript() { + printf '%s\n' "$MAKE_TRANSCRIPT" >&2 + exit 2 +} +export -f emit_make_transcript + +# --- pass path ------------------------------------------------------- +run_gate pre-commit bash -c 'echo stage output; echo stage warning >&2' +expect_eq 'pass exit status' 0 "$rc" +expect_eq 'pass verdict is the last stdout line' \ + 'BCA_GATE: pass (gate=pre-commit)' "$(tail -n 1 "$out")" +expect_eq 'pass emits exactly one verdict' 1 "$(grep -c 'BCA_GATE:' "$out")" +expect_eq 'stdout is forwarded' 'stage output' "$(head -n 1 "$out")" +expect_eq 'stderr is forwarded, and stays on stderr' \ + 'stage warning' "$(cat "$err")" +expect_eq 'the verdict does not leak onto stderr' \ + 0 "$(grep -c 'BCA_GATE:' "$err")" + +# --- failure path: the exit status must survive ---------------------- +run_gate pre-commit bash -c 'exit 42' +expect_eq 'a non-zero exit status is preserved' 42 "$rc" +expect_eq 'fail verdict carries the real exit status' \ + 'BCA_GATE: fail (gate=pre-commit, exit=42, stage=unknown)' \ + "$(tail -n 1 "$out")" + +# --- failure path: stage extraction ---------------------------------- +run_gate pre-commit bash -c emit_make_transcript +expect_eq 'transcript exit status is preserved' 2 "$rc" +expect_eq 'every failing stage is named once, in report order' \ + 'BCA_GATE: fail (gate=pre-commit, exit=2, stage=_pc-fmt,_pc-markdown-lint)' \ + "$(tail -n 1 "$out")" +expect_eq 'fail emits exactly one verdict' 1 "$(grep -c 'BCA_GATE:' "$out")" +expect_eq 'the make transcript still reaches stderr intact' \ + "$MAKE_TRANSCRIPT" "$(cat "$err")" + +# --- failure path: a broken `tee` is not a gate failure --------------- +# The case ${PIPESTATUS[0]} exists for, and the only one where it and a +# plain `$?` disagree. Pointing TMPDIR at a missing directory makes the +# `mktemp` fail, so `tee` gets an empty path and exits non-zero while the +# gate itself succeeds. Under `pipefail`, `$?` is then `tee`'s status and +# the wrapper reports a green run as failed. Without this case that +# substitution passes the whole suite. +TMPDIR=/nonexistent-bca-gate-status run_gate pre-commit bash -c 'echo stage output' +expect_eq 'a tee that cannot write is not a gate failure' 0 "$rc" +expect_eq 'the gate status, not the pipeline status, decides the verdict' \ + 'BCA_GATE: pass (gate=pre-commit)' "$(tail -n 1 "$out")" + +# --- usage ----------------------------------------------------------- +run_gate onlyagatename +expect_eq 'a missing command is a usage error' 2 "$rc" +expect_eq 'a usage error emits no verdict' 0 "$(grep -c 'BCA_GATE:' "$out")" +# Which stream the usage line takes is the assertion, not merely that it +# exists: a verdict-free stdout is equally what dropping the `>&2` — or +# the whole printf — produces. +expect_eq 'the usage line goes to stderr' \ + "usage: $GATE_STATUS [args...]" "$(cat "$err")" +expect_eq 'a usage error writes nothing to stdout' 0 "$(wc -c <"$out")" + +if [ "$failures" -ne 0 ]; then + printf '%d gate-status check(s) failed\n' "$failures" >&2 + exit 1 +fi +printf 'gate-status: all checks passed\n' diff --git a/utils/gate-status.sh b/utils/gate-status.sh new file mode 100755 index 000000000..ab318036c --- /dev/null +++ b/utils/gate-status.sh @@ -0,0 +1,92 @@ +#!/usr/bin/env bash +# Run a validation gate and end its output with one machine-readable +# verdict line. +# +# utils/gate-status.sh [args...] +# +# The command's stdout and stderr are forwarded unchanged, on their own +# descriptors, and the command's exit status is this script's exit +# status. After it finishes, exactly one line is printed to stdout: +# +# BCA_GATE: pass (gate=pre-commit) +# BCA_GATE: fail (gate=pre-commit, exit=2, stage=_pc-fmt) +# +# Why this exists (#1172): `make pre-commit` runs a parallel DAG, so +# GNU make reports the first failing stage as +# `make[1]: *** [Makefile:1234: _pc-fmt] Error 2` and then keeps going +# until the stages already running finish. That line is therefore not a +# terminal verdict — it is routinely followed by several more stages' +# successful output — and reading an outcome out of the log by eye or +# by grepping `Error N` has repeatedly produced the wrong answer. +# +# The contract a consumer greps is `^BCA_GATE:` — the token at the start +# of a line. Only this script emits that, and only once per run. The +# anchor matters: gate-status-test.sh quotes both spellings in its own +# failure messages, and those are prefixed, so they cannot match. +# +# Three states, not two. No `BCA_GATE:` line at all means the run never +# finished: it crashed, was killed, or was interrupted. That must not be +# read as either pass or fail. +# +# On the failure path GNU make appends its own +# `make: *** [Makefile:NNNN: pre-commit] Error 2` epilogue after this +# line, because this script exits non-zero and make says so. That is +# unavoidable without swallowing the exit status, which would be a far +# worse defect than the one being fixed. The `BCA_GATE:` line is still +# the last thing the gate itself writes, and it is still unique. +# +# Unlike the other utils/ scripts this one needs no repository root: it +# runs whatever command it is handed, from wherever it is invoked. + +set -uo pipefail + +if [ "$#" -lt 2 ]; then + printf 'usage: %s [args...]\n' "$0" >&2 + exit 2 +fi + +gate=$1 +shift + +captured_stderr=$(mktemp "${TMPDIR:-/tmp}/bca-gate-status.XXXXXX") +trap 'rm -f -- "$captured_stderr"' EXIT + +# Copy stderr aside — make names a failing target only there — while +# keeping the two streams separate, since a caller may redirect them +# independently. `tee` rather than a filter written in awk: awk block- +# buffers its output, which would reorder stderr against stdout in a +# `> log 2>&1` capture. +# +# ${PIPESTATUS[0]} is the gate's status. `$?` is not a substitute even +# with pipefail set: pipefail yields the *last* non-zero status, so a +# `tee` that failed to write the temp file would be reported as a gate +# failure on an otherwise green run. +{ "$@" 2>&1 1>&3 | tee -- "$captured_stderr" >&2; } 3>&1 +status=${PIPESTATUS[0]} + +if [ "$status" -eq 0 ]; then + printf 'BCA_GATE: pass (gate=%s)\n' "$gate" + exit 0 +fi + +# Pull the DAG stages out of `make[N]: *** [Makefile:123: _pc-fmt] Error 2`. +# Every stage make named is listed, in the order it named them: under +# `-j` the first failure only stops *scheduling*, so stages already +# running can fail too, and naming just one would hide the rest. +# +# Only `_`-prefixed targets are DAG stages. That drops both the leaf +# targets a stage delegates to (`fmt-check`, `test`) and the outer +# `make: *** [... pre-commit]` epilogue, neither of which tells a reader +# anything the stage name does not. The `_pc-all` / `_ci-all` aggregate +# this script is pointed at is dropped for the same reason. +stages=$( + sed -n 's/^make\[[0-9]*\]: \*\*\* \[[^]]*: \(_[A-Za-z][A-Za-z0-9_-]*\)\] Error .*/\1/p' \ + "$captured_stderr" \ + | grep -Ev '^_(pc|ci)-all$' \ + | awk '!seen[$0]++' \ + | paste -sd, - +) + +printf 'BCA_GATE: fail (gate=%s, exit=%d, stage=%s)\n' \ + "$gate" "$status" "${stages:-unknown}" +exit "$status" diff --git a/utils/worktree-setup-test.py b/utils/worktree-setup-test.py new file mode 100644 index 000000000..1df311b8b --- /dev/null +++ b/utils/worktree-setup-test.py @@ -0,0 +1,361 @@ +#!/usr/bin/env python3 +"""Tests for worktree-setup.py. + +The classifier decides whether a submodule gets a *destructive* +``--force``, so it is tested against real git repositories rather than a +mock: the whole question is what git's own plumbing reports for each +damaged shape, and a mock would only re-state this file's assumptions. +Each fixture builds a two-repository superproject in a tempdir, damages +the submodule in one specific way, and asserts the resulting state. + +The ``INCOMPLETE`` fixture is the one that matters most — it reproduces +the interrupted checkout from #1171 and also pins the claim that +motivated the whole script: a plain ``git submodule update --init`` over +that state exits 0 and restores nothing. + +Run with: + python3 -m unittest -q utils/worktree-setup-test.py +""" + +from __future__ import annotations + +import contextlib +import importlib.util +import io +import os +import pathlib +import subprocess +import tempfile +import unittest +import unittest.mock + +# `update()` and `main()` shell out to plain `git` with no `-c` flags of +# their own, so the fixture's local (`file://`) submodule origin has to be +# re-enabled through the environment instead — `protocol.file` is denied +# by default since CVE-2022-39253. Applied per test, never to any config. +FILE_PROTOCOL_ENV = { + "GIT_CONFIG_COUNT": "1", + "GIT_CONFIG_KEY_0": "protocol.file.allow", + "GIT_CONFIG_VALUE_0": "always", +} + +UTILS_DIR = pathlib.Path(__file__).resolve().parent +REPO_ROOT = UTILS_DIR.parent +SCRIPT_SRC = UTILS_DIR / "worktree-setup.py" + + +def _load_module(): # type: ignore[no-untyped-def] + spec = importlib.util.spec_from_file_location("worktree_setup", SCRIPT_SRC) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +SETUP = _load_module() + +# Identity and protocol settings the fixtures need. `protocol.file` is +# denied by default since CVE-2022-39253 and a local submodule clone is +# exactly the use it was denied for, so it is re-enabled per invocation +# rather than written into any config. +GIT_ENV_ARGS = [ + "-c", + "user.email=test@example.invalid", + "-c", + "user.name=worktree-setup test", + "-c", + "protocol.file.allow=always", +] + + +def run_git(cwd: pathlib.Path, *args: str) -> str: + return subprocess.run( + ["git", *GIT_ENV_ARGS, *args], + cwd=cwd, + check=True, + capture_output=True, + text=True, + ).stdout + + +class SuperprojectFixture: + """A superproject with one submodule at `vendor/sub`, in a tempdir.""" + + SUBMODULE_PATH = "vendor/sub" + + def __init__(self, tmp: pathlib.Path) -> None: + self.sub_origin = tmp / "origin" + self.root = tmp / "super" + for path in (self.sub_origin, self.root): + path.mkdir(parents=True) + run_git(path, "init", "-q", "-b", "main") + + # Two files, one nested, so a partial deletion is expressible. + (self.sub_origin / "a.txt").write_text("a\n") + (self.sub_origin / "snapshots").mkdir() + (self.sub_origin / "snapshots" / "s.snap").write_text("s\n") + run_git(self.sub_origin, "add", "-A") + run_git(self.sub_origin, "commit", "-qm", "sub") + + (self.root / "root.txt").write_text("root\n") + run_git(self.root, "add", "-A") + run_git(self.root, "commit", "-qm", "root") + run_git( + self.root, + "submodule", + "add", + "-q", + str(self.sub_origin), + self.SUBMODULE_PATH, + ) + run_git(self.root, "commit", "-qm", "add submodule") + + @property + def work(self) -> pathlib.Path: + return self.root / self.SUBMODULE_PATH + + def classify(self) -> object: + return SETUP.classify(self.root, self.SUBMODULE_PATH) + + +class ClassifyTest(unittest.TestCase): + """One damaged shape per test, each with a distinct expected state.""" + + def setUp(self) -> None: + self._tmp = tempfile.TemporaryDirectory() + self.addCleanup(self._tmp.cleanup) + self.fixture = SuperprojectFixture(pathlib.Path(self._tmp.name)) + + def test_freshly_added_submodule_is_ready(self) -> None: + self.assertIs(self.fixture.classify(), SETUP.State.READY) + + def test_deinitialized_submodule_is_missing(self) -> None: + run_git( + self.fixture.root, + "submodule", + "deinit", + "-f", + "--", + self.fixture.SUBMODULE_PATH, + ) + self.assertIs(self.fixture.classify(), SETUP.State.MISSING) + + def test_emptied_worktree_is_incomplete(self) -> None: + run_git(self.fixture.work, "rm", "-rq", ".") + self.assertIs(self.fixture.classify(), SETUP.State.INCOMPLETE) + + def test_partially_emptied_worktree_is_incomplete(self) -> None: + # The #1171 tell: "a corpus directory containing only snapshots/". + run_git(self.fixture.work, "rm", "-q", "a.txt") + self.assertIs(self.fixture.classify(), SETUP.State.INCOMPLETE) + self.assertTrue((self.fixture.work / "snapshots" / "s.snap").exists()) + + def test_deletion_plus_modification_is_blocked(self) -> None: + run_git(self.fixture.work, "rm", "-q", "a.txt") + (self.fixture.work / "snapshots" / "s.snap").write_text("locally accepted\n") + self.assertIs(self.fixture.classify(), SETUP.State.BLOCKED) + + def test_modification_alone_is_ready(self) -> None: + # Accepting a snapshot in the integration-snapshot submodule is + # routine (AGENTS.md); it must not read as damage, or the setup + # target would offer to overwrite work in progress. + (self.fixture.work / "snapshots" / "s.snap").write_text("locally accepted\n") + self.assertIs(self.fixture.classify(), SETUP.State.READY) + + def test_untracked_file_alone_is_ready(self) -> None: + # `.snap.new` files land here constantly during a metric change. + (self.fixture.work / "snapshots" / "s.snap.new").write_text("pending\n") + self.assertIs(self.fixture.classify(), SETUP.State.READY) + + def test_wrong_revision_is_stale(self) -> None: + (self.fixture.work / "a.txt").write_text("moved on\n") + run_git(self.fixture.work, "add", "-A") + run_git(self.fixture.work, "commit", "-qm", "advance submodule") + self.assertIs(self.fixture.classify(), SETUP.State.STALE) + + +class RepairTest(unittest.TestCase): + """The behaviour the script exists for: plain re-run vs `--force`. + + `update()` is called for real rather than having its argv spelled out + here. A test that assembles its own `git submodule update --force` + proves git can repair the tree; it never proves this script asks git + to — the production bookkeeping and a test's replica of it are two + different things (`.claude/rules/testing.md`). + """ + + def setUp(self) -> None: + self._tmp = tempfile.TemporaryDirectory() + self.addCleanup(self._tmp.cleanup) + self.fixture = SuperprojectFixture(pathlib.Path(self._tmp.name)) + run_git(self.fixture.work, "rm", "-rq", ".") + patcher = unittest.mock.patch.dict(os.environ, FILE_PROTOCOL_ENV) + patcher.start() + self.addCleanup(patcher.stop) + + def test_plain_update_is_a_silent_no_op(self) -> None: + # Deliberately git's own argv, not `update()`: the claim under + # test is about git's behaviour, and it is the premise the whole + # script rests on (#1171). + run_git( + self.fixture.root, + "submodule", + "update", + "--init", + "--", + self.fixture.SUBMODULE_PATH, + ) + self.assertFalse((self.fixture.work / "a.txt").exists()) + self.assertIs(self.fixture.classify(), SETUP.State.INCOMPLETE) + + def test_forced_update_restores_the_worktree(self) -> None: + SETUP.update(self.fixture.root, self.fixture.SUBMODULE_PATH, force=True) + self.assertTrue((self.fixture.work / "a.txt").exists()) + self.assertIs(self.fixture.classify(), SETUP.State.READY) + + def test_unforced_update_restores_nothing(self) -> None: + # The other side of `if force:`. Asserting only the forced case + # leaves the branch half-observable: an inverted condition still + # passes some `--force` somewhere. + SETUP.update(self.fixture.root, self.fixture.SUBMODULE_PATH, force=False) + self.assertFalse((self.fixture.work / "a.txt").exists()) + self.assertIs(self.fixture.classify(), SETUP.State.INCOMPLETE) + + +class MainTest(unittest.TestCase): + """`main()` end to end, against a fixture standing in for the repo. + + Patching the two module-level path constants is what makes the exit + code reachable: `make worktree-setup` consumes it, and neither the + BLOCKED refusal nor the repair path is observable from `classify()` + alone. + """ + + def setUp(self) -> None: + self._tmp = tempfile.TemporaryDirectory() + self.addCleanup(self._tmp.cleanup) + # Resolved up front: `verify_repo_root` compares the patched root + # against git's `--show-toplevel`, and a tempdir reached through + # a symlink would fail that comparison for the wrong reason. + self.fixture = SuperprojectFixture(pathlib.Path(self._tmp.name).resolve()) + for patcher in ( + unittest.mock.patch.object(SETUP, "REPO_ROOT", self.fixture.root), + unittest.mock.patch.object( + SETUP, "GITMODULES", self.fixture.root / ".gitmodules" + ), + unittest.mock.patch.dict(os.environ, FILE_PROTOCOL_ENV), + ): + patcher.start() + self.addCleanup(patcher.stop) + + def run_main(self) -> tuple[int, str, str]: + """`main()`'s exit code, its stdout, and its stderr. + + stdout is returned rather than discarded because it is the only + place the READY skip, the `fetched` bookkeeping and the + `nothing to do` line are observable — see the idempotency test + below. + """ + out, err = io.StringIO(), io.StringIO() + with contextlib.redirect_stdout(out), contextlib.redirect_stderr(err): + code = SETUP.main() + return code, out.getvalue(), err.getvalue() + + def test_main_repairs_an_incomplete_checkout_and_exits_zero(self) -> None: + run_git(self.fixture.work, "rm", "-rq", ".") + code, out, err = self.run_main() + self.assertEqual(code, 0, f"stderr was: {err}") + self.assertTrue((self.fixture.work / "a.txt").exists()) + self.assertIs(self.fixture.classify(), SETUP.State.READY) + self.assertIn(SETUP.State.INCOMPLETE.value, out) + + def test_main_is_idempotent_and_says_so_on_the_second_run(self) -> None: + """The `idempotent` half of the bootstrap's contract. + + A single run cannot show it. Both other cases here damage the + fixture before calling `main()`, so its state is never READY at + the loop and the `if state is State.READY: continue` branch is + never taken — deleting that branch outright failed none of this + suite's 17 tests. Running twice is what makes the skip + observable: without it the second pass re-runs `update()`, sets + `fetched`, and reports the nested-submodule note instead of + `nothing to do`. + """ + run_git(self.fixture.work, "rm", "-rq", ".") + first_code, first_out, first_err = self.run_main() + self.assertEqual(first_code, 0, f"stderr was: {first_err}") + # The first run must actually do something, or "the second run + # is a no-op" is a claim about two no-ops. + self.assertIn(SETUP.State.INCOMPLETE.value, first_out) + self.assertNotIn("nothing to do", first_out) + + second_code, second_out, second_err = self.run_main() + self.assertEqual(second_code, 0, f"stderr was: {second_err}") + self.assertIn(SETUP.State.READY.value, second_out) + self.assertIn("all corpora already checked out; nothing to do", second_out) + self.assertNotIn("nested submodules", second_out) + + def test_main_refuses_a_blocked_submodule_and_exits_one(self) -> None: + run_git(self.fixture.work, "rm", "-q", "a.txt") + modified = self.fixture.work / "snapshots" / "s.snap" + modified.write_text("locally accepted\n") + code, _out, err = self.run_main() + self.assertEqual(code, 1, "a refusal must not exit green") + self.assertIn(self.fixture.SUBMODULE_PATH, err) + self.assertIn(SETUP.FORCE_RATIONALE, err) + self.assertEqual( + modified.read_text(), + "locally accepted\n", + "the refusal exists to keep this file; it must survive", + ) + self.assertFalse((self.fixture.work / "a.txt").exists()) + + +class RepoGuardTest(unittest.TestCase): + """The root-resolution guard around the destructive `--force`.""" + + def test_submodule_paths_reject_an_escaping_path(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + root = pathlib.Path(tmp) + run_git(root, "init", "-q", "-b", "main") + (root / ".gitmodules").write_text( + '[submodule "evil"]\n\tpath = ../outside\n\turl = https://example.invalid\n' + ) + with self.assertRaises(SystemExit): + SETUP.submodule_paths(root) + + def test_real_repository_root_is_accepted(self) -> None: + SETUP.verify_repo_root() + self.assertTrue((REPO_ROOT / ".gitmodules").is_file()) + + def test_a_non_checkout_gets_the_tailored_refusal(self) -> None: + # `git()` runs with `check=False`, so the old + # `except CalledProcessError` arm was unreachable and this case + # exited through git()'s generic "git rev-parse ... failed" + # preamble instead of naming the actual problem. + with tempfile.TemporaryDirectory() as tmp: + root = pathlib.Path(tmp).resolve() + (root / ".gitmodules").write_text("") + with ( + unittest.mock.patch.object(SETUP, "REPO_ROOT", root), + unittest.mock.patch.object(SETUP, "GITMODULES", root / ".gitmodules"), + self.assertRaises(SystemExit) as caught, + ): + SETUP.verify_repo_root() + self.assertIn("is not a git checkout", str(caught.exception)) + + def test_a_missing_git_binary_is_diagnosed_as_such(self) -> None: + # The only input that ever reached the old handler, where + # "is not a git checkout" was the wrong diagnosis. + with ( + unittest.mock.patch.object( + SETUP.subprocess, "run", side_effect=FileNotFoundError("git") + ), + self.assertRaises(SystemExit) as caught, + ): + SETUP.git(["rev-parse", "--show-toplevel"], cwd=REPO_ROOT) + self.assertIn("git is not on PATH", str(caught.exception)) + + +if __name__ == "__main__": + unittest.main() diff --git a/utils/worktree-setup.py b/utils/worktree-setup.py new file mode 100755 index 000000000..f45454957 --- /dev/null +++ b/utils/worktree-setup.py @@ -0,0 +1,227 @@ +#!/usr/bin/env python3 +"""worktree-setup + +Idempotently check out the integration corpora under +``tests/repositories/`` for a fresh clone or a fresh ``git worktree``. + +Driven by ``make worktree-setup``, which runs this script and then the +Python-bindings venv bootstrap. This half owns only the submodules. + +Why a script rather than a Makefile recipe +------------------------------------------ + +Recovering an *interrupted* ``git submodule update`` needs ``--force``, +and ``--force`` is destructive: it is ``git checkout --force`` inside the +submodule, so it overwrites locally modified tracked files. The +integration-snapshot submodule is one contributors legitimately edit +(accepting ``.snap`` files, per AGENTS.md), so the escalation has to be +conditional on what the submodule actually looks like. That +classification is what this script exists for. + +The state that motivates it (#1171): an interrupted +``git submodule update --init`` leaves the submodule's ``.git`` in place, +its HEAD already at the recorded SHA, and its working tree missing some +or all of its files. Because the SHA matches, a plain re-run is a +**silent no-op** — it exits 0 and restores nothing. Only ``--force`` +repairs it. + +Classification per submodule +---------------------------- + +``READY`` working tree complete and at the recorded SHA -> skip. +``MISSING`` never initialized (no ``.git``) -> plain ``--init``. +``STALE`` complete, but not at the recorded SHA -> plain ``--init``. +``INCOMPLETE`` tracked files deleted and nothing modified -> ``--force``. +``BLOCKED`` tracked files deleted *and* others modified -> refuse and + print the command, because forcing would discard the + modifications. + +Deliberately non-recursive +-------------------------- + +``DeepSpeech`` carries its own submodules (``tensorflow`` at 246 MB, +``kenlm``, ``doc/examples``). The corpus test excludes +``**/DeepSpeech/tensorflow/**`` and ``**/DeepSpeech/kenlm/**``, so no +test reads them — and fetching them is most of the wall time that gets +interrupted in the first place. The out-of-band benchmark harness +(``make bench-*``) does walk them; see +``docs/development/benchmarking.md`` for its recursive checkout. +""" + +from __future__ import annotations + +import enum +import pathlib +import subprocess +import sys + +# `parents[1]`, not `parent`: this script lives in `utils/` but every +# path it touches is anchored at the repository root. Resolving from +# `__file__` rather than the cwd is what keeps a `--force` from ever +# being aimed at whatever directory the caller happened to be in. +REPO_ROOT = pathlib.Path(__file__).resolve().parents[1] + +GITMODULES = REPO_ROOT / ".gitmodules" + +# Why the by-hand recovery needs `--force`. Named once so the message +# this script prints and the one the test suite prints stay in step. +FORCE_RATIONALE = ( + "the --force is load-bearing: after an interrupted checkout the " + "submodule HEAD already matches the recorded SHA, so a plain re-run " + "is a silent no-op" +) + + +class State(enum.Enum): + """How a submodule's working tree compares to what git recorded.""" + + READY = "ok" + MISSING = "not initialized" + STALE = "at the wrong revision" + INCOMPLETE = "incomplete checkout" + BLOCKED = "incomplete checkout with local modifications" + + +def git(args: list[str], cwd: pathlib.Path, on_failure: str | None = None) -> str: + """Run git in `cwd`, returning stdout. + + Raises `SystemExit` carrying git's own stderr on a non-zero exit, and + on a missing git binary: a traceback would bury the one line that + says what went wrong. `on_failure` replaces the generic preamble + where the caller can name the actual diagnosis — this runs with + ``check=False``, so a caller cannot catch `CalledProcessError` to + supply one itself. + """ + try: + result = subprocess.run( + ["git", *args], + cwd=cwd, + check=False, + capture_output=True, + text=True, + ) + except FileNotFoundError as exc: + raise SystemExit("git is not on PATH; refusing to run") from exc + if result.returncode != 0: + preamble = on_failure or f"git {' '.join(args)} failed in {cwd}" + raise SystemExit(f"{preamble}:\n{result.stderr.strip()}") + return result.stdout + + +def submodule_paths(root: pathlib.Path) -> list[str]: + """Repository-relative paths of every submodule in `.gitmodules`. + + Read through `git config` rather than a hand-rolled INI parse so the + quoting and section-name rules are git's own. + """ + out = git( + ["config", "-f", str(root / ".gitmodules"), "--get-regexp", r"^submodule\..*\.path$"], + cwd=root, + ) + paths = [line.split(" ", 1)[1] for line in out.splitlines() if " " in line] + # These come from a tracked file, but a `--force` is aimed at each of + # them, so refuse anything that could escape the repository. + for path in paths: + if pathlib.PurePosixPath(path).is_absolute() or ".." in pathlib.PurePosixPath(path).parts: + raise SystemExit(f"refusing to act on submodule path outside the repository: {path}") + return paths + + +def recorded_sha(root: pathlib.Path, path: str) -> str | None: + """The gitlink SHA the superproject's HEAD records for `path`.""" + out = git(["ls-tree", "HEAD", "--", path], cwd=root).split() + return out[2] if len(out) >= 3 and out[1] == "commit" else None + + +def classify(root: pathlib.Path, path: str) -> State: + """Decide what, if anything, `path` needs to become usable.""" + work = root / path + # A submodule git creates gets its `.git` file before any file + # content, so `.git` present means "a checkout was started here" — + # which is exactly the case a plain re-run cannot repair. + if not (work / ".git").exists(): + return State.MISSING + + diff = git(["diff", "--name-status", "HEAD"], cwd=work).splitlines() + statuses = {line.split("\t", 1)[0][:1] for line in diff if line} + if "D" in statuses: + return State.BLOCKED if statuses - {"D"} else State.INCOMPLETE + + current = git(["rev-parse", "HEAD"], cwd=work).strip() + return State.READY if current == recorded_sha(root, path) else State.STALE + + +def update(root: pathlib.Path, path: str, force: bool) -> None: + """Check out `path` at the recorded SHA, optionally forcing. + + git's own progress output is inherited rather than captured — the + first checkout of a corpus is slow enough that a silent wait reads + as a hang, and that wait is what gets interrupted (#1171). + """ + args = ["submodule", "update", "--init"] + if force: + args.append("--force") + result = subprocess.run(["git", *args, "--", path], cwd=root, check=False) + if result.returncode != 0: + raise SystemExit(f"checking out {path} failed; see git's output above") + + +def verify_repo_root() -> None: + """Refuse to run anywhere but this repository's own checkout.""" + if not GITMODULES.is_file(): + raise SystemExit(f"no .gitmodules under {REPO_ROOT}; refusing to run") + toplevel = git( + ["rev-parse", "--show-toplevel"], + cwd=REPO_ROOT, + on_failure=f"{REPO_ROOT} is not a git checkout; refusing to run", + ).strip() + if pathlib.Path(toplevel).resolve() != REPO_ROOT: + raise SystemExit( + f"this script lives under {REPO_ROOT} but git reports the checkout root as " + f"{toplevel}; refusing to run" + ) + + +def main() -> int: + verify_repo_root() + print(f"worktree-setup: integration corpora under {REPO_ROOT}") + + blocked: list[str] = [] + fetched = False + for path in submodule_paths(REPO_ROOT): + state = classify(REPO_ROOT, path) + print(f" {path}: {state.value}") + if state is State.READY: + continue + if state is State.BLOCKED: + blocked.append(path) + continue + update(REPO_ROOT, path, force=state is State.INCOMPLETE) + fetched = True + + for path in blocked: + print( + f"\nERROR: {path} is missing tracked files but also has local " + f"modifications. Repairing it means `git checkout --force` inside " + f"the submodule, which would discard them, so this script will " + f"not. Commit or stash them, then run:\n" + f" git submodule update --init --force -- {path}\n" + f"({FORCE_RATIONALE}.)", + file=sys.stderr, + ) + if blocked: + return 1 + + if fetched: + print( + "\nNote: nested submodules (DeepSpeech's tensorflow and kenlm) are\n" + "not fetched — the corpus tests exclude them. `make bench-*` does\n" + "walk them; see docs/development/benchmarking.md for its checkout." + ) + else: + print(" all corpora already checked out; nothing to do") + return 0 + + +if __name__ == "__main__": + sys.exit(main())