{"schema_version":"mcp-selection-lab.agent-trust-lab.selection-evidence-status-v3","track":"EXTERNAL_GOLD_EXPANSION_V2","status":"MEASURED","benchmark":"External Gold Expansion V2 Batch 3","benchmark_version":"external-gold-expansion-v2-batch3","batch_characterization":"BOUNDARY_RISK_VALIDATION_BATCH","benchmark_type":"EXTERNAL_INDEPENDENT_GOLD_BENCHMARK","benchmark_provenance":"EXTERNAL_INDEPENDENT_GOLD_BENCHMARK","promotion_level":"S2","promotion_status":"S2 external-selection-tested","subject_level_promotions":[],"case_count":36,"strict_correct":34,"strict_score":"34/36","strict_accuracy":0.944444,"strict_accuracy_display":"94.4444%","wilson_95":{"lower":0.818553,"upper":0.98463},"wilson_95_display":"81.8553%–98.4630%","provider_valid_count":36,"provider_valid_coverage":1.0,"semantic_coverage":1.0,"infrastructure_failure_count":0,"risk_bucket_metrics":{"HIGH":{"cases":15,"provider_valid_cases":15,"infrastructure_failures":0,"semantic_valid_selection_cases":15,"semantic_coverage":1.0,"correct":13,"misses":2,"strict_accuracy":0.866667,"miss_rate":0.133333,"wilson_95":{"lower":0.62118,"upper":0.962639},"provider_coverage":1.0},"MEDIUM":{"cases":12,"provider_valid_cases":12,"infrastructure_failures":0,"semantic_valid_selection_cases":12,"semantic_coverage":1.0,"correct":12,"misses":0,"strict_accuracy":1.0,"miss_rate":0.0,"wilson_95":{"lower":0.757506,"upper":1.0},"provider_coverage":1.0},"LOW":{"cases":9,"provider_valid_cases":9,"infrastructure_failures":0,"semantic_valid_selection_cases":9,"semantic_coverage":1.0,"correct":9,"misses":0,"strict_accuracy":1.0,"miss_rate":0.0,"wilson_95":{"lower":0.700855,"upper":1.0},"provider_coverage":1.0}},"risk_score_ordering":{"method":"Spearman rank correlation between frozen risk score and binary semantic failure; exploratory only.","spearman_rho":0.299238,"high_vs_low_miss_rate_difference":0.133333,"high_vs_low_risk_ratio":null,"interpretation":"HIGH had two misses while MEDIUM and LOW had none; the zero denominator and two total misses make the association exploratory, not statistically conclusive."},"scanner_evaluation":{"scanner_id":"BOUNDARY_RISK_SCAN_V1","decision":"PROMISING_BUT_MORE_DATA_REQUIRED","decision_reason":"Both observed Batch 3 misses were in HIGH, while MEDIUM and LOW had no misses, but the total of two misses and concentration in one subject is too small for validation or production-grade prediction claims.","statistical_claim":false},"taxonomy_signal_metrics":{"PAIRWISE_DESCRIPTION_LEXICAL_OVERLAP":{"exposure_count":13,"miss_count":1,"miss_rate":0.076923,"case_ids":["b3-dg-003","b3-gm-002","b3-gm-003","b3-gm-004","b3-gm-006","b3-dw-001","b3-dw-002","b3-dw-003","b3-dw-005","b3-dw-006","b3-mrc-001","b3-mrc-002","b3-mrc-006"],"interpretation":"descriptive exposure analysis; not a causal estimate"},"REQUIRED_INPUT_SCHEMA_OVERLAP":{"exposure_count":15,"miss_count":2,"miss_rate":0.133333,"case_ids":["b3-dg-003","b3-dg-004","b3-dg-005","b3-dg-006","b3-gm-001","b3-gm-002","b3-gm-003","b3-gm-004","b3-gm-005","b3-gm-006","b3-dw-001","b3-dw-002","b3-mrc-001","b3-mrc-002","b3-mrc-006"],"interpretation":"descriptive exposure analysis; not a causal estimate"},"GENERIC_SPECIFIC_DESCRIPTION_IMBALANCE":{"exposure_count":5,"miss_count":0,"miss_rate":0.0,"case_ids":["b3-dg-001","b3-dg-002","b3-dg-003","b3-mrc-001","b3-mrc-002"],"interpretation":"descriptive exposure analysis; not a causal estimate"},"OUTPUT_GRANULARITY_BOUNDARY_ABSENCE":{"exposure_count":12,"miss_count":2,"miss_rate":0.166667,"case_ids":["b3-gm-001","b3-gm-002","b3-gm-003","b3-gm-004","b3-gm-005","b3-gm-006","b3-dw-001","b3-dw-002","b3-dw-003","b3-dw-004","b3-dw-005","b3-dw-006"],"interpretation":"descriptive exposure analysis; not a causal estimate"},"BOUNDARY_LANGUAGE_COVERAGE":{"exposure_count":29,"miss_count":2,"miss_rate":0.068966,"case_ids":["b3-dg-001","b3-dg-002","b3-dg-003","b3-dg-004","b3-dg-005","b3-dg-006","b3-gm-001","b3-gm-002","b3-gm-003","b3-gm-004","b3-gm-005","b3-gm-006","b3-dw-001","b3-dw-002","b3-dw-003","b3-dw-004","b3-dw-005","b3-dw-006","b3-cf-001","b3-cf-002","b3-cf-003","b3-cf-004","b3-cf-005","b3-cf-006","b3-mrc-001","b3-mrc-002","b3-mrc-003","b3-mrc-004","b3-mrc-005"],"interpretation":"descriptive exposure analysis; not a causal estimate"}},"taxonomy_validation":[{"stable_id":"RETRIEVAL_GRANULARITY_CONFUSION","status":"INSUFFICIENT_EVIDENCE","batch3_evidence":"No Batch 3 miss in the data.gouv resource/list cases; prior Batch 2 observation remains unchanged.","confidence":"low"},{"stable_id":"DEFAULT_GENERAL_TOOL_PREFERENCE","status":"SUPPORTED","batch3_evidence":"Both Batch 3 misses selected GitMCP's broad fetch tool over a more specific neighboring operation.","confidence":"low; two misses in one subject"},{"stable_id":"DESCRIPTION_ASYMMETRY","status":"PARTIALLY_SUPPORTED","batch3_evidence":"Both misses involved the short generic fetch description competing with more specific retrieval modes; the scanner exposed incomplete boundary language.","confidence":"low"},{"stable_id":"QUESTION_VS_STRUCTURE_CONFUSION","status":"INSUFFICIENT_EVIDENCE","batch3_evidence":"DeepWiki Batch 3 cases were all correct; Batch 1 miss remains preserved and unchanged.","confidence":"low"},{"stable_id":"SCHEMA_SEMANTIC_GAP","status":"INSUFFICIENT_EVIDENCE","batch3_evidence":"No controlled schema rewrite or parameter-ablation experiment was performed.","confidence":"low"}],"semantic_miss_count":2,"semantic_misses":[{"case_id":"b3-gm-005","subject_id":"gitmcp-docs","frozen_prompt":"Read the external documentation URL linked from the GitMCP repository documentation and summarize that page.","expected_tool":"fetch_generic_url_content","selected_tool":"fetch_generic_documentation","competing_plausible_tools":["fetch_generic_documentation"],"boundary_family":"DOCUMENTATION_VS_REFERENCED_URL","risk_bucket":"HIGH","risk_score":65.5556,"frozen_external_source_id":"gitmcp-readme-c487a29-2026-05-08","frozen_source_url":"https://github.com/idosal/git-mcp/blob/c487a29895dcfcb5b672247e646426a56e2051c1/README.md","frozen_source_pointer":"README.md lines 365-371; referenced URL retrieval workflow","provider_response_hash":"74d2208690baab941886bb7f8147f0b2454222194f5cbde02252d3a3a982e492","review_conclusion":"GOLD_REMAINS_VALID; semantic miss preserved in strict score. The public source supports the requested neighboring operation, and the selected tool is a plausible but incorrect neighboring retrieval mode.","gold_label_changed":false,"original_prediction_preserved":true,"semantic_retry":false},{"case_id":"b3-gm-006","subject_id":"gitmcp-docs","frozen_prompt":"How does GitMCP decide which repository documents to use when llms.txt is absent? Explain the documented fallback behavior; I am not asking where the code implements it.","expected_tool":"search_generic_documentation","selected_tool":"fetch_generic_documentation","competing_plausible_tools":["search_generic_code"],"boundary_family":"DOCUMENTED_BEHAVIOR_VS_IMPLEMENTATION","risk_bucket":"HIGH","risk_score":83.1884,"frozen_external_source_id":"gitmcp-readme-c487a29-2026-05-08","frozen_source_url":"https://github.com/idosal/git-mcp/blob/c487a29895dcfcb5b672247e646426a56e2051c1/README.md","frozen_source_pointer":"README.md lines 365-371 and 463-465; documentation fallback discussion","provider_response_hash":"6676c39505e520f919e7f7747aa86481a74bb947a846580fb0dfc0247411f73c","review_conclusion":"GOLD_REMAINS_VALID; semantic miss preserved in strict score. The public source supports the requested neighboring operation, and the selected tool is a plausible but incorrect neighboring retrieval mode.","gold_label_changed":false,"original_prediction_preserved":true,"semantic_retry":false}],"subjects":[{"canonical_id":"datagouv-mcp","cases":6,"provider_valid_cases":6,"infrastructure_failures":0,"semantic_valid_selection_cases":6,"semantic_coverage":1.0,"correct":6,"misses":0,"strict_accuracy":1.0,"miss_rate":0.0,"wilson_95":{"lower":0.609666,"upper":1.0},"provider_coverage":1.0,"badge":null,"promotion_status":"DESCRIPTIVE_ONLY_NO_SUBJECT_LEVEL_POLICY_PROMOTION"},{"canonical_id":"gitmcp-docs","cases":6,"provider_valid_cases":6,"infrastructure_failures":0,"semantic_valid_selection_cases":6,"semantic_coverage":1.0,"correct":4,"misses":2,"strict_accuracy":0.666667,"miss_rate":0.333333,"wilson_95":{"lower":0.299993,"upper":0.903229},"provider_coverage":1.0,"badge":null,"promotion_status":"DESCRIPTIVE_ONLY_NO_SUBJECT_LEVEL_POLICY_PROMOTION"},{"canonical_id":"deepwiki-mcp","cases":6,"provider_valid_cases":6,"infrastructure_failures":0,"semantic_valid_selection_cases":6,"semantic_coverage":1.0,"correct":6,"misses":0,"strict_accuracy":1.0,"miss_rate":0.0,"wilson_95":{"lower":0.609666,"upper":1.0},"provider_coverage":1.0,"badge":null,"promotion_status":"DESCRIPTIVE_ONLY_NO_SUBJECT_LEVEL_POLICY_PROMOTION"},{"canonical_id":"cloudflare-docs-mcp","cases":6,"provider_valid_cases":6,"infrastructure_failures":0,"semantic_valid_selection_cases":6,"semantic_coverage":1.0,"correct":6,"misses":0,"strict_accuracy":1.0,"miss_rate":0.0,"wilson_95":{"lower":0.609666,"upper":1.0},"provider_coverage":1.0,"badge":null,"promotion_status":"DESCRIPTIVE_ONLY_NO_SUBJECT_LEVEL_POLICY_PROMOTION"},{"canonical_id":"exa-mcp","cases":6,"provider_valid_cases":6,"infrastructure_failures":0,"semantic_valid_selection_cases":6,"semantic_coverage":1.0,"correct":6,"misses":0,"strict_accuracy":1.0,"miss_rate":0.0,"wilson_95":{"lower":0.609666,"upper":1.0},"provider_coverage":1.0,"badge":null,"promotion_status":"DESCRIPTIVE_ONLY_NO_SUBJECT_LEVEL_POLICY_PROMOTION"},{"canonical_id":"microsoft-release-communications","cases":6,"provider_valid_cases":6,"infrastructure_failures":0,"semantic_valid_selection_cases":6,"semantic_coverage":1.0,"correct":6,"misses":0,"strict_accuracy":1.0,"miss_rate":0.0,"wilson_95":{"lower":0.609666,"upper":1.0},"provider_coverage":1.0,"badge":null,"promotion_status":"DESCRIPTIVE_ONLY_NO_SUBJECT_LEVEL_POLICY_PROMOTION"}],"frozen_measurement_identity":{"freeze_input_commit":"8450ed889756784a12ca4ebb8d84974d73b31778","final_freeze_receipt_commit":"a309bda7e8dd2867ccc7894dc7d42efbfc48395e","measurement_commit":"b81cc816158f7c0fbe7f8ac0a6fcf2b3f38f4c86","immutable_measurement_receipt_commit":"9d8290160849a5124d1896fc318d53470585ef41","case_freeze_sha256":"c1595743f48fe8ae6b3f70f2ea401731052b3bd7614297feabaa8172ce190ae5","model_lock_sha256":"48b5ed9c98f16cc2b3b76bc0da2e7ec8072ecd89ba1914683181b1c20074a3a8","promotion_policy_sha256":"6714ca504b2c5679c0f1ba0c7da2e5ee52f61b70b0e814ecaf9071f080f9f61d","raw_predictions_sha256":"c5cfe765f6b72489220cc73f689acb1c1c2c9284ffd8d093de0e597b645f6fbf","measurement_artifact_sha256":"7dcab9d4b71e7de59094c07cbca830ebf674b601ac6de024db0d3e677e7c0067","policy_evaluation_sha256":"3ea5b62044b4da6e92812831ebb00205970849accffe763a270248a24a0adef4"},"cross_batch_separation":{"batch1_untouched":true,"batch2_untouched":true,"expansion_v1_untouched":true,"external_gold_v1_untouched":true,"no_combined_headline_score":true},"strict_score_authoritative":true,"adjusted_accuracy_used":false,"target_mcp_business_tools_called":false,"transport_grade_changed":false,"profile_evidence_grade_changed":false,"scanner_production_validity_claimed":false}