diff --git a/control_plane/tests/accuracy_evidence.rs b/control_plane/tests/accuracy_evidence.rs new file mode 100644 index 00000000..2e6dca8e --- /dev/null +++ b/control_plane/tests/accuracy_evidence.rs @@ -0,0 +1,107 @@ +//! Evidence applicability is independent of prices and observed resource use. +#[path = "support/erp_contract.rs"] +mod fixture; +use control_plane::physical::erp::{ErpAccuracyMode, ErpParameterDecision, ErpPlanningInput}; +use planner_types::post_asap::{SketchAlgorithm, SketchParams}; +use serde_json::json; + +fn decision(policy: &ErpPlanningInput) -> ErpParameterDecision { + policy.select( + SketchAlgorithm::Cms, + 0.01, + SketchParams::Cms { + width: 4096, + depth: 5, + }, + ) +} + +/// Wrong distribution/implementation/metric, insufficient trials and excessive +/// error reject empirical admission even when that record is nearly free. +#[test] +fn inapplicable_accuracy_cannot_be_rescued_by_a_low_price() { + assert!(matches!( + decision(&fixture::policy(ErpAccuracyMode::Empirical)), + ErpParameterDecision::Empirical { .. } + )); + for mismatch in [ + "distribution", + "implementation", + "metric", + "trials", + "error", + "parameters", + ] { + let mut policy = fixture::policy(ErpAccuracyMode::Empirical); + let row = &mut policy.artifact.records[0]; + row.resources.update_cpu_seconds = 1e-30; + row.resources.query_cpu_seconds = 1e-30; + match mismatch { + "distribution" => row.distribution = json!({"different_population":true}), + "implementation" => row.implementation = "different-implementation".into(), + "metric" => row.error_metrics.clear(), + "trials" => row.trials = 1, + "error" => { + row.error_metrics.insert("relative_error".into(), 0.2); + } + "parameters" => row.parameters = json!({"rows":0,"cols":0}), + _ => unreachable!(), + } + assert!( + matches!( + decision(&policy), + ErpParameterDecision::ExactFallback { .. } + ), + "{mismatch}" + ); + } +} + +/// Theoretical fallback retains its identity; it is never relabeled empirical. +#[test] +fn missing_evidence_preserves_the_declared_fallback_mode() { + for mode in [ErpAccuracyMode::Hybrid, ErpAccuracyMode::Empirical] { + let mut policy = fixture::policy(mode); + policy.artifact.records.clear(); + match (mode, decision(&policy)) { + (ErpAccuracyMode::Hybrid, ErpParameterDecision::TheoreticalFallback { params, .. }) => { + assert_eq!( + params, + SketchParams::Cms { + width: 4096, + depth: 5 + } + ) + } + (ErpAccuracyMode::Empirical, ErpParameterDecision::ExactFallback { .. }) => {} + unexpected => panic!("wrong evidence/fallback provenance: {unexpected:?}"), + } + } +} + +/// Both parameter choice and error must be traceable to the same admitted row. +#[test] +fn admitted_accuracy_retains_record_identity() { + let policy = fixture::policy(ErpAccuracyMode::Empirical); + let ErpParameterDecision::Empirical { + record_id, + observed_error, + params, + .. + } = decision(&policy) + else { + panic!("matching evidence should be admitted") + }; + assert_eq!(record_id, policy.artifact.records[0].id); + assert_eq!( + observed_error, + policy.artifact.records[0].error_metrics["relative_error"] + ); + assert_eq!( + params, + SketchParams::Cms { + width: 512, + depth: 3 + } + ); +} diff --git a/control_plane/tests/support/erp_contract.rs b/control_plane/tests/support/erp_contract.rs new file mode 100644 index 00000000..704c4bb6 --- /dev/null +++ b/control_plane/tests/support/erp_contract.rs @@ -0,0 +1,44 @@ +//! Synthetic evidence for admission contracts, not a measured calibration run. +use asap_aware_mapping::erp::{ErpArtifact, ErpRecord, ErpResourceProfile, ERP_SCHEMA_VERSION}; +use control_plane::physical::erp::{ErpAccuracyMode, ErpPlanningInput, ErpRuntimeCapabilities}; +use std::collections::BTreeMap; +pub fn policy(mode: ErpAccuracyMode) -> ErpPlanningInput { + ErpPlanningInput { + artifact: ErpArtifact { + schema_version: ERP_SCHEMA_VERSION, + producer_version: "bench-rev".into(), + records: vec![ErpRecord { + id: "cms-512".into(), + sketch: "cms-fastpath-vector2d".into(), + implementation: "oxide".into(), + parameters: serde_json::json!({"rows": 3, "cols": 512}), + distribution: serde_json::json!({"synthetic":{"kind":"zipf","s":1.1}}), + trials: 20, + error_metrics: BTreeMap::from([("relative_error".into(), 0.009)]), + resources: ErpResourceProfile { + memory_bytes: 12_288.0, + update_cpu_seconds: 1e-7, + merge_cpu_seconds: 1e-5, + query_cpu_seconds: 1e-6, + }, + }], + }, + distribution: serde_json::json!({"synthetic":{"kind":"zipf","s":1.1}}), + implementation: Some("oxide".into()), + error_metric: "relative_error".into(), + min_trials: 10, + expected_updates: 1_000.0, + expected_queries: 100.0, + expected_merges: 0.0, + retention_seconds: 60.0, + cpu_weight: 1.0, + byte_second_weight: 1e-9, + mode, + observed_shape: None, + observed_populations: None, + resolved_data_descriptor: None, + observed_shape_source: None, + shape_match: None, + runtime: ErpRuntimeCapabilities::default(), + } +} diff --git a/docs/design_docs/accuracy-evidence-validation.md b/docs/design_docs/accuracy-evidence-validation.md new file mode 100644 index 00000000..03d09ce2 --- /dev/null +++ b/docs/design_docs/accuracy-evidence-validation.md @@ -0,0 +1,14 @@ +# Accuracy evidence validation + +This PR isolates evidence applicability from ranking. The public ERP adapter is +exercised with explicitly synthetic records in `accuracy_evidence.rs`: mismatched +distribution, implementation, metric, parameters, too few trials and excessive +error cannot admit a cheap candidate. Successful decisions retain record identity; +Hybrid theoretical fallback is distinguishable from empirical admission. + +These tests validate the decision contract, not the quality of a real benchmark. +Level 3 needs applicable measured sketch errors, query-specific accuracy checks +against exact results, and provenance for the implementation/data/parameters. +Observed mean error alone cannot establish a requested tail-probability guarantee. +Existing readout-specific and online evidence freshness checks remain in `erp.rs`; +this PR neither replaces those checks nor adds a second semantic matcher.