{"id":"W4416048252","doi":"10.48550/arxiv.2505.22356","title":"Suitability Filter: A Statistical Framework for Classifier Evaluation in Real-World Deployment Settings","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research","keywords":"Classifier (UML); Ground truth; Covariate; Modular design; Filter (signal processing); Software deployment; Statistical hypothesis testing; Test data; Statistical model","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02709227,0.002426325,0.001666856,0.003349818,0.001033234,0.003397962,0.00274947,0.003073208,0.002262212],"category_scores_gemma":[0.09835494,0.0007765682,0.001286583,0.001428,0.002927494,0.004027616,0.004176447,0.003606748,0.001077789],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001824501,"about_ca_system_score_gemma":0.002770578,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002209623,"about_ca_topic_score_gemma":0.00225711,"domain_scores_codex":[0.9851342,0.008042216,0.0008221693,0.001785019,0.003654854,0.0005615288],"domain_scores_gemma":[0.9529247,0.03093115,0.004659938,0.005723344,0.004795075,0.0009657567],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000535558,0.0003078597,0.01867777,0.0003043949,0.0004364645,0.0003009853,0.0003159484,0.7539774,0.008399689,0.0427714,0.009030737,0.1649418],"study_design_scores_gemma":[0.00001709331,0.0001983718,0.001167316,0.00003009921,0.00001741952,0.00007761635,0.00003461512,0.9732656,0.002949001,0.02129133,0.0009200172,0.00003166638],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008390685,0.0001358066,0.9887912,0.0002757218,0.00004028831,0.0001325546,0.0001551959,0.001441401,0.0006370853],"genre_scores_gemma":[0.603839,0.0002983203,0.3910187,0.000593775,0.0002682777,0.0009515641,0.0009707179,0.0007027912,0.001356889],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02709227,"threshold_uncertainty_score":0.1432793,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08535396798824715,"score_gpt":0.3860397162206754,"score_spread":0.3006857482324282,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}