{"id":"W4414922313","doi":"10.1109/iccv51701.2025.00437","title":"Consensus-Driven Active Model Selection","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Control Systems Optimization","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Model selection; Selection (genetic algorithm); Inference; Benchmark (surveying); Probabilistic logic; Bayesian inference; Active learning (machine learning); Statistical model","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00001425195,0.00006046757,0.00007466381,0.00006162855,0.00002695867,0.000007938046,0.00002950242,0.00004252278,0.00001016657],"category_scores_gemma":[0.00001206334,0.00006291447,0.0000167005,0.0001374257,0.000005657923,0.00005296254,0.000005466183,0.00005220164,0.00001225294],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009886569,"about_ca_system_score_gemma":0.00001341766,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004815412,"about_ca_topic_score_gemma":0.00002450404,"domain_scores_codex":[0.999704,0.000006209422,0.00008984686,0.00007551671,0.00003523534,0.00008917815],"domain_scores_gemma":[0.9998521,0.00002005358,0.000008906528,0.00006268026,0.00004171972,0.00001457957],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004576409,0.000002128302,0.00003449287,0.000008056992,0.0000220924,1.028007e-7,0.000019468,0.9837883,0.008188056,0.005128419,0.0008814448,0.00192293],"study_design_scores_gemma":[0.0002022733,0.000002663992,0.00004787483,0.00000798657,0.000006606353,6.27035e-7,0.00001890663,0.9940934,0.004431275,0.0008135551,0.0003177511,0.00005703804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003793904,0.00002105572,0.9079744,0.00005535561,0.0001075427,0.0001310561,0.000001597412,0.0005084685,0.08740663],"genre_scores_gemma":[0.9788151,0.000004917436,0.01781375,0.00003871442,0.00001489177,0.00002325926,0.000002161254,0.0000100375,0.003277178],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9750212,"threshold_uncertainty_score":0.2565578,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004662843691046648,"score_gpt":0.21524582263793,"score_spread":0.2105829789468834,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}