{"id":"W6966430195","doi":"10.48448/3eyy-3h68","title":"Learning to Identify Top Elo Ratings with A Dueling Bandits Approach","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Regret; Scheduling (production processes); Maximization; Variety (cybernetics); Sample (material); Convergence (economics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002849093,0.0005954048,0.0005749734,0.002352383,0.001098937,0.0006282232,0.001901786,0.0001568022,0.004889604],"category_scores_gemma":[0.0006207532,0.0005265456,0.00006480762,0.005139945,0.0007757515,0.0003498635,0.001003317,0.001213169,0.002047051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006973734,"about_ca_system_score_gemma":0.00113666,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000415522,"about_ca_topic_score_gemma":0.0001890418,"domain_scores_codex":[0.9936298,0.0001717439,0.0004349913,0.001846564,0.002752625,0.001164332],"domain_scores_gemma":[0.9978647,0.00007608039,0.0005653551,0.0008666768,0.0002011602,0.0004259779],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003323295,0.001323927,0.02071532,0.0008545115,0.0004840876,0.0003605455,0.01762869,0.1851386,0.04808804,0.002640746,0.6675159,0.05491733],"study_design_scores_gemma":[0.00209364,0.00155626,0.0009191417,0.0008631898,0.0002451012,0.0003369339,0.0109541,0.07415114,0.0009100742,0.00006597485,0.904193,0.003711413],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0163458,0.0002294367,0.01091629,0.0001777398,0.000695041,0.002069561,0.00008084872,0.002350017,0.9671353],"genre_scores_gemma":[0.1641321,0.00001443165,0.09974782,0.0005844469,0.0009883395,0.000337289,0.0003878173,0.0027137,0.731094],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.2366772,"threshold_uncertainty_score":0.9997186,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02029795967491655,"score_gpt":0.3085787126627098,"score_spread":0.2882807529877933,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}