{"id":"W3168124941","doi":"10.48550/arxiv.2106.00922","title":"An Empirical Comparison of Off-policy Prediction Learning Algorithms on the Collision Task","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Backup; Reinforcement learning; Computer science; Algorithm; Bootstrapping (finance); Artificial intelligence; Task (project management); Machine learning; Tree (set theory); Collision; Function (biology); Temporal difference learning; Mathematics; Econometrics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005362726,0.0002965283,0.0004116247,0.0003202793,0.0003492213,0.0002161515,0.0019203,0.0003336959,0.00001876637],"category_scores_gemma":[0.0002166634,0.0002762739,0.0002003064,0.001011378,0.0001497581,0.0003354969,0.001623211,0.001349056,0.0000224377],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003075332,"about_ca_system_score_gemma":0.000405116,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001047234,"about_ca_topic_score_gemma":0.00000558241,"domain_scores_codex":[0.9974343,0.0006324625,0.0003672694,0.0009006056,0.0003114241,0.0003540005],"domain_scores_gemma":[0.9972128,0.0003462994,0.0005517196,0.001455802,0.0002918075,0.0001416069],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001652364,0.0001220523,0.01138061,0.00002770497,0.00005734092,0.00002676237,0.001254082,0.9758385,0.00004874044,0.00991462,0.0001756882,0.00113733],"study_design_scores_gemma":[0.0002336493,0.0003778896,0.004304321,0.0001430272,0.00003986375,0.000001725137,0.0005644931,0.9928299,0.0003828156,0.0002471162,0.0006497749,0.0002254273],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2724724,0.00001883755,0.7254957,0.0001757915,0.0003641198,0.0002187587,0.000003703708,0.0001594737,0.001091165],"genre_scores_gemma":[0.9972942,0.0001169417,0.001655684,0.00009228462,0.0001217348,9.624212e-7,0.00005102498,0.00001888447,0.0006482915],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7248217,"threshold_uncertainty_score":0.9999689,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1028594484792466,"score_gpt":0.2634438898198402,"score_spread":0.1605844413405937,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}