{"id":"W3168124941","doi":"10.48550/arxiv.2106.00922","title":"An Empirical Comparison of Off-policy Prediction Learning Algorithms on the Collision Task","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Backup; Reinforcement learning; Computer science; Algorithm; Bootstrapping (finance); Artificial intelligence; Task (project management); Machine learning; Tree (set theory); Collision; Function (biology); Temporal difference learning; Mathematics; Econometrics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01580174,0.001870633,0.001680455,0.001765723,0.0008361588,0.001218383,0.002342292,0.003104699,0.001011119],"category_scores_gemma":[0.06055029,0.0005248341,0.0007826732,0.001297147,0.001801622,0.003231624,0.002024015,0.003252151,0.0005243351],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001837911,"about_ca_system_score_gemma":0.002193088,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00740293,"about_ca_topic_score_gemma":0.006006274,"domain_scores_codex":[0.9929762,0.003387073,0.0006580864,0.001241578,0.001169002,0.0005680809],"domain_scores_gemma":[0.9164985,0.0691158,0.002075188,0.005480339,0.005249715,0.001580503],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003593012,0.002757233,0.02622644,0.0006976983,0.0003999705,0.0001325197,0.0003509713,0.6360968,0.001793563,0.003379552,0.005471458,0.3191009],"study_design_scores_gemma":[0.0002026323,0.001337388,0.005025506,0.00007762918,0.00005571863,0.00009510055,0.0002041118,0.9866005,0.002464896,0.003014168,0.0008858093,0.00003654047],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8891436,0.004583986,0.09451246,0.001125346,0.0003089268,0.0005310047,0.0004876248,0.002178045,0.007128958],"genre_scores_gemma":[0.9374929,0.0007288251,0.05802603,0.0002935763,0.00005943203,0.0002959793,0.001216013,0.0002723981,0.001614812],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01580174,"threshold_uncertainty_score":0.08356857,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1028594484792466,"score_gpt":0.2634438898198402,"score_spread":0.1605844413405937,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}