{"id":"W4285061979","doi":"","title":"Graph augmented Deep Reinforcement Learning in the GameRLand3D environment","year":2021,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Reinforcement learning; Graph; Computer science; Artificial intelligence; Reinforcement; Theoretical computer science; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.007024352,0.0004503702,0.0004209247,0.0002948092,0.0004249881,0.00127618,0.003736287,0.0002632267,0.000142846],"category_scores_gemma":[0.0008221903,0.0004155066,0.000270553,0.0006092024,0.0001867411,0.000278095,0.003606528,0.001637392,0.00005413509],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002474577,"about_ca_system_score_gemma":0.0002374916,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005371497,"about_ca_topic_score_gemma":0.0002333939,"domain_scores_codex":[0.9893785,0.007060946,0.0008372315,0.001024408,0.001107661,0.000591322],"domain_scores_gemma":[0.9939266,0.001301227,0.0007169234,0.00341887,0.0004986111,0.0001378003],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004115813,0.000232488,0.002069532,0.0001022382,0.00008334438,0.00003324871,0.02641603,0.9275152,0.0002382483,0.0295103,0.0001630213,0.01363225],"study_design_scores_gemma":[0.0006580378,0.000001641019,0.00392042,0.00110277,0.00003460668,0.00001651476,0.0006193204,0.9809197,0.002439354,0.0006426828,0.009049707,0.0005952943],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006596854,0.0007270515,0.9636363,0.006691156,0.0002272168,0.0005961142,7.517017e-7,0.0002056574,0.02131887],"genre_scores_gemma":[0.9383854,0.001605831,0.05357092,0.0003758547,0.00002096003,0.0001781484,0.0003910396,0.0000403135,0.005431591],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9317885,"threshold_uncertainty_score":0.9998296,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01290767516935561,"score_gpt":0.2150617302733661,"score_spread":0.2021540551040105,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}