{"id":"W4399729340","doi":"10.1109/syscon61195.2024.10553598","title":"Deep Reinforcement Learning Agents for Decision Making for Gameplay","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Computer science; Human–computer interaction; Artificial intelligence; Reinforcement; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006176462,0.0001847672,0.0001639035,0.0002026277,0.000259469,0.0007283938,0.0006903261,0.00007705544,0.00009503686],"category_scores_gemma":[0.0004553054,0.0001610872,0.0001780359,0.0002778402,0.00001647864,0.0005421124,0.0002786267,0.0001516399,0.0001133277],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001231437,"about_ca_system_score_gemma":0.00005744354,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000001501775,"about_ca_topic_score_gemma":0.000001129609,"domain_scores_codex":[0.998255,0.00001796137,0.0003974558,0.0004845927,0.0003722378,0.0004727698],"domain_scores_gemma":[0.9981784,0.001185948,0.00007936288,0.0003785321,0.0001068784,0.00007092162],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001255581,0.000003494007,0.00002752341,0.0000818822,0.00002891782,0.000002654395,0.0003618436,0.8421382,0.00003385329,0.08924555,0.003992436,0.06407109],"study_design_scores_gemma":[0.0002267083,0.0002572081,0.00002009743,0.0001960142,0.00001280896,0.000004749575,0.00003237291,0.8685865,0.0001500423,0.00150926,0.1288165,0.0001877475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00007425722,0.0001049657,0.9923337,0.0002079504,0.001312819,0.000763459,1.843274e-7,0.0005442775,0.004658435],"genre_scores_gemma":[0.5422846,0.00001653891,0.444408,0.0003913471,0.0001384733,0.000143632,0.000008754382,0.00003169635,0.01257701],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5479257,"threshold_uncertainty_score":0.7023918,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03155200176925009,"score_gpt":0.3237776472711069,"score_spread":0.2922256455018568,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}