{"id":"W2098954282","doi":"10.1109/icsmc.2011.6084216","title":"Reinforcement learning and the effects of parameter settings in the game of Chung Toi","year":2011,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Reinforcement learning; Artificial neural network; Computer science; Reinforcement; Artificial intelligence; Recurrent neural network; Temporal difference learning; Machine learning; Action (physics); Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007696152,0.00005896886,0.0001154128,0.00003628107,0.00002569505,0.00001995547,0.0004753177,0.00002172311,0.00001347232],"category_scores_gemma":[0.000410755,0.00002878433,0.00003304095,0.0001499401,0.0001930804,0.0001207933,0.0001677619,0.0001014154,0.000003341488],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000004144565,"about_ca_system_score_gemma":0.000008762375,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003073877,"about_ca_topic_score_gemma":0.00001182975,"domain_scores_codex":[0.9992417,0.0001364897,0.0002319785,0.0001115156,0.000159528,0.0001188592],"domain_scores_gemma":[0.9984471,0.001164025,0.0001125028,0.00023501,0.00002974683,0.00001166251],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00009089564,0.00009998,0.01999188,0.0002124386,0.00005212509,0.00001084212,0.2565705,0.0006128598,0.004799246,0.5758557,0.0001887174,0.1415148],"study_design_scores_gemma":[0.0006360017,0.0008366396,0.01161251,0.0002540613,0.00002609733,0.00001290114,0.004508442,0.2711884,0.6675052,0.04238224,0.0007605217,0.0002769377],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7943562,0.0001997522,0.1904588,0.0005597666,0.0001157413,0.0005240212,3.109935e-8,0.00002657218,0.01375911],"genre_scores_gemma":[0.9964516,0.00002446948,0.003109752,0.0002460305,0.000005238991,0.00001201526,3.750016e-8,0.000001977365,0.0001489206],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.662706,"threshold_uncertainty_score":0.1173791,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01934206153522624,"score_gpt":0.2529023497526191,"score_spread":0.2335602882173929,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}