{"id":"W4386735338","doi":"10.1007/978-3-031-43520-1_13","title":"A Relaxed Variant of Distributed Q-Learning Algorithm for Cooperative Matrix Games","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in networks and systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université du Québec à Rimouski","funders":"","keywords":"Algorithm; Computer science; Convergence (economics); Population-based incremental learning; Matrix (chemical analysis); Distributed algorithm; Q-learning; Weighted Majority Algorithm; Function (biology); Distributed learning; Artificial intelligence; Wake-sleep algorithm; Machine learning; Unsupervised learning; Reinforcement learning; Distributed computing; Genetic algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006582132,0.0004021736,0.000847075,0.0001885734,0.0001297738,0.0002279668,0.0004542994,0.0006199948,0.000002964207],"category_scores_gemma":[0.0001991354,0.0003464777,0.0001271903,0.0001982946,0.00007900858,0.00008744645,0.0002392138,0.000800179,0.000003111094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000762246,"about_ca_system_score_gemma":0.00007183641,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000366306,"about_ca_topic_score_gemma":0.00001028337,"domain_scores_codex":[0.9978741,0.00009782355,0.0007495906,0.0005741296,0.0003095682,0.0003948195],"domain_scores_gemma":[0.9971587,0.001553602,0.0005793747,0.0004202499,0.0002131742,0.00007490973],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007072085,0.000003139758,0.00002530773,0.0001369802,0.000104905,0.00003306567,0.0002333634,0.9570495,0.00000175465,0.03412095,0.000193252,0.008090639],"study_design_scores_gemma":[0.000416936,0.0002654612,0.00001031496,0.00107148,0.0000338424,0.00002687173,0.00000674777,0.9906,0.000002443766,0.0009367491,0.006283106,0.0003459768],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.000001333496,0.002439569,0.9943643,0.00006057449,0.001163729,0.0008833242,0.00003499149,0.0001342621,0.000917878],"genre_scores_gemma":[0.5454763,0.008866083,0.1644842,0.0003695321,0.006496724,0.0008723319,0.003562699,0.001153404,0.2687187],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8298802,"threshold_uncertainty_score":0.9998987,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01637866089937328,"score_gpt":0.245987240694854,"score_spread":0.2296085797954807,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}