{"id":"W7124304825","doi":"10.65109/vzns8734","title":"A Model-Based Solution to the Offline Multi-Agent Reinforcement Learning Coordination Problem","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Observability; Online and offline; Simple (philosophy); Offline learning; Online algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001105928,0.001067912,0.001397654,0.0003265992,0.0004839157,0.000884809,0.001585173,0.001375993,0.003077361],"category_scores_gemma":[0.003402312,0.0005606496,0.0005710484,0.0002956926,0.001135323,0.0009007672,0.001670455,0.002163377,0.0005816305],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007460752,"about_ca_system_score_gemma":0.002029609,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003739989,"about_ca_topic_score_gemma":0.003426855,"domain_scores_codex":[0.9994393,0.00018867,0.00002151267,0.0001584383,0.0001159991,0.00007598808],"domain_scores_gemma":[0.998534,0.0007886937,0.0001992871,0.0001826308,0.0001547005,0.0001407323],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003005107,0.00003153914,0.0002052464,0.00004010438,0.0000145245,0.00004355848,0.00003385748,0.9757043,0.0006304328,0.009365603,0.0007491882,0.01315167],"study_design_scores_gemma":[0.000009200754,0.00001920616,0.00003063247,0.000003476091,0.000002320571,0.00001008831,0.000006155466,0.9950103,0.0001598307,0.004444549,0.0003018827,0.000002371575],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005716526,0.00005813141,0.991352,0.0001948372,0.00002319492,0.00004178069,0.00003310971,0.0002516483,0.002328788],"genre_scores_gemma":[0.7022082,0.000116468,0.2920786,0.0001952508,0.00005255375,0.0003028797,0.0001460059,0.0001439336,0.004756158],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003739989,"threshold_uncertainty_score":0.0102948,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04538121631456204,"score_gpt":0.2906629308777778,"score_spread":0.2452817145632157,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}