{"id":"W1784946991","doi":"","title":"Empirically Evaluating Multiagent Reinforcement Learning Algorithms","year":2005,"lang":"en","type":"article","venue":"","topic":"Game Theory and Applications","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reinforcement learning; Computer science; Regret; Artificial intelligence; Convergence (economics); Variety (cybernetics); Machine learning; Test (biology); Nash equilibrium; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.003951954,0.00009303335,0.0001279433,0.00009026243,0.0002954753,0.0001475518,0.0004563669,0.00003568397,0.01052548],"category_scores_gemma":[0.001688752,0.00006412436,0.00008015661,0.0003875427,0.00004574098,0.0001900742,0.0001398131,0.0001417888,0.006608425],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003627802,"about_ca_system_score_gemma":0.0000353594,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000008183733,"about_ca_topic_score_gemma":0.000006872764,"domain_scores_codex":[0.9977127,0.000168597,0.000543794,0.0003312977,0.00102624,0.0002173507],"domain_scores_gemma":[0.9983056,0.0008178736,0.0001450388,0.0004010221,0.0002158008,0.0001146486],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007489985,0.00004041533,0.0004142742,4.198062e-7,0.000006503418,4.187677e-7,0.0007907549,0.1585642,0.002305961,0.009771637,0.003043471,0.8250545],"study_design_scores_gemma":[0.0003858301,0.0001138551,0.002139874,0.000004337559,0.000008890887,0.000004470382,0.001215427,0.6892925,0.004473745,0.008862239,0.2933171,0.0001817831],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2676253,0.00003761715,0.5698738,0.005253154,0.00008960169,0.0003426016,8.089067e-7,0.0001679839,0.1566092],"genre_scores_gemma":[0.905879,0.000002581402,0.03363121,0.0008036603,0.0001275665,0.0000318581,0.000002698437,0.000005865111,0.05951556],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8248727,"threshold_uncertainty_score":0.9941651,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2610912538203148,"score_gpt":0.4943785065925657,"score_spread":0.2332872527722508,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}