{"id":"W4405266245","doi":"10.31219/osf.io/nbm4f","title":"CoordiLang: Assessing Multi-Agent Coordination Skills in Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University; Université de Montréal","funders":"","keywords":"Benchmark (surveying); Computer science; Coordination game; Motor coordination; Comprehension; Inference; Cognitive science; Knowledge management; Artificial intelligence; Psychology; Microeconomics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006127919,0.001579938,0.0006927989,0.001396029,0.0006320709,0.002707537,0.002558867,0.001980791,0.003517127],"category_scores_gemma":[0.03351036,0.0004410072,0.001083969,0.0007487005,0.001140526,0.003545033,0.003256213,0.002192891,0.0009757464],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001433596,"about_ca_system_score_gemma":0.002622423,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01076646,"about_ca_topic_score_gemma":0.01197796,"domain_scores_codex":[0.9950348,0.002905533,0.0003477553,0.0007868133,0.0007137285,0.0002114051],"domain_scores_gemma":[0.9775718,0.01712939,0.001252574,0.002011734,0.001038876,0.0009955998],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001475171,0.00174749,0.02597224,0.001299345,0.0005583233,0.0004178134,0.001534524,0.7454161,0.01119897,0.01917374,0.01389668,0.1773096],"study_design_scores_gemma":[0.0001003145,0.0004330665,0.002162948,0.00004169703,0.00003949694,0.000079466,0.0003369171,0.9805859,0.003429533,0.0100367,0.00270927,0.00004485905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5431966,0.001109467,0.4150001,0.001079262,0.0002672167,0.00110323,0.004479182,0.01552723,0.01823765],"genre_scores_gemma":[0.7997583,0.000157688,0.191939,0.0002075424,0.00002511712,0.0005876928,0.005377781,0.0004419741,0.00150501],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01076646,"threshold_uncertainty_score":0.03240794,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03513912502458995,"score_gpt":0.323754367793417,"score_spread":0.2886152427688271,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}