{"id":"W4415428152","doi":"10.3233/faia251197","title":"The Heterogeneous Multi-Agent Challenge","year":2025,"lang":"","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thales (Canada)","funders":"","keywords":"Testbed; Benchmarking; Reinforcement learning; Suite; Homogeneous; Domain (mathematical analysis); Class (philosophy); Field (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00316512,0.0009804378,0.001136952,0.0003913911,0.0009336816,0.001541206,0.001830634,0.001625877,0.003835093],"category_scores_gemma":[0.008248064,0.0002911312,0.0006979672,0.000342306,0.001141525,0.002002509,0.002285414,0.002255185,0.0006996891],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009721353,"about_ca_system_score_gemma":0.001631549,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002907823,"about_ca_topic_score_gemma":0.003390094,"domain_scores_codex":[0.9971943,0.001534872,0.00008626194,0.0004001883,0.0004472574,0.0003371653],"domain_scores_gemma":[0.9955675,0.002539215,0.0003152022,0.0005899815,0.0004497092,0.000538454],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004521496,0.0003280192,0.003513471,0.0004209672,0.0001597961,0.0003995886,0.000159748,0.8470894,0.003202658,0.06963816,0.02079253,0.05384352],"study_design_scores_gemma":[0.0001207015,0.0001703027,0.00100115,0.00005224391,0.00002256526,0.000117387,0.0001458793,0.9301485,0.002545939,0.0499257,0.015714,0.00003554289],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3226215,0.005608414,0.5977567,0.011333,0.001265329,0.0005531593,0.001861374,0.002906864,0.05609374],"genre_scores_gemma":[0.9022019,0.000754492,0.08882923,0.00114643,0.0001499334,0.0003454837,0.001236683,0.0002756683,0.005060318],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003835093,"threshold_uncertainty_score":0.01673901,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06666242016830062,"score_gpt":0.2948962038883428,"score_spread":0.2282337837200421,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}