{"id":"W4415428152","doi":"10.3233/faia251197","title":"The Heterogeneous Multi-Agent Challenge","year":2025,"lang":"","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thales (Canada)","funders":"","keywords":"Testbed; Benchmarking; Reinforcement learning; Suite; Homogeneous; Domain (mathematical analysis); Class (philosophy); Field (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.000734673,0.0006147641,0.0006038222,0.0003926229,0.001482619,0.0005898111,0.001470249,0.0004822075,0.00003168642],"category_scores_gemma":[0.00003995372,0.000545301,0.0002492802,0.0003319975,0.0005920519,0.0002111839,0.0004700593,0.0006466493,0.0001812096],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002038925,"about_ca_system_score_gemma":0.0001635413,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001045619,"about_ca_topic_score_gemma":0.0005073647,"domain_scores_codex":[0.9958085,0.0001118342,0.001614041,0.001386149,0.0004355215,0.0006439951],"domain_scores_gemma":[0.9973397,0.0002548718,0.0005738679,0.001392336,0.0002271401,0.0002120613],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001142197,0.0001262456,0.00002757582,0.00005560058,0.00006151883,0.000004569614,0.0004754045,0.0006709662,0.00000842971,0.4132787,0.0003132542,0.5849663],"study_design_scores_gemma":[0.00008115059,0.00009622584,0.00003490161,0.0003730391,0.00007725012,0.000007882129,0.0005584437,0.4162562,0.000617923,0.1207245,0.4603258,0.0008467204],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00001012662,0.01466733,0.9722704,0.001912528,0.002396666,0.002709463,0.00004718061,0.00006777293,0.005918496],"genre_scores_gemma":[0.4892566,0.2674229,0.07529839,0.001341298,0.003527788,0.00752381,0.0001926231,0.0003064137,0.1551302],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8969721,"threshold_uncertainty_score":0.9998173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06666242016830062,"score_gpt":0.2948962038883428,"score_spread":0.2282337837200421,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}