{"id":"W4417130675","doi":"10.2139/ssrn.5887028","title":"MAC: Multi-Agent LLM Coder is All You Need","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Correctness; Robustness (evolution); Coding (social sciences); Code (set theory); Inference; Reliability (semiconductor)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","research_integrity"],"consensus_categories":["metaepi_narrow"],"category_scores_codex":[0.005930657,0.001393926,0.001377033,0.0008473594,0.0008738644,0.00135028,0.006878706,0.00102247,0.0002802599],"category_scores_gemma":[0.0001688409,0.001437205,0.001301193,0.0006656104,0.0001475656,0.0006337016,0.003989764,0.01804987,0.0003167375],"about_ca_system_candidate":true,"about_ca_system_consensus":true,"about_ca_system_score_codex":0.006751444,"about_ca_system_score_gemma":0.02423805,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006839313,"about_ca_topic_score_gemma":0.000663847,"domain_scores_codex":[0.9834925,0.0006877329,0.002350891,0.002452613,0.00170001,0.009316259],"domain_scores_gemma":[0.9943047,0.000161984,0.001437282,0.00268875,0.0007679772,0.0006393782],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002050935,0.001755091,0.001184372,0.0003927867,0.005902657,0.0001721137,0.01347373,0.06098776,0.0004173928,0.2501523,0.001894295,0.6634623],"study_design_scores_gemma":[0.00323364,0.0004424056,0.0001309214,0.0006790736,0.0004907522,0.001390926,0.001179795,0.7720098,0.0002987403,0.1850003,0.03328491,0.001858667],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007184532,0.02055457,0.9508032,0.01245387,0.006708019,0.0008640909,0.00002597639,0.0001653041,0.001240394],"genre_scores_gemma":[0.6970166,0.09509341,0.0963461,0.00817044,0.003639331,0.0001131498,0.00002480671,0.000213173,0.09938296],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8544571,"threshold_uncertainty_score":0.9998811,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03289518674378736,"score_gpt":0.2928385054906594,"score_spread":0.259943318746872,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}