{"id":"W4413533219","doi":"10.20944/preprints202508.1640.v1","title":"Collective Intelligence: On the Promise and Reality of Multi-Agent Systems for AI-Driven Scientific Discovery","year":2025,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Bundesministerium für Bildung und Forschung; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Collective intelligence; Scientific discovery; Computer science; Data science; Cognitive science; Artificial intelligence; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01976328,0.0009522212,0.001173269,0.002227334,0.004331823,0.01549158,0.002680125,0.00520045,0.004076394],"category_scores_gemma":[0.02535144,0.0007160194,0.001074328,0.002249038,0.01629759,0.02072955,0.008965092,0.006261847,0.00132514],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003119502,"about_ca_system_score_gemma":0.004185367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002830462,"about_ca_topic_score_gemma":0.001984896,"domain_scores_codex":[0.9911278,0.004945426,0.0003398577,0.001106412,0.001946271,0.0005341269],"domain_scores_gemma":[0.9787652,0.01391551,0.001096558,0.00347125,0.001599983,0.001151466],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006165051,0.00003977967,0.001141434,0.0003346979,0.00004860696,0.0001338484,0.002304386,0.009053133,0.0007042681,0.9266905,0.004717785,0.05476984],"study_design_scores_gemma":[0.00002278573,0.00005150135,0.0003810952,0.0003087863,0.00002436167,0.00008170459,0.0009685788,0.0256038,0.0004812633,0.9066362,0.06540073,0.00003918458],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02811037,0.02264952,0.7497479,0.1166036,0.00139881,0.0002500201,0.0002277597,0.0009937077,0.08001827],"genre_scores_gemma":[0.6581412,0.01753136,0.3073488,0.005551302,0.001672695,0.000588853,0.0002752838,0.0003295121,0.008560977],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9802367,"threshold_uncertainty_score":0.1045194,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4931222233681156,"score_gpt":0.4770870636988176,"score_spread":0.01603515966929792,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}