{"id":"W4384561507","doi":"10.1038/s42256-023-00652-2","title":"Federated benchmarking of medical artificial intelligence with MedPerf","year":2023,"lang":"en","type":"article","venue":"Nature Machine Intelligence","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":149,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"National Institutes of Health; Agence Nationale de la Recherche; National Cancer Institute; National Research Foundation Singapore; National Research Foundation","keywords":"Benchmarking; Software deployment; Computer science; Process (computing); Health care; Cloud computing; Knowledge management; Data science; Big data; Process management; Engineering management; Artificial intelligence; Business; Software engineering; Engineering; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01995781,0.001377686,0.001197836,0.002865744,0.0008941508,0.004149504,0.003384594,0.001841625,0.00380952],"category_scores_gemma":[0.04475361,0.0005685374,0.001125579,0.002500913,0.001928532,0.00567852,0.005448978,0.002420703,0.001617827],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0021412,"about_ca_system_score_gemma":0.003548233,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00303524,"about_ca_topic_score_gemma":0.00219522,"domain_scores_codex":[0.9826063,0.008463682,0.001624343,0.002406093,0.003894815,0.001004921],"domain_scores_gemma":[0.9757663,0.00707761,0.001892657,0.008649427,0.004587214,0.00202687],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005450267,0.003147271,0.05762209,0.001744989,0.001290894,0.0009593843,0.001764925,0.2237154,0.01220086,0.0824521,0.1442582,0.4653937],"study_design_scores_gemma":[0.000428123,0.001539569,0.01851981,0.0003604461,0.0001302232,0.0006100127,0.0004333491,0.8102524,0.02051963,0.06403343,0.08292753,0.0002454123],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2405349,0.002718315,0.5548639,0.008258177,0.002018166,0.001898877,0.01019316,0.141156,0.03835852],"genre_scores_gemma":[0.7213216,0.0005862734,0.2455428,0.002067867,0.0002368634,0.0007443617,0.022358,0.003700418,0.003441684],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01995781,"threshold_uncertainty_score":0.1055483,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08459942258698776,"score_gpt":0.4346746750443853,"score_spread":0.3500752524573975,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}