{"id":"W4414028042","doi":"10.1101/2025.09.01.673319","title":"BioML-bench: Evaluation of AI Agents for End-to-End Biomedical ML","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"End-to-end principle; Bench to bedside; Computer science; Artificial intelligence; Medicine; Medical physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01212642,0.00231468,0.0008615552,0.001776407,0.0006719573,0.002138479,0.004254341,0.003200128,0.004807218],"category_scores_gemma":[0.02367215,0.0006113499,0.0009972759,0.001101217,0.001462534,0.002598101,0.002632166,0.003161341,0.003274382],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001978608,"about_ca_system_score_gemma":0.002600672,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00801897,"about_ca_topic_score_gemma":0.007022295,"domain_scores_codex":[0.9918629,0.004147693,0.0007250255,0.001251792,0.001653357,0.0003592104],"domain_scores_gemma":[0.9797084,0.01289128,0.0009277633,0.002106287,0.003002045,0.001364195],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00907118,0.007339342,0.0227141,0.00355621,0.00170726,0.0008062482,0.0009123377,0.514834,0.02329237,0.007586967,0.1290736,0.2791063],"study_design_scores_gemma":[0.0008241512,0.002372061,0.003983072,0.0001078537,0.0001279224,0.0001542199,0.0002282666,0.9484983,0.02293089,0.003121099,0.01756159,0.00009054897],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6418972,0.007177942,0.1281396,0.003710212,0.001722898,0.002172285,0.01223842,0.17305,0.0298914],"genre_scores_gemma":[0.7708191,0.0009301648,0.1855064,0.002052494,0.0002559658,0.000937406,0.02887346,0.002910282,0.007714654],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9878736,"threshold_uncertainty_score":0.06413138,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1302289515264287,"score_gpt":0.3852932150268801,"score_spread":0.2550642635004513,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}