{"id":"W4413139576","doi":"10.1038/s41598-025-15617-1","title":"Revisiting model scaling with a U-net benchmark for 3D medical image segmentation","year":2025,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Key Research and Development Program of China; National Science and Technology Major Project; Shanghai Jiao Tong University","keywords":"Benchmark (surveying); Computer science; Scaling; Artificial intelligence; Segmentation; Image (mathematics); Image segmentation; Net (polyhedron); Pattern recognition (psychology); Cartography; Mathematics; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004288263,0.00211875,0.001174925,0.001499071,0.0009728873,0.002912135,0.003283471,0.002729421,0.003709106],"category_scores_gemma":[0.01593333,0.0008049625,0.001466846,0.00164907,0.0013831,0.003761731,0.002273769,0.002307218,0.001381811],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002388209,"about_ca_system_score_gemma":0.00217225,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01576664,"about_ca_topic_score_gemma":0.02308748,"domain_scores_codex":[0.9987148,0.0004913666,0.00008828524,0.0003246957,0.0002287703,0.0001519339],"domain_scores_gemma":[0.9953784,0.002968225,0.000231061,0.0007179864,0.000495964,0.0002083498],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001354434,0.0003604566,0.009529477,0.001168329,0.000471431,0.0003574807,0.0003214032,0.7631788,0.00620593,0.01424297,0.03294178,0.1698676],"study_design_scores_gemma":[0.00008723918,0.000192369,0.0007995396,0.0000949011,0.00005987295,0.0001168839,0.0001116927,0.9730083,0.005978663,0.01520577,0.004321341,0.00002340111],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6031918,0.01791687,0.299946,0.009087637,0.001526604,0.000619942,0.007752639,0.0275679,0.03239056],"genre_scores_gemma":[0.7738379,0.002195074,0.2069953,0.002096534,0.0002462795,0.0003491692,0.00849831,0.001921594,0.003859734],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01576664,"threshold_uncertainty_score":0.03134978,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008355165634917497,"score_gpt":0.3132938013573507,"score_spread":0.3049386357224332,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}