{"id":"W4404481676","doi":"10.1007/s10278-024-01282-9","title":"RIDGE: Reproducibility, Integrity, Dependability, Generalizability, and Efficiency Assessment of Medical Image Segmentation Models","year":2024,"lang":"en","type":"article","venue":"Journal of Imaging Informatics in Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; University of Calgary","funders":"Lunit; Radiological Society of North America; Gordon and Betty Moore Foundation; National Cancer Institute; National Institutes of Health; National Science Foundation","keywords":"Generalizability theory; Dependability; Checklist; Segmentation; Computer science; Artificial intelligence; Reproducibility; Deep learning; Data mining; Machine learning; Software engineering; Psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4284262,0.001877898,0.002360859,0.009782773,0.004030421,0.009540879,0.004324774,0.004820748,0.004613549],"category_scores_gemma":[0.7324775,0.001388984,0.005663428,0.006885108,0.01103367,0.006942704,0.01048181,0.005994548,0.001710424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00449681,"about_ca_system_score_gemma":0.01534574,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00428818,"about_ca_topic_score_gemma":0.004411931,"domain_scores_codex":[0.5121644,0.2868213,0.07873912,0.01927776,0.09910972,0.003887605],"domain_scores_gemma":[0.1229366,0.5776668,0.06760898,0.09926375,0.1295998,0.002923921],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003408754,0.0006645145,0.1411022,0.01114588,0.004030079,0.0005891597,0.01324826,0.0322869,0.007699506,0.1164254,0.06555534,0.6038441],"study_design_scores_gemma":[0.001543672,0.006162585,0.1903693,0.01994696,0.003930674,0.003159189,0.01010027,0.1891949,0.04255741,0.2968457,0.2341,0.002089387],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05040625,0.003473964,0.9039139,0.007264768,0.0009159928,0.007892637,0.004729559,0.004133082,0.01726991],"genre_scores_gemma":[0.4339187,0.001012902,0.5346493,0.002389572,0.0003830062,0.01918043,0.004126963,0.001898446,0.002440668],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5715737,"threshold_uncertainty_score":0.7048522,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1363594050263097,"score_gpt":0.5029321918665249,"score_spread":0.3665727868402152,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}