{"id":"W4391345880","doi":"10.1038/s41598-023-50382-z","title":"Edge roughness quantifies impact of physician variation on training and performance of deep learning auto-segmentation models for the esophagus","year":2024,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Cancer Institute; National Institutes of Health","keywords":"Segmentation; Esophagus; Artificial intelligence; Computer science; Deep learning; Variation (astronomy); Surface roughness; Medicine; Computer vision; Pattern recognition (psychology); Anatomy; Materials science; Composite material","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003670953,0.00110078,0.0007114889,0.0007718211,0.0003550862,0.0007609588,0.0007933714,0.001172925,0.0006065736],"category_scores_gemma":[0.0104668,0.0005080168,0.0007792536,0.0003841553,0.0005700595,0.000692384,0.0007451259,0.0009022682,0.0002009174],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009785239,"about_ca_system_score_gemma":0.0008935984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009073926,"about_ca_topic_score_gemma":0.007386449,"domain_scores_codex":[0.998771,0.0003360233,0.0001240624,0.0004441012,0.0001847087,0.0001401544],"domain_scores_gemma":[0.9941978,0.003688397,0.0004909819,0.0005518874,0.0008739687,0.0001968908],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009513739,0.0002399864,0.05718662,0.0001016781,0.0003033689,0.0001469961,0.0001752038,0.8383617,0.01343794,0.0002334762,0.0008417419,0.08801996],"study_design_scores_gemma":[0.0000122976,0.0002132239,0.005187062,0.00001108571,0.00003425193,0.00003950635,0.00002610122,0.9886621,0.005549041,0.0001506867,0.0001044851,0.00001015336],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9461525,0.0004374647,0.05136935,0.000160752,0.00005546199,0.00005891586,0.0001715908,0.0009795992,0.0006143734],"genre_scores_gemma":[0.987234,0.00004545983,0.01204628,0.00006451779,0.00000794168,0.00001940402,0.0002770263,0.00005609756,0.0002494254],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009073926,"threshold_uncertainty_score":0.01941407,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02435172172744584,"score_gpt":0.3212167419278237,"score_spread":0.2968650202003779,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}