{"id":"W4210656052","doi":"10.1101/2022.01.15.22269276","title":"SCARF: Auto-Segmentation Clinical Acceptability &amp; Reproducibility Framework for Benchmarking Essential Radiation Therapy Targets in Head and Neck Cancer","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; Princess Margaret Cancer Centre; Ontario Institute for Cancer Research; University of Toronto; University Health Network","funders":"","keywords":"Benchmarking; Segmentation; Computer science; Artificial intelligence; Reproducibility; Generalizability theory; Benchmark (surveying); Medical physics; Machine learning; Medicine; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01954822,0.001699018,0.00093371,0.003580366,0.000988126,0.003659499,0.003470911,0.00225895,0.002920243],"category_scores_gemma":[0.05822144,0.0005611521,0.002522889,0.001030248,0.001505591,0.001797478,0.003664405,0.001985254,0.000954951],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003151985,"about_ca_system_score_gemma":0.005687247,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01651626,"about_ca_topic_score_gemma":0.02069509,"domain_scores_codex":[0.9883183,0.004801017,0.001102342,0.00168239,0.003626296,0.0004697749],"domain_scores_gemma":[0.9811674,0.009777196,0.002080373,0.002093501,0.00431415,0.0005674525],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0009141546,0.0004266389,0.03219539,0.002073495,0.000736517,0.0006856597,0.002079267,0.3701218,0.0106696,0.03054316,0.03893646,0.5106179],"study_design_scores_gemma":[0.0001768965,0.0006741975,0.01298724,0.000827698,0.0002285524,0.0008731105,0.0004719007,0.8924059,0.01817611,0.03565077,0.03728507,0.0002424869],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0412575,0.001728131,0.9109017,0.001260925,0.0001586244,0.001560109,0.004084558,0.03183615,0.007212375],"genre_scores_gemma":[0.3785735,0.0004197054,0.6057193,0.0007177442,0.00007711618,0.001653176,0.008113226,0.003312266,0.001413936],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01954822,"threshold_uncertainty_score":0.1033821,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05033500927863903,"score_gpt":0.4301700932714776,"score_spread":0.3798350839928386,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}