{"id":"W4402424385","doi":"10.2139/ssrn.4934721","title":"Scarf: Auto-Segmentation Clinical Acceptability &amp; Reproducibility Framework for Benchmarking Essential Radiation Therapy Targets in Head and Neck Cancer","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Princess Margaret Cancer Centre; University of Toronto; University Health Network","funders":"","keywords":"Benchmarking; Head and neck cancer; Reproducibility; Segmentation; Radiation therapy; Medical physics; Head and neck; Medicine; Computer science; Artificial intelligence; Radiology; Business; Surgery; Mathematics; Statistics; Marketing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01937692,0.002023092,0.002291825,0.003620364,0.0009898946,0.003713424,0.004338458,0.003560287,0.004215594],"category_scores_gemma":[0.03633042,0.0009119209,0.002931481,0.001865902,0.001295803,0.001602801,0.003898669,0.002451527,0.001583387],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001462157,"about_ca_system_score_gemma":0.004076841,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009866803,"about_ca_topic_score_gemma":0.01216134,"domain_scores_codex":[0.9883788,0.004766173,0.0008003565,0.002293612,0.003174553,0.0005864726],"domain_scores_gemma":[0.9873927,0.006581694,0.001152926,0.00228223,0.002147186,0.0004432593],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001407332,0.0004862518,0.02353373,0.001040387,0.00138712,0.0004271354,0.0006943269,0.2371574,0.01345984,0.01083741,0.03912705,0.6704419],"study_design_scores_gemma":[0.0001794395,0.0005803098,0.01054286,0.0001184786,0.0002205308,0.0009837634,0.0001647927,0.9414963,0.0180188,0.01532859,0.01225145,0.0001147823],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0328058,0.001081292,0.9345187,0.0004046984,0.0001122424,0.0003900307,0.002686815,0.02607001,0.001930356],"genre_scores_gemma":[0.3776504,0.000228943,0.604013,0.000382717,0.00016104,0.0006334128,0.007371859,0.007082838,0.002475837],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9806231,"threshold_uncertainty_score":0.1024762,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0268282643985124,"score_gpt":0.4120697710673899,"score_spread":0.3852415066688775,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}