{"id":"W3001067515","doi":"10.1016/j.ejmp.2020.01.009","title":"Overlooked pitfalls in multi-class machine learning classification in radiation oncology and how to avoid them","year":2020,"lang":"en","type":"article","venue":"Physica Medica","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research","keywords":"Surrogate endpoint; Correlation; Clinical endpoint; Statistical significance; Ordinal Scale; Medicine; Radiation oncology; Statistics; Machine learning; Artificial intelligence; Radiation therapy; Clinical trial; Mathematics; Computer science; Internal medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004905062,0.0001588894,0.0004323593,0.0001268905,0.00004739046,0.00002144926,0.0001078895,0.00009830233,0.00003617008],"category_scores_gemma":[0.002233318,0.0001431249,0.00004153726,0.000396449,0.00009010662,0.0000859986,0.00006356373,0.0009922991,0.00003515569],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001536001,"about_ca_system_score_gemma":0.0001175243,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009667935,"about_ca_topic_score_gemma":0.00001835031,"domain_scores_codex":[0.9985459,0.0002032979,0.0002472158,0.0004070586,0.0003061328,0.0002903215],"domain_scores_gemma":[0.9990854,0.0002538652,0.0001119189,0.0001331399,0.00002818701,0.0003874859],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005636145,0.0008447346,0.2209163,0.0004064164,0.00008462868,0.0002268137,0.02103156,0.0004702137,0.1725749,0.001144678,0.004737816,0.5769984],"study_design_scores_gemma":[0.005682881,0.0004300906,0.1769495,0.0001855952,0.00003971998,0.00001636833,0.0005892932,0.7095606,0.0004303213,0.00006043978,0.1058277,0.0002274672],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7345054,0.0002397457,0.003343554,0.259389,0.0001131625,0.0005548339,0.000002539362,0.00009213963,0.001759565],"genre_scores_gemma":[0.9925804,0.0003129219,0.001567227,0.004997647,0.0002646663,0.00004045491,0.00004951375,0.00002948859,0.0001576587],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7090904,"threshold_uncertainty_score":0.5836465,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0376093825906948,"score_gpt":0.3165520865646233,"score_spread":0.2789427039739285,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}