{"id":"W2902154261","doi":"10.1038/s41467-018-07619-7","title":"Why rankings of biomedical image analysis competitions should be interpreted with care","year":2018,"lang":"en","type":"article","venue":"Nature Communications","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":367,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Institute of Neurological Disorders and Stroke; Engineering and Physical Sciences Research Council; Medical Research Council; Ministry of Science and Technology, Taiwan; National Institute of Biomedical Imaging and Bioengineering; Grantová Agentura České Republiky; Deutsches Krebsforschungszentrum; Deutsche Forschungsgemeinschaft; Australian Research Council; Wellcome Trust; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Institutes of Health; National Science Foundation","keywords":"Ranking (information retrieval); Computer science; Rank (graph theory); Data science; Quality (philosophy); Information retrieval; Image (mathematics); Data mining; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1725964,0.001779349,0.003029478,0.008595834,0.004859373,0.01971735,0.004963858,0.004693905,0.008642953],"category_scores_gemma":[0.5637906,0.0010015,0.001876318,0.005798176,0.008878561,0.01384176,0.005958495,0.006521878,0.006755857],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005635572,"about_ca_system_score_gemma":0.007108797,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01056522,"about_ca_topic_score_gemma":0.01034158,"domain_scores_codex":[0.7872573,0.1140479,0.01441178,0.01890745,0.05987372,0.005501846],"domain_scores_gemma":[0.4702255,0.2569391,0.03731015,0.05198671,0.1748581,0.008680375],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001461101,0.0002911167,0.06468026,0.003026243,0.001904477,0.0003423044,0.005657042,0.005288931,0.004402908,0.07652579,0.4571036,0.3793162],"study_design_scores_gemma":[0.0006235125,0.001190187,0.1503759,0.00528839,0.000944041,0.001621097,0.01685851,0.04665772,0.01270766,0.4604288,0.302096,0.001208133],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.1536312,0.03154315,0.3301682,0.3367801,0.03027101,0.001962309,0.009864707,0.009412159,0.09636726],"genre_scores_gemma":[0.7335756,0.003620504,0.193434,0.03669357,0.008400479,0.001408728,0.005275285,0.006266554,0.0113254],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.8274037,"threshold_uncertainty_score":0.9127877,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.013589213883682,"score_gpt":0.3453327389953813,"score_spread":0.3317435251116993,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}