{"id":"W3179608547","doi":"10.48550/arxiv.2107.04540","title":"Objective task-based evaluation of artificial intelligence-based medical imaging methods: Framework, strategies and role of the physician","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Task (project management); Modalities; Context (archaeology); Artificial intelligence; Medical imaging; Focus (optics); Engineering; Systems engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07118463,0.002308176,0.001466019,0.003349881,0.0009960519,0.006215826,0.001659148,0.003493297,0.001942311],"category_scores_gemma":[0.1332608,0.0003990687,0.001037214,0.001464429,0.00294437,0.002892707,0.003650988,0.00233971,0.0006205991],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00213919,"about_ca_system_score_gemma":0.002831565,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002114286,"about_ca_topic_score_gemma":0.002348767,"domain_scores_codex":[0.9254929,0.05848582,0.003411508,0.002446259,0.009355675,0.0008078656],"domain_scores_gemma":[0.8813385,0.08832147,0.009499614,0.006129103,0.01183407,0.002877165],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.009112666,0.003049952,0.0551215,0.005889286,0.002981066,0.0003261072,0.001620052,0.1840641,0.01330245,0.06267618,0.02000938,0.6418473],"study_design_scores_gemma":[0.002029321,0.009435391,0.05727758,0.002307285,0.00120921,0.0008557013,0.001256677,0.7268491,0.03443607,0.1369561,0.02678062,0.0006070297],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1838907,0.03129742,0.7353876,0.007995806,0.0009635602,0.004071593,0.001699626,0.0011196,0.03357403],"genre_scores_gemma":[0.7881169,0.002474044,0.1997411,0.00152186,0.0004892751,0.002758099,0.00139673,0.0003244179,0.003177624],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9288154,"threshold_uncertainty_score":0.3764648,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05626120817646408,"score_gpt":0.3078303269020632,"score_spread":0.2515691187255992,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}