{"id":"W3198929365","doi":"10.1016/j.cpet.2021.06.013","title":"Objective Task-Based Evaluation of Artificial Intelligence-Based Medical Imaging Methods","year":2021,"lang":"en","type":"review","venue":"PET Clinics","topic":"Medical Imaging Techniques and Applications","field":"Medicine","cited_by":42,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"National Institutes of Health","keywords":"Medicine; Artificial intelligence; Task (project management); Medical physics; Machine learning; Systems engineering; Computer science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007492365,0.001521366,0.002909076,0.002504793,0.0001870233,0.001709728,0.001530251,0.001013763,0.0020906],"category_scores_gemma":[0.01204163,0.000289036,0.001497648,0.001628408,0.0004619598,0.0009557611,0.0007495082,0.0007832727,0.0006052295],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00086394,"about_ca_system_score_gemma":0.001235248,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001968965,"about_ca_topic_score_gemma":0.003209252,"domain_scores_codex":[0.9975249,0.0009062678,0.0003583223,0.000296357,0.0008433283,0.00007081636],"domain_scores_gemma":[0.9917675,0.005721993,0.0009303265,0.0001490456,0.001315057,0.0001159664],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.001259519,0.0002791407,0.001695433,0.03322137,0.002828872,0.00004342746,0.0000579134,0.002138152,0.001639158,0.0007707492,0.006520377,0.9495459],"study_design_scores_gemma":[0.004249333,0.02018985,0.1276518,0.1378528,0.06511082,0.006331609,0.0009563138,0.07201142,0.03482654,0.02033642,0.5094364,0.001046608],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.004698012,0.9888695,0.003414229,0.0002567521,0.0002146208,0.0002146677,0.0004199439,0.00004954729,0.001862782],"genre_scores_gemma":[0.06575436,0.9126018,0.01519575,0.0008015361,0.0008015067,0.0006264537,0.002393562,0.00006231693,0.001762751],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9925076,"threshold_uncertainty_score":0.03962386,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2486077522918173,"score_gpt":0.594007935415539,"score_spread":0.3454001831237217,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}