{"id":"W4308368314","doi":"10.1177/08465371221135760","title":"Assessment of Radiology Artificial Intelligence Software: A Validation and Evaluation Framework","year":2022,"lang":"en","type":"article","venue":"Canadian Association of Radiologists Journal","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":55,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University; Centre Intégré de Santé et de Services Sociaux des Laurentides; Vancouver General Hospital; University of British Columbia; University of Toronto; Dalhousie University; Université Laval; Trillium Health Centre; Université de Montréal; Centre intégré de santé et de services sociaux de Chaudière-Appalaches; Centre Hospitalier de l’Université de Montréal","funders":"Fonds de recherche du Québec","keywords":"Workflow; Medicine; Software; Computer science; Standardization; Software development; Software engineering; Verification and validation; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2484359,0.003081706,0.002527858,0.03071202,0.004346163,0.02460904,0.007938805,0.006124447,0.002476658],"category_scores_gemma":[0.2482969,0.00145798,0.003809351,0.01265588,0.01618998,0.01696069,0.01059349,0.005362391,0.001705932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01772241,"about_ca_system_score_gemma":0.03954566,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01769593,"about_ca_topic_score_gemma":0.01008104,"domain_scores_codex":[0.6875849,0.1768841,0.04535284,0.009470571,0.07617401,0.004533598],"domain_scores_gemma":[0.6318458,0.2073227,0.0281965,0.02177277,0.1053192,0.00554288],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001606519,0.0004669188,0.01100448,0.004388089,0.0002424162,0.0003721497,0.003980162,0.01841866,0.002055901,0.7047942,0.01414156,0.2399749],"study_design_scores_gemma":[0.0002602416,0.001512842,0.01200257,0.02412444,0.0005913196,0.001372856,0.005851682,0.164422,0.01090346,0.5753351,0.2029715,0.0006520344],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009283954,0.0071376,0.9264002,0.01674163,0.0002934977,0.006810512,0.001200462,0.00173421,0.03039789],"genre_scores_gemma":[0.07066852,0.002112772,0.9188789,0.0008448192,0.0001882057,0.004640501,0.001165926,0.0001908626,0.001309461],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7515641,"threshold_uncertainty_score":0.9268123,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1291509090875911,"score_gpt":0.4291231438017092,"score_spread":0.2999722347141182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}