{"id":"W4220964562","doi":"10.1007/s11548-022-02578-3","title":"Quantifying uncertainty in machine learning classifiers for medical imaging","year":2022,"lang":"en","type":"article","venue":"International Journal of Computer Assisted Radiology and Surgery","topic":"COVID-19 diagnosis using AI","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"St. Francis Xavier University; Ontario Institute for Cancer Research; University of Toronto","funders":"","keywords":"Interpretability; Artificial intelligence; Machine learning; Computer science; Metric (unit); Context (archaeology); Medical imaging; Convolutional neural network; Boundary (topology); Test set; Set (abstract data type); Confidence interval; Pattern recognition (psychology); Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002177008,0.0001014011,0.0004296727,0.0006999247,0.0001068254,0.00002145612,0.0001617177,0.00005610879,0.00009519579],"category_scores_gemma":[0.0007704143,0.00009238003,0.0002021496,0.0001187312,0.00009066325,0.00007065816,0.00009674012,0.0006822696,2.636156e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002559206,"about_ca_system_score_gemma":0.0004086415,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003107704,"about_ca_topic_score_gemma":0.00000928035,"domain_scores_codex":[0.9982958,0.0003155618,0.0006214009,0.0001675196,0.000427379,0.0001723203],"domain_scores_gemma":[0.9957544,0.003577825,0.0003127704,0.00005609066,0.0001898386,0.0001090783],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001435895,0.0002671503,0.8304902,0.00004047607,0.0004304203,0.002983615,0.0003238569,0.009027076,0.0001988972,0.0001803108,0.0156431,0.138979],"study_design_scores_gemma":[0.004323611,0.0002513293,0.2939325,0.0004347384,0.00007620321,0.02209324,0.0001838679,0.499922,0.00001615328,0.0001621726,0.1783883,0.0002158443],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8398738,0.002267937,0.0284108,0.1243994,0.004861195,0.0001322068,0.000008961397,0.00002935481,0.00001639523],"genre_scores_gemma":[0.9850817,0.0002015619,0.0009268998,0.01319289,0.0005292087,0.000008528309,0.00003442488,0.00001238956,0.0000124519],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5365577,"threshold_uncertainty_score":0.3767148,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04543149845366259,"score_gpt":0.3438277369703598,"score_spread":0.2983962385166972,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}