{"id":"W3156251590","doi":"10.1088/1361-6560/ac3e0e","title":"Understanding machine learning classifier decisions in automated radiotherapy quality assurance","year":2021,"lang":"en","type":"article","venue":"Physics in Medicine and Biology","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Princess Margaret Cancer Centre; University of Toronto","funders":"Canadian Institutes of Health Research","keywords":"Computer science; Classifier (UML); Artificial intelligence; Machine learning; Quality assurance; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01123104,0.0006092512,0.0006847754,0.001634979,0.0007176669,0.002540769,0.001628747,0.001619931,0.00156165],"category_scores_gemma":[0.04920803,0.0004802288,0.0007851005,0.0009619967,0.00224031,0.00358801,0.00147677,0.002359078,0.0001751619],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003434701,"about_ca_system_score_gemma":0.002312764,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004787016,"about_ca_topic_score_gemma":0.004269635,"domain_scores_codex":[0.9923718,0.004797041,0.0004244762,0.0009887794,0.001132934,0.000285004],"domain_scores_gemma":[0.9565109,0.03368288,0.004084028,0.002630886,0.002675987,0.0004153613],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001899735,0.0001635356,0.01507395,0.0002917112,0.0001671228,0.0001877258,0.001873337,0.6478502,0.003089056,0.1630586,0.001833549,0.1662213],"study_design_scores_gemma":[0.0000178318,0.00004605147,0.001697889,0.00003766024,0.00001935805,0.00002523321,0.0001162715,0.8435726,0.001961888,0.1514575,0.001023274,0.00002433734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08709075,0.0002550809,0.9090232,0.001363367,0.00002287949,0.0001064876,0.0001123766,0.0003789183,0.00164689],"genre_scores_gemma":[0.7534486,0.0001180347,0.2453715,0.000184611,0.00002842019,0.000123942,0.0002014245,0.00005995363,0.0004635431],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01123104,"threshold_uncertainty_score":0.05939621,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5385558467910203,"score_gpt":0.4658012825977877,"score_spread":0.07275456419323262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}