{"id":"W4307233259","doi":"10.1007/s00330-022-09165-9","title":"Algorithmic transparency and interpretability measures improve radiologists’ performance in BI-RADS 4 classification","year":2022,"lang":"en","type":"article","venue":"European Radiology","topic":"AI in cancer detection","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Deutschen Konsortium für Translationale Krebsforschung; Technische Universität München; Deutsche Forschungsgemeinschaft","keywords":"Interpretability; Certainty; Neuroticism; Artificial intelligence; Pearson product-moment correlation coefficient; Medicine; Correlation; Machine learning; Neuroradiology; Personality; Transparency (behavior); Big Five personality traits; Radiology; Computer science; Psychology; Mathematics; Statistics; Social psychology; Neurology; Psychiatry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001615459,0.0001460594,0.0002026166,0.0001639354,0.0001854701,0.00002637921,0.000789887,0.00003541754,0.00001401225],"category_scores_gemma":[0.00005072524,0.0001482037,0.00003770375,0.0003245797,0.0001805189,0.000230185,0.0002411419,0.0004494549,0.000009566165],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002810713,"about_ca_system_score_gemma":0.00005187226,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001868346,"about_ca_topic_score_gemma":0.000008024335,"domain_scores_codex":[0.9971796,0.00138122,0.0003595217,0.0006626357,0.0001366727,0.0002803392],"domain_scores_gemma":[0.999176,0.00007520805,0.0001216342,0.0005512636,0.00002404705,0.00005187375],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001503452,0.0001169272,0.1026155,0.00005795509,0.00003280142,0.00005549119,0.006024464,0.002086089,0.02350485,0.002874461,0.0002388961,0.8622422],"study_design_scores_gemma":[0.0008446583,0.001140404,0.7989748,0.000009680665,0.000007097634,0.0004072307,0.0001315888,0.1871548,0.0004790657,0.0005814457,0.009907002,0.0003622387],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7470968,0.00128637,0.2448111,0.0008291682,0.001705652,0.0003963686,0.000008158964,0.0002594966,0.003606978],"genre_scores_gemma":[0.9974698,0.0001786709,0.001989778,0.00015897,0.00008453603,0.00006568825,0.0000035274,0.00001360307,0.00003540155],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.86188,"threshold_uncertainty_score":0.6043571,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02440863810332993,"score_gpt":0.2311921117057944,"score_spread":0.2067834736024644,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}