{"id":"W3125451872","doi":"10.15353/jcvis.v6i1.3548","title":"Where do Clinical Language Models Break Down? A Critical Behavioural Exploration of the ClinicalBERT Deep Transformer Model","year":2021,"lang":"en","type":"article","venue":"Journal of Computational Vision and Imaging Systems","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"University of Waterloo","keywords":"Computer science; Transformer; Inference; Artificial intelligence; Language model; Leverage (statistics); Realm; Deep learning; Encoder; Language understanding; Machine learning; Data science; Natural language processing; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007125102,0.001082587,0.0009163181,0.0006932502,0.0007842892,0.003031757,0.001609318,0.001870488,0.004535473],"category_scores_gemma":[0.04163048,0.0007104276,0.0007062138,0.0005401191,0.002073739,0.00650134,0.002852844,0.006116778,0.001638256],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001818874,"about_ca_system_score_gemma":0.002525538,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00816421,"about_ca_topic_score_gemma":0.01311792,"domain_scores_codex":[0.9959763,0.00223747,0.0001420032,0.0008717852,0.0004761257,0.0002963663],"domain_scores_gemma":[0.9869435,0.009624465,0.0005266718,0.001450546,0.000938372,0.0005164592],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002131675,0.0007701751,0.05527911,0.000982892,0.0005176509,0.0012808,0.003896421,0.3029705,0.02202518,0.1383354,0.04558307,0.4262272],"study_design_scores_gemma":[0.0000682085,0.0003158333,0.002657405,0.0001364457,0.00005146812,0.0003344972,0.0006016578,0.8567953,0.005110386,0.1283484,0.005509609,0.00007091182],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4018157,0.004939382,0.5268959,0.03528731,0.0006103718,0.0003624835,0.003468899,0.003886194,0.02273373],"genre_scores_gemma":[0.9173648,0.0009417251,0.07141751,0.002786667,0.0001080049,0.0001656735,0.002280644,0.0005161192,0.004418711],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00816421,"threshold_uncertainty_score":0.03768158,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1765323993807969,"score_gpt":0.4868758544396559,"score_spread":0.310343455058859,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}