{"id":"W4376113940","doi":"10.1259/bjr.20220769","title":"Transformer versus traditional natural language processing: how much data is enough for automated radiology report classification?","year":2023,"lang":"en","type":"article","venue":"British Journal of Radiology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"National Institute of Biomedical Imaging and Bioengineering","keywords":"Artificial intelligence; Machine learning; Random forest; Computer science; Deep learning; McNemar's test; Medicine; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009950126,0.001360423,0.001056465,0.001964974,0.0004555232,0.002682599,0.001901707,0.001653154,0.002084665],"category_scores_gemma":[0.0387695,0.0004800137,0.0009224619,0.001693615,0.001562495,0.009776201,0.00199708,0.001641571,0.001916637],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001187827,"about_ca_system_score_gemma":0.001542524,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003974713,"about_ca_topic_score_gemma":0.004786387,"domain_scores_codex":[0.9959656,0.001541215,0.0003275716,0.001024463,0.0009455388,0.0001956036],"domain_scores_gemma":[0.983152,0.01176803,0.001109672,0.001816443,0.001762225,0.0003915612],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.004247985,0.0005925362,0.0395784,0.001252224,0.0003182518,0.0002665339,0.0004973038,0.04183941,0.02151676,0.003020923,0.00941887,0.8774507],"study_design_scores_gemma":[0.0004203331,0.002681496,0.02985424,0.0005104891,0.000516189,0.001192381,0.00178689,0.8522575,0.06026357,0.03603913,0.01432069,0.0001570914],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6638475,0.009540295,0.292914,0.008246718,0.000500529,0.0005308341,0.004125318,0.01186558,0.008429139],"genre_scores_gemma":[0.9280347,0.001600732,0.06453513,0.0007291799,0.0001472685,0.0001195481,0.003677301,0.0001840566,0.0009719771],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009950126,"threshold_uncertainty_score":0.05262196,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3207568730192971,"score_gpt":0.4595928361469281,"score_spread":0.1388359631276311,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}