{"id":"W4392238058","doi":"10.7717/peerj-cs.1888","title":"exKidneyBERT: a language model for kidney transplant pathology reports and the crucial role of extended vocabularies","year":2024,"lang":"en","type":"article","venue":"PeerJ Computer Science","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Vocabulary; Natural language processing; Artificial intelligence; Information extraction; Information retrieval; Language model; Domain (mathematical analysis); Field (mathematics); Pathology; Medicine; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001708199,0.001155295,0.0005559561,0.001359009,0.0003840123,0.00129988,0.002016729,0.0008607477,0.003551358],"category_scores_gemma":[0.00426623,0.0005255826,0.001071168,0.0006028484,0.000347995,0.002736073,0.001061385,0.001788864,0.001703366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001954227,"about_ca_system_score_gemma":0.001841439,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02672049,"about_ca_topic_score_gemma":0.02639147,"domain_scores_codex":[0.9994532,0.0001634584,0.00005899603,0.0001940415,0.00007840811,0.00005192383],"domain_scores_gemma":[0.9983858,0.001051307,0.0001271862,0.0001152048,0.0002575355,0.00006291409],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001680921,0.0004345442,0.009794366,0.0007865668,0.0004210229,0.0009963886,0.000830232,0.3801797,0.020586,0.0108462,0.03455522,0.5388888],"study_design_scores_gemma":[0.00005472688,0.00008809828,0.001019516,0.00005387796,0.00007522027,0.0001541729,0.0000709555,0.9854299,0.005022272,0.003194845,0.004796907,0.00003953942],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2736341,0.003383613,0.6704863,0.002688802,0.0007720491,0.0008604128,0.01323892,0.02532367,0.009612148],"genre_scores_gemma":[0.7788006,0.0008898631,0.1908268,0.0006839468,0.0001405636,0.000637739,0.01612287,0.0005125183,0.01138509],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02672049,"threshold_uncertainty_score":0.05312991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007218420904766878,"score_gpt":0.2577834305421606,"score_spread":0.2505650096373938,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}