{"id":"W4403888184","doi":"10.2196/52897","title":"Multifaceted Natural Language Processing Task–Based Evaluation of Bidirectional Encoder Representations From Transformers Models for Bilingual (Korean and English) Clinical Notes: Algorithm Development and Validation","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Security token; Encoder; Transformer; Natural language processing; Artificial intelligence; Language model; Context (archaeology); Machine learning; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002034166,0.0001181019,0.0001803473,0.0001547405,0.000116861,0.0001194357,0.0001466992,0.0001477896,0.000008311375],"category_scores_gemma":[0.001056342,0.00009872855,0.00004157011,0.0002429645,0.00009390789,0.0006713819,0.00004582334,0.0003256527,6.552273e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005610207,"about_ca_system_score_gemma":0.001134673,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004391621,"about_ca_topic_score_gemma":0.00002156816,"domain_scores_codex":[0.9977441,0.0001267165,0.0008022621,0.0002073974,0.0009641845,0.0001553334],"domain_scores_gemma":[0.9981449,0.00104909,0.0001438895,0.0001197092,0.0003989793,0.000143451],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006915423,0.00002776869,0.0002019003,0.0003125748,0.00002158566,5.873246e-7,0.05511015,0.000951646,0.000006341438,0.0000846079,0.00001763021,0.9432583],"study_design_scores_gemma":[0.0007605316,0.00004118332,0.0005715662,0.0002613681,0.00002479492,0.000003054674,0.001869788,0.9955707,0.0003983807,0.0002380651,0.000150138,0.0001103931],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2969489,0.0004213787,0.7013307,0.0002885431,0.0003051374,0.0005234093,0.00002139816,0.0001206978,0.00003990423],"genre_scores_gemma":[0.6836689,0.000009092302,0.3158306,0.0001107311,0.00008892929,0.00007223814,0.00020912,0.000007150533,0.000003195696],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9946191,"threshold_uncertainty_score":0.4026034,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05428755884584214,"score_gpt":0.4284199307430597,"score_spread":0.3741323718972175,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}