{"id":"W4395044957","doi":"10.2196/50428","title":"Examining Linguistic Differences in Electronic Health Records for Diverse Patients With Diabetes: Natural Language Processing Analysis","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Health Literacy and Information Accessibility","field":"Health Professions","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Rice University","keywords":"Health records; Natural language processing; Computer science; Linguistics; Linguistic analysis; Natural language; Diabetes mellitus; Artificial intelligence; Medicine; Health care","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01088306,0.0003590904,0.0003536587,0.003934572,0.0007969782,0.001820263,0.0006189523,0.0005903505,0.001091407],"category_scores_gemma":[0.0606106,0.0001946882,0.0009224249,0.002455317,0.0007448001,0.001198156,0.001555953,0.0008166482,0.0002860602],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001151183,"about_ca_system_score_gemma":0.001317674,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003655038,"about_ca_topic_score_gemma":0.007377796,"domain_scores_codex":[0.9888749,0.005758305,0.002292855,0.00150024,0.001278345,0.000295391],"domain_scores_gemma":[0.9375638,0.04049651,0.0153939,0.001959633,0.004151963,0.000434145],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007784971,0.000803361,0.8235452,0.001633147,0.0005843564,0.0007862846,0.02314992,0.001838314,0.00901087,0.001698718,0.005798575,0.1303728],"study_design_scores_gemma":[0.0001287559,0.0004237849,0.9230456,0.0004934671,0.0003503408,0.001037379,0.02139562,0.03344999,0.004400025,0.00660362,0.008490374,0.0001811093],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9704043,0.0004281401,0.01937506,0.001411957,0.00005184914,0.001101982,0.00576369,0.00008909957,0.001373985],"genre_scores_gemma":[0.9496126,0.0001899432,0.04189743,0.000624679,0.00008587882,0.00154218,0.00576188,0.00002008429,0.0002652415],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01088306,"threshold_uncertainty_score":0.05755585,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03443305287128406,"score_gpt":0.4213797797736721,"score_spread":0.3869467269023881,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}