{"id":"W4404887887","doi":"10.2196/65454","title":"Classifying Unstructured Text in Electronic Health Records for Mental Health Prediction Models: Large Language Model Evaluation Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of Mental Health","keywords":"Preprint; Health records; Computer science; Mental health; Language model; Mental model; Natural language processing; Electronic health record; Unstructured data; Artificial intelligence; Data mining; Psychology; World Wide Web; Psychiatry; Health care; Cognitive science; Big data","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03203473,0.002313199,0.001196527,0.002580764,0.001024804,0.00221621,0.002293724,0.001869229,0.001267991],"category_scores_gemma":[0.06498588,0.000528627,0.002130407,0.001943763,0.001038495,0.002970318,0.002337996,0.003317691,0.0008030303],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002601269,"about_ca_system_score_gemma":0.001943609,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01422516,"about_ca_topic_score_gemma":0.01151804,"domain_scores_codex":[0.9835463,0.01125968,0.00153048,0.001668416,0.001575936,0.0004192451],"domain_scores_gemma":[0.8815789,0.1013073,0.003505443,0.004778333,0.007252603,0.00157733],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.008836014,0.0198647,0.3436091,0.003008553,0.004911753,0.001563984,0.00390354,0.1858651,0.00573833,0.001805288,0.04462798,0.3762657],"study_design_scores_gemma":[0.0006681831,0.002063839,0.03806074,0.0001942026,0.0005532326,0.0003008089,0.0008925304,0.9513628,0.002680011,0.001207869,0.001912578,0.0001031082],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9720935,0.002106878,0.01707613,0.001151449,0.000240328,0.0008309943,0.004202311,0.0008719736,0.001426375],"genre_scores_gemma":[0.9491624,0.0006311376,0.03048982,0.000485118,0.0001887217,0.0006942905,0.01752733,0.00009418978,0.0007269377],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03203473,"threshold_uncertainty_score":0.1694179,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03635449339094467,"score_gpt":0.3996838041880542,"score_spread":0.3633293107971095,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}