{"id":"W4400771582","doi":"10.2196/56243","title":"Extraction of Substance Use Information From Clinical Notes: Generative Pretrained Transformer–Based Investigation","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Mental Health via Writing","field":"Psychology","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; National Institute on Drug Abuse; National Heart, Lung, and Blood Institute","keywords":"Computer science; Preprint; Health care; Set (abstract data type); Informatics; Identification (biology); Health informatics; Information extraction; Artificial intelligence; Data science; Machine learning; Public health; Medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002282131,0.001003777,0.0005337783,0.001250165,0.0003977714,0.001232093,0.001459987,0.001300218,0.002824247],"category_scores_gemma":[0.01458204,0.0003955243,0.001184951,0.0007638261,0.0007408547,0.0015966,0.002006681,0.001773146,0.001173942],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008271274,"about_ca_system_score_gemma":0.001642849,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005223341,"about_ca_topic_score_gemma":0.007294445,"domain_scores_codex":[0.9988295,0.0004739618,0.0001051477,0.0003368372,0.0001587084,0.00009583432],"domain_scores_gemma":[0.9888864,0.009555012,0.0001796848,0.000646567,0.0005898229,0.000142579],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001243378,0.0009672771,0.03649106,0.00098883,0.0001827948,0.00250733,0.001685427,0.1725754,0.02000941,0.006198256,0.008930644,0.7482203],"study_design_scores_gemma":[0.00005716973,0.0003004354,0.002793417,0.00005480663,0.00004888671,0.0005661194,0.0005244566,0.974068,0.01211834,0.006837625,0.002597761,0.00003286885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4030703,0.001112305,0.57655,0.001435362,0.0001630962,0.0007413437,0.002715358,0.0105034,0.003708808],"genre_scores_gemma":[0.7614796,0.00032067,0.2293592,0.0003647197,0.00003333504,0.0003026061,0.005968573,0.0001941365,0.001977224],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005223341,"threshold_uncertainty_score":0.01206917,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.108892163245482,"score_gpt":0.4401655376174755,"score_spread":0.3312733743719936,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}