{"id":"W4403063608","doi":"10.2196/58418","title":"Aligning Large Language Models for Enhancing Psychiatric Interviews Through Symptom Delineation and Summarization: Pilot Study","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Mental Health via Writing","field":"Psychology","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; National Research Foundation of Korea; National Research Foundation","keywords":"Preprint; Automatic summarization; Psychology; Computer science; Natural language processing; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005941187,0.001294655,0.000661898,0.0007515971,0.0006551623,0.0009454484,0.001285334,0.001048817,0.003387788],"category_scores_gemma":[0.02012092,0.0004855785,0.00100918,0.0006820562,0.0005001344,0.001795823,0.001747928,0.001866724,0.001502066],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009908406,"about_ca_system_score_gemma":0.001674431,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008016588,"about_ca_topic_score_gemma":0.009271936,"domain_scores_codex":[0.9963757,0.002565379,0.0001625877,0.0005715385,0.0001803965,0.0001442992],"domain_scores_gemma":[0.9860812,0.01149987,0.0003405263,0.0008409488,0.0009664812,0.0002709023],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"nonrandomized_trial","study_design_scores_codex":[0.002852348,0.00250189,0.02904383,0.000928084,0.0004123358,0.001055613,0.006139601,0.1541367,0.03070819,0.001851273,0.008416777,0.7619534],"study_design_scores_gemma":[0.0002352428,0.0009326527,0.007898591,0.00005125722,0.0001843463,0.0001929316,0.001939459,0.9668481,0.01536084,0.002338454,0.0039248,0.0000933089],"study_design_candidate":"nonrandomized_trial","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6741283,0.0006494707,0.3120872,0.0006368156,0.0001562762,0.001037913,0.001738294,0.007583884,0.00198174],"genre_scores_gemma":[0.7540127,0.0001394624,0.2393378,0.0002264387,0.00005888963,0.0006492235,0.004139931,0.000286496,0.001149067],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008016588,"threshold_uncertainty_score":0.03142035,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1496411742518138,"score_gpt":0.5207048682168024,"score_spread":0.3710636939649886,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}