{"id":"W4391484628","doi":"10.1016/j.jid.2024.01.015","title":"Assessment of Correctness, Content Omission, and Risk of Harm in Large Language Model Responses to Dermatology Continuing Medical Education Questions","year":2024,"lang":"en","type":"letter","venue":"Journal of Investigative Dermatology","topic":"Cutaneous Melanoma Detection and Management","field":"Medicine","cited_by":12,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"National Institutes of Health","keywords":"Specialty; Scopus; Board certification; Medicine; Certificate; Editorial board; Harm; Continuing medical education; Family medicine; Library science; Medical education; Psychology; MEDLINE; Computer science; Continuing education; Political science; Algorithm; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005707411,0.000251888,0.001273333,0.001261031,0.00002944823,0.00001119211,0.0001550168,0.0005165447,0.00005939345],"category_scores_gemma":[0.003083432,0.0002046761,0.0001590935,0.0002642036,0.0003603942,0.00005336678,0.0001208656,0.001808796,0.000002412659],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001517678,"about_ca_system_score_gemma":0.001473562,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000987785,"about_ca_topic_score_gemma":0.00008286787,"domain_scores_codex":[0.9969088,0.0007076979,0.001403151,0.0002524769,0.0004813577,0.0002465111],"domain_scores_gemma":[0.9973782,0.0004724614,0.001120871,0.0002102319,0.0005701812,0.0002480226],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007973392,0.0009106085,0.04164504,0.004647938,0.001668401,0.01624661,0.01141115,0.00005623591,0.02909613,0.002611679,0.8882813,0.002627548],"study_design_scores_gemma":[0.0181936,0.003565182,0.1583395,0.04792849,0.008492816,0.2626316,0.03031963,0.04844029,0.1296051,0.008416682,0.2816904,0.002376718],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7352571,0.001437093,0.001334561,0.2607208,0.0005938739,0.0004208239,0.00004506327,0.00001278103,0.0001778494],"genre_scores_gemma":[0.9178849,0.0007251021,0.003434239,0.07680535,0.0002166434,0.00003502497,0.0000338651,0.00004241291,0.0008224714],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6065909,"threshold_uncertainty_score":0.834645,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02330324212452523,"score_gpt":0.3344747654240556,"score_spread":0.3111715232995304,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}