{"id":"W4391484628","doi":"10.1016/j.jid.2024.01.015","title":"Assessment of Correctness, Content Omission, and Risk of Harm in Large Language Model Responses to Dermatology Continuing Medical Education Questions","year":2024,"lang":"en","type":"letter","venue":"Journal of Investigative Dermatology","topic":"Cutaneous Melanoma Detection and Management","field":"Medicine","cited_by":12,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"National Institutes of Health","keywords":"Specialty; Scopus; Board certification; Medicine; Certificate; Editorial board; Harm; Continuing medical education; Family medicine; Library science; Medical education; Psychology; MEDLINE; Computer science; Continuing education; Political science; Algorithm; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03165717,0.0002744874,0.000519976,0.001369189,0.001414145,0.001832291,0.0009006457,0.004077805,0.004599816],"category_scores_gemma":[0.3627804,0.0003282217,0.0005986822,0.0008726004,0.001454264,0.002310614,0.001955348,0.003585648,0.001471624],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003534756,"about_ca_system_score_gemma":0.00279791,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006105005,"about_ca_topic_score_gemma":0.00536439,"domain_scores_codex":[0.9598532,0.0257079,0.004514113,0.001226486,0.007077883,0.001620464],"domain_scores_gemma":[0.5452928,0.378854,0.02931858,0.009205369,0.03424786,0.003081374],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.004664615,0.001908608,0.7995159,0.0003402006,0.0001414844,0.009662294,0.04117813,0.001355596,0.003912749,0.003975263,0.04745131,0.08589396],"study_design_scores_gemma":[0.0008606569,0.004769871,0.7319088,0.001654336,0.0004443323,0.02941456,0.07902153,0.05446225,0.0107416,0.01912629,0.06713578,0.0004599341],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9128022,0.0001454852,0.003378561,0.06442246,0.0007165632,0.0004439891,0.0006249496,0.0001255724,0.0173402],"genre_scores_gemma":[0.9818951,0.00008202693,0.003803263,0.01160805,0.0002775841,0.0002987398,0.0002036251,0.00005704235,0.001774637],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03165717,"threshold_uncertainty_score":0.1674212,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02330324212452523,"score_gpt":0.3344747654240556,"score_spread":0.3111715232995304,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}