{"id":"W4403482299","doi":"10.2196/60827","title":"The Comparative Sufficiency of ChatGPT, Google Bard, and Bing AI in Answering Diagnosis, Treatment, and Prognosis Questions About Common Dermatological Diagnoses","year":2024,"lang":"en","type":"article","venue":"JMIR Dermatology","topic":"Social Media in Health Education","field":"Social Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Medical diagnosis; Medicine; Dermatology; World Wide Web; Computer science; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04129815,0.0005317861,0.0006780244,0.003767643,0.0008672978,0.003038396,0.0009424418,0.001744575,0.00420307],"category_scores_gemma":[0.266468,0.0005475675,0.0007775275,0.00120568,0.00139326,0.004336803,0.002374519,0.001760213,0.001220968],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001080992,"about_ca_system_score_gemma":0.001355192,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003079005,"about_ca_topic_score_gemma":0.004790038,"domain_scores_codex":[0.9646797,0.02707852,0.002159916,0.001893639,0.003369933,0.0008182746],"domain_scores_gemma":[0.5229176,0.4423686,0.0122176,0.007221569,0.01087639,0.004398387],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0184626,0.0032751,0.3879154,0.006088698,0.001148387,0.0005001545,0.01267921,0.004432313,0.005499447,0.001686974,0.007497708,0.550814],"study_design_scores_gemma":[0.001647511,0.02177298,0.8107834,0.005380425,0.003517631,0.003692891,0.01828138,0.09069727,0.01300564,0.008150456,0.02252494,0.0005455128],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.970683,0.003353739,0.00558567,0.003055936,0.0002827798,0.0006366428,0.001388855,0.0008098997,0.01420337],"genre_scores_gemma":[0.9895857,0.0007353593,0.006885797,0.0006353701,0.0001318714,0.0002900139,0.0007000622,0.00009389852,0.0009420012],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04129815,"threshold_uncertainty_score":0.2184081,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08202554106877735,"score_gpt":0.445074765340222,"score_spread":0.3630492242714446,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}