{"id":"W4407637147","doi":"10.2196/67551","title":"Evaluating the Diagnostic Accuracy of ChatGPT-4 Omni and ChatGPT-4 Turbo in Identifying Melanoma: Comparative Study","year":2025,"lang":"en","type":"article","venue":"JMIR Dermatology","topic":"AI in cancer detection","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Melanoma; Computer science; Medicine; Cancer research; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005253907,0.0004839314,0.0007840519,0.00202016,0.0003730994,0.0008514053,0.0006983429,0.001074162,0.002670361],"category_scores_gemma":[0.01795639,0.0002946856,0.0006451359,0.0007470634,0.0006611761,0.0008695762,0.0007486513,0.0004875684,0.0008050734],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005347081,"about_ca_system_score_gemma":0.0004257338,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001608397,"about_ca_topic_score_gemma":0.001843259,"domain_scores_codex":[0.9981059,0.0008909153,0.0001515801,0.0003378722,0.0003759053,0.0001378796],"domain_scores_gemma":[0.9825755,0.01121186,0.001077267,0.0009504864,0.003388899,0.0007959669],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.03945427,0.001220522,0.8222244,0.001109575,0.0008695096,0.0007961557,0.0005872466,0.001491187,0.01377395,0.0002049609,0.001563968,0.1167043],"study_design_scores_gemma":[0.001158635,0.02429097,0.8870886,0.0003703619,0.003535061,0.01215237,0.001452436,0.04195164,0.02297969,0.0007318375,0.004068611,0.0002197941],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.99295,0.002478772,0.0023879,0.00006952132,0.00006802734,0.0002212404,0.0004313089,0.00007831048,0.001314939],"genre_scores_gemma":[0.9935383,0.0005342036,0.004460179,0.00007997385,0.00007009934,0.00009677999,0.0006952375,0.00002443048,0.0005007875],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005253907,"threshold_uncertainty_score":0.02778566,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06754269392955682,"score_gpt":0.4176163842670595,"score_spread":0.3500736903375027,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}