{"id":"W4403046431","doi":"10.2196/64844","title":"Comparative Analysis of Diagnostic Performance: Differential Diagnosis Lists by LLaMA3 Versus LLaMA2 for Case Reports","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Dokkyo Medical University","keywords":"Preprint; Differential (mechanical device); Computer science; World Wide Web; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02093039,0.001449303,0.001405493,0.01016912,0.0008212821,0.002775069,0.0013977,0.0013596,0.004752407],"category_scores_gemma":[0.1059942,0.0004252687,0.002947662,0.003594878,0.0007934085,0.002915066,0.003231311,0.0009608585,0.00207015],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001316482,"about_ca_system_score_gemma":0.001311724,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001500735,"about_ca_topic_score_gemma":0.001861603,"domain_scores_codex":[0.9830945,0.007380581,0.003315342,0.002335412,0.003193092,0.0006811447],"domain_scores_gemma":[0.8420398,0.1244482,0.0105328,0.007485243,0.01230158,0.00319237],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01146342,0.0006643122,0.7366174,0.002062429,0.002030332,0.001095303,0.001961936,0.01846331,0.006340867,0.0008478611,0.009941665,0.2085111],"study_design_scores_gemma":[0.001089312,0.004840916,0.6401191,0.001031405,0.002583827,0.008190292,0.004758914,0.2863689,0.03080079,0.004856794,0.01484785,0.0005119045],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9649778,0.003073677,0.01840473,0.0006037509,0.0002112686,0.0004960947,0.005631471,0.00228276,0.004318422],"genre_scores_gemma":[0.9560645,0.0006226004,0.03178485,0.0001332464,0.0001752737,0.0002716339,0.01012479,0.0002689436,0.000554076],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9790696,"threshold_uncertainty_score":0.1106918,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.298722092382132,"score_gpt":0.5647689784504003,"score_spread":0.2660468860682683,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}