{"id":"W4412711493","doi":"10.2196/72524","title":"Performance of Large Language Models in the Cognitive Analysis of Misinformation: Evaluation Study","year":2025,"lang":"en","type":"article","venue":"JMIR Infodemiology","topic":"Misinformation and Its Impacts","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Misinformation; Cognition; Psychology; Computer science; Computer security; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02514443,0.00264914,0.001237968,0.002743397,0.0008937224,0.002736972,0.002466227,0.002113524,0.00163525],"category_scores_gemma":[0.07066248,0.0005365995,0.001259191,0.001432264,0.001146929,0.004057163,0.002723073,0.002119399,0.001065771],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002407671,"about_ca_system_score_gemma":0.002302564,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01101141,"about_ca_topic_score_gemma":0.0125032,"domain_scores_codex":[0.9844098,0.01082187,0.00110862,0.001654268,0.001675594,0.0003297213],"domain_scores_gemma":[0.8886783,0.09097388,0.003314896,0.005826002,0.009023351,0.002183496],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01439598,0.01031898,0.1246846,0.006584572,0.002259172,0.001131916,0.01037394,0.159453,0.02257316,0.003243872,0.01983394,0.6251467],"study_design_scores_gemma":[0.0006130844,0.003511833,0.0261976,0.000353997,0.0006713634,0.0003324133,0.002029288,0.9451157,0.01317301,0.002560886,0.005213461,0.0002273641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9303978,0.002759611,0.05137365,0.0007259199,0.0002258721,0.001311825,0.00178351,0.006546877,0.004875077],"genre_scores_gemma":[0.9265651,0.0005274967,0.06623393,0.0002387254,0.0000892169,0.000742505,0.004187886,0.0002452411,0.001169944],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02514443,"threshold_uncertainty_score":0.132978,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05879326106579951,"score_gpt":0.4525299982837979,"score_spread":0.3937367372179984,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}