{"id":"W4404911180","doi":"10.2196/63188","title":"Comparing the Accuracy of Two Generated Large Language Models in Identifying Health-Related Rumors or Misconceptions and the Applicability in Health Science Popularization: Proof-of-Concept Study","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Misinformation and Its Impacts","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Natural Science Foundation of China","keywords":"Readability; Social media; Psychology; Social psychology; Computer science; Political science; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04191333,0.001962378,0.001748931,0.001744633,0.0007144167,0.002751451,0.00148675,0.002289914,0.002989274],"category_scores_gemma":[0.1900769,0.000703567,0.002483431,0.0007321443,0.001067874,0.003943786,0.002285132,0.002380742,0.0008637013],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001746596,"about_ca_system_score_gemma":0.002463362,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003505905,"about_ca_topic_score_gemma":0.00215482,"domain_scores_codex":[0.9800601,0.0138242,0.001860078,0.002168875,0.00161358,0.0004731513],"domain_scores_gemma":[0.6611996,0.3110796,0.009183774,0.007391657,0.009504449,0.00164089],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.06566191,0.02090255,0.2236882,0.01275472,0.006259781,0.0006693099,0.008512305,0.1180691,0.02105936,0.004576951,0.008821084,0.5090247],"study_design_scores_gemma":[0.007673325,0.02724594,0.07549093,0.001349281,0.006530025,0.000464133,0.002725628,0.8302369,0.0327844,0.009211393,0.005643685,0.0006442955],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9559409,0.001855107,0.03284353,0.0006979084,0.0003875997,0.002785307,0.001610874,0.0009380015,0.002940819],"genre_scores_gemma":[0.9355056,0.0005796254,0.05777008,0.0004326567,0.0001458358,0.002823443,0.001853325,0.0001072298,0.0007820553],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04191333,"threshold_uncertainty_score":0.2216615,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2037425647841682,"score_gpt":0.528724420201297,"score_spread":0.3249818554171288,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}