{"id":"W4390935881","doi":"10.2139/ssrn.4673743","title":"Assessing the Accuracy of Generative Conversational Artificial Intelligence in Debunking Sleep Health Myths: Comparative Study with Expert Analysis","year":2024,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"","keywords":"Sleep (system call); Likert scale; Psychology; Misinformation; Sleep medicine; Public health; Categorization; Scale (ratio); Medicine; Applied psychology; Artificial intelligence; Cognition; Computer science; Developmental psychology; Psychiatry; Sleep disorder; Nursing; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02226727,0.0004102781,0.0004747484,0.003444104,0.0009786381,0.002650549,0.001095306,0.001578983,0.002233502],"category_scores_gemma":[0.1867062,0.0003260715,0.0005949558,0.001041911,0.001208599,0.002860707,0.002287677,0.001062847,0.0006065842],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007494829,"about_ca_system_score_gemma":0.0008227861,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001865819,"about_ca_topic_score_gemma":0.002065852,"domain_scores_codex":[0.9790501,0.01567393,0.001395775,0.001321037,0.002069586,0.0004896382],"domain_scores_gemma":[0.7029833,0.2665654,0.00976117,0.00814779,0.01138702,0.001155228],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.005312991,0.002362837,0.4455855,0.001286322,0.000720526,0.000533895,0.1809303,0.006144031,0.008275368,0.006398611,0.002353421,0.3400962],"study_design_scores_gemma":[0.0007638999,0.004011457,0.6972009,0.001449436,0.001326764,0.002272748,0.1150474,0.1321709,0.01250782,0.02146389,0.01114028,0.0006445845],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9833546,0.0002214316,0.007095828,0.0002447715,0.00003129327,0.0002779296,0.0001227085,0.00008751575,0.00856402],"genre_scores_gemma":[0.9927147,0.000137691,0.005899537,0.0001288732,0.00002347591,0.0001638935,0.0001381058,0.00002259854,0.0007711875],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02226727,"threshold_uncertainty_score":0.117762,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08992795111351186,"score_gpt":0.4739074884031422,"score_spread":0.3839795372896304,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}