{"id":"W4409157871","doi":"10.1016/j.jeph.2025.202990","title":"Validation of a generative artificial intelligence tool for the critical appraisal of articles on the epidemiology of mental health: Its application in the Middle East and North Africa","year":2025,"lang":"en","type":"article","venue":"Journal of Epidemiology and Population Health","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Middle East; Critical appraisal; Mental health; Generative grammar; Epidemiology; Psychology; North east; Geography; Artificial intelligence; Medicine; Computer science; Psychiatry; History; Alternative medicine; Ethnology; Pathology; Archaeology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3590449,0.002344208,0.002348474,0.01151041,0.002289037,0.007061498,0.003399882,0.002221626,0.004907103],"category_scores_gemma":[0.6812471,0.00166246,0.00547775,0.005746957,0.003181376,0.005923444,0.007464278,0.00345743,0.001389977],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006697246,"about_ca_system_score_gemma":0.01257779,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002142376,"about_ca_topic_score_gemma":0.005190021,"domain_scores_codex":[0.6921756,0.2348031,0.03990918,0.01149326,0.02017163,0.001447187],"domain_scores_gemma":[0.1387995,0.7601899,0.02957383,0.02820309,0.0416993,0.001534408],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.006038084,0.001868373,0.05208724,0.03312652,0.004585748,0.0007392684,0.06347561,0.03817575,0.006561092,0.01796117,0.03393621,0.7414449],"study_design_scores_gemma":[0.009903891,0.005986845,0.0960886,0.03483107,0.00647566,0.002472082,0.01590747,0.5862679,0.02482161,0.08926124,0.1259069,0.002076643],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1418145,0.002261885,0.7721261,0.004340873,0.001254747,0.05455408,0.003467493,0.01183404,0.008346284],"genre_scores_gemma":[0.1712867,0.0004202253,0.7859703,0.0005644007,0.0001072506,0.03964014,0.001056337,0.0004333797,0.0005212908],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6409551,"threshold_uncertainty_score":0.7904117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5333452892192624,"score_gpt":0.5201651295202311,"score_spread":0.0131801596990313,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}