{"id":"W4396668108","doi":"10.2196/51526","title":"Assessing the Efficacy of ChatGPT Versus Human Researchers in Identifying Relevant Studies on mHealth Interventions for Improving Medication Adherence in Patients With Ischemic Stroke When Conducting Systematic Reviews: Comparative Analysis","year":2024,"lang":"en","type":"article","venue":"JMIR mhealth and uhealth","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"CINAHL; Psychological intervention; Systematic review; MEDLINE; Medicine; Identification (biology); mHealth; PsycINFO; Nursing","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6333498,0.004556551,0.02848401,0.04345775,0.003990258,0.01532057,0.005991716,0.008309604,0.01335682],"category_scores_gemma":[0.7865618,0.004472962,0.03262088,0.03553711,0.007716686,0.01742034,0.01331545,0.004233913,0.00140451],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01014933,"about_ca_system_score_gemma":0.03397968,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002994874,"about_ca_topic_score_gemma":0.009005922,"domain_scores_codex":[0.1730438,0.5369813,0.2259673,0.01474514,0.04743967,0.001822865],"domain_scores_gemma":[0.1018971,0.7886574,0.05894302,0.02120367,0.02736801,0.001930864],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"systematic_review","study_design_gemma":"observational","study_design_scores_codex":[0.007091621,0.0002722144,0.008906884,0.79588,0.08897017,0.0003135604,0.00610626,0.0005470592,0.0007311055,0.002469755,0.001898916,0.08681251],"study_design_scores_gemma":[0.02070404,0.00598326,0.01644956,0.6301188,0.2844841,0.0006235742,0.004222386,0.003097168,0.002103356,0.01005048,0.02140519,0.0007579608],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.03425861,0.6143113,0.0730978,0.007277508,0.004783718,0.2472176,0.008298301,0.0007968058,0.009958379],"genre_scores_gemma":[0.2128911,0.07294913,0.2062058,0.004241672,0.00128273,0.4991448,0.002053006,0.0002976202,0.0009342016],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3666502,"threshold_uncertainty_score":0.452145,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7461510337994581,"score_gpt":0.6343592616645106,"score_spread":0.1117917721349475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}