{"id":"W4407873728","doi":"10.7326/annals-24-02189","title":"Development of Prompt Templates for Large Language Model–Driven Screening in Systematic Reviews","year":2025,"lang":"en","type":"article","venue":"Annals of Internal Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":36,"is_retracted":false,"has_abstract":true,"ca_institutions":"South Health Campus; Vector Institute; Health Sciences Centre; Sunnybrook Health Science Centre; University of Toronto; St. Michael's Hospital; University of Calgary","funders":"","keywords":"Medicine; Systematic review; Template; Intensive care medicine; MEDLINE; Medical physics; Programming language; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2505085,0.003548745,0.003967462,0.01296038,0.001261567,0.006017218,0.00421573,0.003097448,0.02090658],"category_scores_gemma":[0.5867974,0.003846511,0.008857727,0.009076936,0.001292378,0.007277018,0.008189754,0.002733114,0.00746983],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006545333,"about_ca_system_score_gemma":0.0246646,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002783126,"about_ca_topic_score_gemma":0.008338818,"domain_scores_codex":[0.8144362,0.1226159,0.04496988,0.006890224,0.009902922,0.001184895],"domain_scores_gemma":[0.3847231,0.4955326,0.04326074,0.03122585,0.04263498,0.002622617],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004905123,0.0004591938,0.0121968,0.105082,0.002798715,0.0006049825,0.007406321,0.0229036,0.005359079,0.01809001,0.1049681,0.7152261],"study_design_scores_gemma":[0.01972965,0.004296544,0.02378851,0.0918744,0.01252559,0.001711808,0.004756037,0.2692279,0.04091411,0.1727823,0.3561668,0.00222647],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01892535,0.005155665,0.7942883,0.00690572,0.000923962,0.0678037,0.03833145,0.0609662,0.006699631],"genre_scores_gemma":[0.02578067,0.0008000564,0.9283049,0.0006903204,0.00009676529,0.0370458,0.005986811,0.0008346121,0.0004601358],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7494916,"threshold_uncertainty_score":0.9242565,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3742693305846784,"score_gpt":0.5288899414395228,"score_spread":0.1546206108548445,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}