{"id":"W4408546444","doi":"10.1186/s12874-025-02528-y","title":"Using artificial intelligence for systematic review: the example of elicit","year":2025,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Centre Intégré Universitaire de Santé et de Services Sociaux du Centre-Sud-de-l'Île-de-Montréal","funders":"","keywords":"MEDLINE; Computer science; Data science; Medicine; Artificial intelligence; Psychology; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.05109407,0.0001044223,0.0008353471,0.0002794349,0.0002195139,0.000009266055,0.0004242981,0.0002327877,0.0002475175],"category_scores_gemma":[0.3110681,0.00006341675,0.0001685592,0.0009752819,0.0007092064,0.00002520629,0.000112268,0.0005789307,0.00002049859],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001193506,"about_ca_system_score_gemma":0.002987505,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003474035,"about_ca_topic_score_gemma":0.0005388916,"domain_scores_codex":[0.9898877,0.006854987,0.001420156,0.0003360129,0.0009733681,0.0005277434],"domain_scores_gemma":[0.866439,0.1297607,0.0002600455,0.001063182,0.002136379,0.0003405934],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006029104,0.0003009648,0.0002992816,0.5936326,0.000129186,0.000005601683,0.001015846,0.00001219015,0.001695915,0.3257249,0.004156228,0.07242438],"study_design_scores_gemma":[0.0001146717,0.001448207,0.0001014928,0.3067757,0.0009006958,0.0001417528,0.01270666,0.05131485,0.08837067,0.533429,0.004392973,0.0003033682],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006567904,0.01523193,0.9583202,0.01497665,0.0004974084,0.004087559,0.000002347174,0.00001820289,0.0002977786],"genre_scores_gemma":[0.3988778,0.03518693,0.5241148,0.02938032,0.003298734,0.006218837,0.00008710382,0.0001303474,0.002705031],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4342054,"threshold_uncertainty_score":0.9770983,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9149833792834483,"score_gpt":0.7109676948478502,"score_spread":0.2040156844355981,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}