{"id":"W4408546444","doi":"10.1186/s12874-025-02528-y","title":"Using artificial intelligence for systematic review: the example of elicit","year":2025,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Centre Intégré Universitaire de Santé et de Services Sociaux du Centre-Sud-de-l'Île-de-Montréal","funders":"","keywords":"MEDLINE; Computer science; Data science; Medicine; Artificial intelligence; Psychology; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7622569,0.004605281,0.01274851,0.04837372,0.005601249,0.02108827,0.005663355,0.008606726,0.00447141],"category_scores_gemma":[0.8637739,0.003137758,0.01057045,0.04089435,0.02037147,0.02309018,0.01813244,0.007600347,0.001377079],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0190635,"about_ca_system_score_gemma":0.03546216,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003350345,"about_ca_topic_score_gemma":0.007045351,"domain_scores_codex":[0.04281973,0.8946677,0.03818399,0.003785076,0.02001479,0.000528666],"domain_scores_gemma":[0.02224233,0.9050408,0.0233671,0.03381722,0.01500624,0.0005262829],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002330987,0.000271319,0.006267857,0.3428608,0.01494725,0.001356484,0.04117997,0.005215613,0.002191427,0.1244055,0.01975,0.4392226],"study_design_scores_gemma":[0.003076566,0.002263759,0.006561156,0.4548228,0.01148179,0.003879472,0.009790061,0.02258982,0.00500853,0.312744,0.1663556,0.001426381],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01557435,0.2513642,0.5911885,0.07176077,0.00455598,0.03980951,0.001937274,0.001583359,0.02222601],"genre_scores_gemma":[0.08055851,0.02884211,0.8519154,0.007403826,0.0008334645,0.02941941,0.0003714482,0.0001843312,0.0004714669],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2377431,"threshold_uncertainty_score":0.2931797,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9149833792834483,"score_gpt":0.7109676948478502,"score_spread":0.2040156844355981,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}