{"id":"W4406419660","doi":"10.1136/bjo-2024-326254","title":"Can large language models fully automate or partially assist paper selection in systematic reviews?","year":2025,"lang":"en","type":"article","venue":"British Journal of Ophthalmology","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Selection (genetic algorithm); Medicine; Systematic review; Data science; Natural language processing; MEDLINE; Artificial intelligence; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4172865,0.004550164,0.007262994,0.01214805,0.001938168,0.0144938,0.005935383,0.004950211,0.00759711],"category_scores_gemma":[0.7748538,0.00409671,0.01371946,0.01164381,0.003023291,0.01419023,0.01001957,0.004887274,0.003955684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006554523,"about_ca_system_score_gemma":0.02813184,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005235952,"about_ca_topic_score_gemma":0.01668481,"domain_scores_codex":[0.466808,0.4636237,0.042385,0.01365028,0.0123801,0.001153056],"domain_scores_gemma":[0.07717383,0.8694602,0.02497779,0.01937253,0.008304921,0.0007107308],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004322172,0.0003236239,0.02346203,0.1895764,0.01976583,0.0005546422,0.008991379,0.04454195,0.003137507,0.02555691,0.03986329,0.6399043],"study_design_scores_gemma":[0.005105147,0.001340083,0.01222959,0.08148998,0.02277434,0.001010848,0.002351866,0.4194945,0.007753246,0.3394177,0.1055852,0.001447493],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01672279,0.02783654,0.8845201,0.02586038,0.001197125,0.01132687,0.01179875,0.0165187,0.004218833],"genre_scores_gemma":[0.08331615,0.002856319,0.8898895,0.003437871,0.0002846877,0.01633921,0.003034149,0.0005550749,0.0002870002],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5827135,"threshold_uncertainty_score":0.7185895,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3990391644726624,"score_gpt":0.4887466810535566,"score_spread":0.08970751658089415,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}