{"id":"W4405750564","doi":"10.1016/j.jval.2024.10.2293","title":"MSR59 Evaluating Systematic Literature Review Screening Performance of Seven Large Language Models in Response to Different Prompting Strategies","year":2024,"lang":"en","type":"article","venue":"Value in Health","topic":"Delphi Technique in Research","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Systematic review; Computer science; Psychology; Natural language processing; MEDLINE; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3321285,0.003358511,0.009944575,0.02223714,0.002100997,0.005267352,0.003295016,0.003478705,0.0074136],"category_scores_gemma":[0.6693542,0.002707995,0.02662082,0.01282724,0.002043475,0.006129061,0.007865368,0.001770347,0.0008468836],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007473188,"about_ca_system_score_gemma":0.01805572,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005827353,"about_ca_topic_score_gemma":0.01611518,"domain_scores_codex":[0.6161383,0.2475695,0.09645807,0.008177472,0.02983363,0.001823063],"domain_scores_gemma":[0.2370042,0.6834225,0.03159746,0.01194331,0.03445989,0.001572662],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.04306388,0.001190771,0.04286662,0.4972996,0.1109333,0.0006639138,0.01444562,0.01103513,0.004689628,0.00294602,0.00605907,0.2648065],"study_design_scores_gemma":[0.04037306,0.04944078,0.1032105,0.2161744,0.4674768,0.001019256,0.01651336,0.04770638,0.01262261,0.0134806,0.02998469,0.001997487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6285009,0.1408508,0.06275435,0.00754876,0.001408236,0.1137608,0.0294571,0.001893317,0.01382575],"genre_scores_gemma":[0.7133871,0.01261647,0.2079839,0.001344634,0.0001797273,0.05733818,0.006124339,0.0001877036,0.0008379377],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6678715,"threshold_uncertainty_score":0.8236045,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1894569042152032,"score_gpt":0.5058841504226887,"score_spread":0.3164272462074855,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}