{"id":"W4410984146","doi":"10.22541/au.174897564.48312932/v1","title":"The STARR Protocol: An Automated LLM Methodology for Enhanced Systematic Literature Review Screening","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Columbia College","funders":"","keywords":"Protocol (science); Systematic review; Computer science; Medicine; Political science; MEDLINE; Alternative medicine; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1446067,0.004065623,0.00453472,0.01816848,0.003026098,0.007375475,0.004177213,0.003637953,0.1614547],"category_scores_gemma":[0.3099107,0.004626906,0.008537398,0.0125243,0.002681561,0.004843952,0.01069608,0.005034384,0.0373646],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003557952,"about_ca_system_score_gemma":0.04090593,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00280768,"about_ca_topic_score_gemma":0.007965868,"domain_scores_codex":[0.8988085,0.0559898,0.03294991,0.006210188,0.005083942,0.0009575414],"domain_scores_gemma":[0.6562918,0.2480539,0.02267705,0.03544151,0.03451808,0.003017655],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003974449,0.0003738005,0.003298789,0.181261,0.00407201,0.0006946925,0.004796968,0.003925118,0.008939686,0.03554404,0.4168502,0.3362694],"study_design_scores_gemma":[0.01149014,0.0008159649,0.006297539,0.03444853,0.006259583,0.0007712846,0.001530857,0.02695116,0.01520743,0.1003716,0.7944803,0.001375528],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003458275,0.002566326,0.5704427,0.004623957,0.001941131,0.1481419,0.1851381,0.07634482,0.007342935],"genre_scores_gemma":[0.004233188,0.0005053136,0.7727998,0.0005034545,0.0001736237,0.1997551,0.01763074,0.002073937,0.002324822],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8553933,"threshold_uncertainty_score":0.7647626,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07456971998791556,"score_gpt":0.440407428649505,"score_spread":0.3658377086615894,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}