{"id":"W4414132264","doi":"10.1017/rsm.2025.10031","title":"StudyTypeTeller—Large language models to automatically classify research study types for systematic reviews","year":2025,"lang":"en","type":"article","venue":"Research Synthesis Methods","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"Universität Zürich; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Generative grammar; Transformer; Systematic review; Language model; Scientific literature; Encoder","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03251117,0.002455671,0.001622894,0.01121899,0.001107896,0.003551535,0.002919576,0.003008346,0.01062108],"category_scores_gemma":[0.1124124,0.00154404,0.005689278,0.006012614,0.001095128,0.005606305,0.003762526,0.00354038,0.004954876],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003384076,"about_ca_system_score_gemma":0.008725713,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006043514,"about_ca_topic_score_gemma":0.032398,"domain_scores_codex":[0.9831432,0.01052922,0.002662632,0.002240828,0.001213145,0.0002110476],"domain_scores_gemma":[0.8341569,0.1503116,0.005007965,0.006112147,0.003749453,0.0006618756],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002052456,0.000320289,0.01653356,0.05688069,0.003493082,0.001547945,0.003993475,0.05044469,0.01198512,0.02201557,0.2046674,0.6260657],"study_design_scores_gemma":[0.002143829,0.0006836707,0.008775045,0.01008652,0.004104834,0.001796097,0.001284694,0.5431052,0.01374597,0.1250321,0.2887154,0.0005266322],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02829699,0.01936872,0.6722265,0.01287596,0.001360596,0.006488914,0.2015493,0.05158072,0.006252355],"genre_scores_gemma":[0.110829,0.003286168,0.7786722,0.002955796,0.0003219756,0.007191924,0.09346777,0.001582575,0.001692569],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9674888,"threshold_uncertainty_score":0.1719376,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4744790487850936,"score_gpt":0.5983134327456537,"score_spread":0.1238343839605601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}