{"id":"W4410223276","doi":"10.2196/63267","title":"Transformer-Based Language Models for Group Randomized Trial Classification in Biomedical Literature: Model Development and Validation","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generalizability theory; Randomized controlled trial; Medical physics; Clinical trial; Computer science; External validity; Medicine; Artificial intelligence; Research design; Public health; Data mining; Machine learning; Natural language processing; Psychology; Pathology; Statistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02135435,0.001436762,0.001587903,0.003205814,0.000505254,0.002116496,0.002677393,0.002330587,0.002675002],"category_scores_gemma":[0.06168493,0.0006633644,0.002678026,0.001595715,0.001050548,0.002261211,0.001741477,0.003888209,0.001686816],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002405616,"about_ca_system_score_gemma":0.003399259,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008483719,"about_ca_topic_score_gemma":0.0120698,"domain_scores_codex":[0.9941856,0.003534807,0.0005305511,0.001053588,0.0004741347,0.0002212282],"domain_scores_gemma":[0.9487739,0.04417803,0.002240711,0.001786096,0.002605441,0.0004159043],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001157928,0.0005789489,0.01824865,0.0009535872,0.0007141037,0.0003628024,0.0005760795,0.6773943,0.002773463,0.01391354,0.008960473,0.2743661],"study_design_scores_gemma":[0.0000784124,0.00006826512,0.0006333314,0.00007318625,0.00006633875,0.00004674736,0.00002693525,0.9834539,0.0005744019,0.01428371,0.0006781922,0.00001666003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06265454,0.003283205,0.9218467,0.002633249,0.0002330483,0.0008296397,0.003037884,0.003914399,0.001567379],"genre_scores_gemma":[0.5802025,0.001184425,0.4068923,0.001999954,0.0002126391,0.001980915,0.005141789,0.0002251728,0.002160233],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9786456,"threshold_uncertainty_score":0.112934,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3751806373914637,"score_gpt":0.4896650803847516,"score_spread":0.114484442993288,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}