{"id":"W3206407893","doi":"10.21203/rs.3.rs-970738/v1","title":"Machine learning algorithms to identify cluster randomized trials from MEDLINE and EMBASE","year":2021,"lang":"en","type":"preprint","venue":"Research Square","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University; Lawson Health Research Institute; McMaster University; London Health Sciences Centre; Ottawa Hospital","funders":"Canadian Institutes of Health Research; Kidney Foundation of Canada; McMaster University; Ottawa Hospital Research Institute","keywords":"MEDLINE; Computer science; Randomized controlled trial; Cluster (spacecraft); Machine learning; Algorithm; Artificial intelligence; Medicine; Internal medicine; Biology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication","open_science","research_integrity"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.0290862,0.0003775185,0.002004714,0.000654604,0.0002950088,0.00201399,0.001570762,0.0003778601,0.0002068079],"category_scores_gemma":[0.02067394,0.0003185058,0.0004522221,0.0004899902,0.0001039994,0.0002601011,0.009410407,0.002617411,0.00004816803],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001615943,"about_ca_system_score_gemma":0.0004481947,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003754761,"about_ca_topic_score_gemma":0.0001482428,"domain_scores_codex":[0.9835371,0.0107566,0.00120647,0.00175498,0.001994175,0.0007506415],"domain_scores_gemma":[0.9878625,0.008921533,0.0002587793,0.001634954,0.0008100747,0.000512113],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01877553,0.0006154894,0.001052706,0.003052205,0.002186209,0.002428551,0.0339447,0.06594622,0.003613994,0.003813265,0.003479772,0.8610914],"study_design_scores_gemma":[0.03058891,0.0000450435,0.0001186772,0.001044007,0.00004044728,0.00000660596,0.0002114521,0.959141,0.0005849628,0.006936038,0.0008808534,0.0004019721],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06233986,0.007866319,0.9188064,0.007083603,0.0008532767,0.002605947,0.00005838341,0.0001893446,0.0001968586],"genre_scores_gemma":[0.5654315,0.002765295,0.4255552,0.0004970045,0.001973501,0.001099156,0.0005025496,0.0001033798,0.002072365],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8931948,"threshold_uncertainty_score":0.9999267,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1562057166206135,"score_gpt":0.4574555004713133,"score_spread":0.3012497838506998,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}