{"id":"W4317467432","doi":"10.1101/2023.01.18.524571","title":"Ensemble of deep learning language models to support the creation of living systematic reviews for the COVID-19 literature","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research; Horizon 2020 Framework Programme; Innosuisse - Schweizerische Agentur für Innovationsförderung; European Commission; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Artificial intelligence; Machine learning; Computer science; Ensemble forecasting; Ranking (information retrieval); Task (project management); Natural language processing; Class (philosophy); Ensemble learning; Systematic review; Data science; Information retrieval; MEDLINE","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02640519,0.001877979,0.002007566,0.006623822,0.0006831442,0.002733302,0.002250883,0.001659927,0.001554277],"category_scores_gemma":[0.06933872,0.000799395,0.003757591,0.002810291,0.0003895987,0.002772105,0.002034394,0.001985269,0.001288059],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001998528,"about_ca_system_score_gemma":0.006215278,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008044721,"about_ca_topic_score_gemma":0.02762159,"domain_scores_codex":[0.9889163,0.0067426,0.001594855,0.001566671,0.0009741853,0.0002053477],"domain_scores_gemma":[0.9485583,0.03871837,0.003841197,0.002614866,0.005545514,0.0007217299],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001919703,0.0006755539,0.04978055,0.01329024,0.008876809,0.0009483261,0.001068866,0.1573285,0.006419004,0.002842631,0.03958213,0.7172676],"study_design_scores_gemma":[0.0004303293,0.0006579785,0.007142331,0.002925611,0.003688556,0.0003814212,0.0002562013,0.9531395,0.004280079,0.01209265,0.01487195,0.0001332911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2423077,0.08052745,0.6002656,0.01412349,0.00209448,0.0040721,0.02786477,0.02222331,0.006521135],"genre_scores_gemma":[0.5264735,0.006875289,0.436736,0.002824895,0.0005488677,0.002512164,0.02116548,0.0004347944,0.002428989],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9735948,"threshold_uncertainty_score":0.1396456,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05301620034303883,"score_gpt":0.2845664423135185,"score_spread":0.2315502419704797,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}