{"id":"W3029811621","doi":"10.1186/s12874-020-01031-w","title":"The semi-automation of title and abstract screening: a retrospective exploration of ways to leverage Abstrackr’s relevance predictions in systematic and rapid reviews","year":2020,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":85,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Canadian Institutes of Health Research; Alberta Innovates","keywords":"Workload; Leverage (statistics); Systematic review; Computer science; Relevance (law); Medicine; Automation; MEDLINE; Machine learning; Data mining; Medical physics; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2395059,0.00008476325,0.001585877,0.0002073827,0.00006791197,0.00005209526,0.0004646987,0.00009052685,0.001053929],"category_scores_gemma":[0.6097248,0.00003936898,0.0001549078,0.000939008,0.0001947625,0.0001097109,0.0001249875,0.0002641435,0.0001625901],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001633028,"about_ca_system_score_gemma":0.00009718067,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001632769,"about_ca_topic_score_gemma":0.00009692601,"domain_scores_codex":[0.9559295,0.03273862,0.005832588,0.0006330779,0.004611648,0.0002545456],"domain_scores_gemma":[0.9103417,0.08583903,0.001619087,0.001032702,0.0008658385,0.0003016094],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005071257,0.0002921407,0.01798523,0.06277563,0.0007750763,0.00003378962,0.03037218,0.0006334927,0.006383398,0.1558728,0.2289895,0.4953796],"study_design_scores_gemma":[0.002511481,0.002402985,0.2705332,0.02065564,0.0006161536,0.00006831413,0.01887204,0.3848736,0.001411487,0.1633699,0.1337967,0.0008885562],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008056253,0.01255652,0.9722869,0.00240189,0.00007947088,0.00166542,0.000006439614,0.000003629873,0.002943448],"genre_scores_gemma":[0.8723917,0.007380442,0.1186132,0.0001866663,0.0001478115,0.0003212525,0.000003523257,0.00001539806,0.0009400254],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8643355,"threshold_uncertainty_score":0.9998592,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.951314378397344,"score_gpt":0.6454746183789096,"score_spread":0.3058397600184344,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}