{"id":"W1540816673","doi":"10.1186/1471-2288-6-33","title":"An alternative to the hand searching gold standard: validating methodological search filters using relative recall","year":2006,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":135,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University; University of Ottawa; Canadian Agency for Drugs and Technologies in Health; University of Alberta; University of Saskatchewan; McGill University; Children's Hospital of Eastern Ontario","funders":"","keywords":"Recall; Gold standard (test); MEDLINE; Systematic review; Information retrieval; Randomized controlled trial; Computer science; Medicine; Medical physics; Psychology; Cognitive psychology; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6951764,0.003551223,0.006552143,0.0297423,0.00347715,0.007649323,0.005897461,0.006936677,0.002836864],"category_scores_gemma":[0.888297,0.002790389,0.009779756,0.02068036,0.006906791,0.01083276,0.00764303,0.003429154,0.0008914784],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00616984,"about_ca_system_score_gemma":0.01134014,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004413374,"about_ca_topic_score_gemma":0.003779465,"domain_scores_codex":[0.2330382,0.535623,0.1590855,0.01776426,0.05296176,0.0015273],"domain_scores_gemma":[0.05325297,0.7831755,0.05218262,0.06937777,0.04138539,0.000625829],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.009938982,0.000629353,0.07799142,0.1018566,0.03075886,0.0007476297,0.02891861,0.01220588,0.01016612,0.07911921,0.01646777,0.6311996],"study_design_scores_gemma":[0.01605464,0.01736319,0.1357292,0.103666,0.06513709,0.004298803,0.00908865,0.08798472,0.08046459,0.323591,0.1520866,0.004535509],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06212611,0.01614429,0.8804021,0.004302489,0.001342415,0.02104328,0.002741432,0.001501773,0.01039601],"genre_scores_gemma":[0.291703,0.001648838,0.6827385,0.002113497,0.0003084559,0.01965404,0.001084708,0.0001991926,0.0005497665],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3048236,"threshold_uncertainty_score":0.3759018,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9791445577134705,"score_gpt":0.7440587898908313,"score_spread":0.2350857678226392,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}