{"id":"W2470750097","doi":"","title":"Enhancing retrieval of best evidence for health care from bibliographic databases: calibration of the hand search of the literature.","year":2001,"lang":"en","type":"article","venue":"PubMed","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University; Hamilton Health Sciences","funders":"","keywords":"Cohen's kappa; Statistic; Information retrieval; Reliability (semiconductor); Computer science; MEDLINE; Health care; Test (biology); Medical education; Database; Medicine; Statistics; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6883293,0.002464194,0.006557555,0.03774231,0.003608851,0.007785964,0.004683338,0.003018969,0.002194203],"category_scores_gemma":[0.907478,0.002733299,0.004469305,0.02037497,0.005249839,0.01359337,0.01257516,0.002880094,0.0008985504],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009159689,"about_ca_system_score_gemma":0.0214805,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003382695,"about_ca_topic_score_gemma":0.006523068,"domain_scores_codex":[0.2192397,0.5681716,0.1422984,0.01217144,0.05655236,0.001566484],"domain_scores_gemma":[0.03695323,0.8111023,0.06555261,0.03664881,0.04846524,0.001277911],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005946592,0.0007221386,0.04841704,0.07319462,0.005231753,0.0002361015,0.02555256,0.003213265,0.006786405,0.00396708,0.008000621,0.8187318],"study_design_scores_gemma":[0.01368791,0.0160264,0.3785789,0.1869423,0.02997992,0.003529121,0.02721867,0.08422091,0.05128494,0.1065944,0.09846916,0.003467456],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2813936,0.07282669,0.5061377,0.01620227,0.002380591,0.09847105,0.002220384,0.003391362,0.01697637],"genre_scores_gemma":[0.3619817,0.007154993,0.5925779,0.001836012,0.0005219351,0.03411905,0.0008893394,0.0002457465,0.0006733999],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3116707,"threshold_uncertainty_score":0.3843455,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.267435042832476,"score_gpt":0.4016326631986881,"score_spread":0.1341976203662121,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}