{"id":"W2024037973","doi":"10.1371/journal.pone.0001350","title":"External Validation of a Measurement Tool to Assess Systematic Reviews (AMSTAR)","year":2007,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":579,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Agency for Drugs and Technologies in Health; University of Ottawa","funders":"","keywords":"Medicine; Kappa; Systematic review; Cohen's kappa; MEDLINE; Internal medicine; Statistics; Mathematics; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"evaluation","study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":["metaresearch"],"domain":"methods","study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5408151,0.003456047,0.00914902,0.02972515,0.003022132,0.008108527,0.003791389,0.003675019,0.004335584],"category_scores_gemma":[0.7886616,0.002735333,0.02278152,0.02725554,0.005020809,0.007014642,0.01017186,0.003832307,0.0009626439],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006583778,"about_ca_system_score_gemma":0.01717306,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001231384,"about_ca_topic_score_gemma":0.002709532,"domain_scores_codex":[0.3035307,0.3737895,0.2248311,0.01960347,0.07580946,0.002435794],"domain_scores_gemma":[0.1124374,0.6619375,0.08228593,0.03360073,0.1078593,0.001879244],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"systematic_review","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00831816,0.0006964678,0.2025027,0.3062955,0.1109662,0.0006478017,0.01445455,0.008155869,0.003293579,0.01330933,0.03088336,0.3004766],"study_design_scores_gemma":[0.02310896,0.01223813,0.435351,0.1506124,0.1294485,0.00332798,0.007790558,0.05131283,0.007486664,0.05087281,0.126304,0.002146143],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2049015,0.07357898,0.4364963,0.01134261,0.00730613,0.1980541,0.03199146,0.005009207,0.0313197],"genre_scores_gemma":[0.3668627,0.004993573,0.363723,0.001935795,0.000630626,0.253984,0.006648354,0.0005138956,0.0007081078],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4591849,"threshold_uncertainty_score":0.5662568,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9088568580287464,"score_gpt":0.5118559510449199,"score_spread":0.3970009069838265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}