{"id":"W2015957719","doi":"10.1002/jrsm.41","title":"Reliability and validity of three quality rating instruments for systematic reviews of observational studies","year":2011,"lang":"en","type":"article","venue":"Research Synthesis Methods","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":172,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Observational study; Intraclass correlation; Reliability (semiconductor); Validity; Inter-rater reliability; Rating scale; Sign (mathematics); Statistics; Psychology; Medicine; Clinical psychology; Psychometrics; Mathematics; Physics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5165881,0.001711414,0.006860164,0.01798309,0.00270034,0.005799064,0.003587581,0.002776418,0.002943161],"category_scores_gemma":[0.7185348,0.001718927,0.01867958,0.01944919,0.003544137,0.005352745,0.006329552,0.003706559,0.0006997872],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008541285,"about_ca_system_score_gemma":0.01767346,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003240143,"about_ca_topic_score_gemma":0.006751957,"domain_scores_codex":[0.3508987,0.3118401,0.2411631,0.01666247,0.07661088,0.002824652],"domain_scores_gemma":[0.2149162,0.4761422,0.1397777,0.03277,0.1340241,0.002369845],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.006290342,0.0006704562,0.2560601,0.2805487,0.07648426,0.0004295537,0.01879415,0.006766801,0.002577683,0.01423864,0.02931739,0.307822],"study_design_scores_gemma":[0.01005702,0.004341216,0.6088508,0.1515237,0.05600429,0.001630674,0.007042869,0.03042144,0.005775251,0.03135238,0.09041157,0.00258888],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1929411,0.1901228,0.3283856,0.01002457,0.005185094,0.2015466,0.04778966,0.002646815,0.02135775],"genre_scores_gemma":[0.4018338,0.01333698,0.3033724,0.001894656,0.0004953363,0.26467,0.01312651,0.0003242242,0.000946049],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4834119,"threshold_uncertainty_score":0.596133,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9900024069594315,"score_gpt":0.7477755849671291,"score_spread":0.2422268219923024,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}