{"id":"W4322767903","doi":"10.21203/rs.3.rs-2206790/v1","title":"WITHDRAWN: Inter-rater Reliability of Risk of Bias Tools for Non-Randomized Studies","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Inter-rater reliability; Reliability (semiconductor); Reliability engineering; Psychology; Computer science; Engineering; Physics; Developmental psychology; Rating scale","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.169805,0.001571267,0.002981564,0.00573242,0.002866446,0.005388374,0.00482476,0.008670943,0.06121685],"category_scores_gemma":[0.7952799,0.001678954,0.004481378,0.006834676,0.003301956,0.004674411,0.002340628,0.009760062,0.02490741],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003266274,"about_ca_system_score_gemma":0.009535361,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003376189,"about_ca_topic_score_gemma":0.004063934,"domain_scores_codex":[0.7649002,0.1149954,0.06173002,0.008132095,0.04745329,0.002788972],"domain_scores_gemma":[0.207967,0.5018547,0.02285966,0.0677027,0.1958565,0.00375952],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.003704634,0.0001371707,0.004223208,0.00671762,0.001252109,0.0003261158,0.001463111,0.0004638826,0.00112842,0.0209084,0.7415388,0.2181366],"study_design_scores_gemma":[0.005523214,0.001127965,0.05647212,0.01347459,0.002252385,0.002075126,0.0009323184,0.007747479,0.008714651,0.07983676,0.8210278,0.0008156085],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"editorial","genre_gemma":"empirical","genre_scores_codex":[0.02319432,0.02309677,0.2529958,0.2062119,0.2888827,0.02160899,0.08432089,0.01021619,0.08947252],"genre_scores_gemma":[0.2936735,0.00753432,0.3688499,0.08062082,0.03991716,0.05054915,0.03742932,0.01084483,0.1105809],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.830195,"threshold_uncertainty_score":0.8980253,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6070368058631517,"score_gpt":0.5470055814174135,"score_spread":0.06003122444573816,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}