{"id":"W2152177373","doi":"10.2106/jbjs.h.01624","title":"Evaluating Agreement: Conducting a Reliability Study","year":2009,"lang":"en","type":"article","venue":"Journal of Bone and Joint Surgery","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":162,"is_retracted":false,"has_abstract":true,"ca_institutions":"Sunnybrook Health Science Centre; University of Toronto; McMaster University","funders":"","keywords":"Reliability (semiconductor); Sample size determination; Computer science; Reliability engineering; Context (archaeology); Test (biology); Psychology; Statistics; Engineering; Mathematics; Power (physics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4181999,0.001163982,0.003235349,0.009726854,0.002329537,0.00524967,0.002200436,0.001824558,0.002311321],"category_scores_gemma":[0.6037973,0.001328217,0.003497946,0.007982668,0.004677529,0.006516665,0.004764488,0.002814936,0.00104392],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002459252,"about_ca_system_score_gemma":0.007148245,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001443193,"about_ca_topic_score_gemma":0.001638001,"domain_scores_codex":[0.5381239,0.3333645,0.05585367,0.01224225,0.05785561,0.002560002],"domain_scores_gemma":[0.235508,0.5752224,0.03693643,0.03622833,0.113873,0.002231798],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.003220085,0.0013213,0.270816,0.01276703,0.00541332,0.0006365091,0.06580012,0.005184701,0.008110365,0.03327017,0.01366263,0.5797978],"study_design_scores_gemma":[0.001380115,0.01523935,0.4840362,0.01513119,0.007364997,0.004174836,0.09179558,0.06823237,0.03892202,0.1625184,0.1092386,0.001966278],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2193921,0.004628196,0.7206406,0.002710921,0.001274124,0.02191811,0.001078803,0.0009971549,0.02735995],"genre_scores_gemma":[0.4933471,0.001227017,0.4901057,0.0004936762,0.0002351409,0.01298414,0.000406032,0.00027841,0.0009227138],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.5818001,"threshold_uncertainty_score":0.7174631,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5353733677564685,"score_gpt":0.4567027472231148,"score_spread":0.07867062053335366,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}