{"id":"W4389615068","doi":"10.1148/radiol.230492","title":"Measuring Interrater Reliability","year":2023,"lang":"en","type":"editorial","venue":"Radiology","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":30,"is_retracted":false,"has_abstract":false,"ca_institutions":"Centre Hospitalier de l’Université de Montréal","funders":"","keywords":"Medicine; Inter-rater reliability; Neuroradiology; Radiology; Scholarship; Medical physics; Nuclear medicine; Psychiatry; Psychology; Rating scale; Neurology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08067399,0.004742318,0.007161469,0.0125573,0.0059306,0.01526783,0.006208111,0.02675047,0.00785357],"category_scores_gemma":[0.3251442,0.003003239,0.002957665,0.005580886,0.01040725,0.007273261,0.004165647,0.02960612,0.008586651],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006536698,"about_ca_system_score_gemma":0.01030154,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004009149,"about_ca_topic_score_gemma":0.01230898,"domain_scores_codex":[0.9142998,0.03698154,0.01118682,0.004474892,0.03160402,0.001452866],"domain_scores_gemma":[0.5899861,0.2522253,0.01212006,0.008540412,0.1293826,0.007745508],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003344781,0.000008199741,0.00006856968,0.0006311594,0.00005901843,0.0000498331,0.0000456946,0.00001746343,0.00003693121,0.0005764033,0.9846681,0.01380525],"study_design_scores_gemma":[0.0002622593,0.00007892655,0.001590114,0.003828337,0.0004831565,0.0006251943,0.0002479316,0.0006194673,0.0002747259,0.008707726,0.9831539,0.0001283807],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"editorial","genre_scores_codex":[0.00003494815,0.0105793,0.0008027032,0.05362589,0.9337773,0.00004885953,0.00006422085,0.00007817841,0.0009885388],"genre_scores_gemma":[0.001128378,0.006348889,0.001160274,0.03587321,0.9514648,0.0001876829,0.00006007878,0.0001047262,0.003672026],"genre_candidate":"editorial","genre_consensus":"editorial","teacher_disagreement_score":0.919326,"threshold_uncertainty_score":0.4266499,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2162369507691027,"score_gpt":0.3877990363725101,"score_spread":0.1715620856034074,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}