{"id":"W2580371235","doi":"10.1177/0146621616684584","title":"An Evaluation of Interrater Reliability Measures on Binary Tasks Using <i>d-Prime</i>","year":2016,"lang":"en","type":"article","venue":"Applied Psychological Measurement","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Inter-rater reliability; Kappa; Prime (order theory); Psychology; Statistics; Reliability (semiconductor); Cohen's kappa; Binary number; Agreement; Psychometrics; Social psychology; Mathematics; Combinatorics; Arithmetic; Rating scale; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2886595,0.001134056,0.001182251,0.004919657,0.002584416,0.002402039,0.001667546,0.001105747,0.001202315],"category_scores_gemma":[0.4241366,0.0009815239,0.00181328,0.004785256,0.003575735,0.0026757,0.003493607,0.001720036,0.0007188705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001606336,"about_ca_system_score_gemma":0.001992717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008587345,"about_ca_topic_score_gemma":0.002157314,"domain_scores_codex":[0.7193463,0.2144912,0.02753669,0.009409014,0.0278793,0.001337506],"domain_scores_gemma":[0.4231398,0.4529221,0.02396384,0.03081702,0.06789482,0.001262418],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005113521,0.0009351372,0.4490764,0.005384285,0.003329968,0.0003702682,0.0527439,0.01153817,0.02583556,0.04490153,0.01013549,0.3906358],"study_design_scores_gemma":[0.0009052373,0.01175264,0.5599464,0.003018348,0.002156993,0.00355986,0.02671421,0.19321,0.08770531,0.07292041,0.03659859,0.001511905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4082651,0.00137811,0.5665505,0.0005003331,0.0004652507,0.003672213,0.0006822516,0.000749335,0.01773684],"genre_scores_gemma":[0.7268816,0.0002713609,0.2677065,0.0001556997,0.00005822937,0.003708395,0.0002984886,0.000152713,0.0007669865],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7113405,"threshold_uncertainty_score":0.8772095,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5204254440661797,"score_gpt":0.4693431790011876,"score_spread":0.05108226506499214,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}