{"id":"W1990424977","doi":"10.1007/s00180-013-0424-7","title":"Evaluation of confidence intervals for the kappa statistic when the assumption of marginal homogeneity is violated","year":2013,"lang":"en","type":"article","venue":"Computational Statistics","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University; McMaster University; Ontario Clinical Oncology Group","funders":"","keywords":"Homogeneity (statistics); Kappa; Statistic; Confidence interval; Statistics; Mathematics; Cohen's kappa; Econometrics; Test statistic; Statistical hypothesis testing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1793167,0.001752677,0.003406157,0.008491982,0.001487866,0.006270259,0.007349959,0.004897726,0.003851673],"category_scores_gemma":[0.7070058,0.000975482,0.003014733,0.005546239,0.003809483,0.005673708,0.005083294,0.004714029,0.0005437222],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001778,"about_ca_system_score_gemma":0.002641228,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002425736,"about_ca_topic_score_gemma":0.001319545,"domain_scores_codex":[0.8245244,0.1199576,0.01055061,0.01157231,0.03111967,0.002275322],"domain_scores_gemma":[0.1213315,0.8309535,0.01365131,0.01658328,0.01627633,0.001204012],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01056765,0.0007392391,0.07471381,0.006034676,0.006857466,0.001807014,0.008683347,0.1568015,0.006310066,0.2267992,0.01026434,0.4904217],"study_design_scores_gemma":[0.001040688,0.002450743,0.04099452,0.003341505,0.00207528,0.003246106,0.003164034,0.6642146,0.01469791,0.2534912,0.01064678,0.0006367468],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07332394,0.002771979,0.9172407,0.0004792771,0.0001760742,0.0004001933,0.0005425734,0.0009459017,0.004119442],"genre_scores_gemma":[0.6775275,0.0007464086,0.3181504,0.0002121802,0.0001539539,0.001127791,0.00107022,0.0004939915,0.0005175939],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8206833,"threshold_uncertainty_score":0.9483287,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3043425736713799,"score_gpt":0.4321275538967644,"score_spread":0.1277849802253845,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}