{"id":"W1976786126","doi":"10.1002/sim.911","title":"A general goodness‐of‐fit approach for inference procedures concerning the kappa statistic","year":2001,"lang":"en","type":"article","venue":"Statistics in Medicine","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Goodness of fit; Statistics; Polytomous Rasch model; Statistic; Cohen's kappa; Inference; Mathematics; Statistical inference; Test statistic; Multinomial distribution; Dirichlet distribution; Outcome (game theory); Kappa; Range (aeronautics); Confidence interval; Econometrics; Statistical hypothesis testing; Computer science; Item response theory; Psychometrics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1990702,0.003046685,0.004806982,0.01446625,0.003031712,0.004993923,0.007185344,0.006524575,0.007205332],"category_scores_gemma":[0.5594519,0.00210546,0.006545209,0.00837079,0.009444383,0.007726405,0.008252952,0.008232064,0.002676801],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002717419,"about_ca_system_score_gemma":0.004752644,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002896447,"about_ca_topic_score_gemma":0.002263235,"domain_scores_codex":[0.7298037,0.2197848,0.009939073,0.01420662,0.02451747,0.001748266],"domain_scores_gemma":[0.5392739,0.3977384,0.01441923,0.02840788,0.01918815,0.000972489],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0005198887,0.0002533474,0.01334338,0.002797485,0.002969351,0.0007033141,0.004285059,0.07976567,0.002831388,0.466375,0.01254229,0.4136139],"study_design_scores_gemma":[0.0002205897,0.0007043714,0.006099482,0.0009915559,0.0004681251,0.001165691,0.000807247,0.1980703,0.002249573,0.7758346,0.01304426,0.0003441496],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001445146,0.0001957294,0.9966989,0.0002042076,0.00005303442,0.0003099452,0.00006888304,0.0002227662,0.0008012805],"genre_scores_gemma":[0.0755984,0.0004124477,0.9193069,0.0004484842,0.0001958756,0.002983732,0.000263592,0.0002431577,0.0005474819],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1990702,"threshold_uncertainty_score":0.987689,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2882924109126018,"score_gpt":0.4485972092231941,"score_spread":0.1603047983105923,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}