{"id":"W2054195447","doi":"10.2466/pr0.103.2.545-565","title":"On the Integrity of Reliability Estimation in Classical Test Theory: The Case for an Additive Coefficient of Stability","year":2008,"lang":"en","type":"article","venue":"Psychological Reports","topic":"Advanced Statistical Methods and Models","field":"Mathematics","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Winnipeg","funders":"","keywords":"Generalizability theory; Estimator; Reliability (semiconductor); Classical test theory; Econometrics; Estimation; Stability (learning theory); Reliability engineering; Computer science; Scale (ratio); Statistics; Test (biology); Mathematics; Item response theory; Psychometrics; Machine learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2893046,0.001805501,0.003935568,0.006134838,0.003591182,0.009314324,0.00515955,0.004954437,0.002589806],"category_scores_gemma":[0.6751145,0.001940755,0.002249996,0.005791037,0.03871521,0.02214191,0.01488215,0.01431255,0.001238885],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003912794,"about_ca_system_score_gemma":0.005080283,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002455929,"about_ca_topic_score_gemma":0.001327985,"domain_scores_codex":[0.7096522,0.2077588,0.01226571,0.01875553,0.04909465,0.002473074],"domain_scores_gemma":[0.2383909,0.6078051,0.02091245,0.1009013,0.03042573,0.00156444],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001482512,0.00003882536,0.00458793,0.0002444501,0.0001670523,0.00009293704,0.002556282,0.00278894,0.0002636506,0.9185659,0.001540079,0.06900559],"study_design_scores_gemma":[0.00005912281,0.0001569422,0.0026386,0.0004017344,0.0001003476,0.000271412,0.000324489,0.01542272,0.0007981179,0.9734578,0.006255467,0.00011336],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01676852,0.003491192,0.9466178,0.01289131,0.0006358903,0.0001629477,0.0001430006,0.0003077472,0.0189816],"genre_scores_gemma":[0.5374141,0.002138444,0.4514984,0.00317288,0.001995796,0.001127122,0.0001417333,0.0004866811,0.002024902],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7106954,"threshold_uncertainty_score":0.876414,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2583452377276964,"score_gpt":0.4824129585210412,"score_spread":0.2240677207933448,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}