{"id":"W2064379877","doi":"10.1177/0013164402238082","title":"A Monte Carlo Comparison of Item and Person Statistics Based on Item Response Theory versus Classical Test Theory","year":2002,"lang":"en","type":"article","venue":"Educational and Psychological Measurement","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":121,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Item response theory; Classical test theory; Monte Carlo method; Statistics; Psychology; Econometrics; Psychometrics; Test theory; Item analysis; Computerized adaptive testing; Test (biology); Differential item functioning; Statistical hypothesis testing; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06936503,0.0005999351,0.0009441296,0.003069585,0.0004456388,0.001928622,0.001299463,0.001379331,0.001585485],"category_scores_gemma":[0.357976,0.000467116,0.0007666678,0.001840892,0.002121706,0.003339872,0.001667483,0.001270363,0.0003102547],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001204262,"about_ca_system_score_gemma":0.0009909576,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001463618,"about_ca_topic_score_gemma":0.001560229,"domain_scores_codex":[0.9578832,0.03395642,0.001022514,0.002687836,0.004068042,0.0003818708],"domain_scores_gemma":[0.5083188,0.4516118,0.007853303,0.02348077,0.007784965,0.0009503145],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0067143,0.0007161052,0.2153496,0.0007124517,0.002018879,0.0002911628,0.002946689,0.3399122,0.006033674,0.1758301,0.003214745,0.2462602],"study_design_scores_gemma":[0.0002483882,0.001628936,0.06803903,0.0001685685,0.000224031,0.0007141035,0.0004880556,0.8655397,0.004122929,0.05705947,0.001599572,0.0001672851],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5916283,0.000503034,0.4023365,0.0002718386,0.00006818189,0.0003875229,0.0002574084,0.0005772701,0.00397],"genre_scores_gemma":[0.8921206,0.000105847,0.106539,0.0001065846,0.00001559774,0.0002783439,0.000480659,0.0001026952,0.0002506773],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06936503,"threshold_uncertainty_score":0.3668417,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7790233134897627,"score_gpt":0.505780321916664,"score_spread":0.2732429915730987,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}