{"id":"W2055526333","doi":"10.7202/1025005ar","title":"Item response theory in educational assessment and evaluation","year":2014,"lang":"en","type":"article","venue":"Mesure et évaluation en éducation","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Equating; Item response theory; Polytomous Rasch model; Differential item functioning; Item bank; Variance (accounting); Test (biology); Computerized adaptive testing; Classical test theory; Econometrics; Computer science; Educational assessment; Psychology; Psychometrics; Statistics; Mathematics education; Mathematics; Rasch model; Clinical psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1812013,0.002726248,0.004529892,0.01255335,0.001328825,0.007524648,0.004204303,0.00394557,0.01088081],"category_scores_gemma":[0.3369714,0.001075895,0.003240009,0.02037282,0.005386808,0.006300787,0.005039027,0.00464041,0.005165244],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006550563,"about_ca_system_score_gemma":0.008269298,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003863591,"about_ca_topic_score_gemma":0.002183585,"domain_scores_codex":[0.6402832,0.3043775,0.01657274,0.005299765,0.03211472,0.001352048],"domain_scores_gemma":[0.6278922,0.3116959,0.01280542,0.01740218,0.02869881,0.001505514],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003005739,0.0003418483,0.01015022,0.007240734,0.001088775,0.0001696559,0.001899909,0.01559799,0.0003582184,0.2041084,0.03251635,0.7262273],"study_design_scores_gemma":[0.0006559687,0.0019675,0.0286778,0.01861172,0.0009126682,0.0009213725,0.004679401,0.06972877,0.001985397,0.5988637,0.2723602,0.0006354437],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006748567,0.04328743,0.8895054,0.004944723,0.001679215,0.005852709,0.001996474,0.001613732,0.04437172],"genre_scores_gemma":[0.135519,0.02435311,0.8066705,0.002710263,0.001419883,0.02158717,0.002582174,0.0006551924,0.004502703],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1812013,"threshold_uncertainty_score":0.9582953,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5154681060706208,"score_gpt":0.5912371517400445,"score_spread":0.07576904566942377,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}