{"id":"W4396978444","doi":"10.21449/ijate.1376160","title":"The difference between estimated and perceived item difficulty: An empirical study","year":2024,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Statistics; Empirical research; Econometrics; Mathematics education; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01432868,0.0003876636,0.0004572233,0.001698114,0.0003756679,0.001431433,0.0008024081,0.0007078693,0.002675716],"category_scores_gemma":[0.1241156,0.0003500449,0.0005610971,0.001577574,0.00126522,0.002087456,0.0009099299,0.001230655,0.0004585141],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006575521,"about_ca_system_score_gemma":0.0005531189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001361479,"about_ca_topic_score_gemma":0.001648733,"domain_scores_codex":[0.9867541,0.006271604,0.001614691,0.001511036,0.003450953,0.0003975526],"domain_scores_gemma":[0.7186046,0.2400931,0.01662597,0.006829373,0.01607181,0.001775059],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000151052,0.00041378,0.9835125,0.00009524104,0.0001189512,0.0001330353,0.003422404,0.0003575368,0.0005071454,0.000190852,0.0001427115,0.01095498],"study_design_scores_gemma":[0.00001759589,0.0005880485,0.9911004,0.00005473735,0.00006016729,0.000335425,0.003192706,0.00325902,0.0007626127,0.0001834251,0.000418335,0.00002754265],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9974852,0.00009657616,0.001452317,0.00003407165,0.000005333918,0.00003409375,0.00008361876,0.00000574964,0.0008029802],"genre_scores_gemma":[0.9987751,0.0000426758,0.0009050508,0.00001117262,0.000004138855,0.00003577165,0.0000808625,0.000004804426,0.0001403352],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01432868,"threshold_uncertainty_score":0.07577825,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08975046533286718,"score_gpt":0.5107784354602035,"score_spread":0.4210279701273363,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}