{"id":"W4409449514","doi":"10.31219/osf.io/tjxab_v2","title":"Fitting item response theory models using deep learning computational frameworks","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Technology and Data Analysis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Connaught Fund","keywords":"Item response theory; Computer science; Deep learning; Artificial intelligence; Econometrics; Machine learning; Mathematics; Statistics; Psychometrics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008544884,0.00164609,0.00138066,0.001682456,0.000589826,0.002427255,0.002866229,0.001803823,0.006230388],"category_scores_gemma":[0.04729971,0.00113555,0.001880915,0.002387887,0.001186213,0.002968144,0.002333849,0.004503597,0.001653826],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002339724,"about_ca_system_score_gemma":0.002264365,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009397237,"about_ca_topic_score_gemma":0.01178311,"domain_scores_codex":[0.9954529,0.00310985,0.0001711819,0.0004889769,0.000559518,0.00021752],"domain_scores_gemma":[0.9833249,0.01272847,0.0007273027,0.001729973,0.001248228,0.0002410983],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001466353,0.0002235866,0.006471374,0.0001573046,0.0002325057,0.00009450618,0.0002413134,0.8103063,0.0007316245,0.04857247,0.004945139,0.1278773],"study_design_scores_gemma":[0.00001004789,0.00001137643,0.0002987016,0.00001202585,0.000005130018,0.000009413912,0.00001601532,0.9692189,0.0001415805,0.02999859,0.0002715733,0.000006635527],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0246881,0.0001849112,0.9718271,0.0007745901,0.00003552536,0.0001178249,0.0004201505,0.000852884,0.001098855],"genre_scores_gemma":[0.4087451,0.0003926322,0.584337,0.0006047541,0.00009400301,0.000926817,0.001629103,0.0002269193,0.003043752],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009397237,"threshold_uncertainty_score":0.04519016,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0222703727686225,"score_gpt":0.2917846797853701,"score_spread":0.2695143070167476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}