{"id":"W4412601135","doi":"10.51357/jei.v6i1.314","title":"Cross-evaluation of Large Language Model Assessment Behaviours in Educational Tasks by Cognitive Level","year":2025,"lang":"en","type":"article","venue":"Journal of Educational Informatics","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Cognition; Psychology; Cognitive psychology; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002063568,0.0001550276,0.0002704459,0.0004359129,0.00007341061,0.00004183667,0.0003055781,0.0001351594,0.004214974],"category_scores_gemma":[0.0004013738,0.0001383361,0.0001167346,0.0004181519,0.00009228943,0.0004160576,0.00003302339,0.0004008208,0.00002191402],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003223181,"about_ca_system_score_gemma":0.002348074,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003678091,"about_ca_topic_score_gemma":0.00001237114,"domain_scores_codex":[0.9973516,0.0001559893,0.001372087,0.0001075864,0.0007867367,0.000225984],"domain_scores_gemma":[0.9964542,0.000775432,0.0009975522,0.0001586248,0.00152383,0.00009036974],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001835626,0.01123831,0.717406,0.0001029288,0.0004390753,5.548584e-7,0.01868063,0.002104914,0.0001617489,0.1382865,0.1050069,0.006388877],"study_design_scores_gemma":[0.002275658,0.00008742771,0.9685531,0.0001888381,0.0001285083,0.00001643844,0.009690474,0.001706434,0.00004736203,0.01690415,0.0002512215,0.0001503566],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9476041,0.0004677738,0.001837459,0.00348581,0.001307669,0.0002770881,0.0004517471,0.000002095127,0.04456623],"genre_scores_gemma":[0.9890376,0.00001119304,0.004193249,0.0009976964,0.0001176194,0.00006589313,0.0004426684,0.000007390737,0.00512668],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2511471,"threshold_uncertainty_score":0.9966953,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1020707945575831,"score_gpt":0.5349247421723857,"score_spread":0.4328539476148026,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}