{"id":"W4412601135","doi":"10.51357/jei.v6i1.314","title":"Cross-evaluation of Large Language Model Assessment Behaviours in Educational Tasks by Cognitive Level","year":2025,"lang":"en","type":"article","venue":"Journal of Educational Informatics","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Cognition; Psychology; Cognitive psychology; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07423183,0.0007852161,0.0008956314,0.005923899,0.0009350204,0.002791138,0.0009863446,0.0006309605,0.001603877],"category_scores_gemma":[0.2209146,0.0003194228,0.001622167,0.002145499,0.001176492,0.004570726,0.005265079,0.001068403,0.0005461786],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001773975,"about_ca_system_score_gemma":0.001699072,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001135839,"about_ca_topic_score_gemma":0.003871485,"domain_scores_codex":[0.9512255,0.03166208,0.004304076,0.003400067,0.008705179,0.0007030682],"domain_scores_gemma":[0.6935984,0.231164,0.0194582,0.02049916,0.03237384,0.002906332],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00108305,0.001011969,0.5276532,0.001170261,0.0007849729,0.0001174772,0.02323006,0.002720319,0.007145079,0.001824731,0.001542259,0.4317167],"study_design_scores_gemma":[0.0001542303,0.00373301,0.9221679,0.001047124,0.0006277603,0.0003903812,0.01530149,0.02025912,0.01873335,0.008081727,0.009148592,0.0003553133],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9510308,0.0008536344,0.04004644,0.0002292728,0.00006879338,0.00106882,0.0002922382,0.0003630078,0.006047041],"genre_scores_gemma":[0.9638871,0.0002829206,0.03310287,0.00009250314,0.00002442514,0.001173845,0.0005487978,0.00008402963,0.0008035759],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07423183,"threshold_uncertainty_score":0.3925801,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1020707945575831,"score_gpt":0.5349247421723857,"score_spread":0.4328539476148026,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}