{"id":"W4394817988","doi":"10.1080/10447318.2024.2338330","title":"Transforming Educational Assessment: Insights Into the Use of ChatGPT and Large Language Models in Grading","year":2024,"lang":"en","type":"article","venue":"International Journal of Human-Computer Interaction","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":47,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Grading (engineering); Computer science; Artificial intelligence; Mathematics education; Psychology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01354379,0.0004707187,0.0003703559,0.00237733,0.0008963482,0.006716608,0.001074632,0.001004889,0.001966648],"category_scores_gemma":[0.08794835,0.0003740533,0.0003648218,0.001563423,0.004775825,0.006458344,0.002787054,0.001679883,0.0003044199],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002476637,"about_ca_system_score_gemma":0.002261772,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004683294,"about_ca_topic_score_gemma":0.006027237,"domain_scores_codex":[0.9815758,0.01566681,0.0004345248,0.0005760914,0.001509582,0.0002371768],"domain_scores_gemma":[0.8850806,0.1022513,0.003825678,0.004232645,0.003498587,0.001111227],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0005243184,0.0008632867,0.09094933,0.0006832831,0.00007088851,0.001692122,0.2406644,0.02303853,0.005026762,0.2252874,0.003593992,0.4076057],"study_design_scores_gemma":[0.00009236283,0.0006462756,0.07233072,0.0009728897,0.0001096166,0.002370161,0.09692083,0.342797,0.007481226,0.4327829,0.04325263,0.0002434537],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5568874,0.0007639509,0.3769121,0.007553354,0.0001260266,0.0003784454,0.0002319733,0.0008211531,0.05632557],"genre_scores_gemma":[0.9620487,0.0001433488,0.03628612,0.0001369083,0.00001581331,0.00009618935,0.00004327906,0.00004826258,0.001181408],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01354379,"threshold_uncertainty_score":0.07162726,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.138589355159479,"score_gpt":0.4668324641159142,"score_spread":0.3282431089564352,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}