{"id":"W3087421864","doi":"10.1007/s40593-020-00211-5","title":"Automated Essay Scoring and the Deep Learning Black Box: How Are Rubric Scores Determined?","year":2020,"lang":"en","type":"article","venue":"International Journal of Artificial Intelligence in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":false,"ca_institutions":"Athabasca University","funders":"","keywords":"Rubric; Grading (engineering); CONTEST; Computer science; Artificial intelligence; Deep learning; Machine learning; Natural language processing; Mathematics education; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01480927,0.000900137,0.001447126,0.003394545,0.0007288796,0.0047222,0.001834864,0.001268807,0.007214185],"category_scores_gemma":[0.09970408,0.0005063989,0.0004145468,0.002534917,0.0006984776,0.003480838,0.002036748,0.001943557,0.007645219],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009939474,"about_ca_system_score_gemma":0.002284458,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004396026,"about_ca_topic_score_gemma":0.00823978,"domain_scores_codex":[0.9818703,0.009970119,0.0011639,0.001775264,0.004545251,0.0006752276],"domain_scores_gemma":[0.9302749,0.0332562,0.005223432,0.007376508,0.02241514,0.001453813],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004058827,0.0003463624,0.05677023,0.0002217604,0.0001531749,0.00002684552,0.0003221197,0.004726457,0.003330621,0.004090146,0.02908901,0.9005173],"study_design_scores_gemma":[0.0002759325,0.0008767145,0.1800326,0.001187358,0.0002578741,0.0003946137,0.001877684,0.6604035,0.03864,0.06368503,0.05198728,0.0003813698],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3600026,0.004416818,0.5528107,0.007816601,0.001183056,0.0005536709,0.005856837,0.01415618,0.05320356],"genre_scores_gemma":[0.8018166,0.0005795779,0.1809396,0.0006786513,0.0002775136,0.0003837885,0.003235352,0.0007587845,0.01133015],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01480927,"threshold_uncertainty_score":0.07831985,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03866308912839007,"score_gpt":0.3147992035990448,"score_spread":0.2761361144706547,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}