{"id":"W7122019785","doi":"10.1093/teamat/hraf020","title":"Using Möbius for automated assessment in mathematics: a case study","year":2025,"lang":"en","type":"article","venue":"Teaching Mathematics and its Applications An International Journal of the IMA","topic":"Mathematics Education and Programs","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Queen's University; Queen's University Belfast","keywords":"Summative assessment; Formative assessment; Enthusiasm; Grading (engineering); Alphanumeric; Context (archaeology); Grade inflation","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02024173,0.001066734,0.0006827576,0.00151972,0.004331249,0.003536716,0.003577179,0.003779591,0.002415827],"category_scores_gemma":[0.07516462,0.0006776797,0.0006452974,0.001127022,0.004698379,0.003214753,0.003912678,0.003576647,0.001401257],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002315832,"about_ca_system_score_gemma":0.002563803,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003923357,"about_ca_topic_score_gemma":0.007433839,"domain_scores_codex":[0.9761564,0.01440098,0.001342931,0.001371634,0.004767518,0.001960597],"domain_scores_gemma":[0.9183259,0.05187331,0.005128388,0.007615308,0.009594673,0.007462408],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.001410992,0.01528497,0.1687508,0.001648987,0.0001551606,0.05174477,0.3744471,0.009859947,0.01751239,0.008566364,0.0122266,0.338392],"study_design_scores_gemma":[0.0005794173,0.02458849,0.1830563,0.001992086,0.0002778234,0.05672687,0.412336,0.06763606,0.0667043,0.01423552,0.1710341,0.0008331399],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9779193,0.0001499899,0.01493644,0.002162874,0.00006393442,0.0003520759,0.00008098935,0.0003427776,0.003991585],"genre_scores_gemma":[0.9787055,0.000146229,0.01744233,0.000403471,0.00003709712,0.000199894,0.00007361983,0.0001484376,0.002843551],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02024173,"threshold_uncertainty_score":0.1070498,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1037123727496623,"score_gpt":0.4885086332382877,"score_spread":0.3847962604886254,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}