{"id":"W2893953242","doi":"10.1177/1541931218621082","title":"Usability Analysis of Freeform Marking on Engineering Problem Solving","year":2018,"lang":"en","type":"article","venue":"Proceedings of the Human Factors and Ergonomics Society Annual Meeting","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Usability; Formative assessment; Consistency (knowledge bases); Computer science; Test (biology); Perspective (graphical); Group (periodic table); Sample (material); Mathematics education; Human–computer interaction; Psychology; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001220218,0.0001443125,0.0003148585,0.00006669594,0.0007894969,0.00007870943,0.0003643167,0.0001001348,0.00001583539],"category_scores_gemma":[0.0001400769,0.0001118318,0.0003241347,0.0004758609,0.0004013379,0.0002617938,0.0002204241,0.0001466597,1.82016e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001259497,"about_ca_system_score_gemma":0.00002290832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007194352,"about_ca_topic_score_gemma":0.0001461638,"domain_scores_codex":[0.9988461,0.000008077154,0.000354476,0.0002364347,0.0002685649,0.0002863718],"domain_scores_gemma":[0.9991111,0.0001309939,0.0003862235,0.00007325022,0.0002410639,0.00005732468],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00001277495,0.00004399237,0.876689,0.00006767343,0.0004133272,4.662737e-9,0.1112053,0.00003540207,0.004920421,0.006330433,0.000154841,0.0001268834],"study_design_scores_gemma":[0.0002422573,0.0001210561,0.8774284,0.0002103028,0.0005060349,4.30801e-8,0.1137529,0.001418268,0.004855487,0.0004515485,0.0006859389,0.0003278021],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9933854,0.00001805611,0.000003389776,0.00005505054,0.0001001849,0.0001576822,0.0000100555,0.0000279093,0.006242323],"genre_scores_gemma":[0.9987832,0.00002887066,0.0009157294,0.00001670824,0.0001681021,0.000002453989,9.704839e-7,0.00001039348,0.00007353313],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00616879,"threshold_uncertainty_score":0.6072251,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01859004852677291,"score_gpt":0.2736410332256983,"score_spread":0.2550509846989253,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}