{"id":"W4248747928","doi":"10.26434/chemrxiv.13119869.v1","title":"Reasoning, Granularity, and Comparisons: A Unit-Based Method for Characterizing Students’ Arguments on Chemistry Assessments","year":2020,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Rubric; Argumentation theory; Granularity; Argument (complex analysis); Mathematics education; Unit (ring theory); Computer science; Qualitative reasoning; Epistemology; Psychology; Management science; Artificial intelligence; Chemistry; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.003621118,0.0006393682,0.00098011,0.0001299232,0.0002878806,0.0002530584,0.0007739384,0.0006586839,0.0001949879],"category_scores_gemma":[0.0005741008,0.0006823984,0.0002228427,0.0001948579,0.00009965848,0.00003181345,0.0005399336,0.002804119,0.00002146866],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001227457,"about_ca_system_score_gemma":0.0001148612,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007802763,"about_ca_topic_score_gemma":4.363197e-7,"domain_scores_codex":[0.9958324,0.001197701,0.0006026708,0.001388515,0.0004330167,0.0005457035],"domain_scores_gemma":[0.9973897,0.0006938638,0.0007542348,0.0008175824,0.0001547512,0.0001898457],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.004508649,0.006287339,0.7356062,0.007661604,0.007108924,0.0002102609,0.01387174,0.0001359085,0.1451759,0.00576031,0.02536023,0.0483129],"study_design_scores_gemma":[0.01756175,0.001293055,0.7160702,0.004953448,0.001569337,0.0000309127,0.001860714,0.01384996,0.07328553,0.00394195,0.1607656,0.00481746],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6707278,0.0001466425,0.320236,0.001002411,0.001200036,0.001116905,0.00005172281,0.0003601653,0.005158371],"genre_scores_gemma":[0.6985117,0.000006312751,0.2941709,0.002463935,0.0006424182,0.0008569579,0.001117063,0.0002028269,0.002027833],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1354054,"threshold_uncertainty_score":0.9995627,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1390204220203705,"score_gpt":0.4871178097378093,"score_spread":0.3480973877174389,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}