{"id":"W4320498557","doi":"10.1016/j.asw.2022.100691","title":"Comparing summative and dynamic assessments of L2 written argumentative discourse: Microgenetic validity evidence","year":2023,"lang":"en","type":"article","venue":"Assessing Writing","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"","keywords":"Summative assessment; Argumentative; Dialogic; Mediation; Argumentation theory; Psychology; Argument (complex analysis); Dynamic assessment; Protocol analysis; Construct (python library); Formative assessment; Mathematics education; Cognitive psychology; Linguistics; Computer science; Pedagogy; Cognitive science; Developmental psychology; Sociology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05527025,0.0008781487,0.0009474587,0.005507106,0.002793005,0.005034699,0.001835434,0.002208867,0.005079798],"category_scores_gemma":[0.2580529,0.000805492,0.001222046,0.002270491,0.003578336,0.006913702,0.006281943,0.001819176,0.001644825],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002048348,"about_ca_system_score_gemma":0.002400963,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004163735,"about_ca_topic_score_gemma":0.007803008,"domain_scores_codex":[0.9584026,0.01977336,0.003434038,0.004580035,0.01268622,0.001123623],"domain_scores_gemma":[0.5697718,0.3508663,0.02422602,0.01671673,0.03568784,0.002731444],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01160828,0.003155437,0.6439137,0.001834179,0.001956429,0.000293354,0.04915668,0.002120934,0.009276978,0.006377934,0.002599227,0.2677069],"study_design_scores_gemma":[0.001275834,0.007656384,0.9183699,0.001290608,0.001073588,0.0007148384,0.01897269,0.009351228,0.01782287,0.01090636,0.01220368,0.0003619895],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9477623,0.001242393,0.00955729,0.000433005,0.0001759949,0.0009297836,0.0004787963,0.0001613893,0.03925896],"genre_scores_gemma":[0.9880876,0.0003014504,0.00557615,0.0002586392,0.0001098811,0.0009668797,0.0006265339,0.0001115906,0.003961193],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05527025,"threshold_uncertainty_score":0.2923005,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2598054623629933,"score_gpt":0.5350411908807765,"score_spread":0.2752357285177832,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}