{"id":"W4392681534","doi":"10.22318/icls2023.530792","title":"Design and Evaluation of a Conversational Agent for Formative Assessment in Higher Education","year":2023,"lang":"en","type":"article","venue":"Proceedings.","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; Concordia University of Edmonton","funders":"","keywords":"Formative assessment; Conversation; Software walkthrough; Computer science; Psychology; Mathematics education","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01016403,0.000988571,0.0006481619,0.0007859014,0.0007892463,0.00218922,0.002447491,0.001623082,0.003367526],"category_scores_gemma":[0.0155145,0.0005548127,0.0005095911,0.0002691287,0.0006940385,0.001697444,0.001617648,0.001059394,0.001125044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001106459,"about_ca_system_score_gemma":0.002442758,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001214406,"about_ca_topic_score_gemma":0.0009745929,"domain_scores_codex":[0.9917669,0.005142586,0.0007293459,0.0007261122,0.001234212,0.0004009028],"domain_scores_gemma":[0.9867736,0.006286351,0.0007027174,0.0007591303,0.003047518,0.002430648],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"nonrandomized_trial","study_design_scores_codex":[0.01262615,0.01970374,0.0240805,0.004152984,0.0003968919,0.002114839,0.01802358,0.02730625,0.2403088,0.006678362,0.004271733,0.6403362],"study_design_scores_gemma":[0.007984167,0.07770289,0.04539027,0.00101197,0.001459863,0.002325393,0.0088791,0.457052,0.2826499,0.00373818,0.1109442,0.0008620318],"study_design_candidate":"nonrandomized_trial","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6909363,0.0003972879,0.2796494,0.0004559537,0.000313318,0.01539665,0.0003068551,0.005304346,0.007239914],"genre_scores_gemma":[0.5365247,0.0001569303,0.4496105,0.0001988234,0.00005826053,0.007382753,0.0003468928,0.0003107241,0.005410478],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01016403,"threshold_uncertainty_score":0.0537532,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1453634086173477,"score_gpt":0.3622131756128345,"score_spread":0.2168497669954868,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}