{"id":"W4389206342","doi":"10.22215/etd/2023-15821","title":"Evaluating the Effectiveness of Comparison Activities in a Programming Tutor","year":2023,"lang":"en","type":"dissertation","venue":"","topic":"Teaching and Learning Programming","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"TUTOR; Computer science; Test (biology); Mathematics education; Significant difference; Artificial intelligence; Machine learning; Programming language; Psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01342356,0.001212542,0.001517135,0.001342277,0.0005098631,0.002116439,0.001728921,0.001588101,0.00460834],"category_scores_gemma":[0.1292683,0.0004257372,0.0006256209,0.0008267133,0.0005360101,0.002580843,0.001678034,0.0011307,0.001134757],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007773656,"about_ca_system_score_gemma":0.001046881,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005198883,"about_ca_topic_score_gemma":0.0004989039,"domain_scores_codex":[0.9824437,0.00940903,0.002351856,0.002404186,0.002856523,0.0005347021],"domain_scores_gemma":[0.7372116,0.2279411,0.01044663,0.007941916,0.01112687,0.005331981],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0433624,0.04518008,0.0788326,0.004627065,0.0007302754,0.0003237792,0.009053105,0.01219191,0.07517254,0.001173065,0.002186896,0.7271663],"study_design_scores_gemma":[0.01432297,0.3882602,0.2518874,0.001447278,0.003397746,0.00116043,0.006858427,0.09744991,0.2116655,0.003867574,0.0190579,0.0006246726],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9875391,0.0002429634,0.007693403,0.00007098135,0.00006566042,0.001077352,0.0001156457,0.0003694796,0.002825426],"genre_scores_gemma":[0.9591973,0.0002499593,0.03634498,0.00007698212,0.00005059258,0.001399657,0.0004137078,0.00007178822,0.00219508],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01342356,"threshold_uncertainty_score":0.07099146,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06193665027651755,"score_gpt":0.4068142584345004,"score_spread":0.3448776081579828,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}