{"id":"W2468693985","doi":"10.1145/2899415.2899473","title":"Benchmarking Introductory Programming Exams","year":2016,"lang":"en","type":"article","venue":"","topic":"Teaching and Learning Programming","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Benchmark (surveying); Benchmarking; Computer science; Mathematics education; Psychology; Management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01559906,0.001356349,0.001164829,0.007300117,0.00095349,0.003277643,0.001344676,0.001229782,0.007926913],"category_scores_gemma":[0.07075138,0.0003751996,0.0009704275,0.008179829,0.0006686778,0.00167068,0.002933566,0.00151531,0.00431231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001747393,"about_ca_system_score_gemma":0.001433774,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00153558,"about_ca_topic_score_gemma":0.002136213,"domain_scores_codex":[0.9725785,0.01024892,0.003764787,0.003255624,0.007857487,0.002294607],"domain_scores_gemma":[0.9003572,0.0255116,0.007660961,0.01011461,0.04764591,0.008709757],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003321194,0.004689876,0.3475283,0.001147032,0.0006295556,0.000487091,0.003081745,0.03130107,0.01146108,0.008489271,0.0507501,0.5371137],"study_design_scores_gemma":[0.0003018026,0.004854219,0.8446941,0.000333675,0.0001987074,0.0005392169,0.00243737,0.02230093,0.03815956,0.004253495,0.08172594,0.0002009335],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9398311,0.001063064,0.01543177,0.0004195173,0.0003807419,0.0008808408,0.007080914,0.001394722,0.03351746],"genre_scores_gemma":[0.9487607,0.0003491059,0.01797538,0.0001872116,0.000113093,0.0005809923,0.02343678,0.0004331493,0.008163503],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01559906,"threshold_uncertainty_score":0.0824967,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01227547792997245,"score_gpt":0.2325014494734201,"score_spread":0.2202259715434476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}