{"id":"W3173121569","doi":"10.1145/3430895.3460876","title":"Second Workshop on Educational A/B Testing at Scale","year":2021,"lang":"en","type":"article","venue":"","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Toolbox; Computer science; Context (archaeology); Software testing; Scale (ratio); Educational software; Data science; Software engineering; Test strategy; Software; Engineering management; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01221109,0.001199813,0.0005559016,0.001003148,0.001981703,0.005720898,0.002284002,0.003600988,0.0365432],"category_scores_gemma":[0.009785142,0.0004366699,0.001157202,0.0005023446,0.001751209,0.004381043,0.005083101,0.005264611,0.008206743],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002791163,"about_ca_system_score_gemma":0.004822731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004346683,"about_ca_topic_score_gemma":0.00723047,"domain_scores_codex":[0.9959266,0.001407912,0.0001475901,0.0007557758,0.001096183,0.0006659545],"domain_scores_gemma":[0.9885707,0.003392877,0.0001859304,0.0007907489,0.00301996,0.00403977],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004720594,0.001066206,0.001288408,0.0003633522,0.00005024033,0.001168395,0.004498318,0.003500757,0.008624339,0.03961748,0.7192637,0.2200868],"study_design_scores_gemma":[0.0001018401,0.0003321248,0.00148729,0.0003238723,0.00002017983,0.0002573731,0.002556489,0.00235307,0.003243675,0.01482606,0.9744499,0.00004804932],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.06145967,0.01093285,0.2118333,0.2454289,0.06990691,0.002467709,0.002807857,0.004676076,0.3904867],"genre_scores_gemma":[0.2679495,0.006407257,0.06360182,0.02890769,0.009394708,0.001799702,0.002294037,0.002212978,0.6174324],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.0365432,"threshold_uncertainty_score":0.1222491,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0266716211894776,"score_gpt":0.2877790725867897,"score_spread":0.2611074513973121,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}