{"id":"W2763194644","doi":"","title":"A comparative study of the third grade math test at provincial level between China and Canada","year":2017,"lang":"en","type":"article","venue":"2017 Conference of the Canadian Society for the Study of Education","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Mathematics education; China; Mathematics; Degree (music); Reciprocal; Geography; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.0003181467,0.0001053567,0.0001950087,0.00001709345,0.001795586,0.00005355583,0.002150889,0.00004340031,0.000001115572],"category_scores_gemma":[0.0001224762,0.00006142564,0.00006757503,0.00007008293,0.0003273575,0.000107849,0.0003015954,0.0001399246,9.694735e-8],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002284508,"about_ca_system_score_gemma":0.005831786,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.899493,"about_ca_topic_score_gemma":0.99098,"domain_scores_codex":[0.9991673,0.00004391182,0.0002066521,0.0001877172,0.0002508597,0.0001436211],"domain_scores_gemma":[0.9979417,0.0001843571,0.0005735195,0.0009961319,0.0002522185,0.00005205045],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000004154094,0.001089447,0.8909812,0.00004234487,0.0003726753,5.401306e-8,0.07130613,0.000009732037,0.0001140525,0.0171141,0.01715009,0.001815998],"study_design_scores_gemma":[0.0002709082,0.00015001,0.9720458,0.00002510294,0.00007128269,7.465194e-7,0.02517742,0.0002046668,0.0002546682,0.001545486,0.0001876373,0.00006630274],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9870694,0.00002082222,0.00003033839,0.01057538,0.0004727619,0.00149532,0.00008671571,0.000002710645,0.0002465615],"genre_scores_gemma":[0.9989991,0.000001779905,0.0002222143,0.00003878447,0.00003856207,0.00009274635,0.00000138646,0.000002984201,0.0006024141],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.09148699,"threshold_uncertainty_score":0.9998042,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08914793780467989,"score_gpt":0.3304077155300976,"score_spread":0.2412597777254177,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}