{"id":"W2109920068","doi":"10.1080/08957347.2014.905787","title":"Testing Expert-Based Versus Student-Based Cognitive Models for a Grade 3 Diagnostic Mathematics Assessment","year":2014,"lang":"en","type":"article","venue":"Applied Measurement in Education","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"Medical Council of Canada; University of Alberta","funders":"","keywords":"Cognition; Psychology; Test (biology); Consistency (knowledge bases); Construct (python library); Cognitive test; Cognitive psychology; Statistics; Artificial intelligence; Computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02574051,0.0007133285,0.0005020808,0.002333896,0.0004710192,0.002645901,0.001238123,0.0009193409,0.001057837],"category_scores_gemma":[0.1406411,0.0002860547,0.0009680714,0.0008901856,0.0007609454,0.002293194,0.002075996,0.0009412891,0.0003198976],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001206322,"about_ca_system_score_gemma":0.001778819,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002723901,"about_ca_topic_score_gemma":0.004510007,"domain_scores_codex":[0.9849145,0.007167533,0.001162457,0.001242883,0.005019349,0.0004933695],"domain_scores_gemma":[0.9050599,0.06557801,0.01051451,0.007560438,0.009426943,0.001860186],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007928482,0.001322072,0.9124775,0.00009003891,0.000244484,0.0001022558,0.004943299,0.004192852,0.001448942,0.001517044,0.0006047045,0.072264],"study_design_scores_gemma":[0.0003140822,0.006683826,0.8313569,0.0002992983,0.0003004514,0.0008110238,0.01123203,0.1309064,0.009188049,0.005843948,0.002870423,0.0001934796],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9910177,0.00004626317,0.006888364,0.00006579638,0.00001673274,0.0001315641,0.00006745982,0.0000594057,0.001706572],"genre_scores_gemma":[0.9924992,0.00003710112,0.006954953,0.00004104253,0.000006136629,0.0001190884,0.0001403267,0.000009283397,0.000192959],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02574051,"threshold_uncertainty_score":0.1361305,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2185151809927274,"score_gpt":0.4421785572880547,"score_spread":0.2236633762953272,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}