{"id":"W4386172632","doi":"10.4018/978-1-6684-8213-1.ch002","title":"Do Our Test Scores Mean What We Think?","year":2023,"lang":"en","type":"book-chapter","venue":"Advances in educational technologies and instructional design book series","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Christian Studies; University of Toronto; Memorial University of Newfoundland","funders":"","keywords":"Active listening; Construct (python library); Test (biology); Construct validity; Scale (ratio); Language assessment; English language; Psychology; Writing assessment; Cognition; Computer science; Applied psychology; Mathematics education; Psychometrics; Developmental psychology; Communication; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003209975,0.0003693247,0.0003633546,0.0004774204,0.0006755707,0.0003646951,0.000531205,0.0004790884,0.0001911887],"category_scores_gemma":[0.0002726474,0.0003640123,0.00008482977,0.0002008019,0.001219644,0.003899166,0.0001958447,0.0004872731,0.00006954918],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002547249,"about_ca_system_score_gemma":0.0005213451,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002063012,"about_ca_topic_score_gemma":0.0004159153,"domain_scores_codex":[0.9979022,0.00002946045,0.0004165221,0.0005951987,0.0006976599,0.0003589625],"domain_scores_gemma":[0.9986721,0.0005623234,0.0002797433,0.0002446794,0.0001875875,0.00005353144],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002259077,0.00002271037,0.0006211443,0.00003259366,0.00003104389,0.00000344461,0.0009608971,0.00001320112,0.000004895895,0.9294896,0.002615755,0.06618218],"study_design_scores_gemma":[0.00009197115,0.00006112374,0.0004252748,0.0003589126,0.00001310947,0.000008673829,0.03399303,8.309492e-7,0.000007708075,0.5765316,0.388214,0.0002937738],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"review","genre_scores_codex":[0.0004408972,0.2735693,0.0001012212,0.1045939,0.01307363,0.002895057,0.0001787621,0.002628145,0.6025192],"genre_scores_gemma":[0.001514794,0.6584103,0.007825268,0.00008376548,0.0005999038,0.0001303461,0.00004440101,0.00004774372,0.3313435],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.3855982,"threshold_uncertainty_score":0.9998812,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03835766550508211,"score_gpt":0.3299552900022539,"score_spread":0.2915976244971717,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}