{"id":"W4386172632","doi":"10.4018/978-1-6684-8213-1.ch002","title":"Do Our Test Scores Mean What We Think?","year":2023,"lang":"en","type":"book-chapter","venue":"Advances in educational technologies and instructional design book series","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Christian Studies; University of Toronto; Memorial University of Newfoundland","funders":"","keywords":"Active listening; Construct (python library); Test (biology); Construct validity; Scale (ratio); Language assessment; English language; Psychology; Writing assessment; Cognition; Computer science; Applied psychology; Mathematics education; Psychometrics; Developmental psychology; Communication; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01410183,0.0006962446,0.0008081802,0.00230827,0.001366352,0.009540361,0.001165943,0.0009158319,0.007655141],"category_scores_gemma":[0.0865812,0.0002861215,0.0004874202,0.002884235,0.006713198,0.007608278,0.001587167,0.002985698,0.005562329],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004314577,"about_ca_system_score_gemma":0.005396312,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01845462,"about_ca_topic_score_gemma":0.02873739,"domain_scores_codex":[0.9878836,0.003508619,0.0006424459,0.0007505257,0.006758434,0.0004563651],"domain_scores_gemma":[0.9655584,0.01696218,0.00234816,0.00147561,0.01233396,0.001321791],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006907081,0.0001060139,0.06062287,0.0009540461,0.0001043793,0.00008058563,0.03550586,0.0003207489,0.0009670792,0.07309919,0.1711564,0.6570137],"study_design_scores_gemma":[0.00003496567,0.0002903179,0.1548273,0.00513286,0.0001627461,0.0008371706,0.10348,0.001363531,0.003204283,0.1745163,0.555784,0.0003665372],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1241459,0.03337558,0.05445715,0.1624127,0.01346756,0.0003586272,0.003166281,0.001918983,0.6066971],"genre_scores_gemma":[0.8345889,0.02878541,0.05083565,0.01930585,0.001886945,0.000712398,0.002126499,0.001379897,0.06037859],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9858981,"threshold_uncertainty_score":0.07457852,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03835766550508211,"score_gpt":0.3299552900022539,"score_spread":0.2915976244971717,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}