{"id":"W4411157553","doi":"10.1080/10645578.2025.2494485","title":"Implementing Culturally Responsive Evaluation Methods: Reflections on Challenges to Traditional Understandings of Power, Validity, and Rigor","year":2025,"lang":"en","type":"article","venue":"Visitor Studies","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Learning Partnership","funders":"National Aeronautics and Space Administration","keywords":"Power (physics); Rigour; Sociology; Psychology; Engineering ethics; Political science; Epistemology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01045176,0.0001235434,0.0002798758,0.0005003039,0.0005601642,0.00008122113,0.0001620129,0.00003572888,0.0001370784],"category_scores_gemma":[0.004361462,0.0000878853,0.00006709679,0.0006242067,0.00008905727,0.0002052129,0.0001523261,0.00007891459,0.00001190382],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001883671,"about_ca_system_score_gemma":0.0001344023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004290814,"about_ca_topic_score_gemma":0.0001094106,"domain_scores_codex":[0.9969105,0.0007349603,0.0005269422,0.0003824191,0.00126758,0.0001775623],"domain_scores_gemma":[0.9950321,0.003200126,0.0002058186,0.000217454,0.001301365,0.0000431043],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006843905,0.0003681496,0.003476913,0.0001095789,0.0009583873,0.000001561711,0.08169604,0.0004987387,0.01908899,0.5319631,0.06565853,0.2954957],"study_design_scores_gemma":[0.001976008,0.002004139,0.279055,0.0003852357,0.000263628,0.000003866483,0.2740193,0.001211464,0.01012117,0.3312572,0.09924677,0.0004561875],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7871996,0.005286383,0.01275614,0.08444092,0.004015256,0.00196203,0.00008407242,0.00009498088,0.1041607],"genre_scores_gemma":[0.9890832,0.0005623915,0.009052254,0.0004402934,0.00008113309,0.00007971653,0.000002605037,0.00000424879,0.0006941654],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2950395,"threshold_uncertainty_score":0.5221392,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.728286921715105,"score_gpt":0.6583180992551956,"score_spread":0.06996882245990943,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}