{"id":"W2011276392","doi":"10.1080/08957347.2011.607061","title":"An Experimental Test of Student Verbal Reports and Teacher Evaluations as a Source of Validity Evidence for Test Development","year":2011,"lang":"en","type":"article","venue":"Applied Measurement in Education","topic":"Science Education and Pedagogy","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Test (biology); Psychology; CLARITY; Test validity; Set (abstract data type); Association (psychology); Standards for Educational and Psychological Testing; Mathematics education; Content validity; Applied psychology; Social psychology; Psychometrics; Computer science; Higher education; Clinical psychology; Education theory","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05463069,0.001576458,0.001819986,0.004031057,0.001456684,0.002825055,0.001909738,0.00199498,0.006396119],"category_scores_gemma":[0.2414405,0.00112097,0.002087398,0.003486505,0.003422527,0.004712238,0.003703678,0.002887038,0.001816353],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001234445,"about_ca_system_score_gemma":0.001510908,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006247304,"about_ca_topic_score_gemma":0.000884552,"domain_scores_codex":[0.909871,0.05313108,0.009758758,0.008578184,0.01748619,0.001174758],"domain_scores_gemma":[0.4834188,0.4368646,0.03248284,0.0254917,0.01997738,0.001764744],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.03212894,0.03775991,0.379723,0.003876914,0.005478818,0.0004929427,0.02288346,0.003463254,0.02854566,0.02128044,0.01354556,0.4508211],"study_design_scores_gemma":[0.009744899,0.09015408,0.7873656,0.00124932,0.001897428,0.001255752,0.007236168,0.02429931,0.02924515,0.01305741,0.03389464,0.0006002625],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8572423,0.0008498717,0.06876655,0.0006781085,0.00109746,0.02138679,0.002569794,0.0008276808,0.04658147],"genre_scores_gemma":[0.8198565,0.0005716055,0.112037,0.0009787478,0.0004368942,0.0569895,0.002736241,0.0002758413,0.006117909],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05463069,"threshold_uncertainty_score":0.2889181,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4437717804134738,"score_gpt":0.4947924416640539,"score_spread":0.05102066125058008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}