{"id":"W2116070328","doi":"10.1177/0265532210376379","title":"Think-aloud protocols in research on essay rating: An empirical study of their veridicality and reactivity","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":121,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Think aloud protocol; Psychology; Protocol analysis; Perception; Empirical research; Sample (material); Qualitative research; Rating scale; Social psychology; Nomothetic and idiographic; Cognitive psychology; Applied psychology; Developmental psychology; Epistemology; Cognitive science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03743877,0.0007361121,0.0004927889,0.00120755,0.0009173541,0.001642874,0.001102275,0.0009466977,0.00116799],"category_scores_gemma":[0.2199901,0.000555056,0.000320532,0.001070857,0.001426186,0.001562504,0.001957292,0.001280929,0.0006609897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004902453,"about_ca_system_score_gemma":0.0006471811,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001933144,"about_ca_topic_score_gemma":0.0002665671,"domain_scores_codex":[0.9395154,0.04742792,0.003363881,0.002915364,0.006367211,0.0004101973],"domain_scores_gemma":[0.6944738,0.2487216,0.0227628,0.01710609,0.01570473,0.001230987],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00291203,0.001885879,0.1145801,0.002211474,0.0003552911,0.0007919685,0.2230306,0.001853537,0.1190248,0.005478701,0.001786692,0.526089],"study_design_scores_gemma":[0.0007464606,0.02040762,0.5165713,0.002543318,0.0005755922,0.007781459,0.1235017,0.02727211,0.2092096,0.0289061,0.06154568,0.0009391778],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8939719,0.000418444,0.09813229,0.0001922338,0.0001273409,0.001318618,0.000135502,0.0002361061,0.005467661],"genre_scores_gemma":[0.913073,0.0005242794,0.07989421,0.0002689874,0.0001163691,0.003634382,0.0001949199,0.0001571192,0.002136816],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9625612,"threshold_uncertainty_score":0.1979975,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2923764194054679,"score_gpt":0.5510859877479262,"score_spread":0.2587095683424582,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}