{"id":"W2109863812","doi":"10.1191/0265532204lt278oa","title":"A teacher-verification study of speaking and writing prototype tasks for a new TOEFL","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":93,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Test of English as a Foreign Language; Active listening; Psychology; Formative assessment; Mathematics education; Test (biology); CLARITY; Reading (process); Second language writing; Language proficiency; Presentation (obstetrics); Language assessment; Pedagogy; Computer science; Second language; Linguistics; Communication","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01861471,0.0006208867,0.0009603996,0.001292594,0.001853008,0.001735428,0.001565011,0.0009866271,0.001701944],"category_scores_gemma":[0.08789999,0.0007758045,0.0004841319,0.0005552672,0.001156927,0.001861287,0.00123315,0.001889018,0.0007459284],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001631151,"about_ca_system_score_gemma":0.001822153,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003271908,"about_ca_topic_score_gemma":0.009383092,"domain_scores_codex":[0.992465,0.004347793,0.0007357041,0.0007294015,0.001326899,0.0003952288],"domain_scores_gemma":[0.8910272,0.07368827,0.006842709,0.009079237,0.01620084,0.003161734],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.002855938,0.03294211,0.2388501,0.0008384478,0.00007237328,0.002615521,0.4407515,0.001073167,0.04442396,0.0006428777,0.002023237,0.2329106],"study_design_scores_gemma":[0.0018267,0.05766236,0.614072,0.0003974746,0.0001408935,0.005233862,0.2254458,0.01058414,0.05340882,0.001231481,0.02955977,0.0004366987],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.998639,0.00001713757,0.0007689169,0.00002716865,0.000008696666,0.0001215204,0.0000150893,0.00001553825,0.0003868745],"genre_scores_gemma":[0.9916943,0.00006048545,0.0055071,0.0001148041,0.00001614463,0.0003316253,0.0001041072,0.00002439609,0.002147121],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01861471,"threshold_uncertainty_score":0.09844518,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08046825576428888,"score_gpt":0.3114058708120012,"score_spread":0.2309376150477123,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}