{"id":"W2145055698","doi":"10.1177/0265532210368717","title":"Explaining ESL essay holistic scores: A multilevel modeling approach","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":64,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"Educational Testing Service","keywords":"Psychology; Argumentation theory; Context (archaeology); Multilevel model; Set (abstract data type); Multilevel modelling; Social psychology; Epistemology; Statistics; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02143838,0.001149958,0.001252039,0.003385942,0.00127685,0.002500051,0.001755244,0.0008932617,0.003217856],"category_scores_gemma":[0.05999744,0.0005565675,0.00319727,0.002971543,0.0007871265,0.001354862,0.002714962,0.002110571,0.0005126999],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00121941,"about_ca_system_score_gemma":0.001244723,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01274714,"about_ca_topic_score_gemma":0.009672363,"domain_scores_codex":[0.9846848,0.01102033,0.0006814038,0.001720024,0.001386587,0.0005068664],"domain_scores_gemma":[0.9512467,0.03797689,0.003823817,0.003778102,0.002673199,0.0005012836],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000445789,0.0004304921,0.8650311,0.0002404689,0.002834687,0.0003303559,0.007359024,0.01943192,0.001955191,0.01132918,0.001978673,0.08863313],"study_design_scores_gemma":[0.000104547,0.001206616,0.4178872,0.0002116721,0.001107979,0.0003594108,0.003312034,0.5430786,0.002164633,0.02590172,0.004474832,0.0001907043],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7655522,0.0002507368,0.2289283,0.0006523838,0.00006473838,0.000614025,0.001429692,0.0003741534,0.002133884],"genre_scores_gemma":[0.9431535,0.00006960533,0.05454081,0.00004303349,0.00002640606,0.0007340079,0.0006876765,0.00005312483,0.000691813],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02143838,"threshold_uncertainty_score":0.1133783,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1195594706556013,"score_gpt":0.2960164806540859,"score_spread":0.1764570099984846,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}