{"id":"W2995870317","doi":"10.5539/elt.v13n1p124","title":"A Comparative Study of Test Takers’ Performance on Computer-Based Test and Paper-Based Test Across Different CEFR Levels","year":2019,"lang":"en","type":"article","venue":"English Language Teaching","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Task (project management); Psychology; Perspective (graphical); Language assessment; Mathematics education; Computer science; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0007624865,0.0003819581,0.0005791754,0.0001245164,0.0006979426,0.0003159519,0.000276498,0.000064119,0.0002929904],"category_scores_gemma":[0.0005584938,0.0003073057,0.00008295067,0.00004437476,0.0001362788,0.0002496465,0.00008023941,0.001125844,0.00002613121],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006693292,"about_ca_system_score_gemma":0.00002834968,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004095984,"about_ca_topic_score_gemma":0.0005737496,"domain_scores_codex":[0.9980091,0.0003003441,0.0004078754,0.0004824293,0.000376137,0.0004240536],"domain_scores_gemma":[0.996178,0.002934067,0.000266384,0.0004474777,0.00007554201,0.00009854505],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00003373072,0.002432409,0.151664,0.000158605,0.00003885051,0.00001320794,0.8378857,0.001516254,0.001297483,0.0001665576,0.0001066114,0.004686634],"study_design_scores_gemma":[0.009999355,0.01313638,0.1382649,0.001749096,0.0001277823,0.000003393058,0.7721344,0.05464258,0.001890665,0.000005450292,0.006435649,0.001610297],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9919174,0.0000659053,0.00003520726,0.00004118874,0.0003255049,0.0004945807,0.0001311451,0.0002686309,0.006720456],"genre_scores_gemma":[0.9979428,4.134187e-7,0.0001495328,0.0002316375,0.0006143953,0.0000197987,0.00004044755,0.00004789006,0.0009530874],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06575129,"threshold_uncertainty_score":0.9999379,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03000706054974718,"score_gpt":0.2756809433892808,"score_spread":0.2456738828395336,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}