{"id":"W2995870317","doi":"10.5539/elt.v13n1p124","title":"A Comparative Study of Test Takers’ Performance on Computer-Based Test and Paper-Based Test Across Different CEFR Levels","year":2019,"lang":"en","type":"article","venue":"English Language Teaching","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Task (project management); Psychology; Perspective (graphical); Language assessment; Mathematics education; Computer science; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003398265,0.0005538682,0.0005816158,0.002404034,0.0003697762,0.001145285,0.0006916194,0.0004959742,0.001801523],"category_scores_gemma":[0.025566,0.0001781126,0.0006594443,0.0008853231,0.0006779035,0.0007763329,0.001129898,0.0004482842,0.0008339599],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006992639,"about_ca_system_score_gemma":0.0005012935,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004484841,"about_ca_topic_score_gemma":0.004287324,"domain_scores_codex":[0.9956365,0.001005148,0.0006041648,0.0005080632,0.001786633,0.0004594841],"domain_scores_gemma":[0.9823858,0.005130603,0.0035333,0.001126627,0.005375386,0.002448256],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001246356,0.000916779,0.8907971,0.0001936597,0.0002313135,0.000726285,0.01623071,0.0003355162,0.006588715,0.0001413776,0.0008705907,0.08172166],"study_design_scores_gemma":[0.00002072463,0.002023329,0.9901692,0.00004284433,0.00002858264,0.000324649,0.004602883,0.000367163,0.001226371,0.00005707761,0.001106904,0.00003034406],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9986506,0.0001471693,0.0001518747,0.00003147972,0.0000094507,0.0000171694,0.00004845327,0.00001269393,0.000931146],"genre_scores_gemma":[0.9982286,0.0001124334,0.0003089197,0.00002824575,0.00001032267,0.00002629534,0.0001681699,0.000008257523,0.001108672],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004484841,"threshold_uncertainty_score":0.01797199,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03000706054974718,"score_gpt":0.2756809433892808,"score_spread":0.2456738828395336,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}