{"id":"W4408274262","doi":"10.5430/wjel.v15n5p1","title":"Trustworthiness of EFL Assessment of Learning in the Age of AI: Challenges and Solutions","year":2025,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Foreign Language Teaching Methods","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Trustworthiness; Computer science; Mathematics education; Artificial intelligence; Natural language processing; Psychology; Computer security","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1224782,0.0004673421,0.00117487,0.004161728,0.00455556,0.01062178,0.00240908,0.001954541,0.001700174],"category_scores_gemma":[0.3962717,0.0005537291,0.0004680886,0.001829225,0.008808893,0.01003422,0.009083878,0.003003809,0.0005004315],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004911005,"about_ca_system_score_gemma":0.009025332,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007379278,"about_ca_topic_score_gemma":0.004072557,"domain_scores_codex":[0.8012644,0.1159853,0.01657346,0.009242826,0.05336348,0.003570535],"domain_scores_gemma":[0.4967739,0.3126718,0.06702031,0.02830915,0.08707197,0.008152883],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0005502928,0.0002939857,0.2564041,0.001273429,0.0001581922,0.001300185,0.4552253,0.0009826074,0.002936841,0.01710482,0.002956863,0.2608134],"study_design_scores_gemma":[0.00006313415,0.001030785,0.1512628,0.006723227,0.0001519947,0.002775857,0.6828358,0.01557875,0.009879597,0.07182542,0.05731379,0.0005588991],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8814672,0.00362178,0.06457691,0.01882646,0.0005136087,0.0004217502,0.0001665885,0.0002121274,0.03019355],"genre_scores_gemma":[0.9932733,0.0003062518,0.005203652,0.0003435239,0.00004405698,0.00009317287,0.00002792847,0.00002742995,0.0006806773],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1224782,"threshold_uncertainty_score":0.6477344,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02873283754247458,"score_gpt":0.3710610737244341,"score_spread":0.3423282361819596,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}