{"id":"W2909407853","doi":"10.22215/etd/2018-12972","title":"A Corpus-Based Investigation of Academic Vocabulary and Phrasal Verbs in Academic Spoken English","year":2018,"lang":"en","type":"dissertation","venue":"","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"College of Pharmacy, University of Michigan; University of Warwick; Arts and Humanities Research Council; British Academy; University of Michigan","keywords":"Vocabulary; Noun; British National Corpus; Linguistics; Computer science; Corpus linguistics; Natural language processing; Verb; Word list; Artificial intelligence; Word (group theory); Part of speech","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0005485618,0.0002627289,0.0003932814,0.0005529287,0.00003724716,0.00001078823,0.0002119187,0.001325374,0.0157726],"category_scores_gemma":[0.0002485556,0.000274372,0.00006311519,0.000304356,0.0001103262,0.0001131303,0.00001512476,0.001474043,0.0000886269],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004053771,"about_ca_system_score_gemma":0.0001302977,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002637477,"about_ca_topic_score_gemma":0.00009267888,"domain_scores_codex":[0.9981749,0.0002394273,0.0006126461,0.000479144,0.000222817,0.0002710586],"domain_scores_gemma":[0.998888,0.0002076642,0.0004242181,0.0002430417,0.0001276744,0.0001094192],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.006383202,0.0002421002,0.2860965,0.00239813,0.0005888955,0.0002864891,0.4819196,0.00002782655,0.05820367,0.008776681,0.05951776,0.09555914],"study_design_scores_gemma":[0.005969913,0.0005557499,0.9141976,0.001593943,0.0002140687,0.00002001263,0.04695452,0.0003477536,0.01433392,0.0009207471,0.01361541,0.001276361],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9741429,0.008561707,0.0000110085,0.00003774527,0.00100899,0.000322119,0.00001327367,0.0000908048,0.01581147],"genre_scores_gemma":[0.9839674,0.00009580777,0.0001833753,0.00375345,0.0006239588,0.00006642563,0.002238947,0.00007281105,0.008997825],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6281011,"threshold_uncertainty_score":0.9999711,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02221661586677118,"score_gpt":0.336204979187352,"score_spread":0.3139883633205808,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}