{"id":"W2251547673","doi":"10.63317/54n4hrwwu9u9","title":"Measuring Interlanguage: Native Language Identification with L1-influence Metrics","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Bigram; Natural language processing; Artificial intelligence; Interlanguage; Identification (biology); Language identification; Word (group theory); Machine translation; Task (project management); Natural language; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005813814,0.0001485232,0.0001230949,0.0002852207,0.00009292082,0.0002382357,0.0009293006,0.00005591891,0.0000140498],"category_scores_gemma":[0.0002886885,0.0001067899,0.00003000238,0.001102767,0.00004222977,0.002270289,0.0002266124,0.0001998934,0.00004633973],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001024213,"about_ca_system_score_gemma":0.00002968245,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001023067,"about_ca_topic_score_gemma":0.00001389714,"domain_scores_codex":[0.9987151,0.00005646607,0.000188401,0.0002811488,0.0004302711,0.0003286014],"domain_scores_gemma":[0.9989484,0.00008718511,0.0001376272,0.0005212665,0.000203332,0.0001022171],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00004164922,0.0004257897,0.01556816,0.000232651,0.0001355357,0.000110896,0.05876347,0.00002324281,0.185476,0.341074,0.001023111,0.3971255],"study_design_scores_gemma":[0.0002322329,0.0000692258,0.005070806,0.00009379345,0.00001812491,0.0001115084,0.0008973199,0.001940423,0.9889123,0.001916889,0.000208811,0.0005285344],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0945242,0.003179981,0.8990776,0.0001651015,0.00009904506,0.0001660591,9.92933e-7,0.001057397,0.001729649],"genre_scores_gemma":[0.7091554,0.000002658803,0.2901851,0.0001816266,0.00003619838,0.0000167645,0.000001560308,0.000009044616,0.0004116652],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8034363,"threshold_uncertainty_score":0.4354766,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01889991639026808,"score_gpt":0.2762415163566559,"score_spread":0.2573415999663878,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}