{"id":"W2251370914","doi":"","title":"Robust, Lexicalized Native Language Identification","year":2012,"lang":"en","type":"article","venue":"","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Classifier (UML); Training set; Task (project management); Identification (biology); Set (abstract data type); Language identification; Test set; Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006571268,0.001225091,0.001496895,0.002713992,0.00118603,0.003971785,0.001905279,0.002122818,0.006804669],"category_scores_gemma":[0.02881436,0.0004464128,0.0005445026,0.001253026,0.000922802,0.005986751,0.005574277,0.001662388,0.0103405],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004909578,"about_ca_system_score_gemma":0.001208031,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001288412,"about_ca_topic_score_gemma":0.001943539,"domain_scores_codex":[0.9901154,0.00337622,0.0006865365,0.003076663,0.002252609,0.0004925265],"domain_scores_gemma":[0.9843875,0.005176863,0.0008599926,0.004834547,0.004300371,0.0004406546],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008668639,0.000781297,0.01802954,0.0009459354,0.0002756432,0.0006974015,0.00245416,0.01209038,0.144017,0.00706928,0.02851664,0.784256],"study_design_scores_gemma":[0.0002445332,0.001101195,0.07847854,0.0003571915,0.0002459796,0.003776429,0.00569536,0.5561177,0.2518595,0.05046263,0.05107416,0.0005868498],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4496232,0.00142132,0.4876615,0.00073397,0.0007829505,0.0009320139,0.006463311,0.02199879,0.03038307],"genre_scores_gemma":[0.78747,0.0002248843,0.1937339,0.0003024923,0.0001682323,0.0005018577,0.008319804,0.001238031,0.0080408],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006804669,"threshold_uncertainty_score":0.03475255,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0716369537548322,"score_gpt":0.3186020371403878,"score_spread":0.2469650833855556,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}