{"id":"W2250996967","doi":"","title":"Cognate and Misspelling Features for Natural Language Identification","year":2013,"lang":"en","type":"article","venue":"Workshop on Innovative Use of NLP for Building Educational Applications","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Bigram; Computer science; Cognate; Natural language processing; Artificial intelligence; Classifier (UML); Natural language; Word (group theory); Identification (biology); Spelling; Syntax; Language identification; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002527087,0.001327926,0.0008709689,0.003890466,0.001020371,0.001381296,0.000874899,0.0009793027,0.002749864],"category_scores_gemma":[0.0109757,0.00023774,0.0007656406,0.00193756,0.0004682339,0.002097928,0.001537054,0.00150325,0.002163836],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003309601,"about_ca_system_score_gemma":0.0008284198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00158971,"about_ca_topic_score_gemma":0.002710665,"domain_scores_codex":[0.9971951,0.0009006651,0.0003001769,0.0005909849,0.0007678008,0.0002452282],"domain_scores_gemma":[0.9895887,0.005806613,0.0008951103,0.001398646,0.001700785,0.0006100758],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000848415,0.0008313481,0.04739359,0.0003870119,0.000159598,0.0004180543,0.0005387546,0.008026438,0.04497234,0.001149729,0.007068025,0.8882068],"study_design_scores_gemma":[0.0001519073,0.001221826,0.06688757,0.0001859972,0.0003158285,0.001696663,0.001313319,0.7375987,0.1559478,0.01834087,0.01605993,0.0002794624],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7547152,0.001557654,0.2256594,0.0006205891,0.0002668394,0.0002527125,0.003101665,0.007741486,0.006084489],"genre_scores_gemma":[0.8842067,0.0001697323,0.1094606,0.0000904362,0.00009078635,0.0001442146,0.003975436,0.0002351529,0.00162718],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003890466,"threshold_uncertainty_score":0.01336467,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04016925118674335,"score_gpt":0.347697391551569,"score_spread":0.3075281403648256,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}