{"id":"W2508129077","doi":"10.20381/ruor-4638","title":"An Unsupervised Approach to Detecting and Correcting Errors in Text","year":2011,"lang":"en","type":"dissertation","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Error detection and correction; Word (group theory); Spelling; Verb; Punctuation; Algorithm; Mathematics; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004081478,0.001330013,0.001282412,0.005293638,0.001517333,0.003181606,0.003781241,0.001956919,0.00192074],"category_scores_gemma":[0.02150302,0.0008207933,0.001826805,0.003011303,0.003118807,0.004251657,0.002390998,0.001897944,0.002374539],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001419063,"about_ca_system_score_gemma":0.003062684,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002968593,"about_ca_topic_score_gemma":0.004468569,"domain_scores_codex":[0.9890036,0.002984411,0.0009855777,0.003193162,0.003547455,0.0002857947],"domain_scores_gemma":[0.9672687,0.01311607,0.004250378,0.007108479,0.007947802,0.0003086304],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002136591,0.0003865247,0.0102551,0.001010945,0.0003418693,0.0005008299,0.002363799,0.04928382,0.07456814,0.03867341,0.008611588,0.8137904],"study_design_scores_gemma":[0.00005086121,0.0003523838,0.009306198,0.0002718641,0.0002372216,0.001760971,0.001346093,0.7019517,0.1389341,0.09881797,0.04674817,0.0002225226],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00723768,0.0001289433,0.9886155,0.0002797268,0.00004745029,0.0002236238,0.0001596338,0.001958352,0.001349077],"genre_scores_gemma":[0.08363353,0.0002732844,0.9094364,0.0002674017,0.0001367584,0.0003889954,0.0007566409,0.0005039721,0.004602963],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005293638,"threshold_uncertainty_score":0.02158517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0052972131944227,"score_gpt":0.1757599231479749,"score_spread":0.1704627099535522,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}