{"id":"W2188556664","doi":"10.1109/wimob.2015.7347988","title":"Microtext normalization using probably-phonetically-similar word discovery","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Spelling; Normalization (sociology); Computer science; Natural language processing; Rendering (computer graphics); Artificial intelligence; Speech recognition; Word (group theory); Pattern matching; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002577451,0.0001317647,0.0001197583,0.00009498325,0.00006514865,0.000476853,0.0007275313,0.00007974677,0.000007404077],"category_scores_gemma":[0.00009532276,0.0001060102,0.0000326687,0.0004854701,0.00004209804,0.001677207,0.0004068393,0.0001103499,0.00002055854],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008754474,"about_ca_system_score_gemma":0.0001479246,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006552283,"about_ca_topic_score_gemma":0.00001109754,"domain_scores_codex":[0.9989145,0.00004647578,0.0001915513,0.0003092109,0.0002860846,0.0002521174],"domain_scores_gemma":[0.9992241,0.00001700144,0.00006861205,0.000413316,0.0001713269,0.0001056264],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00007275648,0.0004406368,0.001895818,0.0001829997,0.00005638029,0.0001999665,0.003961755,0.000820448,0.1825882,0.6269459,0.0241432,0.158692],"study_design_scores_gemma":[0.0007554311,0.0001633351,0.0001049731,0.000196759,0.00002179269,0.000219863,0.00007916484,0.2383053,0.5298517,0.2236466,0.005563246,0.001091872],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006474513,0.000679545,0.9902372,0.0005377794,0.0001669887,0.0001614,5.603502e-7,0.0006347304,0.0011073],"genre_scores_gemma":[0.1747997,0.000003806379,0.8238887,0.0007265002,0.00005172357,0.000004777342,0.000002501693,0.0000101466,0.000512107],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4032993,"threshold_uncertainty_score":0.4598305,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03083235292961977,"score_gpt":0.2786322494743038,"score_spread":0.247799896544684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}