{"id":"W2160547679","doi":"10.3115/1687878.1687898","title":"Reducing the annotation effort for letter-to-phoneme conversion","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Annotation; Classifier (UML); Natural language processing; Artificial intelligence; Cluster analysis; Speech recognition; Training set; Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002487981,0.00006503459,0.0000563806,0.00005367324,0.0001381359,0.0001262731,0.0005384739,0.00003260852,0.000003154811],"category_scores_gemma":[0.00004234567,0.00004058249,0.00002995814,0.0002221942,0.000008840255,0.0003092715,0.00005479931,0.00006566692,0.00001046876],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002595467,"about_ca_system_score_gemma":0.00001523959,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001572633,"about_ca_topic_score_gemma":8.389939e-7,"domain_scores_codex":[0.9994279,0.00001010368,0.00009769625,0.0001947361,0.0001258037,0.0001437792],"domain_scores_gemma":[0.9995474,0.00004529493,0.00003774535,0.0002701564,0.00007055606,0.00002881439],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0000248315,0.00003857301,0.00002345415,0.00002363834,0.000006050541,0.000005251238,0.00225813,0.00004377982,0.1355491,0.08643134,0.2457905,0.5298053],"study_design_scores_gemma":[0.0003379891,0.0004055171,0.0005316076,0.00006734502,0.000007255435,0.00001563131,0.00003104254,0.05256067,0.8582052,0.06033732,0.02715615,0.0003442464],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004966645,0.00008230103,0.9700648,0.02367155,0.000141963,0.0002876893,2.942303e-7,0.0004405,0.0003442353],"genre_scores_gemma":[0.4520878,5.779619e-7,0.5371,0.01044701,0.00007777836,0.000008392095,0.000001303348,0.000002262038,0.0002748094],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7226561,"threshold_uncertainty_score":0.1654906,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01142126957735392,"score_gpt":0.2732580540577467,"score_spread":0.2618367844803928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}