{"id":"W2136180489","doi":"","title":"Phrase Clustering for Smoothing TM Probabilities - or, How to Extract Paraphrases from Phrase Tables","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; National Research Council Canada","funders":"","keywords":"Phrase; Smoothing; Computer science; Cluster analysis; Artificial intelligence; Natural language processing; Sentence; Machine translation; Language model; Feature (linguistics); Translation (biology); Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004171662,0.0002652022,0.0002741808,0.0001579708,0.000245537,0.0008115587,0.00131304,0.0001407601,0.0001041872],"category_scores_gemma":[0.001054119,0.0002105598,0.00008981062,0.0002984753,0.00007114667,0.001211521,0.0004103475,0.0003276269,0.000007303687],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004161036,"about_ca_system_score_gemma":0.0001209517,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001168419,"about_ca_topic_score_gemma":0.0005198582,"domain_scores_codex":[0.9982558,0.00004037649,0.0002450531,0.0006618992,0.0003040521,0.0004927508],"domain_scores_gemma":[0.9983096,0.0004170969,0.0001240039,0.0008231504,0.0001369371,0.0001892097],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001304468,0.0001363317,0.0001758672,0.0001824489,0.0000279886,0.00007877628,0.003202664,0.00001370954,0.7162339,0.01222101,0.007018269,0.2605786],"study_design_scores_gemma":[0.0008483875,0.0003474423,0.000081864,0.0004267638,0.00003590744,0.00006907379,0.0003389232,0.04156711,0.5752115,0.3471673,0.03262667,0.001278993],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1276489,0.0002513635,0.8652487,0.004043119,0.0005582515,0.0007429853,0.00007238964,0.001186535,0.000247668],"genre_scores_gemma":[0.4106354,0.000002848045,0.5882475,0.000459104,0.0001806193,0.0001570852,0.000008247445,0.00001922364,0.0002899312],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3349463,"threshold_uncertainty_score":0.8586378,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02216712995711584,"score_gpt":0.2810010237187076,"score_spread":0.2588338937615917,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}