{"id":"W2121479142","doi":"","title":"Collocation Extraction for Machine Translation","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Collocation (remote sensing); Computer science; Natural language processing; Translation (biology); Artificial intelligence; Machine translation; Embedding; Information extraction; Extraction (chemistry); Machine translation software usability; Rule-based machine translation; Example-based machine translation; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002058039,0.001249849,0.0009674121,0.003307879,0.001897305,0.00322559,0.001417869,0.001402291,0.03491938],"category_scores_gemma":[0.008089895,0.0006955874,0.0009132783,0.004436946,0.001111176,0.004450906,0.002498799,0.001618229,0.02547152],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001286062,"about_ca_system_score_gemma":0.001436379,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002565479,"about_ca_topic_score_gemma":0.002325825,"domain_scores_codex":[0.9967268,0.001439003,0.0003331258,0.0006448875,0.0007345557,0.0001216526],"domain_scores_gemma":[0.9964881,0.001148055,0.0002193987,0.001123769,0.0009500446,0.00007055403],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001356862,0.000042555,0.000440701,0.001209766,0.0001159127,0.0006317587,0.0006108735,0.004780186,0.01806661,0.1631291,0.08832145,0.7225154],"study_design_scores_gemma":[0.00004610462,0.00005893514,0.001044075,0.0004288381,0.00007677265,0.001331838,0.0003847067,0.06681004,0.02606212,0.2686687,0.6349491,0.0001386941],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002179629,0.004861423,0.9554985,0.001330808,0.0008646075,0.0002105534,0.001452125,0.009943677,0.02365871],"genre_scores_gemma":[0.08743307,0.004612369,0.8795481,0.0005411144,0.0006810057,0.0003791152,0.00582173,0.001969275,0.01901433],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03491938,"threshold_uncertainty_score":0.1168169,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02199688784710613,"score_gpt":0.3105580019753942,"score_spread":0.2885611141282881,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}