{"id":"W2109163800","doi":"10.1145/1187415.1187417","title":"A statistical model for near-synonym choice","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Synonym (taxonomy); Artificial intelligence; Natural language processing; Task (project management); Context (archaeology); Machine translation; Thesaurus; Mutual information; Information retrieval; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007710207,0.0008869519,0.001535097,0.004647051,0.001216191,0.002365637,0.003732417,0.002097383,0.006610387],"category_scores_gemma":[0.02908325,0.0009377501,0.001802021,0.004780365,0.002132369,0.006543473,0.002222133,0.003029281,0.002916869],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001451034,"about_ca_system_score_gemma":0.001957068,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00510651,"about_ca_topic_score_gemma":0.008466461,"domain_scores_codex":[0.9942364,0.002718104,0.0003914164,0.001274879,0.001098008,0.0002813216],"domain_scores_gemma":[0.9796816,0.01514949,0.001525214,0.001619487,0.001658699,0.0003654337],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005407126,0.0005336883,0.01318171,0.0003272263,0.0006339837,0.0007943607,0.0009751529,0.3685607,0.006033978,0.3656934,0.01271676,0.2300084],"study_design_scores_gemma":[0.00002678683,0.00004102195,0.0008477432,0.00001400667,0.00002549104,0.0001286284,0.00002777371,0.8600659,0.0004930675,0.136903,0.001387926,0.00003872755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01697694,0.000212936,0.9791524,0.000506069,0.00006433646,0.00009699557,0.0004963054,0.0005595031,0.001934551],"genre_scores_gemma":[0.5934439,0.0005410455,0.3913545,0.000535332,0.0004672578,0.001017406,0.002852882,0.0005227919,0.00926476],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007710207,"threshold_uncertainty_score":0.04077595,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01971804255985616,"score_gpt":0.3185797563200914,"score_spread":0.2988617137602353,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}