{"id":"W2952197960","doi":"","title":"Word Sense Disambiguation by Web Mining for Word Co-occurrence Probabilities","year":2004,"lang":"en","type":"preprint","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"University of Waterloo","keywords":"Computer science; Natural language processing; SemEval; Word (group theory); Artificial intelligence; Feature (linguistics); Task (project management); Novelty; Sample (material); Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005528465,0.0003863419,0.0003845411,0.0001605735,0.0001656065,0.0005667363,0.001294355,0.0003402811,0.00001678615],"category_scores_gemma":[0.000421984,0.0003720193,0.0001471235,0.0002199957,0.0001367288,0.0003933404,0.0008132595,0.000555574,0.000005915375],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003197964,"about_ca_system_score_gemma":0.0006012723,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001296781,"about_ca_topic_score_gemma":0.000008054766,"domain_scores_codex":[0.997567,0.0000805955,0.0004381789,0.0009876916,0.0004647377,0.0004618241],"domain_scores_gemma":[0.9981087,0.0002444502,0.0003626806,0.0009410395,0.0002429761,0.0001001048],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001069242,0.0003720066,0.0002023413,0.00363692,0.0001027323,0.00003844719,0.011042,0.0002578971,0.02019453,0.04480117,0.0956535,0.8235915],"study_design_scores_gemma":[0.00050522,0.0001155185,0.00001584146,0.001698352,0.00003522121,0.00002223375,0.00008117068,0.03223334,0.03394689,0.9250306,0.005127985,0.001187616],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01677533,0.001583226,0.973913,0.002221609,0.0008129344,0.001491089,0.0004602764,0.001866509,0.0008759979],"genre_scores_gemma":[0.3267058,0.00003330076,0.6719239,0.0001911622,0.000114872,0.0004475719,0.0002876417,0.00002119339,0.0002745997],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8802295,"threshold_uncertainty_score":0.9998732,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02588644249646029,"score_gpt":0.3028768655376259,"score_spread":0.2769904230411656,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}