{"id":"W2118337010","doi":"10.1075/ijcl.10.2.05lem","title":"Two methods for extracting “specific” single-word terms from specialized corpora","year":2005,"lang":"en","type":"article","venue":"International Journal of Corpus Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Word (group theory); Focus (optics); Term (time); Precision and recall; Noun phrase; Recall; Field (mathematics); Noun; Linguistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005126683,0.001319204,0.001432201,0.014356,0.001173268,0.002536035,0.001561218,0.00114636,0.005834443],"category_scores_gemma":[0.02972778,0.0008680384,0.001471969,0.01476644,0.001292303,0.003570419,0.002457273,0.001183036,0.002491993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001016532,"about_ca_system_score_gemma":0.001826582,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002251266,"about_ca_topic_score_gemma":0.00502927,"domain_scores_codex":[0.9939601,0.00127506,0.001092515,0.001753697,0.001698157,0.0002204086],"domain_scores_gemma":[0.9867976,0.004253006,0.001738119,0.003357521,0.003550117,0.0003036679],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005625097,0.0001795704,0.009317426,0.0009573276,0.0003174276,0.0003216084,0.002137924,0.003250317,0.04004342,0.01188041,0.005461734,0.9255704],"study_design_scores_gemma":[0.0008954068,0.001831422,0.1590142,0.0005442246,0.001465403,0.01038955,0.005349222,0.2666615,0.2379858,0.0602417,0.2547033,0.0009183566],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03662081,0.0007078954,0.9525789,0.0001741343,0.0001496058,0.001385854,0.001413311,0.003601949,0.003367472],"genre_scores_gemma":[0.03261529,0.0001930131,0.9623982,0.00003177782,0.00004261302,0.001122404,0.002032875,0.0003079086,0.001256066],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.014356,"threshold_uncertainty_score":0.02711284,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04948392902264137,"score_gpt":0.3942393048380948,"score_spread":0.3447553758154535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}