{"id":"W2118337010","doi":"10.1075/ijcl.10.2.05lem","title":"Two methods for extracting “specific” single-word terms from specialized corpora","year":2005,"lang":"en","type":"article","venue":"International Journal of Corpus Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Word (group theory); Focus (optics); Term (time); Precision and recall; Noun phrase; Recall; Field (mathematics); Noun; Linguistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008607631,0.0001868573,0.0003102064,0.0002744747,0.00007368414,0.0005050654,0.001967699,0.00008464156,0.00002471075],"category_scores_gemma":[0.004652236,0.0001684014,0.0001957436,0.0001344984,0.00005097329,0.000286588,0.0001910534,0.0003294242,0.000003994618],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002559673,"about_ca_system_score_gemma":0.0001131993,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001548458,"about_ca_topic_score_gemma":0.000008131463,"domain_scores_codex":[0.9980213,0.00008868048,0.0008717804,0.0002427996,0.0005629429,0.0002124951],"domain_scores_gemma":[0.9946083,0.001167128,0.001375128,0.0002507347,0.002487405,0.0001113233],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001545249,0.0001718382,0.000242546,0.000004507258,0.00009217959,0.0001419947,0.000336758,0.0000586352,0.02584953,0.1277813,0.001392063,0.8437741],"study_design_scores_gemma":[0.001362421,0.0001054283,0.00005504772,0.0002166799,0.00003237859,0.0002184605,0.00001340941,0.009996871,0.124115,0.2809388,0.5826383,0.0003072126],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0009768362,0.002181303,0.9878409,0.0007980194,0.007183499,0.00009788697,0.00001510908,0.000114951,0.0007915175],"genre_scores_gemma":[0.1397375,0.00005226559,0.8503307,0.0003554514,0.009418515,0.00000221223,0.000006684696,0.00002066445,0.00007597752],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8434669,"threshold_uncertainty_score":0.6867208,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04948392902264137,"score_gpt":0.3942393048380948,"score_spread":0.3447553758154535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}