{"id":"W1560709708","doi":"10.1007/3-540-45153-6_4","title":"A Statistical Corpus-Based Term Extractor","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Perplexity; Extractor; Term (time); Natural language processing; Artificial intelligence; Lexicography; Language model; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001992461,0.001767896,0.002103263,0.009937399,0.001261889,0.002213882,0.002091055,0.00117402,0.01685638],"category_scores_gemma":[0.006690515,0.0009914383,0.00173321,0.009513778,0.000493729,0.003358053,0.002250558,0.001983907,0.02526328],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005581779,"about_ca_system_score_gemma":0.003146542,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005511184,"about_ca_topic_score_gemma":0.01046705,"domain_scores_codex":[0.9983918,0.0002030962,0.0002278886,0.0004284651,0.0006483595,0.0001002823],"domain_scores_gemma":[0.9960024,0.001374208,0.0002292987,0.0005838537,0.001654111,0.0001561546],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003968746,0.000141907,0.001292201,0.0006241666,0.0001829156,0.0002286129,0.00009974767,0.002513322,0.05209751,0.00305984,0.05755585,0.881807],"study_design_scores_gemma":[0.0005014559,0.0007514829,0.01134517,0.0003260416,0.001065213,0.002554473,0.0006176247,0.4936804,0.1835856,0.02459253,0.2805184,0.0004615422],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01059288,0.0017741,0.8878135,0.0003614252,0.0006013984,0.0006038201,0.02101155,0.07293578,0.004305578],"genre_scores_gemma":[0.02867215,0.0009324369,0.9095166,0.0001820601,0.0003373313,0.0007560346,0.04711529,0.002419639,0.01006848],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01685638,"threshold_uncertainty_score":0.05639017,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01606606083387984,"score_gpt":0.2780418839368485,"score_spread":0.2619758231029687,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}