{"id":"W4281688623","doi":"10.1075/tlrp.23.13mar","title":"Knowledge patterns in corpora","year":2022,"lang":"en","type":"book-chapter","venue":"Terminology and lexicography research and practice","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Paralanguage; Identification (biology); Computer science; Relation (database); Natural language processing; Artificial intelligence; Linguistics; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004146311,0.0004848447,0.0006350235,0.01077921,0.001943907,0.005937232,0.001662345,0.001319644,0.01729297],"category_scores_gemma":[0.02133663,0.0006260647,0.0004000186,0.02307184,0.001853025,0.007203036,0.002624573,0.001435195,0.005150616],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002313607,"about_ca_system_score_gemma":0.002226848,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004941222,"about_ca_topic_score_gemma":0.00709078,"domain_scores_codex":[0.9945256,0.001989479,0.0005853054,0.001065005,0.001705326,0.0001293526],"domain_scores_gemma":[0.9820494,0.01104357,0.0008251215,0.003422536,0.00240304,0.000256322],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008433034,0.00007801937,0.003915953,0.001520113,0.00005544629,0.001001472,0.003359918,0.004227831,0.002874744,0.2694606,0.1124306,0.600991],"study_design_scores_gemma":[0.00002284182,0.00002244281,0.003283184,0.0009432133,0.00002979817,0.0008581072,0.002043468,0.01382938,0.004073372,0.1144371,0.8604153,0.00004172298],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05907661,0.01660603,0.4878995,0.01424996,0.002397595,0.001390652,0.05038494,0.008584704,0.3594101],"genre_scores_gemma":[0.2047411,0.01152275,0.6411926,0.001562036,0.0006348868,0.001533548,0.07660954,0.002454226,0.05974934],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01729297,"threshold_uncertainty_score":0.05785078,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1228648343836937,"score_gpt":0.4049428103849939,"score_spread":0.2820779760013001,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}