{"id":"W3042496857","doi":"","title":"Dealing with specialized co-text in text mining: The verbal terminological collocations","year":2019,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"linguistics and terminology studies","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Computer science; Natural language processing; Co-occurrence; Artificial intelligence; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00751991,0.001249979,0.001628491,0.01121496,0.003207013,0.006412188,0.002750867,0.003277883,0.003191092],"category_scores_gemma":[0.03804895,0.0008882966,0.001744711,0.01917778,0.002731492,0.008334344,0.004934659,0.002942452,0.002275888],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008164999,"about_ca_system_score_gemma":0.003635932,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001569081,"about_ca_topic_score_gemma":0.003098398,"domain_scores_codex":[0.9841927,0.00721576,0.002261383,0.003023397,0.002750995,0.0005557167],"domain_scores_gemma":[0.9446377,0.03917318,0.003857165,0.007183292,0.004367407,0.0007811848],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001086998,0.0006317798,0.02840247,0.005185878,0.000847816,0.004434839,0.007669456,0.01419921,0.04506422,0.08955333,0.01891091,0.7840132],"study_design_scores_gemma":[0.0002501614,0.0003699163,0.02109205,0.001833922,0.001331262,0.0104545,0.009580646,0.4607556,0.09171478,0.2962208,0.1060267,0.00036958],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1109186,0.005528152,0.8741509,0.001571628,0.0004851664,0.0002901405,0.00258686,0.001092309,0.003376339],"genre_scores_gemma":[0.5303663,0.002906048,0.4510272,0.0003429137,0.0007561132,0.0004340074,0.009825859,0.0009390697,0.003402513],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01121496,"threshold_uncertainty_score":0.03976959,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04332405443573575,"score_gpt":0.2524065320160365,"score_spread":0.2090824775803008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}