{"id":"W3042496857","doi":"","title":"Dealing with specialized co-text in text mining: The verbal terminological collocations","year":2019,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"linguistics and terminology studies","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Computer science; Natural language processing; Co-occurrence; Artificial intelligence; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001554611,0.0003023468,0.0004241813,0.0001485536,0.0007048114,0.0004900179,0.0009064553,0.0001738171,0.0004953636],"category_scores_gemma":[0.0008140275,0.0002182983,0.0001178777,0.00007335445,0.001004969,0.00004444426,0.0007057238,0.000584131,0.0000702414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008051049,"about_ca_system_score_gemma":0.0002072807,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001208961,"about_ca_topic_score_gemma":0.01399954,"domain_scores_codex":[0.9971057,0.001287312,0.0004643204,0.0005503671,0.0002601802,0.000332183],"domain_scores_gemma":[0.9955595,0.001737719,0.0003915922,0.001176104,0.001079273,0.00005587392],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002975323,0.0003377656,0.007829942,0.00007668573,0.0001506477,0.00001709561,0.08502443,0.00005197755,0.00001519783,0.8992237,0.001973363,0.005269375],"study_design_scores_gemma":[0.002799644,0.000005851147,0.02271588,0.00338376,0.0002993554,0.00002570338,0.01111055,0.006769517,0.001340805,0.01705137,0.9329145,0.00158309],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.359427,0.0009840667,0.0008055049,0.01153656,0.000691291,0.0008143762,0.0001324808,0.0001568116,0.6254519],"genre_scores_gemma":[0.9785694,0.0003022322,0.001350736,0.0001754007,0.0001339588,0.0000998855,0.0001831734,0.00003112809,0.01915408],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9309411,"threshold_uncertainty_score":0.8901948,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04332405443573575,"score_gpt":0.2524065320160365,"score_spread":0.2090824775803008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}