{"id":"W4281688623","doi":"10.1075/tlrp.23.13mar","title":"Knowledge patterns in corpora","year":2022,"lang":"en","type":"book-chapter","venue":"Terminology and lexicography research and practice","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Paralanguage; Identification (biology); Computer science; Relation (database); Natural language processing; Artificial intelligence; Linguistics; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["research_integrity"],"consensus_categories":[],"category_scores_codex":[0.002062277,0.0002417681,0.0003056299,0.001445275,0.0003269186,0.0001965228,0.0007990206,0.000379941,0.0001082687],"category_scores_gemma":[0.0002867593,0.0002305035,0.00004571522,0.0002743143,0.000564572,0.0007711477,0.001412313,0.002692515,0.00000884704],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003309076,"about_ca_system_score_gemma":0.0001350667,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001203819,"about_ca_topic_score_gemma":0.00006426578,"domain_scores_codex":[0.9977569,0.0004498327,0.0002497508,0.0007620106,0.0003193057,0.0004621861],"domain_scores_gemma":[0.9967216,0.002286111,0.0001647013,0.0005285717,0.0001725901,0.0001264839],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000108965,0.00007154088,0.0004524778,0.0001169706,0.00003692933,0.001042708,0.0006277579,6.320933e-9,0.00002159496,0.8012108,0.0008525749,0.1954577],"study_design_scores_gemma":[0.0003169295,0.0009279778,0.0004318485,0.0001372273,0.00001422135,0.001278158,0.00006189814,0.00002595792,0.00003597833,0.5074229,0.4889665,0.0003804491],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.001134131,0.2038887,0.002938739,0.009651118,0.0004160647,0.001248632,0.00005444182,0.0006594224,0.7800088],"genre_scores_gemma":[0.4382157,0.1264185,0.1523104,0.004124388,0.0008199189,0.0007528672,0.00020107,0.0002970195,0.2768601],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5031487,"threshold_uncertainty_score":0.9996083,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1228648343836937,"score_gpt":0.4049428103849939,"score_spread":0.2820779760013001,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}