{"id":"W4376460726","doi":"10.1007/978-3-031-29937-7_9","title":"Language Corpora and Principal Components Analysis","year":2023,"lang":"en","type":"book-chapter","venue":"Studies in big data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Dawson College; Cegep de Thetford; Cegep de Trois-Rivieres; College Ahuntsic; Université du Québec à Montréal; Memorial University of Newfoundland","funders":"","keywords":"Principal (computer security); Computer science; Natural language processing; Linguistics; Principal component analysis; Artificial intelligence; History; Philosophy; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001831405,0.001150138,0.0009261122,0.004024152,0.0009692273,0.003616561,0.001217084,0.0009035462,0.02447671],"category_scores_gemma":[0.009469366,0.00087836,0.0005055165,0.0109428,0.001742702,0.004048512,0.001342054,0.001773266,0.01175638],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008433215,"about_ca_system_score_gemma":0.001063198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002500099,"about_ca_topic_score_gemma":0.004166584,"domain_scores_codex":[0.9984515,0.0006938508,0.00008825296,0.0001750189,0.000549884,0.00004157071],"domain_scores_gemma":[0.9955037,0.003197351,0.000139774,0.0005383386,0.0005701049,0.00005069892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002306754,0.00003594294,0.0003040685,0.000554255,0.00004222028,0.00008035661,0.0002912974,0.003569983,0.0006012446,0.2740839,0.1770854,0.5433283],"study_design_scores_gemma":[0.000007894937,0.00001149664,0.001456216,0.0002867752,0.00002344685,0.0002043044,0.0002501331,0.01369143,0.001455869,0.5331755,0.4493999,0.00003699854],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003100686,0.06775218,0.7398285,0.006445093,0.003116023,0.0001673828,0.003550815,0.005604528,0.1704348],"genre_scores_gemma":[0.05469197,0.06493549,0.6701337,0.001478077,0.004667364,0.0009221291,0.00956269,0.004751026,0.1888575],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02447671,"threshold_uncertainty_score":0.08188272,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2423355936847015,"score_gpt":0.37504984459207,"score_spread":0.1327142509073685,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}