{"id":"W2021254338","doi":"10.1145/1871840.1871841","title":"The nature of noise in linguistic corpora","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Variety (cybernetics); Computer science; Natural language processing; Artificial intelligence; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04327562,0.0008418169,0.002195449,0.009427086,0.003109116,0.008865211,0.003455739,0.003095215,0.002960949],"category_scores_gemma":[0.2956412,0.001741741,0.0008492388,0.01305538,0.006895758,0.009413852,0.004849953,0.004131609,0.002420773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003053995,"about_ca_system_score_gemma":0.00250986,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008701993,"about_ca_topic_score_gemma":0.009605628,"domain_scores_codex":[0.922444,0.02934568,0.004366784,0.0117961,0.03060464,0.001442778],"domain_scores_gemma":[0.7148896,0.2296735,0.009101704,0.02467544,0.02086416,0.0007954656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007268527,0.00023437,0.05404169,0.002656183,0.001002285,0.002270435,0.01526596,0.02802664,0.01034683,0.3280051,0.1047666,0.452657],"study_design_scores_gemma":[0.0001281159,0.000128711,0.05043843,0.001472208,0.0003324878,0.002862338,0.004336492,0.1081387,0.01029569,0.6399316,0.1816261,0.0003090832],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1104925,0.01543517,0.7870433,0.03040725,0.003276735,0.0004459822,0.00644474,0.004031904,0.04242241],"genre_scores_gemma":[0.755438,0.007155419,0.1864695,0.01613009,0.004536275,0.001426656,0.01415523,0.002376624,0.0123122],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04327562,"threshold_uncertainty_score":0.2288661,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.005712326292142829,"score_gpt":0.2647338117014542,"score_spread":0.2590214854093114,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}