{"id":"W4372279189","doi":"10.48550/arxiv.2305.02797","title":"The Elephant in the Room: Analyzing the Presence of Big Tech in Natural Language Processing Research","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; Institute for Work & Health","funders":"Deutscher Akademischer Austauschdienst; Niedersächsische Ministerium für Wissenschaft und Kultur","keywords":"Transparency (behavior); Artificial intelligence; Internship; Metadata; Computer science; Field (mathematics); Natural language processing; Corpus linguistics; Data science; Political science; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.007938411,0.000263938,0.0004203732,0.01690538,0.002205216,0.00627368,0.0007667734,0.001285519,0.002354212],"category_scores_gemma":[0.06884538,0.0003586727,0.0003357178,0.02539674,0.002321897,0.01080349,0.005118496,0.001160565,0.001267216],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001170766,"about_ca_system_score_gemma":0.001671075,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003694524,"about_ca_topic_score_gemma":0.01113831,"domain_scores_codex":[0.9931201,0.002210846,0.0009128033,0.001241975,0.002056804,0.0004575613],"domain_scores_gemma":[0.8737059,0.07412365,0.02988932,0.006946545,0.009945864,0.005388805],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003424135,0.0001676159,0.8290266,0.001126007,0.0001482424,0.0009650772,0.03241622,0.0006002653,0.003110344,0.01236371,0.01664361,0.1030899],"study_design_scores_gemma":[0.00003906984,0.0002085167,0.7762397,0.0008085991,0.0001945004,0.001434831,0.04984932,0.006529465,0.003279488,0.0188917,0.1423788,0.0001459758],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9496217,0.006904485,0.009149914,0.005115591,0.0002630168,0.00008663935,0.006136207,0.0002633723,0.0224591],"genre_scores_gemma":[0.978109,0.002664832,0.008671574,0.0006613411,0.000518624,0.0001383522,0.006415436,0.0001567136,0.002664041],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9920616,"threshold_uncertainty_score":0.04198283,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1539755596109414,"score_gpt":0.2841513684074624,"score_spread":0.130175808796521,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}