{"id":"W4385571788","doi":"10.18653/v1/2023.acl-long.734","title":"The Elephant in the Room: Analyzing the Presence of Big Tech in Natural Language Processing Research","year":2023,"lang":"en","type":"article","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; Institute for Work & Health","funders":"LG Display; Deutscher Akademischer Austauschdienst; Niedersächsische Ministerium für Wissenschaft und Kultur","keywords":"Computer science; Georgia tech; Volume (thermodynamics); Library science; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.004776802,0.00045071,0.0005776621,0.004418871,0.001901875,0.004515755,0.001282161,0.001485525,0.002334166],"category_scores_gemma":[0.03541942,0.0004025054,0.0004655868,0.006546438,0.001626453,0.01057286,0.004974999,0.001464768,0.0008999204],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005805209,"about_ca_system_score_gemma":0.0007971028,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002597148,"about_ca_topic_score_gemma":0.009252749,"domain_scores_codex":[0.9968412,0.00170465,0.0001964208,0.0005515402,0.0005354297,0.000170723],"domain_scores_gemma":[0.966924,0.02408347,0.003206388,0.002412501,0.001781613,0.001592093],"domain_codex":null,"domain_gemma":"incentives","domain_candidate":"incentives","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001335058,0.0003230677,0.5240245,0.00172183,0.0006512239,0.002005633,0.040787,0.001421612,0.004538236,0.02912075,0.07945997,0.314611],"study_design_scores_gemma":[0.0001389362,0.000401212,0.4934318,0.00153712,0.0009507514,0.002537807,0.1092163,0.04996421,0.004986852,0.2099593,0.1265848,0.0002909902],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8561078,0.02344603,0.05854044,0.03167535,0.001234748,0.0002999479,0.007821446,0.0008258069,0.02004842],"genre_scores_gemma":[0.946843,0.002818181,0.04002988,0.001572903,0.0006739712,0.000287002,0.005038029,0.000198599,0.002538395],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9955812,"threshold_uncertainty_score":0.02526248,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1197683283919852,"score_gpt":0.4937736672440102,"score_spread":0.374005338852025,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}