{"id":"W3176075285","doi":"10.1145/3462741.3466809","title":"Bigger Isn’t Better: The Ethical and Scientific Vices of Extra-Large Datasets in Language Models","year":2021,"lang":"en","type":"article","venue":"","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"Social Sciences and Humanities Research Council of Canada; Mitacs; Compute Canada","keywords":"Computer science; Quality (philosophy); Data science; Language model; Control (management); Artificial intelligence; Epistemology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009637202,0.00007985171,0.0001221528,0.00006497689,0.0001366234,0.0003178831,0.0003605446,0.00007439192,0.00002912078],"category_scores_gemma":[0.0000327352,0.00005484086,0.00002674486,0.0004013495,0.0001226788,0.0002281917,0.0003790212,0.0002515049,0.000004771288],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000006871153,"about_ca_system_score_gemma":0.00005924653,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008219638,"about_ca_topic_score_gemma":0.0003769427,"domain_scores_codex":[0.9988528,0.0001307618,0.0001844667,0.0003681253,0.0002437847,0.000220062],"domain_scores_gemma":[0.9990262,0.0001650391,0.00003902692,0.0006852451,0.00003902386,0.00004548486],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002030644,0.0007120224,0.003687518,0.0003958187,0.00008460062,0.0008591615,0.0717817,0.002261778,0.2783146,0.3597086,0.0424386,0.2397353],"study_design_scores_gemma":[0.0007711657,0.00002804743,0.001970477,0.000177388,0.00001413804,0.0001207957,0.002604019,0.9042178,0.0715297,0.00513975,0.01304053,0.0003861357],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8673214,0.001202165,0.1246624,0.005220421,0.0001965599,0.00007738517,0.00003191471,0.00004683474,0.001240962],"genre_scores_gemma":[0.9924411,0.000007263673,0.005984035,0.001252169,0.000015545,0.000002017155,0.00001524149,0.000004055613,0.00027862],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9019561,"threshold_uncertainty_score":0.3065354,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01724107290708424,"score_gpt":0.26430836574723,"score_spread":0.2470672928401458,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}