{"id":"W4399362834","doi":"10.1145/3630106.3659029","title":"Fairness Feedback Loops: Training on Synthetic Data Amplifies Bias","year":2024,"lang":"en","type":"article","venue":"","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Redress; Computer science; Stochastic gradient descent; Ground truth; Representation (politics); Machine learning; Performative utterance; Synthetic data; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002281399,0.0001078591,0.0001503989,0.00006858962,0.000593725,0.001231207,0.0006506828,0.0002318566,0.001085734],"category_scores_gemma":[0.002295556,0.00008867519,0.00005450599,0.000316083,0.000413399,0.0007431832,0.0001131089,0.0004181118,0.0005310434],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000544054,"about_ca_system_score_gemma":0.0004567126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003514828,"about_ca_topic_score_gemma":0.004626271,"domain_scores_codex":[0.9984563,0.0001577079,0.0001545302,0.0003577061,0.0005024128,0.0003713537],"domain_scores_gemma":[0.9982297,0.001119942,0.00002406433,0.0003852696,0.00008607736,0.0001549515],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000003311305,0.00002293543,0.00001692507,0.00002139249,0.00003140339,0.00001498146,0.05475087,0.000002836216,0.00002169575,0.8852181,0.02235352,0.03754207],"study_design_scores_gemma":[0.0001016185,0.00006217655,0.0002781517,0.0003454016,0.00003972371,0.000001272601,0.1086342,0.0005830572,0.00005526503,0.1319751,0.7575249,0.0003991019],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.02383217,0.0003170082,0.0004346094,0.0771174,0.001711098,0.0001779529,0.00005136358,0.0005031907,0.8958552],"genre_scores_gemma":[0.9834056,0.0003391067,0.0001787371,0.001362449,0.0005906019,0.000002852274,0.00001282258,0.00001798229,0.01408983],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9595734,"threshold_uncertainty_score":0.9998274,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.432989751682221,"score_gpt":0.451844179927153,"score_spread":0.018854428244932,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}