{"id":"W4205570006","doi":"10.1145/3492853","title":"Studying Up Machine Learning Data","year":2022,"lang":"en","type":"article","venue":"Proceedings of the ACM on Human-Computer Interaction","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":113,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"International Development Research Centre","keywords":"Documentation; Transparency (behavior); Computer science; Data science; Context (archaeology); Argument (complex analysis); Perspective (graphical); Quality (philosophy); Data quality; Power (physics); Work (physics); Artificial intelligence; Machine learning; Knowledge management; Epistemology; Engineering; Business; Computer security; Marketing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.05570061,0.0008058053,0.001094933,0.004865214,0.00529516,0.01414667,0.003542933,0.004192219,0.005830116],"category_scores_gemma":[0.2048752,0.0009111905,0.001239919,0.005641342,0.02444024,0.02909004,0.01065532,0.009262962,0.001381056],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004161262,"about_ca_system_score_gemma":0.0064935,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001989169,"about_ca_topic_score_gemma":0.002135511,"domain_scores_codex":[0.9371008,0.04352767,0.002413079,0.004315733,0.01143499,0.001207663],"domain_scores_gemma":[0.7226893,0.2159479,0.01045324,0.03785032,0.01100515,0.002054139],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.00003906089,0.00004829258,0.005295814,0.0004613917,0.00005080449,0.0001713793,0.01327788,0.002084984,0.0004948859,0.8990687,0.009701291,0.06930544],"study_design_scores_gemma":[0.00001129974,0.00003878401,0.001348713,0.000704084,0.00001762033,0.0001701624,0.004746724,0.005423057,0.001400165,0.8640431,0.1220636,0.00003271699],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0462235,0.008156033,0.6644479,0.2113343,0.002485359,0.0003823735,0.001184279,0.0005724106,0.06521382],"genre_scores_gemma":[0.6200324,0.00668757,0.3347647,0.02007215,0.003378442,0.0009439817,0.001133225,0.0007013183,0.0122862],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9947048,"threshold_uncertainty_score":0.2945765,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2530857386306079,"score_gpt":0.4353445862956434,"score_spread":0.1822588476650355,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}