{"id":"W4376133361","doi":"10.1126/sciadv.abq0701","title":"Judging facts, judging norms: Training machine learning models to judge humans requires a modified approach to labeling data","year":2023,"lang":"en","type":"article","venue":"Science Advances","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Schwartz/Reisman Emergency Medicine Institute; Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research; Microsoft Research; University of Toronto; University of Southern California","keywords":"Normative; Computer science; Artificial intelligence; Object (grammar); Machine learning; Normative model of decision-making; Core (optical fiber); Psychology; Cognitive psychology; Epistemology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03212,0.001182325,0.001152809,0.00190594,0.001468378,0.006462737,0.002723748,0.002409563,0.002094443],"category_scores_gemma":[0.1413346,0.0007459924,0.0009477698,0.001746384,0.004426793,0.00898782,0.002600031,0.007370858,0.001146861],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002347539,"about_ca_system_score_gemma":0.002689347,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008022054,"about_ca_topic_score_gemma":0.01175605,"domain_scores_codex":[0.9811912,0.01208957,0.0008804928,0.003156704,0.002428212,0.0002537285],"domain_scores_gemma":[0.8850143,0.08282369,0.005135887,0.02079266,0.005187564,0.00104588],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001080575,0.0009106484,0.08545332,0.0009336452,0.001012038,0.0003171658,0.00649852,0.2221713,0.01622077,0.1569874,0.02529588,0.4831188],"study_design_scores_gemma":[0.00007780481,0.0002030933,0.008539476,0.0002208749,0.00007560349,0.0001433508,0.0006407115,0.6593374,0.007327263,0.314281,0.009016559,0.0001368973],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08956336,0.0004638878,0.8961303,0.006515005,0.0003106014,0.0002906104,0.0005926175,0.001406584,0.004727012],"genre_scores_gemma":[0.5434953,0.0003039051,0.4505647,0.001820395,0.0002451598,0.0004254061,0.001250702,0.0003268818,0.001567553],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.96788,"threshold_uncertainty_score":0.1698688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2543949435950519,"score_gpt":0.357115698066272,"score_spread":0.1027207544712201,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}