{"id":"W2891947597","doi":"10.48550/arxiv.1711.06664","title":"Predict Responsibly: Improving Fairness and Accuracy by Learning to Defer","year":2017,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Decision maker; Process (computing); Decision process; Machine learning; Artificial intelligence; Simple (philosophy); Work (physics); Risk analysis (engineering); Management science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02552136,0.0012728,0.001763341,0.0008453062,0.001798591,0.002975498,0.003060906,0.002377557,0.00186374],"category_scores_gemma":[0.08981826,0.0006487315,0.0009303954,0.0006357066,0.004219748,0.00598043,0.00462885,0.003694994,0.000539329],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002006464,"about_ca_system_score_gemma":0.003824504,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003493469,"about_ca_topic_score_gemma":0.00221947,"domain_scores_codex":[0.9824401,0.01070171,0.0005324103,0.002426674,0.002766106,0.001132971],"domain_scores_gemma":[0.9208617,0.05059062,0.006201605,0.01518235,0.004424737,0.002738962],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002007458,0.001001684,0.03405287,0.0002586879,0.000312489,0.0004194729,0.003325687,0.5571686,0.007977122,0.1396761,0.003962686,0.2498372],"study_design_scores_gemma":[0.0001099517,0.0003264544,0.001376866,0.00002826733,0.00004947657,0.0001111677,0.0001141875,0.8566487,0.003302208,0.1368191,0.001068145,0.00004548147],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2459368,0.0004278164,0.7438564,0.002815632,0.0001101213,0.0001663946,0.00005994715,0.000816856,0.005810067],"genre_scores_gemma":[0.9576606,0.00006114816,0.04050272,0.0004069829,0.0000532595,0.0000597809,0.00003251428,0.00005766336,0.001165319],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02552136,"threshold_uncertainty_score":0.1349714,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08803717292815989,"score_gpt":0.2645044095299627,"score_spread":0.1764672366018029,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}