{"id":"W4200303997","doi":"10.1126/science.abi7176","title":"Filling gaps in trustworthy development of AI","year":2021,"lang":"en","type":"article","venue":"Science","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":50,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"Engineering and Physical Sciences Research Council","keywords":"Trustworthiness; Audit; Computer science; Business; Computer security; Accounting","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002674128,0.00003584424,0.00008267412,0.00006907265,0.0005303284,0.0000926416,0.000261486,0.0000453778,0.00008395562],"category_scores_gemma":[0.001645763,0.0000366009,0.00001885035,0.001213148,0.000741369,0.0003754266,0.00006201166,0.0001362731,0.000007371567],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009507324,"about_ca_system_score_gemma":0.004071995,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003075933,"about_ca_topic_score_gemma":0.006517081,"domain_scores_codex":[0.9987855,0.00004941665,0.0001520018,0.0001533815,0.0005818948,0.0002778255],"domain_scores_gemma":[0.9993428,0.00007616223,0.00003992081,0.00008526953,0.0003604357,0.00009541604],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.00000608177,0.0001981547,0.06376703,0.00003393625,0.000006822707,0.00003318994,0.4733165,0.0001322736,0.02823022,0.370011,0.0001773378,0.06408752],"study_design_scores_gemma":[0.0007961396,0.00005833017,0.2924508,0.0005013238,0.000009094247,0.000001639617,0.1073409,0.0002419441,0.1982534,0.1759763,0.2234598,0.0009103331],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9368203,0.0001613049,0.0001009405,0.004529303,0.0002863005,0.00004760375,6.312781e-7,0.00001168107,0.05804198],"genre_scores_gemma":[0.995542,0.00005171445,0.003500462,0.0004266664,0.00003281397,0.000001111798,2.651732e-7,0.000001740213,0.0004432041],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3659755,"threshold_uncertainty_score":0.7223545,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04915249500649999,"score_gpt":0.400799252304235,"score_spread":0.351646757297735,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}