{"id":"W7132995918","doi":"","title":"Towards Trustworthy Machine Learning in High-Stakes Decision-Making Systems","year":2023,"lang":"","type":"dissertation","venue":"TSpace","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Trustworthiness; Task (project management); Heuristic; Strengths and weaknesses; Adversarial system; Focus (optics); Artificial neural network; Function (biology); Cognitive reframing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","research_integrity"],"consensus_categories":["metaepi_narrow"],"category_scores_codex":[0.003789709,0.001571946,0.002166259,0.002229185,0.001140701,0.001573177,0.003466325,0.00124961,0.000282453],"category_scores_gemma":[0.005848229,0.001726265,0.0004580233,0.005077111,0.0001296075,0.0009486153,0.001152963,0.005157187,0.0007467957],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008969349,"about_ca_system_score_gemma":0.001093672,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01195187,"about_ca_topic_score_gemma":0.002772957,"domain_scores_codex":[0.9890108,0.00154854,0.002090586,0.002858759,0.002550986,0.001940327],"domain_scores_gemma":[0.9922153,0.003225681,0.00202108,0.001695742,0.0004947771,0.0003474069],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002771772,0.00007590509,0.004853077,0.0005065214,0.0001140162,0.0008064904,0.03330634,0.8006666,0.000029643,0.005474597,0.0001015633,0.1537881],"study_design_scores_gemma":[0.001336184,0.0002521609,0.01588793,0.0061894,0.0001120974,0.00003880576,0.01651477,0.9549337,0.00001401215,0.0009819777,0.001965039,0.001773926],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3380615,0.005329101,0.6204731,0.0006235821,0.02652632,0.001947852,0.00001290677,0.001933613,0.00509193],"genre_scores_gemma":[0.9616884,0.0005592498,0.01996959,0.00004978505,0.000935681,0.0001378887,0.0002550251,0.0003864654,0.0160179],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6236269,"threshold_uncertainty_score":0.9997029,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02168616051524598,"score_gpt":0.3387278517736889,"score_spread":0.3170416912584429,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}