{"id":"W3017343191","doi":"10.48550/arxiv.2004.07213","title":"Toward Trustworthy AI Development: Mechanisms for Supporting Verifiable Claims","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":125,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Université du Québec à Montréal; Polytechnique Montréal","funders":"","keywords":"Verifiable secret sharing; Trustworthiness; Computer science; Development (topology); Computer security; Data science; Programming language; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04937119,0.001576106,0.001600544,0.004519685,0.0039479,0.01197291,0.007035494,0.009292577,0.00880966],"category_scores_gemma":[0.1995461,0.00194475,0.00231811,0.00183857,0.02137582,0.02571032,0.0185553,0.01281958,0.002566682],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004188198,"about_ca_system_score_gemma":0.005759572,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001697161,"about_ca_topic_score_gemma":0.0009336949,"domain_scores_codex":[0.9577882,0.02236361,0.002280483,0.005086667,0.01049817,0.001982938],"domain_scores_gemma":[0.7119836,0.1854524,0.02133536,0.06373761,0.01387198,0.003619071],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001775145,0.0001110523,0.001619088,0.0002593021,0.0001117231,0.0002815404,0.001091458,0.02782009,0.002748804,0.9194474,0.003614341,0.04271767],"study_design_scores_gemma":[0.00006664107,0.00007305116,0.0003018657,0.0001768525,0.00005241654,0.0001910287,0.0002155009,0.1580373,0.004089922,0.8269403,0.009794343,0.00006078524],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0124696,0.0007502633,0.9628655,0.008873081,0.0001859373,0.0003072064,0.00009999333,0.001100989,0.01334737],"genre_scores_gemma":[0.6668708,0.001164111,0.3216436,0.002382811,0.0005044136,0.0007313641,0.0002181268,0.0003544219,0.006130377],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04937119,"threshold_uncertainty_score":0.261103,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09684774672380762,"score_gpt":0.2238924999907693,"score_spread":0.1270447532669616,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}