{"id":"W4391334942","doi":"10.1145/3630106.3659037","title":"Black-Box Access is Insufficient for Rigorous AI Audits","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; University of Toronto","funders":"","keywords":"Black box; Audit; Computer science; Business; Computer security; Accounting; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.145906,0.001562506,0.002647611,0.002765125,0.003224102,0.0138182,0.003857912,0.005658865,0.007755456],"category_scores_gemma":[0.3835642,0.001769189,0.001492123,0.002036909,0.01308556,0.02264714,0.01046545,0.01072354,0.003835893],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00378353,"about_ca_system_score_gemma":0.01123555,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00291569,"about_ca_topic_score_gemma":0.001640556,"domain_scores_codex":[0.8030759,0.1138932,0.01267478,0.01915927,0.04351573,0.007681056],"domain_scores_gemma":[0.4844972,0.251373,0.04298675,0.1738108,0.04177213,0.005560108],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002417867,0.0006087187,0.01839593,0.001582617,0.0005476185,0.0008719962,0.01059252,0.02288102,0.01146135,0.4202003,0.03817349,0.4722666],"study_design_scores_gemma":[0.0007059705,0.0008257772,0.01246898,0.00265509,0.000241942,0.0007257988,0.002533787,0.06096282,0.01868671,0.784215,0.1154726,0.0005054819],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1269667,0.003151363,0.745761,0.0563635,0.001647062,0.002357261,0.0009257988,0.00888794,0.05393927],"genre_scores_gemma":[0.8497697,0.001025279,0.1313662,0.007682608,0.000647816,0.001334248,0.0003353295,0.001187508,0.006651354],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.854094,"threshold_uncertainty_score":0.7716342,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0312995521654295,"score_gpt":0.3489467783865863,"score_spread":0.3176472262211568,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}