{"id":"W4408605051","doi":"10.1007/s43681-025-00703-x","title":"Metaethical perspectives on ‘benchmarking’ AI ethics","year":2025,"lang":"en","type":"article","venue":"AI and Ethics","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Schwartz/Reisman Emergency Medicine Institute; University of Toronto","funders":"Dalhousie University","keywords":"Benchmarking; Cornerstone; Variety (cybernetics); Set (abstract data type); Value (mathematics); Benchmark (surveying); Epistemology; Engineering ethics; Computer science; Artificial intelligence; Sociology; Philosophy; Management; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1169434,0.001018268,0.001590538,0.007810049,0.00923402,0.02122811,0.004255832,0.01411724,0.005235302],"category_scores_gemma":[0.1097149,0.0008462758,0.001225968,0.005426432,0.08844706,0.02385665,0.01186922,0.01163271,0.0006680221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01884584,"about_ca_system_score_gemma":0.01013665,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002728437,"about_ca_topic_score_gemma":0.00244878,"domain_scores_codex":[0.8230526,0.1503461,0.004761353,0.005943832,0.01248416,0.003412038],"domain_scores_gemma":[0.8517573,0.1018987,0.0118805,0.01302548,0.01736284,0.004075127],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000003596036,0.000006707723,0.00009612169,0.00002210594,0.000006615693,0.00001198559,0.001177476,0.0002435431,0.00001890551,0.997039,0.0004237988,0.0009502693],"study_design_scores_gemma":[0.000007241581,0.000007432267,0.00009685202,0.0001303877,0.000004509691,0.00001776213,0.0009107868,0.0006131741,0.00006956689,0.9907405,0.00739272,0.000009067719],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.04044126,0.01184074,0.1863612,0.3099614,0.001516394,0.0002538989,0.0001799709,0.0001819992,0.4492631],"genre_scores_gemma":[0.9540226,0.001328298,0.03136966,0.008260574,0.0005814199,0.0004238626,0.00005247441,0.0000922843,0.003868869],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1169434,"threshold_uncertainty_score":0.618463,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08713312055306624,"score_gpt":0.4764134192578392,"score_spread":0.389280298704773,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}