{"id":"W4386580576","doi":"10.1007/s44206-023-00063-1","title":"Lessons Learned from Assessing Trustworthy AI in Practice","year":2023,"lang":"en","type":"article","venue":"Digital Society","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Cégep André Laurendeau; Université du Québec à Montréal","funders":"Horizon 2020 Framework Programme; Connecting Europe Facility; National Institutes of Health; European Commission; Foundation for the National Institutes of Health","keywords":"Trustworthiness; Computer science; Psychology; Data science; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3392356,0.001634831,0.002139234,0.007818981,0.006126586,0.02438636,0.007056178,0.006487614,0.004451521],"category_scores_gemma":[0.5052279,0.001466803,0.001233304,0.005251695,0.03810824,0.03736028,0.02004546,0.01289659,0.001060786],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01626636,"about_ca_system_score_gemma":0.02774635,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008560763,"about_ca_topic_score_gemma":0.008884114,"domain_scores_codex":[0.5617507,0.3582151,0.01818263,0.01100367,0.04635816,0.004489687],"domain_scores_gemma":[0.3125304,0.5594904,0.01394698,0.02836407,0.07979017,0.005877931],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001563228,0.00063164,0.01557792,0.005555706,0.0002101648,0.001024871,0.3330949,0.006874649,0.001093256,0.2860417,0.01522029,0.3345186],"study_design_scores_gemma":[0.0001216223,0.0006833489,0.008313796,0.01759787,0.0001061285,0.0007379507,0.18434,0.0146145,0.002636928,0.6983877,0.07216319,0.000296902],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1487342,0.01161013,0.5082557,0.2445126,0.001671262,0.003083606,0.0002140253,0.0006063773,0.08131222],"genre_scores_gemma":[0.73226,0.003723591,0.2541741,0.005857322,0.0002385959,0.001628854,0.000100169,0.0002344769,0.001782892],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3392356,"threshold_uncertainty_score":0.8148401,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1539208411819251,"score_gpt":0.4696216618800328,"score_spread":0.3157008206981077,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}