{"id":"W4389138268","doi":"10.1007/s11943-023-00331-z","title":"Exploring quality dimensions in trustworthy Machine Learning in the context of official statistics: model explainability and uncertainty quantification","year":2023,"lang":"en","type":"article","venue":"AStA Wirtschafts- und Sozialstatistisches Archiv","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Statistics Canada","funders":"","keywords":"Computer science; Quality (philosophy); Context (archaeology); Data science; Trustworthiness; Data collection; Coding (social sciences); Data quality; Quality management; Process management; Artificial intelligence; Machine learning; Data mining; Engineering; Operations management; Statistics; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02705307,0.001086652,0.002283837,0.003994672,0.00154129,0.008189641,0.002223687,0.00250807,0.001674605],"category_scores_gemma":[0.1624605,0.001014088,0.001940975,0.002940848,0.007521715,0.01387275,0.00554978,0.005005877,0.00009929162],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004638195,"about_ca_system_score_gemma":0.004084329,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01198656,"about_ca_topic_score_gemma":0.006439289,"domain_scores_codex":[0.9851854,0.009409153,0.0008998355,0.001644313,0.002082255,0.0007790138],"domain_scores_gemma":[0.7378927,0.2266108,0.01676058,0.01008885,0.006836937,0.00181016],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00008316265,0.00008068007,0.008843996,0.0001745789,0.0002487737,0.0001512847,0.001010071,0.2103407,0.0002250213,0.768593,0.0004499709,0.009798765],"study_design_scores_gemma":[0.00000898747,0.0000242645,0.0008531202,0.00005180294,0.0000307182,0.00001987156,0.0001319201,0.4129634,0.0001206342,0.5855216,0.0002498401,0.00002374794],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1453426,0.001155059,0.8426374,0.006672049,0.00006728631,0.00006296027,0.0001933273,0.0001262938,0.003743008],"genre_scores_gemma":[0.9620138,0.0004370665,0.03663047,0.0001254687,0.0001093814,0.00008138461,0.0001443704,0.00006589444,0.0003922428],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02705307,"threshold_uncertainty_score":0.143072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1553770748700975,"score_gpt":0.3480554712608029,"score_spread":0.1926783963907054,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}