{"id":"W4389518718","doi":"10.18653/v1/2023.emnlp-main.427","title":"Rather a Nurse than a Physician - Contrastive Explanations under Investigation","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Berlin Center for Machine Learning; Novo Nordisk Fonden; Research Executive Agency; Banting and Best Diabetes Centre, University of Toronto; Novo Nordisk; European Commission","keywords":"Contrastive analysis; Computer science; Linguistics; Contrast (vision); Natural language processing; Artificial intelligence; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003213214,0.0006293796,0.0003124764,0.0009921957,0.000820821,0.001519293,0.000848776,0.00153977,0.01719754],"category_scores_gemma":[0.020687,0.0003170164,0.0007262646,0.001085427,0.0007359536,0.002791816,0.0008990642,0.001860263,0.003363024],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001009011,"about_ca_system_score_gemma":0.001371301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003171297,"about_ca_topic_score_gemma":0.00961089,"domain_scores_codex":[0.9979493,0.0008877167,0.0001379169,0.0005291939,0.0003940252,0.0001019099],"domain_scores_gemma":[0.9861014,0.0102645,0.00111869,0.001326018,0.000898285,0.000290967],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0018069,0.0003582178,0.1913515,0.00495637,0.0005091288,0.004283266,0.0128582,0.01063784,0.02500107,0.1416327,0.1946278,0.411977],"study_design_scores_gemma":[0.0002211597,0.0002869652,0.05454872,0.001282571,0.0003929021,0.005136217,0.006089764,0.02894008,0.01669197,0.08988846,0.7963412,0.0001799576],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2683885,0.009413654,0.5112266,0.04537078,0.002742454,0.000732348,0.04082983,0.009356005,0.1119398],"genre_scores_gemma":[0.835893,0.001871089,0.1214444,0.005305815,0.0003999682,0.0002407046,0.01399507,0.0009352699,0.01991476],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01719754,"threshold_uncertainty_score":0.05753154,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0387413242078954,"score_gpt":0.2665800505663476,"score_spread":0.2278387263584522,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}