{"id":"W3115763169","doi":"10.18653/v1/2020.coling-main.456","title":"Explain by Evidence: An Explainable Memory-based Neural Network for Question Answering","year":2020,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Interpretability; Computer science; Artificial intelligence; TRACE (psycholinguistics); Machine learning; Tracing; Artificial neural network; Question answering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004396293,0.000129732,0.0001394392,0.00002142439,0.0001663683,0.0002000429,0.0006528018,0.00005182782,0.00001871647],"category_scores_gemma":[0.0001070581,0.000125826,0.00004839832,0.0002067514,0.00001177567,0.001187975,0.00008917966,0.00009074224,0.000006255534],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003461717,"about_ca_system_score_gemma":0.00005219852,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007434866,"about_ca_topic_score_gemma":0.00001240455,"domain_scores_codex":[0.9986545,0.00008086613,0.0002138645,0.00049349,0.0001963189,0.0003609434],"domain_scores_gemma":[0.9991796,0.0001461089,0.00005448925,0.0003712445,0.00006045624,0.0001881283],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006339192,0.00004173956,0.0006929171,0.000126763,0.000008453962,0.00001340827,0.001359687,0.8846461,0.006903736,0.02012539,0.02158157,0.06443686],"study_design_scores_gemma":[0.0002226359,0.0001912934,0.00001772403,0.00002945211,0.000002939126,0.000001100241,0.00005355944,0.9935997,0.003658645,0.0004914965,0.001568761,0.0001627024],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02614595,0.0002104937,0.9657337,0.006821151,0.0002602107,0.0003010468,9.169884e-7,0.0004239704,0.0001025436],"genre_scores_gemma":[0.671546,0.000002589186,0.3241284,0.003821243,0.0003174236,0.00006187923,0.000005270628,0.00001129185,0.0001058758],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6454001,"threshold_uncertainty_score":0.5131034,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05916506338980908,"score_gpt":0.2892817277727288,"score_spread":0.2301166643829197,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}