{"id":"W3109752747","doi":"10.18653/v1/2020.coling-main.230","title":"Improving Conversational Question Answering Systems after Deployment using Feedback-Weighted Learning","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"CHIST-ERA; Ministerio de Educación, Cultura y Deporte; Nvidia; Samsung Advanced Institute of Technology; Canadian Institute for Advanced Research; Eusko Jaurlaritza; Agencia Estatal de Investigación; Samsung; Agence Nationale de la Recherche","keywords":"Computer science; Software deployment; Exploit; Domain (mathematical analysis); Binary number; Binary classification; Recommender system; Question answering; Matching (statistics); Artificial intelligence; Machine learning; Information retrieval; Human–computer interaction; Software engineering; Support vector machine","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01076885,0.001743276,0.001290696,0.0009622675,0.0008009521,0.001362355,0.00232708,0.00181125,0.001916785],"category_scores_gemma":[0.03789665,0.000715899,0.0007354933,0.0005299175,0.0007231034,0.004339773,0.002749306,0.002792218,0.00208248],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001006965,"about_ca_system_score_gemma":0.00139613,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004476488,"about_ca_topic_score_gemma":0.006205467,"domain_scores_codex":[0.991262,0.005308904,0.0003459302,0.001660197,0.0009951371,0.0004279081],"domain_scores_gemma":[0.9769613,0.01363295,0.0007537364,0.003353348,0.004380698,0.0009178431],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002040595,0.00214578,0.01220793,0.000920771,0.00039385,0.0003087242,0.002477774,0.09896893,0.09707049,0.001899748,0.01609398,0.7654715],"study_design_scores_gemma":[0.000119855,0.0005241397,0.00204598,0.00002524983,0.0000602238,0.00009532557,0.0002534301,0.9640558,0.02654944,0.002842635,0.003378997,0.00004901164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2572792,0.001335797,0.7027858,0.001202243,0.0002584084,0.0007599893,0.0005335044,0.03280365,0.003041346],"genre_scores_gemma":[0.7130038,0.0001617484,0.2806659,0.0004336687,0.0001369619,0.0003777459,0.001828785,0.000788302,0.002603225],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01076885,"threshold_uncertainty_score":0.05695188,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02939263299873394,"score_gpt":0.2517073156640895,"score_spread":0.2223146826653555,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}