{"id":"W3022592851","doi":"10.18653/v1/2020.acl-main.220","title":"Learning an Unreferenced Metric for Online Dialogue Evaluation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"McGill University; Canadian Institute for Advanced Research","keywords":"Inference; Computer science; Metric (unit); Task (project management); Artificial intelligence; Domain (mathematical analysis); Natural language processing; Quality (philosophy); Machine learning; Open domain; Question answering; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01482646,0.002520668,0.00184495,0.003423371,0.0009523492,0.002447745,0.002630862,0.002569502,0.002722317],"category_scores_gemma":[0.07216726,0.0005518529,0.0008352005,0.001519565,0.00130127,0.004649567,0.003825561,0.002936927,0.002030697],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001743941,"about_ca_system_score_gemma":0.001662691,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002421866,"about_ca_topic_score_gemma":0.003201035,"domain_scores_codex":[0.9702912,0.01839627,0.001619379,0.0046982,0.004130652,0.0008644185],"domain_scores_gemma":[0.9454311,0.03347637,0.003713947,0.00692676,0.008868887,0.001582891],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001923317,0.001015439,0.01608231,0.00105017,0.0004934046,0.0003242617,0.001457099,0.1550081,0.03628891,0.01539614,0.02000805,0.7509528],"study_design_scores_gemma":[0.00005958326,0.0006395678,0.004560431,0.00008354366,0.00006362853,0.0002526188,0.0002424802,0.9590496,0.01423503,0.01548957,0.005217308,0.0001066635],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04026425,0.00118183,0.9488636,0.0002861712,0.0001711053,0.0002672295,0.0007748635,0.004798219,0.003392815],"genre_scores_gemma":[0.6502303,0.0002717965,0.3407606,0.0003025051,0.0002581159,0.0008964114,0.003627312,0.001024145,0.002628691],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01482646,"threshold_uncertainty_score":0.07841074,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2226464616326511,"score_gpt":0.3771446736430029,"score_spread":0.1544982120103518,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}