{"id":"W3154584133","doi":"10.48550/arxiv.2104.06335","title":"On the Use of Linguistic Features for the Evaluation of Generative Dialogue Systems","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Generalization; Computer science; Relevance (law); Task (project management); Metric (unit); Proposition; Generative grammar; Natural language processing; Artificial intelligence; Measure (data warehouse); Generative model; Machine learning; Linguistics; Mathematics; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02690569,0.001742524,0.001642081,0.005100169,0.001101703,0.005351638,0.001851397,0.003325453,0.001297276],"category_scores_gemma":[0.1169548,0.0005428329,0.0008628314,0.002553811,0.001684063,0.005001176,0.002926095,0.002257948,0.0006380116],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001365338,"about_ca_system_score_gemma":0.0008940277,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002792477,"about_ca_topic_score_gemma":0.004335744,"domain_scores_codex":[0.9744092,0.01745675,0.001038573,0.002833743,0.003743903,0.0005178644],"domain_scores_gemma":[0.8626966,0.1192344,0.004829082,0.006023563,0.005909124,0.001307249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002234935,0.00109169,0.05190707,0.001208032,0.0009059085,0.0002873946,0.002236274,0.1583737,0.04070558,0.008636544,0.004284497,0.7281284],"study_design_scores_gemma":[0.00008626113,0.001287649,0.0283784,0.0001909521,0.000163952,0.0002968213,0.0004927092,0.9347811,0.01804899,0.01439028,0.001709702,0.0001731857],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5506573,0.002433094,0.4327753,0.001259334,0.0001434124,0.0004452712,0.0007976134,0.004186392,0.007302342],"genre_scores_gemma":[0.911846,0.0001548774,0.08640181,0.0001052916,0.00004472123,0.000140988,0.0006098518,0.0002269964,0.0004694391],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02690569,"threshold_uncertainty_score":0.1422926,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3171293089957628,"score_gpt":0.2439817698250702,"score_spread":0.07314753917069264,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}