{"id":"W2977000626","doi":"10.1007/978-3-030-32236-6_76","title":"Overview of the NLPCC 2019 Shared Task: Open Domain Conversation Evaluation","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Blackberry (Canada)","funders":"","keywords":"Computer science; Conversation; Task (project management); Domain (mathematical analysis); Open domain; Human–computer interaction; Artificial intelligence; Programming language; Systems engineering; Linguistics; Question answering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01591009,0.00407099,0.003267462,0.005317968,0.004506465,0.006412704,0.006164842,0.004173349,0.03527458],"category_scores_gemma":[0.02442817,0.001525666,0.001789433,0.004243686,0.001467778,0.007619202,0.01224568,0.005853656,0.04211405],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003968615,"about_ca_system_score_gemma":0.007813534,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02888783,"about_ca_topic_score_gemma":0.03209838,"domain_scores_codex":[0.9789256,0.01067555,0.001185029,0.003076553,0.004794769,0.001342567],"domain_scores_gemma":[0.9871613,0.004018723,0.0002493007,0.002453623,0.004485487,0.001631642],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001527591,0.001399856,0.00170042,0.002344657,0.0003291373,0.0002029049,0.001067908,0.006487677,0.01472502,0.005475651,0.4727614,0.4919778],"study_design_scores_gemma":[0.001188236,0.001656677,0.009428459,0.0009754109,0.0003679855,0.0009873856,0.002410473,0.1653875,0.04982304,0.02905546,0.7380636,0.0006557027],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05502947,0.02345569,0.4408486,0.00781,0.004121237,0.0152776,0.1640189,0.1433894,0.1460491],"genre_scores_gemma":[0.1199582,0.002889913,0.3647449,0.002046706,0.0009644299,0.01275499,0.4310168,0.01199255,0.05363154],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03527458,"threshold_uncertainty_score":0.1180052,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03685488199454683,"score_gpt":0.3154158096821468,"score_spread":0.2785609276875999,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}