{"id":"W2985067290","doi":"10.18653/v1/d19-1459","title":"Taskmaster-1: Toward a Realistic and Diverse Dialog Dataset","year":2019,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":139,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Dialog box; Joint (building); Computer science; Natural language; Natural (archaeology); Natural language processing; Artificial intelligence; Engineering; World Wide Web; History; Architectural engineering; Archaeology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004290163,0.001800591,0.001087516,0.002290415,0.002281229,0.001942381,0.003559803,0.003658131,0.008901863],"category_scores_gemma":[0.01163531,0.0006755318,0.001568168,0.001819079,0.0007624475,0.003329237,0.00415392,0.00296816,0.01441175],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00155796,"about_ca_system_score_gemma":0.002201736,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01620426,"about_ca_topic_score_gemma":0.03163166,"domain_scores_codex":[0.9962,0.001811991,0.0002838093,0.0008223338,0.0005762173,0.0003055709],"domain_scores_gemma":[0.9939451,0.001707129,0.0003045558,0.001816794,0.001179351,0.001047262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001610331,0.001993241,0.01076102,0.001699451,0.0003315367,0.0005587834,0.001107238,0.008077226,0.005183357,0.003545617,0.9086308,0.05650144],"study_design_scores_gemma":[0.001733344,0.001525584,0.04266193,0.0006512259,0.0002980682,0.001817466,0.004738601,0.1081114,0.01281561,0.0132163,0.8119903,0.0004401417],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.1150854,0.003631646,0.03301553,0.005336209,0.002127551,0.002823171,0.800756,0.01902854,0.01819598],"genre_scores_gemma":[0.07230058,0.0002953564,0.03181538,0.000909049,0.0002347444,0.001773907,0.8849177,0.0004447869,0.007308501],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01620426,"threshold_uncertainty_score":0.03221989,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0398766274616048,"score_gpt":0.2522937653920621,"score_spread":0.2124171379304572,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}