{"id":"W3190342989","doi":"10.48550/arxiv.2104.01604","title":"Timers and Such: A Practical Benchmark for Spoken Language Understanding\\n with Numbers","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Benchmark (surveying); Spoken language; Computer science; Natural language processing; Linguistics; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002184188,0.002389424,0.000892051,0.002778688,0.001103162,0.002218059,0.002682817,0.002738646,0.01553766],"category_scores_gemma":[0.01671495,0.0004548487,0.001348394,0.001988468,0.0007762719,0.004589285,0.003415613,0.002368529,0.01877031],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001218998,"about_ca_system_score_gemma":0.001997985,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01246923,"about_ca_topic_score_gemma":0.02000292,"domain_scores_codex":[0.9961976,0.001117195,0.0004973972,0.001080024,0.0008203469,0.0002873974],"domain_scores_gemma":[0.9937637,0.002244339,0.0003427716,0.001587291,0.001465326,0.000596629],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001917323,0.00110193,0.01016782,0.002311621,0.0002915339,0.0005157063,0.001723721,0.02024977,0.01017826,0.008586315,0.6886376,0.2543184],"study_design_scores_gemma":[0.0006759681,0.001443241,0.025928,0.0005967614,0.0002187883,0.001455277,0.003901128,0.2595185,0.03374682,0.02872743,0.6433789,0.0004092539],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.1567345,0.003767314,0.1085924,0.002461566,0.002476439,0.001581517,0.5130758,0.1649466,0.04636378],"genre_scores_gemma":[0.1298707,0.0005158473,0.06583894,0.0007407523,0.0002079244,0.001237784,0.7845188,0.003452631,0.0136167],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01553766,"threshold_uncertainty_score":0.05197859,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1272331395489106,"score_gpt":0.222093520618043,"score_spread":0.09486038106913239,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}