{"id":"W4389523781","doi":"10.18653/v1/2023.emnlp-main.99","title":"Evaluating Cross-Domain Text-to-SQL Models and Benchmarks","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); SQL; Ranking (information retrieval); Field (mathematics); Query by Example; Domain (mathematical analysis); Matching (statistics); Information retrieval; Programming language; Data mining; Web search query; Search engine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01847349,0.002803878,0.001151017,0.003559663,0.0008882221,0.003188898,0.00418893,0.001909831,0.002962074],"category_scores_gemma":[0.0642911,0.0005638096,0.001481013,0.004394195,0.001352818,0.004998291,0.003290393,0.002345772,0.002120978],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002869623,"about_ca_system_score_gemma":0.002591898,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01189527,"about_ca_topic_score_gemma":0.01489444,"domain_scores_codex":[0.9698509,0.01296243,0.00324519,0.004217935,0.008673256,0.001050249],"domain_scores_gemma":[0.9526342,0.02467088,0.001890347,0.009436723,0.009961664,0.0014062],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0040989,0.00539292,0.0499529,0.006557011,0.001712,0.001568559,0.002008558,0.4075428,0.04059866,0.01678011,0.150344,0.3134436],"study_design_scores_gemma":[0.0003975997,0.002089113,0.01210872,0.0004117255,0.0002313671,0.0006336959,0.001452145,0.8886206,0.05221108,0.009403858,0.03225017,0.0001900928],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7413418,0.008957331,0.1514381,0.002590128,0.001492905,0.001571863,0.02244389,0.05226279,0.01790114],"genre_scores_gemma":[0.7117868,0.00153397,0.1940901,0.001376799,0.0001759959,0.0006556017,0.08236746,0.004407899,0.003605437],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01847349,"threshold_uncertainty_score":0.09769833,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07326356775616878,"score_gpt":0.3835356209260535,"score_spread":0.3102720531698847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}