{"id":"W6966779364","doi":"10.48448/1kye-5e47","title":"Evaluating Cross-Domain Text-to-SQL Models and Benchmarks","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Benchmark (surveying); SQL; Ranking (information retrieval); Field (mathematics); Matching (statistics); Code (set theory); Standard Model (mathematical formulation)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01565904,0.002577066,0.0009058875,0.003255882,0.0008900851,0.003248932,0.003912667,0.001732468,0.004028144],"category_scores_gemma":[0.0579542,0.0005096384,0.001331199,0.003598395,0.001365049,0.005070867,0.003288137,0.002233412,0.003210674],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002530379,"about_ca_system_score_gemma":0.002696337,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01424765,"about_ca_topic_score_gemma":0.01753056,"domain_scores_codex":[0.9745258,0.01020498,0.002256652,0.003513174,0.008602948,0.0008964511],"domain_scores_gemma":[0.963155,0.01859179,0.001424293,0.008097523,0.007540818,0.001190557],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003416391,0.004837997,0.0438423,0.005747215,0.001235452,0.001414959,0.002175683,0.2888255,0.03106124,0.0262351,0.2572688,0.3339393],"study_design_scores_gemma":[0.0003951725,0.001656845,0.01051557,0.0005034643,0.0001895787,0.0006297411,0.001652338,0.8552396,0.04777239,0.01493541,0.06629664,0.000213312],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6288401,0.008591574,0.1803724,0.003556577,0.001875738,0.001649613,0.03485726,0.1024337,0.03782306],"genre_scores_gemma":[0.6297646,0.001479415,0.2304041,0.001655236,0.0001807963,0.0006639246,0.1208642,0.008495123,0.006492598],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01565904,"threshold_uncertainty_score":0.08281386,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1428459743605008,"score_gpt":0.4486470291044169,"score_spread":0.3058010547439161,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}