{"id":"W3202137189","doi":"10.48550/arxiv.2109.13066","title":"Prefix-to-SQL: Text-to-SQL Generation from Incomplete User Questions","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; SQL; Prefix; Task (project management); Data definition language; Benchmark (surveying); Construct (python library); SQL injection; Stored procedure; Metric (unit); SQL/PSM; Programming language; Database; Query by Example; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002124908,0.0003722336,0.000389803,0.0003427091,0.0002406839,0.0004243092,0.002088435,0.0003011919,0.0000783321],"category_scores_gemma":[0.00007767371,0.0004737685,0.0001885782,0.0007387269,0.00003187245,0.0005021788,0.003940933,0.0005565451,0.0002937274],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003842598,"about_ca_system_score_gemma":0.0003316996,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001913167,"about_ca_topic_score_gemma":0.0009941334,"domain_scores_codex":[0.9969649,0.0002411766,0.0003109501,0.001893071,0.0001816266,0.000408312],"domain_scores_gemma":[0.996832,0.00008363731,0.0001629221,0.002242929,0.0003003035,0.0003782035],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007465107,0.0000742513,0.001755344,0.00001596652,0.00008335171,0.0001778416,0.0007691787,0.8881383,0.001344854,0.1054602,0.0008371087,0.001336089],"study_design_scores_gemma":[0.0003054708,0.0000380165,0.005339073,0.0001732707,0.00006935425,0.000003127521,0.0000525185,0.9785738,0.000496491,0.008668789,0.005516617,0.0007634525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.380424,0.00002947785,0.6171614,0.000489125,0.0008620677,0.0002755662,0.00002464243,0.000222895,0.0005108495],"genre_scores_gemma":[0.9094115,0.00002734119,0.08747139,0.0006885901,0.000368265,0.000004803199,0.00007077091,0.00002586254,0.001931439],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.52969,"threshold_uncertainty_score":0.9997714,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1141670447659727,"score_gpt":0.2023945811433634,"score_spread":0.08822753637739075,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}