{"id":"W4413978785","doi":"10.14778/3749646.3749727","title":"ParSEval: Plan-Aware Test Database Generation for SQL Equivalence Evaluation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Database; SQL; Parseval's theorem; Plan (archaeology); Equivalence (formal languages); Programming language; Mathematics; Discrete mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009009007,0.0001275901,0.0001730058,0.0002404551,0.0003277262,0.0003736748,0.00140317,0.00003026,0.00008697494],"category_scores_gemma":[0.0081132,0.00008153628,0.0001062243,0.0009047411,0.0000748065,0.0003430352,0.0007893633,0.00006250628,0.00001793175],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001002074,"about_ca_system_score_gemma":0.00008794876,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002139203,"about_ca_topic_score_gemma":0.00001214397,"domain_scores_codex":[0.9967093,0.00002302719,0.0006235163,0.0006938383,0.001718182,0.0002320708],"domain_scores_gemma":[0.9972804,0.0006029849,0.0004131126,0.0005738692,0.001085279,0.00004434347],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005309268,0.0003058749,0.01115418,0.0001311003,0.00004648163,1.119062e-7,0.0004586636,0.0008512009,0.08448385,0.02001898,0.7785161,0.1039804],"study_design_scores_gemma":[0.002088569,0.000169585,0.005885048,0.0004032553,0.0002868336,0.00000254091,0.002475654,0.6765835,0.1923485,0.06170914,0.0576867,0.0003607717],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8814117,0.000798238,0.05190799,0.01688861,0.01163706,0.009902502,0.001002314,0.0002429801,0.02620863],"genre_scores_gemma":[0.9942284,0.000009655902,0.002612843,0.0002779416,0.0001159211,0.00017928,0.0000312577,0.000005546544,0.002539205],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7208294,"threshold_uncertainty_score":0.9712844,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3019404009545655,"score_gpt":0.4383997879009069,"score_spread":0.1364593869463414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}