{"id":"W4413978785","doi":"10.14778/3749646.3749727","title":"ParSEval: Plan-Aware Test Database Generation for SQL Equivalence Evaluation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Database; SQL; Parseval's theorem; Plan (archaeology); Equivalence (formal languages); Programming language; Mathematics; Discrete mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004721154,0.001820195,0.001014382,0.002305634,0.0005254734,0.002058699,0.003503745,0.001086431,0.006224447],"category_scores_gemma":[0.01844391,0.0008737901,0.002201742,0.001236615,0.001332199,0.002748169,0.002789869,0.001827476,0.001513773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001327608,"about_ca_system_score_gemma":0.002976613,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006306109,"about_ca_topic_score_gemma":0.007195491,"domain_scores_codex":[0.9949184,0.001686723,0.0004521807,0.0008607434,0.001731855,0.0003499735],"domain_scores_gemma":[0.9900918,0.00586262,0.0005360389,0.002154443,0.001160618,0.0001944573],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001510545,0.0007873001,0.0138422,0.001810819,0.000502769,0.0008344909,0.0007127408,0.2098179,0.02680487,0.0480394,0.07347863,0.6218582],"study_design_scores_gemma":[0.0002120395,0.0002412289,0.0009923831,0.00008189465,0.00006813174,0.000272456,0.0001179294,0.9404092,0.02135638,0.02326893,0.01292199,0.00005743531],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01290208,0.0004184019,0.8915955,0.0002511535,0.00008036447,0.0005759221,0.00217678,0.09002889,0.001971044],"genre_scores_gemma":[0.2306956,0.000252457,0.7520573,0.000372976,0.00004372477,0.0007938365,0.009752013,0.004916668,0.001115505],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006306109,"threshold_uncertainty_score":0.02496815,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3019404009545655,"score_gpt":0.4383997879009069,"score_spread":0.1364593869463414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}