{"id":"W4393486226","doi":"10.5281/zenodo.10119773","title":"Benchmarks for POPL'24 Paper \"Commutativity Simplifies Proofs of Parameterized Programs\"","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Logic, programming, and type systems","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Parameterized complexity; Mathematical proof; Commutative property; Programming language; Computer science; Mathematics; Algorithm; Discrete mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002814292,0.004001988,0.002006118,0.005617921,0.0018048,0.004557119,0.005677588,0.003010676,0.05242987],"category_scores_gemma":[0.01827079,0.001210579,0.002108179,0.009983825,0.0008140315,0.003321601,0.003629156,0.003712579,0.06168978],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002676915,"about_ca_system_score_gemma":0.004677203,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01820105,"about_ca_topic_score_gemma":0.02259099,"domain_scores_codex":[0.9959272,0.0009035699,0.0004225565,0.0007777081,0.001548531,0.0004205141],"domain_scores_gemma":[0.9903797,0.003548989,0.000667577,0.003061736,0.001667562,0.0006745028],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000145471,0.00005874248,0.0003752272,0.00103859,0.00006265365,0.0000288826,0.00002041529,0.0008995933,0.0001750749,0.001601443,0.9918247,0.003769206],"study_design_scores_gemma":[0.001692427,0.00007589552,0.003554381,0.0005804455,0.0001113461,0.0002046453,0.00008358051,0.005833964,0.001839914,0.01118566,0.9747764,0.00006138896],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0008792132,0.0005949749,0.0008524354,0.0003311739,0.0001071713,0.00005107038,0.987323,0.007283499,0.002577367],"genre_scores_gemma":[0.001026366,0.0001883697,0.0009610966,0.00009245685,0.00001529932,0.0001194819,0.9965854,0.0004923994,0.0005191934],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.05242987,"threshold_uncertainty_score":0.1753954,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06007767842485172,"score_gpt":0.2778392554667943,"score_spread":0.2177615770419425,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}