{"id":"W1984067593","doi":"10.1145/2768577.2768578","title":"The SDEval benchmarking toolkit","year":2015,"lang":"en","type":"article","venue":"ACM communications in computer algebra","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Executable; Benchmarking; Computer science; Focus (optics); Field (mathematics); Task (project management); Symbolic computation; Programming language; Software engineering; Software; Code (set theory); Systems engineering; Engineering; Set (abstract data type); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007028366,0.001814658,0.001309505,0.003161683,0.0007579593,0.003523223,0.005240604,0.001358648,0.01454547],"category_scores_gemma":[0.01987383,0.00111742,0.001406226,0.002676676,0.001096402,0.004179526,0.005243197,0.003314498,0.007293443],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001231089,"about_ca_system_score_gemma":0.002665891,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003347312,"about_ca_topic_score_gemma":0.002951726,"domain_scores_codex":[0.9892959,0.00412292,0.001149174,0.001088454,0.003502498,0.0008411002],"domain_scores_gemma":[0.9920262,0.003109003,0.0003567627,0.002514373,0.001527982,0.0004656061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002111519,0.001138199,0.01043009,0.004584481,0.0006004711,0.00067477,0.001023845,0.1435034,0.01006298,0.189211,0.3064502,0.3302091],"study_design_scores_gemma":[0.00053976,0.0004425011,0.003260371,0.00084109,0.0001216544,0.000525226,0.0003762276,0.3736223,0.02416371,0.1010715,0.494753,0.0002826909],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.03339278,0.003658926,0.5448973,0.00121572,0.00119229,0.0007879773,0.01705554,0.3266084,0.07119118],"genre_scores_gemma":[0.3287816,0.002791378,0.4737297,0.001109645,0.0001947061,0.001869049,0.08491301,0.08540942,0.02120158],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.01454547,"threshold_uncertainty_score":0.04865944,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.114504119115066,"score_gpt":0.355765799854737,"score_spread":0.241261680739671,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}