{"id":"W6930672160","doi":"10.5281/zenodo.15012085","title":"Results of the 14th Intl. Competition on Software Verification (SV-COMP 2025)","year":2025,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"","field":"","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Validator; XML; Competition (biology); Software; Benchmark (surveying); Table (database); Set (abstract data type); Software verification","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03883103,0.004669628,0.003178718,0.006588564,0.003431563,0.01681963,0.005873011,0.006590199,0.2007698],"category_scores_gemma":[0.03947569,0.001547634,0.004550678,0.004840278,0.001895653,0.007230073,0.009134641,0.006213211,0.2061664],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006359728,"about_ca_system_score_gemma":0.01024071,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01534557,"about_ca_topic_score_gemma":0.0164502,"domain_scores_codex":[0.959884,0.007229419,0.002251665,0.003241333,0.0235171,0.003876432],"domain_scores_gemma":[0.9319028,0.008570137,0.001171745,0.008966945,0.03857144,0.01081696],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002411052,0.000114748,0.0001601763,0.0003160967,0.00002498215,0.00004916823,0.00004027828,0.0007426981,0.0004002168,0.004560668,0.9546176,0.03873226],"study_design_scores_gemma":[0.0001482552,0.0002414743,0.001050797,0.0004936737,0.00004307202,0.0001435207,0.00008279396,0.002971603,0.002356835,0.007511264,0.9848968,0.00005998854],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"dataset","genre_scores_codex":[0.01054448,0.03129565,0.1105039,0.04388827,0.09274896,0.002709629,0.1409762,0.04769643,0.5196365],"genre_scores_gemma":[0.03650375,0.009605893,0.05802827,0.008031659,0.009259609,0.002046042,0.3253707,0.03907972,0.5120745],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.2007698,"threshold_uncertainty_score":0.6716418,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03174916285699964,"score_gpt":0.2601869409768824,"score_spread":0.2284377781198828,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}