{"id":"W6949326520","doi":"10.5281/zenodo.15012096","title":"SV-Benchmarks: Benchmark Set for Software Verification (SV-COMP 2025)","year":2025,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"","field":"","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Software verification; Benchmarking; Benchmark (surveying); Intelligent verification; Software; Verification; Metadata; Functional verification; Runtime verification","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005626401,0.003627654,0.001131317,0.003703813,0.0009860938,0.00254399,0.004193217,0.002548091,0.0306936],"category_scores_gemma":[0.01782804,0.001002464,0.002020874,0.004071118,0.0007073858,0.002180912,0.002401909,0.002655801,0.0226799],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00195661,"about_ca_system_score_gemma":0.003625454,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01014727,"about_ca_topic_score_gemma":0.0131454,"domain_scores_codex":[0.9930021,0.002129785,0.0007610162,0.0006989583,0.002860593,0.000547478],"domain_scores_gemma":[0.9912099,0.003033285,0.0004614376,0.001628295,0.003224387,0.0004427651],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006143979,0.0002939926,0.001333516,0.003217085,0.0001543416,0.000139997,0.0000690001,0.0254404,0.002947524,0.008006015,0.8966509,0.06113276],"study_design_scores_gemma":[0.001081518,0.0008226641,0.006590383,0.001651592,0.0002498487,0.0006446631,0.0001968252,0.1411773,0.0220393,0.02676476,0.7986127,0.000168564],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.05071598,0.01804414,0.1050859,0.003717884,0.00357838,0.002612775,0.4765669,0.167133,0.172545],"genre_scores_gemma":[0.0726901,0.002455193,0.07622806,0.001026322,0.0001638602,0.001861685,0.8077919,0.01905832,0.01872455],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.0306936,"threshold_uncertainty_score":0.1026803,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03688140675897977,"score_gpt":0.2828759407641014,"score_spread":0.2459945340051216,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}