{"id":"W4415002701","doi":"10.1145/3759425.3763382","title":"Towards Repository-Level Program Verification with Large Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Benchmark (surveying); Focus (optics); Software verification; Software; Code (set theory); Formal verification; Formal methods; Functional verification","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001397282,0.00008137574,0.00008257994,0.00008348991,0.0001142153,0.0001655781,0.0004553181,0.00004275212,0.000001763763],"category_scores_gemma":[0.000006765145,0.00006301798,0.0000232389,0.0003983815,0.00001469177,0.0002469433,0.0001052009,0.00006394937,0.000002893156],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002548753,"about_ca_system_score_gemma":0.00008771675,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004559528,"about_ca_topic_score_gemma":0.000003366342,"domain_scores_codex":[0.9993016,0.00003172801,0.0001268279,0.000259866,0.0001298507,0.0001501212],"domain_scores_gemma":[0.9993477,0.00001121029,0.00004279283,0.0004677373,0.0001004169,0.00003008303],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001205113,0.0003446738,0.0002059908,0.0000292567,0.00003316777,0.00001205123,0.001123099,0.00818797,0.0002937601,0.8603028,0.007092776,0.1223624],"study_design_scores_gemma":[0.0002054515,0.0000642831,0.0005754109,0.00002625302,0.000003615667,0.000004725871,0.00003128248,0.9800802,0.01524542,0.001899547,0.001745225,0.0001186086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0008065252,0.00005387977,0.8985479,0.0003426121,0.00005330028,0.0001865934,5.203902e-7,0.001545283,0.09846338],"genre_scores_gemma":[0.4536579,0.000003189703,0.5424529,0.0001606272,0.00000810882,0.00003506326,0.000002981561,0.00000281111,0.003676401],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9718922,"threshold_uncertainty_score":0.2569799,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01716724560965954,"score_gpt":0.2836060432290026,"score_spread":0.266438797619343,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}