{"id":"W4415002701","doi":"10.1145/3759425.3763382","title":"Towards Repository-Level Program Verification with Large Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Benchmark (surveying); Focus (optics); Software verification; Software; Code (set theory); Formal verification; Formal methods; Functional verification","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0144092,0.001839415,0.001601526,0.002550367,0.001132455,0.005439242,0.005124643,0.002173987,0.003905951],"category_scores_gemma":[0.06705427,0.001455709,0.00265248,0.001953904,0.003369404,0.01067427,0.008009012,0.004902348,0.001593636],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002587114,"about_ca_system_score_gemma":0.007599581,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003993346,"about_ca_topic_score_gemma":0.00771224,"domain_scores_codex":[0.9715286,0.0143793,0.001469398,0.002549231,0.00883462,0.001238929],"domain_scores_gemma":[0.9547869,0.02414967,0.002404642,0.01337546,0.00469358,0.0005897388],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006256385,0.0005308272,0.009338469,0.002010602,0.0003943118,0.0007935785,0.0009085432,0.5501775,0.02629254,0.219138,0.01761038,0.1721796],"study_design_scores_gemma":[0.00008779456,0.0002030435,0.0004199299,0.0002395757,0.00009014093,0.0001930953,0.0002138228,0.8170581,0.02352844,0.1452655,0.01264732,0.00005321831],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03912963,0.0007415832,0.9339382,0.001334789,0.0001153286,0.0002209115,0.0009152938,0.01917262,0.00443166],"genre_scores_gemma":[0.3825859,0.0006535857,0.6053796,0.000695827,0.0001065032,0.00054615,0.003460532,0.004377071,0.002194698],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0144092,"threshold_uncertainty_score":0.07620406,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01716724560965954,"score_gpt":0.2836060432290026,"score_spread":0.266438797619343,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}