{"id":"W4415007625","doi":"10.1145/3763174","title":"AutoVerus: Automated Proof Generation for Rust Code","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Correctness; Mathematical proof; Debugging; Code (set theory); Proof of concept; Suite; Benchmark (surveying); Rust (programming language); Automated theorem proving","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003897014,0.001053354,0.0005557426,0.001858938,0.0006184386,0.001660285,0.00224407,0.001002495,0.01301753],"category_scores_gemma":[0.02027068,0.0008879031,0.001401471,0.0007778654,0.001863544,0.002549415,0.003128,0.001526603,0.003070772],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009082966,"about_ca_system_score_gemma":0.002491988,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001570941,"about_ca_topic_score_gemma":0.002070665,"domain_scores_codex":[0.9962332,0.00146582,0.000228843,0.0004503536,0.001413572,0.0002082422],"domain_scores_gemma":[0.9842461,0.009826516,0.0006828907,0.003444162,0.001592206,0.0002082174],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006412062,0.0003885838,0.003768337,0.002243288,0.0002098857,0.0009840003,0.001022346,0.09939758,0.04613129,0.1262078,0.06466213,0.6543435],"study_design_scores_gemma":[0.00040235,0.0003380548,0.0009084113,0.0004143236,0.00006714062,0.001125259,0.0001920205,0.7176665,0.09202041,0.1029084,0.08385533,0.0001018288],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.01454071,0.0004084486,0.9045817,0.0004080707,0.00009837454,0.0002926451,0.0009229124,0.07282961,0.00591753],"genre_scores_gemma":[0.1807435,0.0003866868,0.8005425,0.0003330628,0.00004715104,0.0003471865,0.003072695,0.01121317,0.003314081],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.01301753,"threshold_uncertainty_score":0.04354799,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02026451834181121,"score_gpt":0.3060811090185903,"score_spread":0.2858165906767791,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}