{"id":"W4399851533","doi":"10.1145/3656438","title":"Towards Trustworthy Automated Program Verifiers: Formally Validating Translations into an Intermediate Verification Language","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Security and Verification in Computing","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Trustworthiness; Computer science; Programming language; Software engineering; Natural language processing; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03077989,0.002012511,0.00159718,0.002312079,0.001242909,0.007101581,0.00521571,0.003724247,0.005161744],"category_scores_gemma":[0.09148731,0.002561819,0.004319216,0.001015666,0.008627309,0.01079069,0.007262468,0.007289019,0.002760144],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002769823,"about_ca_system_score_gemma":0.008251173,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001839887,"about_ca_topic_score_gemma":0.001375451,"domain_scores_codex":[0.9568416,0.02063904,0.003364803,0.003998432,0.0123576,0.002798443],"domain_scores_gemma":[0.9009984,0.05713693,0.006709515,0.02465438,0.009809126,0.0006916601],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004862031,0.0004799272,0.004071773,0.001795966,0.0003750217,0.001183063,0.003319977,0.1702399,0.0445144,0.6562523,0.003417532,0.113864],"study_design_scores_gemma":[0.0002768143,0.0004103541,0.0004706919,0.0008663154,0.0002333199,0.0004625931,0.0004147963,0.4996082,0.115303,0.358926,0.02286236,0.0001655519],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003936838,0.0000526609,0.993601,0.0001516834,0.00002940217,0.0001213118,0.0000410758,0.001516781,0.0005492065],"genre_scores_gemma":[0.1413494,0.0002691268,0.853807,0.0003621307,0.00007972165,0.0005237815,0.000434143,0.00166265,0.001512069],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03077989,"threshold_uncertainty_score":0.1627815,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0193190425675604,"score_gpt":0.3301303058016578,"score_spread":0.3108112632340974,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}