{"id":"W4403024154","doi":"10.1109/lad62341.2024.10691745","title":"Toward Hardware Security Benchmarking of LLMs","year":2024,"lang":"en","type":"article","venue":"","topic":"Security and Verification in Computing","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Intel Corporation","keywords":"Benchmarking; Computer science; Computer security; Business","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0135204,0.001064754,0.0005513167,0.002838899,0.0005907934,0.002408747,0.001954271,0.001088545,0.003030028],"category_scores_gemma":[0.0547177,0.0005071145,0.0006878515,0.001538293,0.001129093,0.003092487,0.002015024,0.001828648,0.0007296233],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0016768,"about_ca_system_score_gemma":0.001922822,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001445079,"about_ca_topic_score_gemma":0.001994294,"domain_scores_codex":[0.9834907,0.00846792,0.00102054,0.0006490047,0.005745048,0.000626737],"domain_scores_gemma":[0.9647816,0.01438627,0.002138822,0.00987929,0.008377375,0.0004365893],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0015293,0.001760424,0.03975047,0.001833891,0.0002822244,0.0007916953,0.00217298,0.3349539,0.08160003,0.09007598,0.02164643,0.4236026],"study_design_scores_gemma":[0.0002134024,0.002232234,0.007723517,0.0005796829,0.000109661,0.0005048296,0.0008306989,0.7892607,0.138006,0.02529899,0.03512654,0.0001138248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.553889,0.001428512,0.3984122,0.001774671,0.0002882465,0.0009569703,0.001651025,0.01436677,0.02723263],"genre_scores_gemma":[0.7438428,0.0004574913,0.2487485,0.0003410258,0.00002942907,0.0004785914,0.002343543,0.001210394,0.002548106],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0135204,"threshold_uncertainty_score":0.07150358,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03368345729125524,"score_gpt":0.2739716618867946,"score_spread":0.2402882045955393,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}