{"id":"W4404569041","doi":"10.48550/arxiv.2411.09772","title":"Beyond Static Tools: Evaluating Large Language Models for Cryptographic Misuse Detection","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Cryptography; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0007874892,0.000310249,0.0002744761,0.0005123788,0.0002611401,0.0004697428,0.0009285051,0.0002990142,0.00001036027],"category_scores_gemma":[0.0000965287,0.0003616899,0.0003323586,0.0008289289,0.00003518584,0.0005935687,0.001058225,0.0006494991,0.00003882369],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002007883,"about_ca_system_score_gemma":0.0001499333,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001482624,"about_ca_topic_score_gemma":0.0002088648,"domain_scores_codex":[0.9978966,0.0001571455,0.0002290255,0.001158064,0.0001494563,0.0004097153],"domain_scores_gemma":[0.9983286,0.0002577139,0.0002022491,0.0009041657,0.0001857658,0.0001214555],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001481529,0.0001889644,0.00005666906,0.001373262,0.0004422907,0.0002040522,0.005898066,0.6475059,0.002575342,0.3114612,0.0003510753,0.02979499],"study_design_scores_gemma":[0.000318766,0.00009462259,0.00002035658,0.00009052128,0.0001117959,0.000002888742,0.0001531271,0.6649203,0.0004771072,0.333498,0.00004689944,0.0002656015],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3244585,0.0002080737,0.672592,0.00003947282,0.001240946,0.0004559126,0.00004667453,0.0004527201,0.0005056349],"genre_scores_gemma":[0.9939676,0.00005033806,0.00523727,0.00008420939,0.0001642562,0.00001111967,0.00002437218,0.00003492022,0.0004259376],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6695091,"threshold_uncertainty_score":0.9998835,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1040626596519416,"score_gpt":0.2435424170057431,"score_spread":0.1394797573538016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}