{"id":"W4404569041","doi":"10.48550/arxiv.2411.09772","title":"Beyond Static Tools: Evaluating Large Language Models for Cryptographic Misuse Detection","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Cryptography; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01001082,0.002130753,0.0009275808,0.004811554,0.0007582245,0.003050621,0.00219216,0.001777812,0.001561029],"category_scores_gemma":[0.05242215,0.0005840578,0.00156032,0.002572465,0.001023876,0.005795082,0.002467866,0.002402365,0.00141191],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00210455,"about_ca_system_score_gemma":0.002663883,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01306968,"about_ca_topic_score_gemma":0.0185971,"domain_scores_codex":[0.9903941,0.005128413,0.0008007446,0.001365334,0.001986128,0.0003254566],"domain_scores_gemma":[0.9395072,0.04875443,0.002229715,0.00504731,0.003621975,0.0008393178],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003858764,0.003541308,0.1329289,0.003295977,0.001487848,0.0005909274,0.002570215,0.2743026,0.01236301,0.01122184,0.05737823,0.4964605],"study_design_scores_gemma":[0.0001442245,0.0006627976,0.007699803,0.0001980953,0.0001552113,0.0001981119,0.00058507,0.9689804,0.005855126,0.006790488,0.008656361,0.00007431364],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.829596,0.004712568,0.1021309,0.002047894,0.0004328009,0.0007412552,0.01460268,0.03769973,0.008036281],"genre_scores_gemma":[0.8380553,0.0008845698,0.127722,0.000579307,0.0001201718,0.0004823713,0.02879642,0.001439315,0.001920591],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01306968,"threshold_uncertainty_score":0.05294293,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1040626596519416,"score_gpt":0.2435424170057431,"score_spread":0.1394797573538016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}