{"id":"W4410041812","doi":"10.1007/978-3-031-87499-4_11","title":"Evaluating Large Language Models on Cybersecurity Knowledge and Skills: A Comparative Analysis","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Information and Cyber Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université du Québec en Outaouais","funders":"","keywords":"Computer science; Computer security; Natural language processing; Software engineering; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03977222,0.001264454,0.001522612,0.00445433,0.0009350882,0.005898515,0.003684806,0.002710958,0.009731876],"category_scores_gemma":[0.190564,0.0007648953,0.002506817,0.003868544,0.002796485,0.01343423,0.003448016,0.003277802,0.001260871],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004913223,"about_ca_system_score_gemma":0.001662968,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01044596,"about_ca_topic_score_gemma":0.008083645,"domain_scores_codex":[0.979995,0.01573254,0.0007887,0.001371077,0.001672094,0.000440585],"domain_scores_gemma":[0.2797357,0.7059074,0.003704396,0.005129834,0.00442982,0.001092978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.04022761,0.008752159,0.2306775,0.00253653,0.007536792,0.0007378242,0.008754796,0.3714197,0.003705129,0.05002374,0.009099823,0.2665284],"study_design_scores_gemma":[0.002012466,0.006662844,0.1220101,0.0006671175,0.004602876,0.0003839053,0.007204283,0.7655661,0.002792391,0.08419646,0.003522576,0.0003787743],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9675066,0.003106671,0.01849203,0.0009701456,0.00006941611,0.0001824532,0.001022646,0.0002260328,0.008423966],"genre_scores_gemma":[0.9912241,0.0004712733,0.005135754,0.0001298018,0.00004182254,0.0001670434,0.001699429,0.0001118181,0.001019081],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03977222,"threshold_uncertainty_score":0.2103382,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02611560327253232,"score_gpt":0.3291824211190175,"score_spread":0.3030668178464851,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}