{"id":"W4392405447","doi":"10.1109/tifs.2024.3372809","title":"(Security) Assertions by Large Language Models","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Information Forensics and Security","topic":"Topic Modeling","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Office of Naval Research; Intel Corporation","keywords":"Computer science; Programming language; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008669131,0.001605205,0.0005853594,0.002018589,0.0007829816,0.003795202,0.002478983,0.00145919,0.005678999],"category_scores_gemma":[0.04247289,0.001167498,0.002765731,0.001218732,0.002132152,0.006719755,0.003363739,0.00264258,0.001565886],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001786074,"about_ca_system_score_gemma":0.002641697,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004542458,"about_ca_topic_score_gemma":0.008139155,"domain_scores_codex":[0.9883795,0.00555519,0.0008261862,0.001187108,0.003664903,0.0003870652],"domain_scores_gemma":[0.9561854,0.03073922,0.00257687,0.006948658,0.003223866,0.0003259752],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003601798,0.0003050286,0.004093342,0.0007879883,0.0001475578,0.0006493682,0.001977353,0.3101143,0.009341063,0.5622144,0.01331762,0.09669181],"study_design_scores_gemma":[0.0000947432,0.0001111625,0.0002825929,0.000201199,0.00006883314,0.0002285698,0.0002024006,0.7328943,0.008105324,0.2279138,0.02985405,0.0000430063],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01222057,0.0002319965,0.9772988,0.0007478909,0.00006130342,0.0003144423,0.001156197,0.004717401,0.003251503],"genre_scores_gemma":[0.2094962,0.0004733701,0.7809645,0.0003910759,0.0001135857,0.0009503891,0.00308537,0.001325519,0.003200044],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008669131,"threshold_uncertainty_score":0.04584736,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00911134536940814,"score_gpt":0.2310153578091204,"score_spread":0.2219040124397123,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}