{"id":"W4414413212","doi":"10.21203/rs.3.rs-7511791/v1","title":"Fine-grained Insider Threat Detection with Large Language Models: A Comparative Study","year":2025,"lang":"en","type":"preprint","venue":"Research Square","topic":"Information and Cyber Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"Compute Canada","keywords":"Insider threat; Insider; Generative grammar; Generative model; Behavioral analysis; Stability (learning theory); Behavioral economics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009750758,0.0009300683,0.001066625,0.001103222,0.0008141692,0.002940703,0.001828502,0.001649605,0.003809934],"category_scores_gemma":[0.05675405,0.0007060987,0.001010088,0.00101896,0.001127879,0.01101902,0.001906455,0.003077322,0.001344908],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001670095,"about_ca_system_score_gemma":0.001205609,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009442389,"about_ca_topic_score_gemma":0.007463843,"domain_scores_codex":[0.9939907,0.00418576,0.0002064058,0.0008413488,0.0005306946,0.0002451494],"domain_scores_gemma":[0.8583419,0.128842,0.002705641,0.006626239,0.002495212,0.0009890138],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01106074,0.005691331,0.1300765,0.001429683,0.002090886,0.0009592552,0.008060766,0.3849354,0.01749058,0.01901631,0.01173029,0.4074582],"study_design_scores_gemma":[0.0002127699,0.001047356,0.02846532,0.00006235709,0.0003971454,0.0003072536,0.001669334,0.945546,0.003108338,0.01678981,0.002272745,0.0001215946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9416169,0.001242361,0.04896776,0.001270697,0.0000628302,0.000132437,0.0008172899,0.001138212,0.004751467],"genre_scores_gemma":[0.9895585,0.0001664029,0.008611053,0.0001273282,0.00002587811,0.00003696306,0.0007622414,0.0001148981,0.0005968346],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009750758,"threshold_uncertainty_score":0.05156755,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0821520498102914,"score_gpt":0.3995551456815371,"score_spread":0.3174030958712457,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}