{"id":"W4400499505","doi":"10.1007/978-3-031-64171-8_19","title":"Extended Abstract: Assessing Language Models for Semantic Textual Similarity in Cybersecurity","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Network Security and Intrusion Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Semantic similarity; Similarity (geometry); Natural language processing; Artificial intelligence; Information retrieval; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003237943,0.000939226,0.0007450517,0.002443482,0.0007571056,0.00375634,0.001530077,0.00150093,0.03271544],"category_scores_gemma":[0.04159312,0.0002780593,0.00130927,0.002218586,0.0006409013,0.005621073,0.001736597,0.001154795,0.008514442],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009733666,"about_ca_system_score_gemma":0.0009999046,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003273813,"about_ca_topic_score_gemma":0.00188831,"domain_scores_codex":[0.9968976,0.001097642,0.0002532866,0.0005490271,0.001089585,0.0001128928],"domain_scores_gemma":[0.9705849,0.02099849,0.001003853,0.002463714,0.00438468,0.000564363],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002171725,0.001834535,0.02932194,0.001979834,0.0004564999,0.0004063172,0.001647154,0.05380389,0.02158263,0.02987582,0.104819,0.7521006],"study_design_scores_gemma":[0.0002523661,0.001477131,0.03389451,0.0005165104,0.0003946312,0.001262998,0.001883526,0.785948,0.03651957,0.1104378,0.02716736,0.0002456227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2965102,0.002233024,0.6301408,0.002802312,0.002013165,0.001231842,0.01642586,0.008892618,0.03975013],"genre_scores_gemma":[0.8228187,0.0006126549,0.1354744,0.0004743753,0.0007249222,0.0009141304,0.02366746,0.0009892659,0.01432401],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03271544,"threshold_uncertainty_score":0.109444,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02162344388602977,"score_gpt":0.2768957237447787,"score_spread":0.2552722798587489,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}