{"id":"W4416961966","doi":"10.1109/pst65910.2025.11268838","title":"RefPentester: A Knowledge-Informed Self-Reflective Penetration Testing Framework Based on Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; Concordia University","funders":"Concordia University","keywords":"Process (computing); Hacker; Baseline (sea); Fuzz testing; Soundness","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003127968,0.002006563,0.0007633177,0.001276464,0.0004974408,0.001808304,0.003919564,0.001663826,0.005398311],"category_scores_gemma":[0.01499851,0.001055877,0.001960748,0.0004744459,0.001756031,0.005066982,0.003876431,0.002889922,0.001589992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001112967,"about_ca_system_score_gemma":0.002676454,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006641383,"about_ca_topic_score_gemma":0.01275848,"domain_scores_codex":[0.9973549,0.001189355,0.000150186,0.0004748134,0.0006418177,0.0001889745],"domain_scores_gemma":[0.9933468,0.004221919,0.0005170785,0.00118194,0.000524405,0.0002078566],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004333526,0.0006705,0.007135754,0.0008856354,0.0002258056,0.001048077,0.001352783,0.4794097,0.01299315,0.05290028,0.01891852,0.4240264],"study_design_scores_gemma":[0.00003397859,0.00008535808,0.0002293335,0.00004897875,0.00002409583,0.0001780096,0.00005454332,0.9668605,0.003358212,0.02266419,0.006429221,0.00003366501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00677505,0.0002122404,0.9566146,0.0003829648,0.00003096374,0.0002462417,0.0002506632,0.03353546,0.001951847],"genre_scores_gemma":[0.2859227,0.0003263704,0.7056496,0.000588985,0.00003401929,0.0006003071,0.001357114,0.002420604,0.003100357],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006641383,"threshold_uncertainty_score":0.01805913,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03270531685287602,"score_gpt":0.3436117691507119,"score_spread":0.3109064522978359,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}