{"id":"W7127444902","doi":"10.1109/ccece64018.2025.11364472","title":"Advanced Feature Engineering for Twitter Bot Detection: Utilizing Metadata, NLP, and Transformers","year":2025,"lang":"","type":"article","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"New York Institute of Technology","funders":"","keywords":"Transformer; Feature extraction; Feature engineering; Feature (linguistics); Natural language; Random forest; Deep learning; Feature learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000840758,0.0007326381,0.0006647175,0.002955273,0.0003286964,0.0006755048,0.0005127892,0.0003930686,0.00104319],"category_scores_gemma":[0.003214291,0.00015135,0.0004679806,0.001734698,0.0002756066,0.001879966,0.0007706599,0.0005412571,0.0008588319],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003992709,"about_ca_system_score_gemma":0.0006060131,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003219471,"about_ca_topic_score_gemma":0.004003596,"domain_scores_codex":[0.9995441,0.00009408867,0.00004175685,0.00008300635,0.0001736476,0.00006349429],"domain_scores_gemma":[0.9989561,0.0004054238,0.0001431402,0.0001449298,0.0002974231,0.00005295603],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006712578,0.0005908911,0.02994078,0.0001918379,0.00008172316,0.0002829189,0.00009012466,0.02979878,0.04013006,0.002398786,0.009860835,0.885962],"study_design_scores_gemma":[0.00006379942,0.0003407236,0.01379184,0.0000267228,0.00006559972,0.0004125548,0.0001290495,0.9419266,0.03239815,0.006162109,0.004628968,0.00005390361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3298994,0.001085616,0.6461758,0.0007337466,0.0001338366,0.0002476258,0.002674081,0.01564327,0.00340671],"genre_scores_gemma":[0.9034694,0.0002721934,0.09130817,0.0000857077,0.00007244088,0.00009541408,0.003426734,0.00009967112,0.001170165],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003219471,"threshold_uncertainty_score":0.006401479,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01242306746153325,"score_gpt":0.2450554857543261,"score_spread":0.2326324182927929,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}