{"id":"W4391718096","doi":"10.31234/osf.io/4nmxh","title":"Unsupervised [randomly responding] survey bot detection: In search of high classification accuracy","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Survey Sampling and Estimation Techniques","field":"Mathematics","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Variety (cybernetics); Set (abstract data type); Data mining; Sample (material); Machine learning; Data set; Data quality; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0215671,0.0009450683,0.001503138,0.001314646,0.001224948,0.001822335,0.002703115,0.002512191,0.001200145],"category_scores_gemma":[0.108944,0.0005273027,0.0009683185,0.00125684,0.001758205,0.00277992,0.001690537,0.002463657,0.001081023],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0012458,"about_ca_system_score_gemma":0.001876908,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002246379,"about_ca_topic_score_gemma":0.003052024,"domain_scores_codex":[0.9816155,0.01158156,0.0007729107,0.003512831,0.001984826,0.000532389],"domain_scores_gemma":[0.8754002,0.07179179,0.01107011,0.03150538,0.009208687,0.001023765],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002532707,0.002231495,0.3387352,0.000903625,0.00093851,0.0003695231,0.001935075,0.2073588,0.0120191,0.02168609,0.02141626,0.3898737],"study_design_scores_gemma":[0.0001596457,0.0004821788,0.02460057,0.00009164831,0.00007076659,0.0002912186,0.0002966598,0.9431656,0.00670586,0.02161233,0.002466427,0.00005710209],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5054016,0.0003926299,0.4826934,0.001695332,0.000178571,0.000999712,0.0007256567,0.003099612,0.004813539],"genre_scores_gemma":[0.86376,0.00007841765,0.1327878,0.0007990176,0.00007897904,0.0004131004,0.0008672046,0.000103572,0.001111972],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9784329,"threshold_uncertainty_score":0.1140592,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2643761707962933,"score_gpt":0.426084819815171,"score_spread":0.1617086490188777,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}