{"id":"W4410524059","doi":"10.2139/ssrn.5261048","title":"Unsupervised, Robust, and Lightweight Detection of Data Pattern Anomalies and Outliers","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Outlier; Anomaly detection; Pattern recognition (psychology); Computer science; Artificial intelligence; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001647094,0.000860368,0.001411985,0.002270585,0.0005512811,0.00166503,0.001804892,0.001231977,0.0008411638],"category_scores_gemma":[0.01073995,0.0005328681,0.0009808855,0.001936423,0.0007819724,0.002050578,0.002579765,0.001829763,0.001236584],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004036472,"about_ca_system_score_gemma":0.001347943,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009562429,"about_ca_topic_score_gemma":0.001597411,"domain_scores_codex":[0.9968644,0.0004597345,0.0002236601,0.0006874047,0.001515863,0.000248947],"domain_scores_gemma":[0.9929367,0.002041657,0.001359438,0.002049396,0.001392374,0.0002204512],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007261214,0.0004326219,0.02223962,0.0003068828,0.0002821852,0.0005031795,0.0003035294,0.05631086,0.1361635,0.01028506,0.008187374,0.7642591],"study_design_scores_gemma":[0.00003191104,0.0001966855,0.008602328,0.00002653295,0.00005992148,0.0009485026,0.0001356948,0.9132407,0.05160876,0.02129388,0.003805034,0.00004998784],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03481195,0.0001199555,0.9616141,0.0001616029,0.0000475601,0.00005199817,0.0002210385,0.002515665,0.0004561663],"genre_scores_gemma":[0.4602471,0.000210991,0.5348719,0.0001174129,0.0001512801,0.0001570249,0.001353713,0.0005086637,0.002381836],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002270585,"threshold_uncertainty_score":0.008710802,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01986977007842456,"score_gpt":0.2571072211651208,"score_spread":0.2372374510866962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}