{"id":"W7081581119","doi":"","title":"Anti-Money Laundering Machine Learning Pipelines; A Technical Analysis on Identifying High-risk Bank Clients with Supervised Learning","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Toronto","keywords":"Pipeline (software); Technical analysis; Feature (linguistics); Feature engineering; Supervised learning; Task (project management); Big data","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005211123,0.0009640271,0.0005618942,0.001874422,0.0005867925,0.001160108,0.001554526,0.0008925284,0.001648855],"category_scores_gemma":[0.01259565,0.000398675,0.0007322695,0.0009570281,0.0005778648,0.001783657,0.001774694,0.001617567,0.001440575],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009591796,"about_ca_system_score_gemma":0.002005348,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003912753,"about_ca_topic_score_gemma":0.005553674,"domain_scores_codex":[0.9977824,0.0008329769,0.0001324615,0.0005304763,0.0005450093,0.0001767029],"domain_scores_gemma":[0.9931117,0.003286856,0.0006498569,0.001312329,0.001356847,0.0002824808],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008092609,0.001262276,0.06651153,0.0003142059,0.0002866747,0.0003495043,0.0004837772,0.08137154,0.01985445,0.008828388,0.03104262,0.7888858],"study_design_scores_gemma":[0.00004323329,0.0002330759,0.00940668,0.00002206011,0.0000311453,0.000114785,0.00008679172,0.9637096,0.01439418,0.007832908,0.004095407,0.00003016225],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.18941,0.0006824399,0.7653072,0.002264573,0.000128483,0.000647501,0.002665488,0.03418008,0.00471423],"genre_scores_gemma":[0.6634685,0.0001416954,0.327433,0.0004417143,0.00009300871,0.0003516292,0.005153623,0.0002601708,0.00265654],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005211123,"threshold_uncertainty_score":0.0275594,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01906870289495001,"score_gpt":0.2476202153990997,"score_spread":0.2285515125041497,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}