{"id":"W4411436346","doi":"10.1007/978-3-031-87496-3_12","title":"Is Expert-Labeled Data Worth the Cost? Exploring Active and Semi-supervised Learning Across Imbalance Scenarios in Financial Crime Detection","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Supervised learning; Finance; Labeled data; Data science; Artificial neural network; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001182519,0.0005312658,0.0005280591,0.0005073483,0.0006284685,0.0008047377,0.005375485,0.0003174369,0.000004488116],"category_scores_gemma":[0.0005797892,0.0004566247,0.00005407964,0.001161002,0.0006371358,0.002113952,0.004528206,0.001684748,0.000006208366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004292613,"about_ca_system_score_gemma":0.0005340411,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001086603,"about_ca_topic_score_gemma":0.000347763,"domain_scores_codex":[0.9955584,0.0001051502,0.0005862812,0.002241578,0.0007517352,0.0007568746],"domain_scores_gemma":[0.9959663,0.0007063072,0.0003301843,0.002676091,0.0002240888,0.00009709216],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001960047,0.00002038783,0.0002401145,0.00002625345,0.000006634834,0.00001942443,0.003892556,0.0006363085,0.001086984,0.0007379581,0.00004350365,0.9932703],"study_design_scores_gemma":[0.0006310561,0.0001209647,0.004163099,0.001114743,0.000007346319,0.0000358758,0.000008070558,0.9501751,0.02823408,0.006922337,0.007625917,0.000961417],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0009825273,0.0004476155,0.9954799,0.000955042,0.0008427275,0.0007938006,0.00003357307,0.0002673389,0.0001974857],"genre_scores_gemma":[0.8142642,0.001503489,0.1784149,0.004495844,0.0005602643,0.0002892817,0.00006667433,0.00006576004,0.0003396375],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9923089,"threshold_uncertainty_score":0.9997885,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05898058754057987,"score_gpt":0.3038318852418694,"score_spread":0.2448512977012895,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}