{"id":"W2128322880","doi":"10.4018/jdwm.2005040101","title":"The Use of Smart Tokens in Cleaning Integrated Warehouse Data","year":2005,"lang":"en","type":"article","venue":"International Journal of Data Warehousing and Mining","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Data warehouse; Security token; Database; Alphanumeric; Matching (statistics); Data mining; Unique identifier; Identifier; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002575184,0.0007611071,0.001578684,0.003447952,0.001365228,0.003118864,0.003744211,0.001074043,0.001242942],"category_scores_gemma":[0.01205782,0.0008173598,0.001414351,0.006191608,0.001455021,0.006390649,0.003494951,0.0013657,0.00191091],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001114868,"about_ca_system_score_gemma":0.002661657,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003532039,"about_ca_topic_score_gemma":0.002655116,"domain_scores_codex":[0.9938956,0.0006992731,0.001098014,0.00137277,0.002602998,0.0003313158],"domain_scores_gemma":[0.989056,0.002447736,0.00236171,0.003879423,0.001895846,0.000359282],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001858164,0.0003640988,0.02347237,0.0008374796,0.0002505691,0.0009476087,0.002139027,0.02854486,0.07004926,0.03152947,0.007502207,0.8325049],"study_design_scores_gemma":[0.0002040747,0.001351347,0.01167416,0.0002548191,0.0004227506,0.004150433,0.002179639,0.351941,0.4796521,0.04729123,0.100316,0.0005623276],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04962937,0.0005188232,0.9429004,0.0001735346,0.0001409286,0.0002190135,0.0005592613,0.005003414,0.0008552823],"genre_scores_gemma":[0.1514124,0.0002597928,0.8447666,0.00009751943,0.00002591159,0.0001393031,0.00137337,0.0004404357,0.001484697],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003744211,"threshold_uncertainty_score":0.01361907,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5821457750325668,"score_gpt":0.474360714999418,"score_spread":0.1077850600331489,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}