{"id":"W2148260866","doi":"10.4236/jsea.2014.78059","title":"D-IMPACT: A Data Preprocessing Algorithm to Improve the Performance of Clustering","year":2014,"lang":"en","type":"article","venue":"Journal of Software Engineering and Applications","topic":"Advanced Clustering Algorithms Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute of Genetics; University of Tokyo; Institute of Medical Science, University of Tokyo; Research Organization of Information and Systems","keywords":"Cluster analysis; Computer science; Preprocessor; Data mining; Outlier; Data pre-processing; CURE data clustering algorithm; Noise (video); Algorithm; Canopy clustering algorithm; Correlation clustering; Pattern recognition (psychology); Artificial intelligence; Image (mathematics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00426694,0.002350806,0.002027575,0.007405015,0.002839458,0.002628867,0.004073218,0.00184546,0.002598603],"category_scores_gemma":[0.01520426,0.0009099707,0.002050524,0.007863746,0.001111962,0.003444087,0.002947398,0.002808624,0.003096516],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001352775,"about_ca_system_score_gemma":0.002909611,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006554496,"about_ca_topic_score_gemma":0.007236664,"domain_scores_codex":[0.9960175,0.0005870814,0.0004236829,0.0009289716,0.001741666,0.0003010631],"domain_scores_gemma":[0.992239,0.001925514,0.0004849983,0.001432045,0.003665518,0.0002529524],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000668858,0.0003856441,0.00872312,0.000631666,0.0003407039,0.0001976043,0.0006280171,0.06885713,0.02839701,0.01029331,0.02080225,0.8600746],"study_design_scores_gemma":[0.0001350235,0.000320006,0.005388234,0.00007886568,0.0001528896,0.0004692323,0.0003780936,0.8883071,0.06103262,0.01355386,0.02997903,0.0002050323],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00762015,0.0003158445,0.985873,0.0001842389,0.0001651385,0.0002167127,0.0002702017,0.004459593,0.0008951004],"genre_scores_gemma":[0.0538119,0.0002690767,0.9417221,0.0001714987,0.00008851502,0.0003167941,0.001755558,0.0007249297,0.001139636],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007405015,"threshold_uncertainty_score":0.02256596,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01387619459314898,"score_gpt":0.2891263948725445,"score_spread":0.2752502002793955,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}