{"id":"W4233585779","doi":"10.22215/etd/2014-10523","title":"Empirical Study of Performance of Classification and Clustering Algorithms on Binary Data with Real-World Applications","year":2014,"lang":"en","type":"dissertation","venue":"","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Cluster analysis; Single-linkage clustering; Hierarchical clustering; Computer science; CURE data clustering algorithm; Correlation clustering; Data mining; Pattern recognition (psychology); Centroid; Medoid; Canopy clustering algorithm; Fuzzy clustering; Artificial intelligence; Entropy (arrow of time); Rand index","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01998707,0.0007393179,0.0007248828,0.0038881,0.000897045,0.002958821,0.001331427,0.001409089,0.001199201],"category_scores_gemma":[0.1664047,0.0002492311,0.0006713968,0.005815866,0.001570959,0.003235831,0.001042138,0.001332678,0.0007543586],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001216192,"about_ca_system_score_gemma":0.0006843592,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002118107,"about_ca_topic_score_gemma":0.001636982,"domain_scores_codex":[0.9869515,0.00650597,0.001343098,0.001451524,0.003329895,0.0004180731],"domain_scores_gemma":[0.6872771,0.2758054,0.009822661,0.0110978,0.0143383,0.001658633],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00423941,0.002382035,0.4717976,0.002082346,0.001367026,0.0003565612,0.00261244,0.1590362,0.00451528,0.008103647,0.01670778,0.3267998],"study_design_scores_gemma":[0.0001812957,0.002731863,0.3846204,0.0003717507,0.0002470209,0.001113793,0.002791465,0.5813274,0.008096286,0.01033066,0.008009528,0.0001785313],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9817331,0.00183074,0.01186601,0.0005506253,0.00008530858,0.00009788768,0.001042143,0.000237392,0.002556778],"genre_scores_gemma":[0.9774821,0.0004428826,0.01782952,0.00005359583,0.00006818095,0.00009162823,0.003366245,0.00008247881,0.0005832886],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01998707,"threshold_uncertainty_score":0.1057031,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0676863383469383,"score_gpt":0.3537118901673301,"score_spread":0.2860255518203918,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}