{"id":"W4206536433","doi":"10.22215/etd/2021-14757","title":"Automated Discovery of Big Data Workload Types","year":2021,"lang":"en","type":"dissertation","venue":"","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Workload; Big data; Computer science; SPARK (programming language); DBSCAN; Cluster analysis; Data mining; Machine learning; Fuzzy clustering; CURE data clustering algorithm; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00151663,0.001166731,0.001015479,0.003898448,0.0008212622,0.003181098,0.001818958,0.0006956311,0.001442961],"category_scores_gemma":[0.00645952,0.0005167693,0.0008837972,0.002003885,0.0003046984,0.002138989,0.001171763,0.001122949,0.001622432],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006696373,"about_ca_system_score_gemma":0.001481843,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002652996,"about_ca_topic_score_gemma":0.004419712,"domain_scores_codex":[0.9979894,0.0002159367,0.0001648368,0.0004890619,0.000972938,0.0001677833],"domain_scores_gemma":[0.9954927,0.001049428,0.0004318243,0.001097725,0.001677972,0.0002502548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004667174,0.0006010493,0.05920338,0.0003597504,0.000127812,0.0002800005,0.00065149,0.01727047,0.05129912,0.006597161,0.01625514,0.8468879],"study_design_scores_gemma":[0.00004691981,0.000137817,0.03546747,0.0001017675,0.00005082913,0.0005416301,0.0008633287,0.8557191,0.07151394,0.01695963,0.01849627,0.0001012579],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1858812,0.0007418161,0.7767161,0.0009467076,0.0002485003,0.0008591155,0.00365489,0.02028203,0.01066957],"genre_scores_gemma":[0.4277435,0.0004692806,0.5586945,0.0002143131,0.0001073305,0.0003010674,0.007193495,0.0007212677,0.004555297],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003898448,"threshold_uncertainty_score":0.008020759,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03133673140251048,"score_gpt":0.2741875819253041,"score_spread":0.2428508505227936,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}