{"id":"W4388078721","doi":"10.36227/techrxiv.24438271.v1","title":"SparkSim: Performance Modeling for Resource Allocation of Spark Data Science Projects","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Big Data and Business Intelligence","field":"Business, Management and Accounting","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"SPARK (programming language); Computer science; Data science; Process (computing); Analytics; Data analysis; Resource (disambiguation); Big data; Predictive modelling; Resource allocation; Multitude; Machine learning; Industrial engineering; Artificial intelligence; Data mining; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002487544,0.001104285,0.0006444582,0.001170776,0.0004710385,0.001449976,0.001685783,0.00108189,0.003656283],"category_scores_gemma":[0.00894828,0.0005163072,0.0007673184,0.001501911,0.0004702054,0.001396657,0.0006837558,0.001436976,0.0009221771],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001993722,"about_ca_system_score_gemma":0.002213904,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02562813,"about_ca_topic_score_gemma":0.01790781,"domain_scores_codex":[0.9992133,0.0002515011,0.00004715072,0.0001854403,0.0001681029,0.0001344284],"domain_scores_gemma":[0.9966881,0.00198569,0.0003124287,0.0003155188,0.0005089115,0.0001894022],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009798457,0.00006396703,0.003785325,0.00004206045,0.00002234849,0.00002876354,0.00004922014,0.9802562,0.0003123487,0.00327561,0.004378741,0.007687458],"study_design_scores_gemma":[0.00000437189,0.000007857021,0.0003219062,0.000002399698,0.000002213783,0.000004881524,0.000008761916,0.9977726,0.0001728306,0.001300053,0.0003986403,0.000003413564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4796906,0.0007382451,0.4683222,0.003388356,0.000248113,0.0003519457,0.0143108,0.01680123,0.01614841],"genre_scores_gemma":[0.927746,0.0002977471,0.06246857,0.0001630206,0.00006401534,0.000292948,0.005417197,0.0005494338,0.003000971],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02562813,"threshold_uncertainty_score":0.05095792,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4044783927831498,"score_gpt":0.3639950429362652,"score_spread":0.04048334984688462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}