{"id":"W4388086255","doi":"10.36227/techrxiv.24438271","title":"SparkSim: Performance Modeling for Resource Allocation of Spark Data Science Projects","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Big Data and Business Intelligence","field":"Business, Management and Accounting","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"SPARK (programming language); Computer science; Data science; Process (computing); Analytics; Data analysis; Resource (disambiguation); Big data; Predictive modelling; Resource allocation; Machine learning; Artificial intelligence; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001994884,0.0009286587,0.0005424176,0.001048186,0.000467802,0.001294965,0.001606198,0.0009893096,0.003307595],"category_scores_gemma":[0.006828561,0.0004551922,0.000723968,0.001245372,0.0004174007,0.00124731,0.0006420586,0.001201638,0.0007788215],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002116965,"about_ca_system_score_gemma":0.002184536,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02932742,"about_ca_topic_score_gemma":0.02371731,"domain_scores_codex":[0.9993808,0.0001982553,0.00003680684,0.0001468783,0.0001344192,0.0001028659],"domain_scores_gemma":[0.9974841,0.001465192,0.0002550781,0.0002290827,0.0003952585,0.0001713013],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008684974,0.00006411678,0.00380677,0.00003498021,0.00001998093,0.00002999893,0.00004413685,0.979021,0.000328115,0.002979763,0.004582862,0.009001492],"study_design_scores_gemma":[0.00000304924,0.000006596291,0.0002222883,0.000001831065,0.00000159085,0.000003279004,0.000006801832,0.9984475,0.0001398896,0.0008456936,0.000318832,0.000002629484],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4791922,0.0007145197,0.4704247,0.003698165,0.0002625247,0.0003112415,0.01140995,0.01657887,0.01740797],"genre_scores_gemma":[0.9314984,0.0002778196,0.05966998,0.0001767899,0.00005976492,0.0002504246,0.004334113,0.0004221557,0.003310544],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02932742,"threshold_uncertainty_score":0.05831343,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4044783927831498,"score_gpt":0.3639950429362652,"score_spread":0.04048334984688462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}