{"id":"W3011588630","doi":"10.1109/escience.2019.00023","title":"Evaluation of pilot jobs for Apache Spark applications on HPC clusters","year":2019,"lang":"","type":"article","venue":"Espace ÉTS (ETS)","topic":"Distributed and Parallel Computing Systems","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"SPARK (programming language); Computer science; Big data; Supercomputer; Scheduling (production processes); Debugging; Software deployment; Operating system; Pipeline transport; Database; Distributed computing; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.005453609,0.0004573862,0.0006329704,0.0002632324,0.0002758352,0.0002971555,0.001496474,0.0002028891,0.00009063337],"category_scores_gemma":[0.0002414315,0.0004702205,0.0002819846,0.0009675811,0.000101204,0.0002995214,0.0002582718,0.0002845035,0.0009238572],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003232011,"about_ca_system_score_gemma":0.0006430225,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005996905,"about_ca_topic_score_gemma":0.0000300525,"domain_scores_codex":[0.9947476,0.0007034143,0.000821077,0.001174178,0.001846846,0.0007068786],"domain_scores_gemma":[0.9949916,0.0006998791,0.000888971,0.001905198,0.001285159,0.0002292211],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006561344,0.002661915,0.003283091,0.001497168,0.0008395436,0.000001766835,0.008335574,0.6498915,0.003151935,0.1164692,0.03509735,0.1781148],"study_design_scores_gemma":[0.003222496,0.001627309,0.002632797,0.0004742806,0.0002443045,0.000007528981,0.0002423917,0.9361153,0.0006323985,0.001311168,0.05286844,0.0006216299],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0742696,0.0008860884,0.894132,0.0029876,0.003635121,0.007675749,0.0001611625,0.0001615021,0.01609118],"genre_scores_gemma":[0.9936247,0.00001286297,0.002808856,0.0001992726,0.0004352043,0.0003365196,0.00006045578,0.00004046701,0.002481677],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9193551,"threshold_uncertainty_score":0.999854,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05107269823017081,"score_gpt":0.3051105241221561,"score_spread":0.2540378258919853,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}