{"id":"W2735418241","doi":"10.1109/icdcs.2017.317","title":"Learning from Failure Across Multiple Clusters: A Trace-Driven Approach to Understanding, Predicting, and Mitigating Job Terminations","year":2017,"lang":"en","type":"article","venue":"","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"Computer science; Limiting; Usability; Task (project management); Cluster (spacecraft); Predictive power; Power consumption; Scale (ratio); Resource (disambiguation); TRACE (psycholinguistics); Machine learning; Power (physics); Engineering; Human–computer interaction","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003853987,0.001207271,0.001072426,0.002881363,0.00085998,0.001469059,0.002578885,0.001117902,0.0005026852],"category_scores_gemma":[0.0177988,0.0006123171,0.00068289,0.001519344,0.0007761994,0.002017016,0.001384925,0.001919422,0.0002515391],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001088401,"about_ca_system_score_gemma":0.00206281,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01893923,"about_ca_topic_score_gemma":0.02225865,"domain_scores_codex":[0.9984933,0.0004003754,0.0001345839,0.000414618,0.0003865509,0.00017048],"domain_scores_gemma":[0.9852438,0.008093403,0.002288519,0.001729279,0.00172136,0.0009236906],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000503055,0.0008775172,0.1576675,0.0002041654,0.0002504645,0.0005433653,0.0009714897,0.688545,0.005944947,0.0036523,0.002415676,0.1384246],"study_design_scores_gemma":[0.00000791432,0.00007011757,0.005700239,0.00001159264,0.0000164618,0.00004570869,0.00009223376,0.9884601,0.0009894039,0.004290999,0.0002973466,0.00001792147],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5052021,0.0005408231,0.4865609,0.001337201,0.00006966391,0.0002876347,0.001174911,0.00334195,0.001484855],"genre_scores_gemma":[0.9444001,0.0001279877,0.0536661,0.00007213519,0.00003757832,0.00008897591,0.0009685915,0.00007883603,0.0005596985],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01893923,"threshold_uncertainty_score":0.03765798,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0399174546020463,"score_gpt":0.2682651498599565,"score_spread":0.2283476952579102,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}