{"id":"W4411689634","doi":"10.1016/j.jpdc.2025.105138","title":"Topology-aware GPU job scheduling with deep reinforcement learning and heuristics","year":2025,"lang":"en","type":"article","venue":"Journal of Parallel and Distributed Computing","topic":"Distributed and Parallel Computing Systems","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"IBM (Canada); York University","funders":"IBM Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Anesthesiologists' Society","keywords":"Computer science; Heuristics; Reinforcement learning; Scheduling (production processes); Parallel computing; Job shop scheduling; General-purpose computing on graphics processing units; Artificial intelligence; Mathematical optimization; Computer graphics (images); Operating system; Graphics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008660475,0.0008560218,0.001505341,0.0008232147,0.0007379428,0.001172681,0.002199105,0.001266219,0.004303117],"category_scores_gemma":[0.003980598,0.0006766004,0.0005702143,0.000737078,0.000777987,0.001323354,0.001210768,0.001705655,0.0006357836],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001403113,"about_ca_system_score_gemma":0.003094984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01398022,"about_ca_topic_score_gemma":0.01884567,"domain_scores_codex":[0.9995071,0.0001243484,0.00002127878,0.000112759,0.0000859882,0.0001484809],"domain_scores_gemma":[0.9980672,0.001078283,0.00013539,0.0001729301,0.0002828322,0.0002633491],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002015189,0.000146486,0.0007394746,0.00003847091,0.00002833428,0.00003402267,0.0000268326,0.9499803,0.0009894886,0.002497679,0.002540274,0.04277715],"study_design_scores_gemma":[0.000007400934,0.000008127367,0.0000288576,0.000001497912,0.000001920159,0.000002319677,0.000004075293,0.9988148,0.00009665566,0.000968589,0.00006416148,0.000001643645],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1205937,0.0008686553,0.8624172,0.0009781533,0.0004496449,0.000159377,0.000264263,0.003738167,0.01053096],"genre_scores_gemma":[0.8916704,0.000104756,0.1047743,0.0002557426,0.00007524329,0.00008939998,0.0002225152,0.0002533039,0.002554242],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01398022,"threshold_uncertainty_score":0.0277977,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009154864165384203,"score_gpt":0.2529626781795075,"score_spread":0.2438078140141233,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}