{"id":"W7115041457","doi":"","title":"HarMoEny: Efficient multi-GPU inference of Mixture of Experts models by scheduling experts across accelerators","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Inference; Scheduling (production processes); Mixture model; Job shop scheduling; Expert system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001082167,0.001117191,0.001504901,0.0005111707,0.001067979,0.0002777046,0.002504308,0.001174807,0.00002210615],"category_scores_gemma":[0.0008493682,0.001157735,0.000604152,0.001356705,0.0001267017,0.0009254853,0.000668106,0.001164135,0.000009005524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003817545,"about_ca_system_score_gemma":0.000226524,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004151744,"about_ca_topic_score_gemma":0.0002086389,"domain_scores_codex":[0.9933786,0.0004210928,0.001793394,0.001921428,0.001348476,0.001137023],"domain_scores_gemma":[0.994408,0.0004942577,0.00139383,0.002004191,0.00131279,0.0003869077],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002261121,0.001444294,0.00004146438,0.002195509,0.0004859473,0.00006956052,0.001488629,0.03959168,0.6970131,0.03024312,0.00005848424,0.2271421],"study_design_scores_gemma":[0.001438368,0.0001342628,0.00004135465,0.003385119,0.00009026114,0.00001168147,0.001595202,0.06675834,0.9218051,0.001280467,0.001940602,0.001519235],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9889687,0.002563037,0.001777466,0.00001296749,0.001954653,0.0007565578,0.0007408274,0.000356891,0.002868877],"genre_scores_gemma":[0.9823437,0.0001873002,0.01476963,0.0001179343,0.00002430206,0.000108445,0.0002723301,0.0001166713,0.002059647],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2256229,"threshold_uncertainty_score":0.9990873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02738314471006957,"score_gpt":0.2804819732843613,"score_spread":0.2530988285742917,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}