Yiheng Tao, Yihe Zhang, Matthew Dearing, Xin Wang, Yuping Fan, Michael E. Papka, and Zhiling Lan, Ranking Before Serving: Low-Latency LLM Serving via Pairwise Learning-to-Rank, In ISC High Performance 2026 Research Paper Proceedings (41st International Conference), 2026
@inproceedings{TaZhDe26,
author = {Tao, Yiheng and Zhang, Yihe and Dearing, Matthew and Wang, Xin and Fan, Yuping and Papka, Michael E. and Lan, Zhiling},
title = {Ranking Before Serving: Low-Latency LLM Serving via Pairwise Learning-to-Rank},
booktitle = {ISC High Performance 2026 Research Paper Proceedings (41st International Conference)},
year = {2026},
pages = {1--13},
doi = {10.23919/isc.2026.11520485},
keywords = {Modeling;Ranking (statistics);Large language models;Scheduling;Schedules;Learning (artificial intelligence);Cognition;Cognitive systems;Training;Generative Pre-trained transformer;Large Language Models;Inference Scheduling;Shortest Job First;Reasoning Capabilities;Serving Systems},
}
Yiheng Tao, Yuping Fan, Michael E. Papka, and Zhiling Lan, Optimizing LLM Batch Inference with Workload-Aware DVFS, In 13th Greater Chicago Area Systems Research Workshop (GCASR), 2026 Poster
@inproceedings{TaFaPa26,
author = {Tao, Yiheng and Fan, Yuping and Papka, Michael E. and Lan, Zhiling},
title = {Optimizing LLM Batch Inference with Workload-Aware DVFS},
booktitle = {13th Greater Chicago Area Systems Research Workshop (GCASR)},
year = {2026},
note = {Poster},
url = {https://gcasr.org/2026/posters},
}
Boyang Li, Yuping Fan, Michael E. Papka, and Zhiling Lan, Encoding for Reinforcement Learning Driven Scheduling, In Job Scheduling Strategies for Parallel Processing, 2023
Reinforcement learning (RL) is exploited for cluster scheduling in the field of high-performance computing (HPC). One of the key challenges for RL driven scheduling is state representation for RL agent (i.e., capturing essential features of dynamic scheduling environment for decision making). Existing state encoding approaches either lack critical scheduling information or suffer from poor scalability. In this study, we present SEM (Scalable and Efficient encoding Model) for general RL driven scheduling in HPC. It captures system resource and waiting job state, both being critical information for scheduling. It encodes these pieces of information into a fixed-sized vector as an input to the agent. A typical agent is built on deep neural network, and its training/inference cost grows exponentially with the size of its input. Production HPC systems contain a large number of computer nodes. As such, a direct encoding of each of the system resources would lead to poor scalability of the RL agent. SEM uses two techniques to transform the system resource state into a small-sized vector, hence being capable of representing a large number of system resources in a vector of 100–200. Our trace-based simulations demonstrate that compared to the existing state encoding methods, SEM can achieve 9X training speedup and 6X inference speedup while maintaining comparable scheduling performance.
@inproceedings{LiFaPa23,
author = {Li, Boyang and Fan, Yuping and Papka, Michael E. and Lan, Zhiling},
editor = {Klus{\'a}{\v{c}}ek, Dalibor and Julita, Corbal{\'a}n and Rodrigo, Gonzalo P.},
title = {Encoding for Reinforcement Learning Driven Scheduling},
booktitle = {Job Scheduling Strategies for Parallel Processing},
year = {2023},
pages = {68--87},
publisher = {Springer Nature Switzerland},
address = {Cham},
isbn = {978-3-031-22698-4},
}
Yuping Fan, Boyang Li, Dustin Favorite, Naunidh Singh, Taylor Childers, Paul Rich, William Allcock, Michael E. Papka, and Zhiling Lan, DRAS: Deep Reinforcement Learning for Cluster Scheduling in High Performance Computing, IEEE Transactions on Parallel and Distributed Systems, 2022
@article{FaLiFa22,
author = {Fan, Yuping and Li, Boyang and Favorite, Dustin and Singh, Naunidh and Childers, Taylor and Rich, Paul and Allcock, William and Papka, Michael E. and Lan, Zhiling},
title = {DRAS: Deep Reinforcement Learning for Cluster Scheduling in High Performance Computing},
journal = {IEEE Transactions on Parallel and Distributed Systems},
year = {2022},
volume = {33},
number = {12},
pages = {4903--4917},
publisher = {Institute of Electrical and Electronics Engineers (IEEE)},
doi = {10.1109/tpds.2022.3205325},
url = {https://doi.org/10.1109/tpds.2022.3205325},
}
Boyang Li, Yuping Fan, Matthew Dearing, Zhiling Lan, Paul Rich, William Allcock, and Michael E. Papka, MRSch: Multi-Resource Scheduling for HPC, In 2022 IEEE International Conference on Cluster Computing (CLUSTER), 2022
@inproceedings{LiFaDe22,
author = {Li, Boyang and Fan, Yuping and Dearing, Matthew and Lan, Zhiling and Rich, Paul and Allcock, William and Papka, Michael E.},
title = {MRSch: Multi-Resource Scheduling for HPC},
booktitle = {2022 IEEE International Conference on Cluster Computing (CLUSTER)},
year = {2022},
pages = {47--57},
publisher = {IEEE},
doi = {10.1109/cluster51413.2022.00020},
url = {https://doi.org/10.1109/cluster51413.2022.00020},
}
Yuping Fan, Zhiling Lan, J. Taylor Childers, Paul Rich, William Allcock, and Michael E. Papka, Deep reinforcement agent for scheduling in HPC, In 2021 IEEE International Parallel and Distributed Processing Symposium (IPDPS), 2021
@inproceedings{FaLaCh21,
author = {Fan, Yuping and Lan, Zhiling and Childers, J. Taylor and Rich, Paul and Allcock, William and Papka, Michael E.},
title = {Deep reinforcement agent for scheduling in HPC},
booktitle = {2021 IEEE International Parallel and Distributed Processing Symposium (IPDPS)},
year = {2021},
pages = {807--816},
organization = {IEEE},
}
Yuping Fan, Paul Rich, William Allcock, Michael Papka, and Zhiling Lan, Hybrid Workload Scheduling on HPC Systems, arXiv preprint arXiv:2109.05412, 2021
@article{FaRiAl21,
author = {Fan, Yuping and Rich, Paul and Allcock, William and Papka, Michael and Lan, Zhiling},
title = {Hybrid Workload Scheduling on HPC Systems},
journal = {arXiv preprint arXiv:2109.05412},
year = {2021},
}
Yuping Fan, Zhiling Lan, Paul Rich, William E. Allcock, Michael E. Papka, Brian Austin, and David Paul , Scheduling beyond CPUs for HPC, In Proceedings of the 28th International Symposium on High-Performance Parallel and Distributed Computing, 2019
@inproceedings{FaLaRi19,
author = {Fan, Yuping and Lan, Zhiling and Rich, Paul and Allcock, William E. and Papka, Michael E. and Austin, Brian and Paul, David},
title = {Scheduling beyond CPUs for HPC},
booktitle = {Proceedings of the 28th International Symposium on High-Performance Parallel and Distributed Computing},
year = {2019},
pages = {97--108},
}
Li Yu, Zhou Zhou, Yuping Fan, Michael E. Papka, and Zhiling Lan, System-wide trade-off modeling of performance, power, and resilience on petascale systems, The Journal of Supercomputing, 2018
@article{YuZhFa18,
author = {Yu, Li and Zhou, Zhou and Fan, Yuping and Papka, Michael E. and Lan, Zhiling},
title = {System-wide trade-off modeling of performance, power, and resilience on petascale systems},
journal = {The Journal of Supercomputing},
year = {2018},
volume = {74},
number = {7},
pages = {3168--3192},
publisher = {Springer US},
}
Yuping Fan, Paul Rich, William E. Allcock, Michael E. Papka, and Zhiling Lan, Trade-off between prediction accuracy and underestimation rate in job runtime estimates, In 2017 IEEE International Conference on Cluster Computing (CLUSTER), 2017
@inproceedings{FaRiAl17,
author = {Fan, Yuping and Rich, Paul and Allcock, William E. and Papka, Michael E. and Lan, Zhiling},
title = {Trade-off between prediction accuracy and underestimation rate in job runtime estimates},
booktitle = {2017 IEEE International Conference on Cluster Computing (CLUSTER)},
year = {2017},
pages = {530--540},
organization = {IEEE},
}