{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T07:22:29Z","timestamp":1774596149739,"version":"3.50.1"},"reference-count":54,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,5,1]],"date-time":"2023-05-01T00:00:00Z","timestamp":1682899200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,5,1]],"date-time":"2023-05-01T00:00:00Z","timestamp":1682899200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,5]]},"DOI":"10.1109\/ccgrid57682.2023.00036","type":"proceedings-article","created":{"date-parts":[[2023,7,10]],"date-time":"2023-07-10T18:26:14Z","timestamp":1689013574000},"page":"299-310","source":"Crossref","is-referenced-by-count":3,"title":["FreeTrain: A Framework to Utilize Unused Supercomputer Nodes for Training Neural Networks"],"prefix":"10.1109","author":[{"given":"Zhengchun","family":"Liu","sequence":"first","affiliation":[{"name":"Argonne National Laboratory,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rajkumar","family":"Kettimuthu","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael E.","family":"Papka","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ian","family":"Foster","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3480859"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/369028.369109"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2019.00016"},{"key":"ref14","article-title":"RFaaS: RDMA-enabled faas platform for serverless high-performance computing","author":"copik","year":"2021","journal-title":"ArXiv Preprint"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2003.1189582"},{"key":"ref52","first-page":"6848","article-title":"ShuffieNet: An extremely efficient convolutional neural network for mobile devices","author":"zhang","year":"0","journal-title":"IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3320060"},{"key":"ref10","first-page":"99","article-title":"Special facilities in a general math-ematical programming system for non-convex problems using ordered sets of variables","volume":"69","author":"beale","year":"1970","journal-title":"OR"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/HPDC.1999.805303"},{"key":"ref17","author":"glockner","year":"2015","journal-title":"Parallel and distributed optimization with Gurobi optimizer"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1023\/A:1015617019423"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3064176.3064182"},{"key":"ref18","article-title":"Accurate, large minibatch SGD: Training ImageNet in 1 hour","author":"goyal","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref51","article-title":"Machine learning on volatile instances: Convergence, runtime, and cost tradeoffs","author":"zhang","year":"2021","journal-title":"IEEE\/ACM Transactions on Networking"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CLOUD.2012.59"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00293"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/IPPS.1999.760525"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01128"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.938"},{"key":"ref42","article-title":"Very deep convolutional networks for large-scale image recognition","author":"simonyan","year":"2014","journal-title":"ArXiv Preprint"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CSIE.2009.950"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICPPW.2002.1039773"},{"key":"ref43","article-title":"Don't decay the learning rate, increase the batch size","author":"smith","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1145\/3225058.3225069"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/581571.581573"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/1465482.1465560"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/BF01580653"},{"key":"ref4","year":"0","journal-title":"Summit user guide"},{"key":"ref3","year":"0","journal-title":"A scheduler log replayer for FreeTrain evaluation"},{"key":"ref6","article-title":"Spot instances","year":"0","journal-title":"Amazon Web Services"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-77398-8_1"},{"key":"ref40","article-title":"Horovod: fast and easy distributed deep learning in TensorFlow","author":"sergeev","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref35","first-page":"4095","article-title":"Efficient neural architecture search via parameters sharing","author":"pham","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref34","article-title":"PyTorch: An imperative style, high-performance deep learning library","author":"paszke","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/1362622.1362680"},{"key":"ref36","article-title":"Pollux: Co-adaptive cluster scheduling for goodput-optimized deep learning","author":"qiao","year":"2020","journal-title":"ArXiv Preprint"},{"key":"ref31","article-title":"Scaling distributed training with adaptive summation","volume":"3","author":"maleki","year":"2021","journal-title":"Machine Learning and Systems"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2010.91"},{"key":"ref33","article-title":"Analysis and exploitation of dynamic pricing in the public cloud for ML training","author":"narayanan","year":"0","journal-title":"VLDB DISPA Workshop 2020"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid.2011.56"},{"key":"ref2","year":"0","journal-title":"Kubernetes"},{"key":"ref1","year":"0","journal-title":"FreeTrain Implementation based on Elastic Horovod"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2014.10.003"},{"key":"ref38","first-page":"11","article-title":"A212: An application aware flexible HPCscheduling model for low-latency allocation","author":"rodrigo alvarez","year":"0","journal-title":"International Workshop on Virtualization Technologies in Distributed Computing"},{"key":"ref24","article-title":"Hyp-RL: Hyperparameter optimization by reinforcement learning","author":"jomaa","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref26","first-page":"1097","article-title":"ImageNet classification with deep convolutional neural networks","volume":"25","author":"krizhevsky","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/s12532-011-0025-9"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref22","article-title":"MobileNets: Efficient convolutional neural networks for mobile vision applications","author":"howard","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3400302.3415699"},{"key":"ref28","article-title":"DARTS: Differentiable architecture search","author":"liu","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICAC.2019.00024"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3392717.3392774"}],"event":{"name":"2023 IEEE\/ACM 23rd International Symposium on Cluster, Cloud and Internet Computing (CCGrid)","location":"Bangalore, India","start":{"date-parts":[[2023,5,1]]},"end":{"date-parts":[[2023,5,4]]}},"container-title":["2023 IEEE\/ACM 23rd International Symposium on Cluster, Cloud and Internet Computing (CCGrid)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10171437\/10171438\/10171512.pdf?arnumber=10171512","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,1]],"date-time":"2023-08-01T17:51:52Z","timestamp":1690912312000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10171512\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5]]},"references-count":54,"URL":"https:\/\/doi.org\/10.1109\/ccgrid57682.2023.00036","relation":{},"subject":[],"published":{"date-parts":[[2023,5]]}}}