@inproceedings{eec7f35fe3314129af0347f98458635e,
title = "Exploiting Input Tensor Dynamics in Activation Checkpointing for Efficient Training on GPU",
abstract = "Larger deep learning models usually lead to higher model quality, however with an ever-increasing GPU memory footprint. Although several tensor checkpointing techniques have been proposed to enable training under a restricted GPU memory budget, they fail to exploit the input tensor dynamics due to diverse datasets and subsequent data augmentation, and thus leave the training optimization on table. In this paper, we propose Mimose, an input-aware tensor checkpointing planner respecting the memory budget while enabling efficient model training on GPU. Mimose builds a lightweight but accurate prediction model of GPU memory usage online, without pre-analyzing the model. It generates a tensor checkpointing plan based on per-layer memory prediction and applies it to the training process on the fly. Our experiments show that Mimose achieves superior training throughput compared to state-of-the-art checkpointing frameworks under the same GPU memory budgets.",
keywords = "GPU memory, input dynamics, model training, tensor checkpointing",
author = "Jianjin Liao and Mingzhen Li and Hailong Yang and Qingxiao Sun and Biao Sun and Jiwei Hao and Tianyu Feng and Fengwei Yu and Shengdong Chen and Ye Tao and Zicheng Zhang and Zhongzhi Luan and Depei Qian",
note = "Publisher Copyright: {\textcopyright} 2023 IEEE.; 37th IEEE International Parallel and Distributed Processing Symposium, IPDPS 2023 ; Conference date: 15-05-2023 Through 19-05-2023",
year = "2023",
doi = "10.1109/IPDPS54959.2023.00025",
language = "英语",
series = "Proceedings - 2023 IEEE International Parallel and Distributed Processing Symposium, IPDPS 2023",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "156--166",
booktitle = "Proceedings - 2023 IEEE International Parallel and Distributed Processing Symposium, IPDPS 2023",
address = "美国",
}