@inproceedings{bb749aef233c476b8c4025718c015fc9,
title = "QDLoRA: Enhanced LoRA Fine-Tuning on Quantized LLMs via Integrated Low-Rank Decomposition",
abstract = "We propose QDLoRA, a parameter-efficient fine-tuning (PEFT) framework that integrates low-rank decomposition and quantization into the LoRA fine-tuning process for pretrained large language models (LLMs). Unlike prior methods such as LoftQ and ApiQ that rely solely on quantization and suffer performance degradation under extreme compression, QDLoRA preserves more informative structure at the same compression ratio, thereby improving fine-tuning results. To further enhance robustness, QDLoRA introduces a similarity-aware rank selection strategy and a quantization-aware initialization scheme. Experimental results on various model architectures across diverse NLP benchmarks demonstrate that QDLoRA achieves superior accuracy and efficiency compared to existing methods, particularly under limited resource budgets. The proposed method offers a practical and scalable solution for efficient fine-tuning of large language models.",
keywords = "Model Low-rank Decomposition, Model Quantization, PEFT",
author = "Xingyi Su and Rui Wang and Zhongzhi Luan and Yi Liu and Depei Qian",
note = "Publisher Copyright: {\textcopyright} The Author(s), under exclusive license to Springer Nature Singapore Pte Ltd. 2026.; 16th International Symposium on Advanced Parallel Processing Technologies, APPT 2025 ; Conference date: 13-07-2025 Through 16-07-2025",
year = "2026",
doi = "10.1007/978-981-95-1021-4\_36",
language = "英语",
isbn = "9789819510207",
series = "Lecture Notes in Computer Science",
publisher = "Springer Science and Business Media Deutschland GmbH",
pages = "432--437",
editor = "Chao Li and Xuehai Qian and Dimitris Gizopoulos and Boris Grot",
booktitle = "Advanced Parallel Processing Technologies - 16th International Symposium, APPT 2025, Proceedings",
address = "德国",
}