@inproceedings{6c83d820d39a4c558c82c0e17345e75a,
title = "MixMixQ: Quantization with Mixed Bit-Sparsity and Mixed Bit-Width for CIM Accelerators",
abstract = "Quantization is vital for deploying neural networks on Computing-In-Memory (CIM) based accelerators due to inherent limitations in memory devices and data interfaces' representational capacities. However, traditional quantization algorithms often overlook CIM's unique computing paradigm, leading to suboptimal performance. To address this, we introduce MixMixQ, a novel quantization algorithm specifically designed for CIM accelerators that strategically integrates mixed bit-sparsity and mixed bit-width, enhancing overall hardware efficiency while preserving high accuracy. Notably, our method can enhance hardware efficiency by up to 294\% compared to traditional quantization methods, with only a minimal 0.13\% decrease in accuracy compared to a full-precision network.",
keywords = "Bit-Sparsity, Computing-In-Memory, Evolutionary Algorithm, Gradient Estimation, Neural Network Quantization",
author = "Jinyu Bai and He Zhang and Liu, \{Long Chao\} and Pengfei Li and Wang Kang",
note = "Publisher Copyright: {\textcopyright} 2024 ACM.; 34th Great Lakes Symposium on VLSI 2024, GLSVLSI 2024 ; Conference date: 12-06-2024 Through 14-06-2024",
year = "2024",
month = jun,
day = "12",
doi = "10.1145/3649476.3658809",
language = "英语",
series = "Proceedings of the ACM Great Lakes Symposium on VLSI, GLSVLSI",
publisher = "Association for Computing Machinery ",
pages = "537--540",
booktitle = "GLSVLSI 2024 - Proceedings of the Great Lakes Symposium on VLSI 2024",
address = "美国",
}