@inproceedings{88bf3c2918cf4523babb5d3e3c2b1c18,
title = "MV-ViT: High-Quality Features and Simple Fusion for Superior Multi-View 3D Object Classification",
abstract = "We present MV-ViT, a novel and efficient approach for multi-view 3D object classification. MV-ViT leverages the power of pre-trained Vision Transformers (ViTs) for robust single-view feature extraction, coupled with a two-stage training strategy that enhances performance while minimizing computational overhead. Unlike methods relying on complex multi-view fusion mechanisms, we demonstrate that high-quality features extracted by a fine-tuned ViT, when aggregated via simple mean pooling, achieve state-of-the-art results. MV-ViT achieves 97.23\% accuracy with 12 views on the ModelNet40 benchmark, surpassing methods employing significantly more elaborate fusion architectures. Ablation studies highlight the critical role of strong ViT-derived features in achieving superior performance, making complex fusion redundant.",
keywords = "3D Object Recognition, Feature Extraction, Fusion Strategy, Multi-View Classification, Vision Transformer",
author = "Hailang Peng and Maolin Cai and Xuwen Zhao and Xiaohua Wang and Xiaomeng Tong",
note = "Publisher Copyright: {\textcopyright} 2025 IEEE.; 2025 China Automation Congress, CAC 2025 ; Conference date: 26-09-2025 Through 28-09-2025",
year = "2025",
doi = "10.1109/CAC67268.2025.11487106",
language = "英语",
series = "Proceedings - 2025 China Automation Congress, CAC 2025",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "120--125",
booktitle = "Proceedings - 2025 China Automation Congress, CAC 2025",
address = "美国",
}