@inproceedings{bf2b7e470ab54b8b996d1c3eff6dd2fb,
title = "GFENet: Group-Free Enhancement Network for Indoor Scene 3D Object Detection",
abstract = "The state-of-the-art group-free network (GFNet) has achieved superior performance for indoor scene 3D object detection. However, we find there is still room for improvement in the following three aspects. Firstly, seed point features extracted by multi-layer perception (MLP) in the backbone (PointNet++) neglect to consider the different importance of each level feature. Second, the single-scale transformer module in GFNet to handle hand-crafted grouping via Hough Voting cannot adequately model the relationship between points and objects. Finally, GFNet directly utilizes the decoders to predict detection results disregarding the different contributions of decoders at each stage. In this paper, we propose the group-free enhancement network (GFENet) to tackle the above issues. Specifically, our network mainly consists of three lifting modules: the weighted MLP (WMLP) module, the hierarchical-aware module, and the stage-aware module. The WMLP module adaptively combines features of different levels in the backbone before max-pooling for informative feature learning. The hierarchical-aware module formulates a hierarchical way to mitigate the negative impact of insufficient modeling of points and objects. The stage-aware module aggregates multi-stage predictions adaptively for better detection performance. Extensive experiments on ScanNet V2 and SUN RGB-D datasets demonstrate the effectiveness and advantages of our method against existing 3D object detection methods.",
keywords = "3D Object Detection, Group-free, Hough Voting, Point Cloud, Transformers",
author = "Feng Zhou and Ju Dai and Junjun Pan and Mengxiao Zhu and Xingquan Cai and Bin Huang and Chen Wang",
note = "Publisher Copyright: {\textcopyright} 2024, The Author(s), under exclusive license to Springer Nature Switzerland AG.; 40th Computer Graphics International Conference, CGI 2023 ; Conference date: 28-08-2023 Through 01-09-2023",
year = "2024",
doi = "10.1007/978-3-031-50075-6\_10",
language = "英语",
isbn = "9783031500749",
series = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)",
publisher = "Springer Science and Business Media Deutschland GmbH",
pages = "119--136",
editor = "Bin Sheng and Lei Bi and Jinman Kim and Nadia Magnenat-Thalmann and Daniel Thalmann",
booktitle = "Advances in Computer Graphics - 40th Computer Graphics International Conference, CGI 2023, Proceedings",
address = "德国",
}