@inproceedings{a459a0417ec649bb8d7ebf8a1a090894,
title = "CLIP-HOI: Vision-Language Semantic Alignment for Human-Object Interaction Detection",
abstract = "We propose CLIP-HOI, a vision-language enhanced framework for end-to-end Human-Object Interaction (HOI) detection. While transformer-based HOI methods have shown strong performance, they still struggle with rare and complex interactions that demand deeper semantic understanding beyond visual cues. To address this limitation, we integrate Contrastive Language-Image Pre-Training (CLIP) into the HoiTransformer, leveraging its powerful vision-language alignment capability. Specifically, CLIP-HOI adopts a dual-stream architecture: the original HoiTransformer branch encodes instance-level visual and spatial features, while a parallel CLIP branch extracts global, context-aware semantic representations from image-text alignment space. A cross-attention fusion module adaptively refines interaction features by infusing CLIP-guided semantic cues into visual reasoning. Extensive experiments on the HICO-DET benchmark demonstrate that CLIP-HOI consistently out-performs the baseline HoiTransformer. Our approach shows that integrating vision-language knowledge substantially improves the comprehension of human-object interactions within an end-to-end framework.",
keywords = "CLIP, Human-Object Interaction Detection, Semantic Alignment, Transformer, Vision-Language Integration",
author = "Congcong Geng and Tian Wang and Yutong Jiang and Jian Wang and Mali Xing and Deyuan Liu",
note = "Publisher Copyright: {\textcopyright} 2025 IEEE.; 2025 China Automation Congress, CAC 2025 ; Conference date: 26-09-2025 Through 28-09-2025",
year = "2025",
doi = "10.1109/CAC67268.2025.11487774",
language = "英语",
series = "Proceedings - 2025 China Automation Congress, CAC 2025",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "7623--7628",
booktitle = "Proceedings - 2025 China Automation Congress, CAC 2025",
address = "美国",
}