@inproceedings{63989786980f458e99740729d9fb93d7,
title = "Research on Image Caption Method Based on Mixed Image Features",
abstract = "With the continuous development of deep learning in the field of image caption, the effect of it has improved. However, the traditional method for feature extraction is based on the whole picture, ignoring the local features and the relationships of global and local features. Another problem is that the text description is broad and not targeted. Considering the above problems, we propose a new method based on mixed image features. The method uses an improved ResNet to extract global features. Local features are extracted using a deep RetinaNet. The global and the local features of the image are merged by an attention mechanism, which are mapped into embedding vectors. To obtain the mapping relationship between images and descriptions, a long short-term memory (LSTM) based on an attention mechanism is used as a language generation model. Image features and semantic features are combined to generate content description of images.",
keywords = "image caption, mixed feature, text generation attention",
author = "Yang Zhenyu and Zhang Jiao",
note = "Publisher Copyright: {\textcopyright} 2019 IEEE.; 4th IEEE Advanced Information Technology, Electronic and Automation Control Conference, IAEAC 2019 ; Conference date: 20-12-2019 Through 22-12-2019",
year = "2019",
month = dec,
doi = "10.1109/IAEAC47372.2019.8998010",
language = "英语",
series = "Proceedings of 2019 IEEE 4th Advanced Information Technology, Electronic and Automation Control Conference, IAEAC 2019",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "1572--1576",
editor = "Bing Xu and Kefen Mou",
booktitle = "Proceedings of 2019 IEEE 4th Advanced Information Technology, Electronic and Automation Control Conference, IAEAC 2019",
address = "美国",
}