@inproceedings{26edadf7e83143f3a3d0adf94a67fc7b,
title = "Improved graph-based bilingual corpus selection with sentence pair ranking for statistical machine translation",
abstract = "In statistical machine translation, the number of sentence pairs in the bilingual corpus is very important to the quality of translation. However, when the quantity reaches some extent, enlarging corpus has less effect on the translation; whereas increasing greatly the time and space complexity to building translation systems, which hinders the development of statistical machine translation. In this paper, we propose several ranking approaches to measure the quantity of information of each sentence pair, and apply them into a graph-based bilingual corpus selection framework to form an improved corpus selection approach, which now considers the difference of the initial quantities of information between the sentence pairs. Our experiments in a Chinese-English translation task show that, selecting only 50\% of the whole corpus via the graph-based selection approach as training set, we can obtain the near translation result with the one using the whole corpus, and we obtain better results than the baselines after using the IDF-related ranking approach.",
keywords = "Corpus selection, Graph, Ranking, SMT",
author = "Chao, \{Wen Han\} and Li, \{Zhou Jun\}",
year = "2011",
doi = "10.1109/ICTAI.2011.73",
language = "英语",
isbn = "9780769545967",
series = "Proceedings - International Conference on Tools with Artificial Intelligence, ICTAI",
pages = "446--451",
booktitle = "Proceedings - 2011 23rd IEEE International Conference on Tools with Artificial Intelligence, ICTAI 2011",
note = "23rd IEEE International Conference on Tools with Artificial Intelligence, ICTAI 2011 ; Conference date: 07-11-2011 Through 09-11-2011",
}