@inproceedings{861, author = {Linh Truong Dieu and Thuan Nguyen and Nguyen Vo and Tam Nguyen and Khang Nguyen}, title = {Parsing Digitized Vietnamese Paper Documents}, abstract = {In recent years, the need to exploit digitized document data has been increasing. In this paper, we address the problem of parsing digitized Vietnamese paper documents. The digitized Vietnamese documents are mainly in the form of scanned images with diverse layouts and special characters introducing many challenges. To this end, we first collect the UIT-DODV dataset, a novel Vietnamese document image dataset that includes scientific papers in Vietnamese derived from different scientific conferences. We compile both images that were converted from PDF and scanned by a smartphone in addition a physical scanner that poses many new challenges. Additionally, we further leverage the state-of-the-art object detector along with the fused loss function to efficiently parse the Vietnamese paper documents. Extensive experiments conducted on the UIT-DODV dataset provide a comprehensive evaluation and insightful analysis.}, year = {2021}, journal = {International Conference on Computer Analysis of Images and Patterns}, month = {01}, url = {https://par.nsf.gov/biblio/10277210}, }