BibTeX

@inProceedings{hamalainen-hengchen-2019-from-293917,
	title        = {From the paft to the fiiture: A fully automatic NMT and word embeddings method for OCR post-correction},
	abstract     = {A great deal of historical corpora suffer from errors introduced by the OCR (optical character recognition) methods used in the digitization process. Correcting these errors manually is a time-consuming process and a great part of the automatic approaches have been relying on rules or supervised machine learning. We present a fully automatic unsupervised way of extracting parallel data for training a character-based sequence-to-sequence NMT (neural machine translation) model to conduct OCR error correction.},
	booktitle    = {International Conference Recent Advances in Natural Language Processing, RANLP, Varna, Bulgaria, 2–4 September, 2019 },
	author       = {Hämäläinen, Mika and Hengchen, Simon},
	year         = {2019},
	ISBN         = {978-954-452-056-4 },
}
Sidansvarig: sb-webb