Title |
Parallel Creation of Gigaword Corpora for Medium Density Languages - an Interim Report |
Authors |
Péter Halácsy, András Kornai, Péter Németh and Dániel Varga |
Abstract |
For increased speed in developing gigaword language resources for medium resource density languages we integrated several FOSS tools in the HUN* toolkit. While the speed and efficiency of the resulting pipeline has surpassed our expectations, our experience in developing LDC-style resource packages for Uzbek and Kurdish makes clear that neither the data collection nor the subsequent processing stages can be fully automated. |
Language |
|
Topics |
Corpus (creation, annotation, etc.), Multilinguality, Tools, systems, applications |
Full paper |
Parallel Creation of Gigaword Corpora for Medium Density Languages - an Interim Report |
Slides |
- |
Bibtex |
@InProceedings{HALCSY08.858,
author = {Péter Halácsy, András Kornai, Péter Németh and Dániel Varga},
title = {Parallel Creation of Gigaword Corpora for Medium Density Languages - an Interim Report},
booktitle = {Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC'08)},
year = {2008},
month = {may},
date = {28-30},
address = {Marrakech, Morocco},
editor = {Nicoletta Calzolari (Conference Chair), Khalid Choukri, Bente Maegaard, Joseph Mariani, Jan Odijk, Stelios Piperidis, Daniel Tapias},
publisher = {European Language Resources Association (ELRA)},
isbn = {2-9517408-4-0},
note = {http://www.lrec-conf.org/proceedings/lrec2008/},
language = {english}
} |