Title |
Identification of Naturally Occurring Numerical Expressions in Arabic |
Authors |
Nizar Habash and Ryan Roth |
Abstract |
In this paper, we define the task of Number Identification in natural context. We present and validate a language-independent semi-automatic approach to quickly building a gold standard for evaluating number identification systems by exploiting hand-aligned parallel data. We also present and extensively evaluate a robust rule-based system for number identification in natural context for Arabic for a variety of number formats and types. The system is shown to have strong performance, achieving, on a blind test, a 94.8% F-score for the task of correctly identifying number expression spans in natural text, and a 92.1% F-score for the task of correctly determining the core numerical value. |
Language |
|
Topics |
MultiWord Expressions & Collocations, Corpus (creation, annotation, etc.), Tools, systems, applications |
Full paper |
Identification of Naturally Occurring Numerical Expressions in Arabic |
Slides |
- |
Bibtex |
@InProceedings{HABASH08.843,
author = {Nizar Habash and Ryan Roth},
title = {Identification of Naturally Occurring Numerical Expressions in Arabic},
booktitle = {Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC'08)},
year = {2008},
month = {may},
date = {28-30},
address = {Marrakech, Morocco},
editor = {Nicoletta Calzolari (Conference Chair), Khalid Choukri, Bente Maegaard, Joseph Mariani, Jan Odijk, Stelios Piperidis, Daniel Tapias},
publisher = {European Language Resources Association (ELRA)},
isbn = {2-9517408-4-0},
note = {http://www.lrec-conf.org/proceedings/lrec2008/},
language = {english}
} |