%Aigaion2 BibTeX export from Idiap Publications %Saturday 21 December 2024 05:42:56 PM @INPROCEEDINGS{Popescu-Belis_LREC_2008, author = {Popescu-Belis, Andrei}, projects = {Idiap, IM2}, title = {Reference-based vs. task-based evaluation of human language technology}, booktitle = {LREC 2008 ELRA Workshop on Evaluation}, year = {2008}, location = {Marrakech, Morocco}, organization = {ELRA}, abstract = {This paper starts from the ISO distinction of three types of evaluation procedures {\^{a}}€“ internal, external and in use {\^{a}}€“ and proposes to match these types to the three types of human language technology (HLT) systems: analysis, generation, and interactive. The paper explains why internal evaluation is not suitable to measure the qualities of HLT systems, and shows that reference-based external evaluation is best adapted to {\^{a}}€˜analysis{\^{a}}€™ systems, task-based evaluation to {\^{a}}€˜interactive{\^{a}}€™ systems, while {\^{a}}€˜generation{\^{a}}€™ systems can be subject to both types of evaluation. In particular, some limits of reference-based external evaluation are shown in the case of generation systems. Finally, the paper shows that contextual evaluation, as illustrated by the FEMTI framework for MT evaluation, is an effective method for getting reference-based evaluation closer to the users of a system.}, pdf = {https://publications.idiap.ch/attachments/papers/2008/Popescu-Belis_LREC_2008.pdf} }