@online{Hedderich2311.10920,
TITLE = {Understanding and Mitigating Classification Errors Through Interpretable Token Patterns},
AUTHOR = {Hedderich, Michael A. and Fischer, Jonas and Klakow, Dietrich and Vreeken, Jilles},
LANGUAGE = {eng},
URL = {https://arxiv.org/abs/2311.10920},
EPRINT = {2311.10920},
EPRINTTYPE = {arXiv},
YEAR = {2023},
MARGINALMARK = {$\bullet$},
ABSTRACT = {State-of-the-art NLP methods achieve human-like performance on many tasks,<br>but make errors nevertheless. Characterizing these errors in easily<br>interpretable terms gives insight into whether a classifier is prone to making<br>systematic errors, but also gives a way to act and improve the classifier. We<br>propose to discover those patterns of tokens that distinguish correct and<br>erroneous predictions as to obtain global and interpretable descriptions for<br>arbitrary NLP classifiers. We formulate the problem of finding a succinct and<br>non-redundant set of such patterns in terms of the Minimum Description Length<br>principle. Through an extensive set of experiments, we show that our method,<br>Premise, performs well in practice. Unlike existing solutions, it recovers<br>ground truth, even on highly imbalanced data over large vocabularies. In VQA<br>and NER case studies, we confirm that it gives clear and actionable insight<br>into the systematic errors made by NLP classifiers.<br>},
}
