STEALTH is a method for using some AI-generated model, without suffering from malicious attacks (i.e. lying) or associated unfairness issues. After recursively bi-clustering the data, STEALTH system asks the AI model a limited number of queries about class labels. STEALTH asks so few queries (1 per data cluster) that malicious algorithms (a) cannot detect its operation, nor (b) know when to lie.
@article{arxiv.2301.10407,
title = {Don't Lie to Me: Avoiding Malicious Explanations with STEALTH},
author = {Lauren Alvarez and Tim Menzies},
journal= {arXiv preprint arXiv:2301.10407},
year = {2023}
}