publications
publications by categories in reversed chronological order. generated by jekyll-scholar.
2026
- A Systematic Comparison between Extractive Self-Explanations and Human Rationales in Text ClassificationStephanie Brandl and Oliver EberleIn Proceedings of the 6th Workshop on Trustworthy NLP (TrustNLP 2026), Jul 2026
Instruction-tuned LLMs are able to provide an explanation about their output to users by generating self-explanations, without requiring the application of complex interpretability techniques. In this paper, we analyse whether this ability results in a good explanation. We evaluate self-explanations in the form of input rationales with respect to their plausibility to humans. We study three text classification tasks: sentiment classification, forced labour detection and claim verification. We include Danish and Italian translations of the sentiment classification task and compare self-explanations to human annotations. For this, we collected human rationale annotations for Climate-Fever, a claim verification dataset. We furthermore evaluate the faithfulness of human and self-explanation rationales with respect to correct model predictions, and extend the study by incorporating post-hoc attribution-based explanations. We analyse four open-weight LLMs and find that alignment between self-explanations and human rationales highly depends on text length and task complexity. Nevertheless, self-explanations yield faithful subsets of token-level rationales, whereas post-hoc attribution methods tend to emphasize structural and formatting tokens, reflecting fundamentally different explanation strategies.
@inproceedings{brandl_systematic_2026, address = {San Diego, California}, title = {A {Systematic} {Comparison} between {Extractive} {Self}-{Explanations} and {Human} {Rationales} in {Text} {Classification}}, isbn = {9798891764187}, url = {https://aclanthology.org/2026.trustnlp-main.44/}, doi = {10.18653/v1/2026.trustnlp-main.44}, urldate = {2026-08-19}, booktitle = {Proceedings of the 6th {Workshop} on {Trustworthy} {NLP} ({TrustNLP} 2026)}, publisher = {Association for Computational Linguistics}, author = {Brandl, Stephanie and Eberle, Oliver}, editor = {Chang, Kai-Wei and Mehrabi, Ninareh and Krishna, Satyapriya and Das, Anubrata and Dhamala, Jwala and Cao, Yang Trista and Kumarage, Tharindu and Ramakrishna, Anil and Christodoulopoulos, Christos and Wan, Yixin and Galystan, Aram and Kumar, Anoop and Gupta, Rahul}, month = jul, year = {2026}, pages = {563--583}, } - MedIABeyond attention heatmaps: How to get better explanations for multiple instance learning models in histopathologyMina Jamshidi Idaji, Julius Hense, Tom Neuhäuser, and 12 more authorsMedical Image Analysis, Sep 2026
@article{jamshidi_idaji_beyond_2026, title = {Beyond attention heatmaps: {How} to get better explanations for multiple instance learning models in histopathology}, volume = {113}, issn = {13618415}, shorttitle = {Beyond attention heatmaps}, url = {https://linkinghub.elsevier.com/retrieve/pii/S1361841526002173}, doi = {10.1016/j.media.2026.104148}, language = {en}, urldate = {2026-08-19}, journal = {Medical Image Analysis}, author = {Jamshidi Idaji, Mina and Hense, Julius and Neuhäuser, Tom and Krause, Augustin and Luo, Yanqing and Eberle, Oliver and Schnake, Thomas and Ciernik, Laure and Rezaei Jafari, Farnoush and Vahidimajd, Reza and Dippel, Jonas and Walz, Christoph and Klauschen, Frederick and Mock, Andreas and Müller, Klaus-Robert}, month = sep, year = {2026}, pages = {104148}, } - PreprintDistributed Sparse Interventions in Language ModelsMaximilian S. Ernst, Lorenz Linhardt, Aaron Peikert, and 1 more authorJul 2026arXiv:2607.07128 [cs.LG]
Language models perform a wide range of tasks at varying levels of abstraction with the capacity to flexibly infer tasks from context, execute multiple tasks simultaneously, and select among competing tasks. To study the role of model components in task behaviour, their causal influence can be investigated through interventions. Prior work on model steering has largely focused on interventions along global directions in activation space, modeling task representations as approximately linear and additive. By studying interventions at the neuron level, we find substantial, neuron-specific nonlinear effects on model outputs that are not captured by current steering approaches. We introduce Distributed Sparse Interventions (DSI), an intervention approach that considers nonlinearities and interactions between neurons across layers to identify sparse sets of neurons that elicit task-relevant computations. Across a range of tasks, we demonstrate that DSI can activate task behaviour in instruction-tuned language models by localising and intervening on as few as 0.01% of neurons, highlighting the effectiveness of sparse, distributed interventions in the neuron basis. Additionally, adopting a set-based perspective enables computations over the identified neuron sets, offering insights into the roles of individual neurons by analysing their effects across tasks. Through sparse interventions, DSI enables fine-grained control over model behaviour, localisation of task-relevant neuron sets, and furthers our understanding of task composition.
@misc{ernst_distributed_2026, title = {Distributed {Sparse} {Interventions} in {Language} {Models}}, url = {http://arxiv.org/abs/2607.07128}, doi = {10.48550/arXiv.2607.07128}, urldate = {2026-08-19}, publisher = {arXiv}, author = {Ernst, Maximilian S. and Linhardt, Lorenz and Peikert, Aaron and Eberle, Oliver}, month = jul, year = {2026}, note = {arXiv:2607.07128 [cs.LG]}, keywords = {Computer Science - Machine Learning}, } - ICMLAlgoTrace: Algorithmic Primitives and Compositional Geometry of Reasoning in Language ModelsSamuel Lippl, Thomas Austin McGee, Kimberly Lopez, and 5 more authorsIn Forty-third International Conference on Machine Learning, 2026
How do inference time and latent computations enable large language models (LLMs) to solve multi-step reasoning problems? We introduce AlgoTrace, a framework for tracing and steering algorithmic operations in the model latent space for multi-step reasoning. We operationalize primitives by clustering latent activations of the model when solving four benchmarks: Traveling Salesperson Problem (TSP), 3SAT, AIME, and Graph Navigation. We annotate the clusters using their corresponding tokens in the reasoning trace. We then apply function vector methods to extract primitive vectors as reusable compositional building blocks of reasoning. We find that a) injecting a primitive vector into models (Phi, Qwen, Llama) elicits the associated algorithmic operation in the reasoning trace, b) injecting primitives can steer behavior across tasks, c) primitive vectors can be composed through algebraic operations, revealing a geometric logic in activation space, and d) a fine-tuned model exhibits improved composition of primitives (Phi-4-Reasoning vs. Phi-4). These findings demonstrate that LLM reasoning can be understood as a walk through algorithmic primitives in the latent space governed by compositional geometry. These primitives transfer across tasks, and reasoning finetuning strengthens algorithmic generalization and composition across domains.
@inproceedings{lippl_algotrace_2026, title = {{AlgoTrace}: {Algorithmic} {Primitives} and {Compositional} {Geometry} of {Reasoning} in {Language} {Models}}, url = {https://openreview.net/forum?id=ppirVueEj6}, booktitle = {Forty-third {International} {Conference} on {Machine} {Learning}}, author = {Lippl, Samuel and McGee, Thomas Austin and Lopez, Kimberly and Pan, Ziwen and Zhang, Pierce and Ziadi, Salma and Eberle, Oliver and Momennejad, Ida}, year = {2026}, }
2025
- ICMLPosition: We Need An Algorithmic Understanding of Generative AIOliver Eberle, Thomas Austin Mcgee, Hamza Giaffar, and 2 more authorsIn Proceedings of the 42nd International Conference on Machine Learning, Oct 2025
What algorithms do LLMs actually learn and use to solve problems? Studies addressing this question are sparse, as research priorities are focused on improving performance through scale, leaving a theoretical and empirical gap in understanding emergent algorithms. This position paper proposes AlgEval: a framework for systematic research into the algorithms that LLMs learn and use. AlgEval aims to uncover algorithmic primitives, reflected in latent representations, attention, and inference-time compute, and their algorithmic composition to solve task-specific problems. We highlight potential methodological paths and a case study toward this goal, focusing on emergent search algorithms. Our case study illustrates both the formation of top-down hypotheses about candidate algorithms, and bottom-up tests of these hypotheses via circuit-level analysis of attention patterns and hidden states. The rigorous, systematic evaluation of how LLMs actually solve tasks provides an alternative to resource-intensive scaling, reorienting the field toward a principled understanding of underlying computations. Such algorithmic explanations offer a pathway to human-understandable interpretability, enabling comprehension of the model’s internal reasoning performance measures. This can in turn lead to more sample-efficient methods for training and improving performance, as well as novel architectures for end-to-end and multi-agent systems.
@inproceedings{eberle_position_2025, title = {Position: {We} {Need} {An} {Algorithmic} {Understanding} of {Generative} {AI}}, issn = {2640-3498}, shorttitle = {Position}, url = {https://proceedings.mlr.press/v267/eberle25a.html}, language = {en}, urldate = {2026-08-19}, booktitle = {Proceedings of the 42nd {International} {Conference} on {Machine} {Learning}}, publisher = {PMLR}, author = {Eberle, Oliver and Mcgee, Thomas Austin and Giaffar, Hamza and Webb, Taylor Whittington and Momennejad, Ida}, month = oct, year = {2025}, pages = {81292--81314}, } - PreprintRelP: Faithful and Efficient Circuit Discovery in Language Models via Relevance PatchingFarnoush Rezaei Jafari, Oliver Eberle, Ashkan Khakzar, and 1 more authorOct 2025arXiv:2508.21258 [cs.LG]
Activation patching is a standard method in mechanistic interpretability for localizing the components of a model responsible for specific behaviors, but it is computationally expensive to apply at scale. Attribution patching offers a faster, gradient-based approximation, yet suffers from noise and reduced reliability in deep, highly non-linear networks. In this work, we introduce Relevance Patching (RelP), which replaces the local gradients in attribution patching with propagation coefficients derived from Layer-wise Relevance Propagation (LRP). LRP propagates the network’s output backward through the layers, redistributing relevance to lower-level components according to local propagation rules that ensure properties such as relevance conservation or improved signal-to-noise ratio. Like attribution patching, RelP requires only two forward passes and one backward pass, maintaining computational efficiency while improving faithfulness. We validate RelP across a range of models and tasks, showing that it more accurately approximates activation patching than standard attribution patching, particularly when analyzing residual stream and MLP outputs in the Indirect Object Identification (IOI) task. For instance, for MLP outputs in GPT-2 Large, attribution patching achieves a Pearson correlation of 0.006, whereas RelP reaches 0.956, highlighting the improvement offered by RelP. Additionally, we compare the faithfulness of sparse feature circuits identified by RelP and Integrated Gradients (IG), showing that RelP achieves comparable faithfulness without the extra computational cost associated with IG.
@misc{jafari_relp_2025, title = {{RelP}: {Faithful} and {Efficient} {Circuit} {Discovery} in {Language} {Models} via {Relevance} {Patching}}, shorttitle = {{RelP}}, url = {http://arxiv.org/abs/2508.21258}, doi = {10.48550/arXiv.2508.21258}, urldate = {2026-08-19}, publisher = {arXiv}, author = {Jafari, Farnoush Rezaei and Eberle, Oliver and Khakzar, Ashkan and Nanda, Neel}, month = oct, year = {2025}, note = {arXiv:2508.21258 [cs.LG]}, keywords = {Computer Science - Machine Learning}, } - NeurIPSCapturing Polysemanticity with PRISM: A Multi-Concept Feature Description FrameworkLaura Kopf, Nils Feldhus, Kirill Bykov, and 4 more authorsIn Advances in Neural Information Processing Systems, 2025
Automated interpretability research aims to identify concepts encoded in neural network features to enhance human understanding of model behavior. Within the context of large language models (LLMs) for natural language processing (NLP), current automated neuron-level feature description methods face two key challenges: limited robustness and the assumption that each neuron encodes a single concept (monosemanticity), despite increasing evidence of polysemanticity. This assumption restricts the expressiveness of feature descriptions and limits their ability to capture the full range of behaviors encoded in model internals. To address this, we introduce Polysemantic FeatuRe Identification and Scoring Method (PRISM), a novel framework specifically designed to capture the complexity of features in LLMs. Unlike approaches that assign a single description per neuron, common in many automated interpretability methods in NLP, PRISM produces more nuanced descriptions that account for both monosemantic and polysemantic behavior. We apply PRISM to LLMs and, through extensive benchmarking against existing methods, demonstrate that our approach produces more accurate and faithful feature descriptions, improving both overall description quality (via a description score) and the ability to capture distinct concepts when polysemanticity is present (via a polysemanticity score).
@inproceedings{kopf_capturing_2025, title = {Capturing {Polysemanticity} with {PRISM}: {A} {Multi}-{Concept} {Feature} {Description} {Framework}}, volume = {38, Main Conference}, shorttitle = {Capturing {Polysemanticity} with {PRISM}}, url = {https://proceedings.neurips.cc/paper_files/paper/2025/hash/77a404d99d58544293763afb769a16e4-Abstract-Conference.html}, doi = {10.52202/085713-2780}, urldate = {2026-08-19}, booktitle = {Advances in {Neural} {Information} {Processing} {Systems}}, publisher = {Curran Associates, Inc.}, author = {Kopf, Laura and Feldhus, Nils and Bykov, Kirill and Bommer, Philine L and Hedström, Anna and Höhne, Marina and Eberle, Oliver}, year = {2025}, pages = {82938--82974}, } - ACLTrick or Neat: Adversarial Ambiguity and Language Model EvaluationAntonia Karamolegkou, Oliver Eberle, Phillip Rust, and 2 more authorsIn Findings of the Association for Computational Linguistics: ACL 2025, 2025
@inproceedings{karamolegkou_trick_2025, address = {Vienna, Austria}, title = {Trick or {Neat}: {Adversarial} {Ambiguity} and {Language} {Model} {Evaluation}}, shorttitle = {Trick or {Neat}}, url = {https://aclanthology.org/2025.findings-acl.954}, doi = {10.18653/v1/2025.findings-acl.954}, language = {en}, urldate = {2026-08-19}, booktitle = {Findings of the {Association} for {Computational} {Linguistics}: {ACL} 2025}, publisher = {Association for Computational Linguistics}, author = {Karamolegkou, Antonia and Eberle, Oliver and Rust, Phillip and Kauf, Carina and Søgaard, Anders}, year = {2025}, pages = {18542--18561}, }
2024
- Sci AdvHistorical insights at scale: A corpus-wide machine learning analysis of early modern astronomic tablesOliver Eberle, Jochen Büttner, Hassan el-Hajj, and 3 more authorsScience Advances, Oct 2024
Understanding the evolution and dissemination of human knowledge over time faces challenges due to the abundance of historical materials and limited specialist resources. However, the digitization of historical archives presents an opportunity for AI-supported analysis. This study advances historical analysis by using an atomization-recomposition method that relies on unsupervised machine learning and explainable AI techniques. Focusing on the “Sacrobosco Collection,” consisting of 359 early modern printed editions of astronomy textbooks from European universities (1472–1650), totaling 76,000 pages, our analysis uncovers temporal and geographic patterns in knowledge transformation. We highlight the relevant role of astronomy textbooks in shaping a unified mathematical culture, driven by competition among educational institutions and market dynamics. This approach deepens our understanding by grounding insights in historical context, integrating with traditional methodologies. Case studies illustrate how communities embraced scientific advancements, reshaping astronomic and geographical views and exploring scientific roots amidst a changing world. , An unsupervised ML model analyzes historical sources beyond human capacities through an atomization-recomposition approach.
@article{eberle_historical_2024, title = {Historical insights at scale: {A} corpus-wide machine learning analysis of early modern astronomic tables}, volume = {10}, issn = {2375-2548}, shorttitle = {Historical insights at scale}, url = {https://www.science.org/doi/10.1126/sciadv.adj1719}, doi = {10.1126/sciadv.adj1719}, language = {en}, number = {43}, urldate = {2026-08-19}, journal = {Science Advances}, author = {Eberle, Oliver and Büttner, Jochen and el-Hajj, Hassan and Montavon, Grégoire and Müller, Klaus-Robert and Valleriani, Matteo}, month = oct, year = {2024}, pages = {eadj1719}, } - NeurIPSxMIL: Insightful Explanations for Multiple Instance Learning in HistopathologyJulius Hense, Mina Jamshidi Idaji, Oliver Eberle, and 7 more authorsIn Advances in Neural Information Processing Systems, 2024
Multiple instance learning (MIL) is an effective and widely used approach for weakly supervised machine learning. In histopathology, MIL models have achieved remarkable success in tasks like tumor detection, biomarker prediction, and outcome prognostication. However, MIL explanation methods are still lagging behind, as they are limited to small bag sizes or disregard instance interactions. We revisit MIL through the lens of explainable AI (XAI) and introduce xMIL, a refined framework with more general assumptions. We demonstrate how to obtain improved MIL explanations using layer-wise relevance propagation (LRP) and conduct extensive evaluation experiments on three toy settings and four real-world histopathology datasets. Our approach consistently outperforms previous explanation attempts with particularly improved faithfulness scores on challenging biomarker prediction tasks. Finally, we showcase how xMIL explanations enable pathologists to extract insights from MIL models, representing a significant advance for knowledge discovery and model debugging in digital histopathology.
@inproceedings{hense_xmil_2024, title = {{xMIL}: {Insightful} {Explanations} for {Multiple} {Instance} {Learning} in {Histopathology}}, volume = {37}, shorttitle = {{xMIL}}, url = {https://proceedings.neurips.cc/paper_files/paper/2024/hash/0f9e0309d8a947ca44463a9b7e8b6a3f-Abstract-Conference.html}, doi = {10.52202/079017-0266}, urldate = {2026-08-19}, booktitle = {Advances in {Neural} {Information} {Processing} {Systems}}, publisher = {Curran Associates, Inc.}, author = {Hense, Julius and Jamshidi Idaji, Mina and Eberle, Oliver and Schnake, Thomas and Dippel, Jonas and Ciernik, Laure and Buchstab, Oliver and Mock, Andreas and Klauschen, Frederick and Müller, Klaus-Robert}, year = {2024}, pages = {8300--8328}, } - NeurIPSMambaLRP: Explaining Selective State Space Sequence ModelsFarnoush Rezaei Jafari, Grégoire Montavon, Klaus-Robert Müller, and 1 more authorIn Advances in Neural Information Processing Systems 37, 2024
@inproceedings{jafari_mambalrp_2024, address = {Vancouver, BC, Canada}, title = {{MambaLRP}: {Explaining} {Selective} {State} {Space} {Sequence} {Models}}, isbn = {9798331314385}, shorttitle = {{MambaLRP}}, url = {http://www.proceedings.com/079017-3764.html}, doi = {10.52202/079017-3764}, urldate = {2026-08-19}, booktitle = {Advances in {Neural} {Information} {Processing} {Systems} 37}, publisher = {Neural Information Processing Systems Foundation, Inc. (NeurIPS)}, author = {Jafari, Farnoush Rezaei and Montavon, Grégoire and Müller, Klaus-Robert and Eberle, Oliver}, year = {2024}, pages = {118540--118570}, } - ACLExplaining Text Similarity in Transformer ModelsAlexandros Vasileiou and Oliver EberleIn Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), 2024
@inproceedings{vasileiou_explaining_2024, address = {Mexico City, Mexico}, title = {Explaining {Text} {Similarity} in {Transformer} {Models}}, url = {https://aclanthology.org/2024.naacl-long.435}, doi = {10.18653/v1/2024.naacl-long.435}, language = {en}, urldate = {2026-08-19}, booktitle = {Proceedings of the 2024 {Conference} of the {North} {American} {Chapter} of the {Association} for {Computational} {Linguistics}: {Human} {Language} {Technologies} ({Volume} 1: {Long} {Papers})}, publisher = {Association for Computational Linguistics}, author = {Vasileiou, Alexandros and Eberle, Oliver}, year = {2024}, pages = {7859--7873}, } - LREC-COLINGEvaluating Webcam-based Gaze Data as an Alternative for Human Rationale AnnotationsStephanie Brandl, Oliver Eberle, Tiago Ribeiro, and 2 more authorsIn Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024), May 2024
Rationales in the form of manually annotated input spans usually serve as ground truth when evaluating explainability methods in NLP. They are, however, time-consuming and often biased by the annotation process. In this paper, we debate whether human gaze, in the form of webcam-based eye-tracking recordings, poses a valid alternative when evaluating importance scores. We evaluate the additional information provided by gaze data, such as total reading times, gaze entropy, and decoding accuracy with respect to human rationale annotations. We compare WebQAmGaze, a multilingual dataset for information-seeking QA, with attention and explainability-based importance scores for 4 different multilingual Transformer-based language models (mBERT, distil-mBERT, XLMR, and XLMR-L) and 3 languages (English, Spanish, and German). Our pipeline can easily be applied to other tasks and languages. Our findings suggest that gaze data offers valuable linguistic insights that could be leveraged to infer task difficulty and further show a comparable ranking of explainability methods to that of human rationales.
@inproceedings{brandl_evaluating_2024, title = {Evaluating Webcam-based Gaze Data as an Alternative for Human Rationale Annotations}, author = {Brandl, Stephanie and Eberle, Oliver and Ribeiro, Tiago and S{\o}gaard, Anders and Hollenstein, Nora}, booktitle = {Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)}, month = may, year = {2024}, address = {Torino, Italia}, publisher = {ELRA and ICCL}, url = {https://aclanthology.org/2024.lrec-main.580/}, pages = {6544--6556}, }
2023
- IJDHExplainability and transparency in the realm of digital humanities: toward a historian XAIHassan El-Hajj, Oliver Eberle, Anika Merklein, and 7 more authorsInternational Journal of Digital Humanities, Nov 2023
The recent advancements in the field of Artificial Intelligence (AI) translated to an increased adoption of AI technology in the humanities, which is often challenged by the limited amount of annotated data, as well as its heterogeneity. Despite the scarcity of data it has become common practice to design increasingly complex AI models, usually at the expense of human readability, explainability, and trust. This in turn has led to an increased need for tools to help humanities scholars better explain and validate their models as well as their hypotheses. In this paper, we discuss the importance of employing Explainable AI (XAI) methods within the humanities to gain insights into historical processes as well as ensure model reproducibility and a trustworthy scientific result. To drive our point, we present several representative case studies from the Sphaera project where we analyze a large, well-curated corpus of early modern textbooks using an AI model, and rely on the XAI explanatory outputs to generate historical insights concerning their visual content. More specifically, we show that XAI can be used as a partner when investigating debated subjects in the history of science, such as what strategies were used in the early modern period to showcase mathematical instruments and machines.
@article{el-hajj_explainability_2023, title = {Explainability and transparency in the realm of digital humanities: toward a historian {XAI}}, volume = {5}, issn = {2524-7840}, shorttitle = {Explainability and transparency in the realm of digital humanities}, url = {https://doi.org/10.1007/s42803-023-00070-1}, doi = {10.1007/s42803-023-00070-1}, language = {en}, number = {2}, urldate = {2026-08-19}, journal = {International Journal of Digital Humanities}, author = {El-Hajj, Hassan and Eberle, Oliver and Merklein, Anika and Siebold, Anna and Shlomi, Noga and Büttner, Jochen and Martinetz, Julius and Müller, Klaus-Robert and Montavon, Grégoire and Valleriani, Matteo}, month = nov, year = {2023}, keywords = {Computational humanities, Digital humanities, Early modern printing, Explainable AI, Sphaera}, pages = {299--331}, } - EMNLPRather a Nurse than a Physician - Contrastive Explanations under InvestigationOliver Eberle, Ilias Chalkidis, Laura Cabello, and 1 more authorIn Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, 2023
@inproceedings{eberle_rather_2023, address = {Singapore}, title = {Rather a {Nurse} than a {Physician} - {Contrastive} {Explanations} under {Investigation}}, url = {https://aclanthology.org/2023.emnlp-main.427}, doi = {10.18653/v1/2023.emnlp-main.427}, language = {en}, urldate = {2026-08-19}, booktitle = {Proceedings of the 2023 {Conference} on {Empirical} {Methods} in {Natural} {Language} {Processing}}, publisher = {Association for Computational Linguistics}, author = {Eberle, Oliver and Chalkidis, Ilias and Cabello, Laura and Brandl, Stephanie}, year = {2023}, pages = {6907--6920}, }
2022
- ICMLXAI for Transformers: Better Explanations through Conservative PropagationAmeen Ali, Thomas Schnake, Oliver Eberle, and 3 more authorsIn Proceedings of the 39th International Conference on Machine Learning, Jun 2022
Transformers have become an important workhorse of machine learning, with numerous applications. This necessitates the development of reliable methods for increasing their transparency. Multiple interpretability methods, often based on gradient information, have been proposed. We show that the gradient in a Transformer reflects the function only locally, and thus fails to reliably identify the contribution of input features to the prediction. We identify Attention Heads and LayerNorm as main reasons for such unreliable explanations and propose a more stable way for propagation through these layers. Our proposal, which can be seen as a proper extension of the well-established LRP method to Transformers, is shown both theoretically and empirically to overcome the deficiency of a simple gradient-based approach, and achieves state-of-the-art explanation performance on a broad range of Transformer models and datasets.
@inproceedings{ali_xai_2022, title = {{XAI} for {Transformers}: {Better} {Explanations} through {Conservative} {Propagation}}, issn = {2640-3498}, shorttitle = {{XAI} for {Transformers}}, url = {https://proceedings.mlr.press/v162/ali22a.html}, language = {en}, urldate = {2026-08-19}, booktitle = {Proceedings of the 39th {International} {Conference} on {Machine} {Learning}}, publisher = {PMLR}, author = {Ali, Ameen and Schnake, Thomas and Eberle, Oliver and Montavon, Grégoire and Müller, Klaus-Robert and Wolf, Lior}, month = jun, year = {2022}, pages = {435--451}, } - SpringerAn Ever-Expanding Humanities Knowledge Graph: The Sphaera Corpus at the Intersection of Humanities, Data Management, and Machine LearningHassan El-Hajj, Maryam Zamani, Jochen Büttner, and 8 more authorsDatenbank-Spektrum, Jul 2022
Abstract The Sphere project stands at the intersection of the humanities and information sciences. The project aims to better understand the evolution of knowledge in the early modern period by studying a collection of 359 textbook editions published between 1472 and 1650 which were used to teach geocentric cosmology and astronomy at European universities. The relatively large size of the corpus at hand presents a challenge for traditional historical approaches, but provides a great opportunity to explore such a large collection of historical data using computational approaches. In this paper, we present a review of the different computational approaches, used in this project over the period of the last three years, that led to a better understanding of the dynamics of knowledge transfer and transformation in the early modern period.
@article{el-hajj_ever-expanding_2022, title = {An {Ever}-{Expanding} {Humanities} {Knowledge} {Graph}: {The} {Sphaera} {Corpus} at the {Intersection} of {Humanities}, {Data} {Management}, and {Machine} {Learning}}, volume = {22}, issn = {1618-2162, 1610-1995}, shorttitle = {An {Ever}-{Expanding} {Humanities} {Knowledge} {Graph}}, url = {https://link.springer.com/10.1007/s13222-022-00414-1}, doi = {10.1007/s13222-022-00414-1}, language = {en}, number = {2}, urldate = {2026-08-19}, journal = {Datenbank-Spektrum}, author = {El-Hajj, Hassan and Zamani, Maryam and Büttner, Jochen and Martinetz, Julius and Eberle, Oliver and Shlomi, Noga and Siebold, Anna and Montavon, Grégoire and Müller, Klaus-Robert and Kantz, Holger and Valleriani, Matteo}, month = jul, year = {2022}, pages = {153--162}, } - ACLDo Transformer Models Show Similar Attention Patterns to Task-Specific Human Gaze?Oliver Eberle, Stephanie Brandl, Jonas Pilot, and 1 more authorIn Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2022
@inproceedings{eberle_transformer_2022, address = {Dublin, Ireland}, title = {Do {Transformer} {Models} {Show} {Similar} {Attention} {Patterns} to {Task}-{Specific} {Human} {Gaze}?}, url = {https://aclanthology.org/2022.acl-long.296}, doi = {10.18653/v1/2022.acl-long.296}, language = {en}, urldate = {2026-08-19}, booktitle = {Proceedings of the 60th {Annual} {Meeting} of the {Association} for {Computational} {Linguistics} ({Volume} 1: {Long} {Papers})}, publisher = {Association for Computational Linguistics}, author = {Eberle, Oliver and Brandl, Stephanie and Pilot, Jonas and Søgaard, Anders}, year = {2022}, pages = {4295--4309}, } - PAMIHigher-Order Explanations of Graph Neural Networks via Relevant WalksThomas Schnake, Oliver Eberle, Jonas Lederer, and 4 more authorsIEEE Transactions on Pattern Analysis and Machine Intelligence, Nov 2022
@article{schnake_higher-order_2022, title = {Higher-{Order} {Explanations} of {Graph} {Neural} {Networks} via {Relevant} {Walks}}, volume = {44}, copyright = {https://creativecommons.org/licenses/by/4.0/legalcode}, issn = {0162-8828, 2160-9292, 1939-3539}, url = {https://ieeexplore.ieee.org/document/9547794/}, doi = {10.1109/TPAMI.2021.3115452}, number = {11}, urldate = {2026-08-19}, journal = {IEEE Transactions on Pattern Analysis and Machine Intelligence}, author = {Schnake, Thomas and Eberle, Oliver and Lederer, Jonas and Nakajima, Shinichi and Schutt, Kristof T. and Muller, Klaus-Robert and Montavon, Gregoire}, month = nov, year = {2022}, pages = {7581--7596}, } - PAMIBuilding and Interpreting Deep Similarity ModelsOliver Eberle, Jochen Buttner, Florian Krautli, and 3 more authorsIEEE Transactions on Pattern Analysis and Machine Intelligence, Mar 2022
@article{eberle_building_2022, title = {Building and {Interpreting} {Deep} {Similarity} {Models}}, volume = {44}, copyright = {https://creativecommons.org/licenses/by/4.0/legalcode}, issn = {0162-8828, 2160-9292, 1939-3539}, url = {https://ieeexplore.ieee.org/document/9183996/}, doi = {10.1109/TPAMI.2020.3020738}, number = {3}, urldate = {2026-08-19}, journal = {IEEE Transactions on Pattern Analysis and Machine Intelligence}, author = {Eberle, Oliver and Buttner, Jochen and Krautli, Florian and Muller, Klaus-Robert and Valleriani, Matteo and Montavon, Gregoire}, month = mar, year = {2022}, pages = {1149--1161}, }