Commit be4a7540 authored by Delvallez Delvallez's avatar Delvallez Delvallez

V1 sota.Explicabilité

parent 31a19213
......@@ -11,6 +11,36 @@
year = {2024},
}
@article{somvanshi_bridging_2026,
title = {Bridging the {Black} {Box}: {A} {Survey} on {Mechanistic} {Interpretability} in {AI}},
volume = {58},
issn = {0360-0300, 1557-7341},
shorttitle = {Bridging the {Black} {Box}},
url = {https://dl.acm.org/doi/10.1145/3787104},
doi = {10.1145/3787104},
abstract = {Mechanistic interpretability seeks to reverse-engineer the internal logic of neural networks by uncovering human-understandable circuits, algorithms, and causal structures that drive model behavior. Unlike post hoc explanations that describe what models do, this paradigm focuses on why and how they compute, tracing information flow through neurons, attention heads, and activation pathways. This survey provides a high-level synthesis of the field-highlighting its motivation, conceptual foundations, and methodological taxonomy rather than enumerating individual techniques. We organize mechanistic interpretability across three abstraction layers—
neurons
,
circuits
, and
algorithms
—and three evaluation perspectives:
behavioral
,
counterfactual
, and
causal
. We further discuss representative approaches and toolchains that enable structural analysis of modern AI systems, outlining how mechanistic interpretability bridges theoretical insights with practical transparency. Despite rapid progress, challenges persist in scaling these analyses to frontier models, resolving polysemantic representations, and establishing standardized causal benchmarks. By connecting historical evolution, current methodologies, and emerging research directions, this survey aims to provide an integrative framework for understanding how mechanistic interpretability can support transparency, reliability, and governance in large-scale AI.},
language = {en},
number = {8},
urldate = {2026-06-01},
journal = {ACM Computing Surveys},
author = {Somvanshi, Shriyank and Islam, Md Monzurul and Rafe, Amir and Tusti, Anannya Ghosh and Chakraborty, Arka and Baitullah, Anika and Chowdhury, Tausif Islam and Alnawmasi, Nawaf and Dutta, Anandi and Das, Subasish},
month = jun,
year = {2026},
pages = {1--35},
}
@inproceedings{chen_axiomatic_2024,
address = {Washington DC USA},
title = {Axiomatic {Causal} {Interventions} for {Reverse} {Engineering} {Relevance} {Computation} in {Neural} {Retrieval} {Models}},
......@@ -169,3 +199,35 @@
author = {Liu, Pei and Liu, Xin and Yao, Ruoyu and Liu, Junming and Meng, Siyuan and Wang, Ding and Ma, Jun},
year = {2025},
}
@inproceedings{bell_its_2022,
address = {Seoul Republic of Korea},
title = {It’s {Just} {Not} {That} {Simple}: {An} {Empirical} {Study} of the {Accuracy}-{Explainability} {Trade}-off in {Machine} {Learning} for {Public} {Policy}},
isbn = {9781450393522},
shorttitle = {It’s {Just} {Not} {That} {Simple}},
url = {https://dl.acm.org/doi/10.1145/3531146.3533090},
doi = {10.1145/3531146.3533090},
language = {en},
urldate = {2025-09-11},
booktitle = {2022 {ACM} {Conference} on {Fairness} {Accountability} and {Transparency}},
publisher = {ACM},
author = {Bell, Andrew and Solano-Kamaiko, Ian and Nov, Oded and Stoyanovich, Julia},
month = jun,
year = {2022},
pages = {248--266},
}
@inproceedings{garouani_investigating_2024,
title = {Investigating the {Duality} of {Interpretability} and {Explainability} in {Machine} {Learning}},
url = {http://arxiv.org/abs/2503.21356},
doi = {10.1109/ICTAI62512.2024.00125},
abstract = {The rapid evolution of machine learning (ML) has led to the widespread adoption of complex "black box" models, such as deep neural networks and ensemble methods. These models exhibit exceptional predictive performance, making them invaluable for critical decision-making across diverse domains within society. However, their inherently opaque nature raises concerns about transparency and interpretability, making them untrustworthy decision support systems. To alleviate such a barrier to high-stakes adoption, research community focus has been on developing methods to explain black box models as a means to address the challenges they pose. Efforts are focused on explaining these models instead of developing ones that are inherently interpretable. Designing inherently interpretable models from the outset, however, can pave the path towards responsible and beneficial applications in the field of ML. In this position paper, we clarify the chasm between explaining black boxes and adopting inherently interpretable models. We emphasize the imperative need for model interpretability and, following the purpose of attaining better (i.e., more effective or efficient w.r.t. predictive performance) and trustworthy predictors, provide an experimental evaluation of latest hybrid learning methods that integrates symbolic knowledge into neural network predictors. We demonstrate how interpretable hybrid models could potentially supplant black box ones in different domains.},
urldate = {2025-08-21},
booktitle = {2024 {IEEE} 36th {International} {Conference} on {Tools} with {Artificial} {Intelligence} ({ICTAI})},
author = {Garouani, Moncef and Mothe, Josiane and Barhrhouj, Ayah and Aligon, Julien},
month = oct,
year = {2024},
note = {arXiv:2503.21356 [cs]},
keywords = {Computer Science - Artificial Intelligence, Computer Science - Machine Learning},
pages = {861--867},
}
......@@ -103,7 +103,7 @@ and Technology (US)) propose différents axes ou composantes de l'IA de confianc
\begin{description}
\item[Fiabilité] Le comportement d'un IA de confiance doit être systématiquement adapté, l'incertitude sur son comportement doit être quantifiable et quantifié et ses capacité de généralisation doivent être robustes.
\item[Intimité] Une IA de confiance doit prévenir des risques pour les données des utilisateurs ainsi que pour les données sensibles contenues dans le modèle (issues de l'apprentissage ou de la base de donnée)
\item[Explicabilité] Le processus de décision d'un IA de confiance doit être transparent et compréhensible pour l'utilisateur.
\item[Explicabilité] Le processus de décision d'une IA de confiance doit être transparent et compréhensible pour l'utilisateur.
\item[Justesse] Une IA de confiance doit minimiser les biais.
\item[Justifiabilité] Une IA de confiance doit pouvoir justifier son bon comportement et notamment sa capacité à respecter les réglementations
\item[Sécurité] Une IA de confiance doit détecter et prévenir les usages malicieux et cherchant à nuire.
......@@ -120,6 +120,21 @@ de l'association/intégration des deux)
- Métrique ??
\end{verbatim}
L'explicabilité de l'intelligence artificielle, vise à rendre humainement compréhensible les raisons pour lesquelles un modèle ou système génère une certaine sortie. \cite{garouani_investigating_2024,bell_its_2022}.
\color{gray}
Dans le contexte de l'explicabilité des systèmes RAG, \cite{ni_towards_2025} distingue trois temps : l'explication de l'extracteur, l'explication du générateur et la mise en perspective de ces deux éléments. \md{Sachant qu'il faut aussi compter les autres composants, il ne faut peut être pas garder cette idée.}
\paragraph{Explicabilité pour les extracteurs} Lorsqu'on se concentre sur l'extraction des document dans un modèle de RAG, on retrouve une tâche d'extraction d'information. \cite{ni_towards_2025} distingue quatre approches d'explication des modèles d'extraction d'information. Les méthodes \textit{post-hoc} construisent des explications à partir des résultats fournis par le modèle sans s'intéresser à l'entraînement ni la conception du modèle étudié. Les méthodes \textit{basées sur les axiomes}, exportent les méthode d'extraction d'information axiomatique pour expliquer d'autres modèles. Les méthodes de \textit{probing} cherchent à localiser les connaissances et les compétences dans les poids et l'espace latent d'un modèle. Pour finir, Certains modèles sont \textit{interprétables by-design} : leur définition est une explication de leur fonctionnement.
\color{black}
Dans le contexte de l'explicabilité des systèmes RAG, \cite{ni_towards_2025} propose de distinguer l'explication du modèle d'extraction et l'explication du modèle génératif. De façon plus générale, une approche naïve d'explication des systèmes RAG est de construire des explications pour chaque composant du modèle (extraction, génération, améliorations, ...). La difficulté intervient au moment de mettre en perspective les explications fournies entre elles. \cite{ni_towards_2025} souligne l'absence de de méthode d'explication pour l'extraction d'information spécifique contexte du RAG ni d'étude de l'application des méthodes génériques à ce cas. La même question se pose pour l'explication des modèle génératifs : les méthodes d'explications classiques sont-elles capable de prendre en compte la présence des autres composants?
Au delà de l'objectif de transparence de l'explicabilité, \cite{ni_towards_2025} pointe la possible utilisation d'explication d'un composant comme indication supplémentaire pour améliorer les performances d'un autre composant. \md{développer avec un exemple?}
\subsection{Interprétabilité Mécaniste}
\begin{verbatim}
......@@ -129,6 +144,9 @@ de l'association/intégration des deux)
*Intervention-based technique*, Representation analysis, Toy Model and synthetic tasks)
\end{verbatim}
\cite{somvanshi_bridging_2026} distingue trois paradigmes d'explicabilité. L'\textit{interprétabilité post-hoc} construit des explication à partir de l'étude du comportement des modèles. Les modèles \textit{interprétables by-design} dont l'explication est auto-contenue. L'\textit{interprétabilité mécaniste} cherche à identifier les (groupes) de composants qui induisent les différents comportements des modèles.
\md{à finir}
\subsection{Activation Patching}
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment