Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models. Marks, S., Rager, C., Michaud, E. J., Belinkov, Y., Bau, D., & Mueller, A. CoRR, 2024.
Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models [link]Paper  doi  bibtex   
@article{DBLP:journals/corr/abs-2403-19647,
  author       = {Samuel Marks and
                  Can Rager and
                  Eric J. Michaud and
                  Yonatan Belinkov and
                  David Bau and
                  Aaron Mueller},
  title        = {Sparse Feature Circuits: Discovering and Editing Interpretable Causal
                  Graphs in Language Models},
  journal      = {CoRR},
  volume       = {abs/2403.19647},
  year         = {2024},
  url          = {https://doi.org/10.48550/arXiv.2403.19647},
  doi          = {10.48550/ARXIV.2403.19647},
  eprinttype    = {arXiv},
  eprint       = {2403.19647},
  timestamp    = {Mon, 03 Mar 2025 00:00:00 +0100},
  biburl       = {https://dblp.org/rec/journals/corr/abs-2403-19647.bib},
  bibsource    = {dblp computer science bibliography, https://dblp.org}
}

Downloads: 0