[{"data":1,"prerenderedAt":42},["ShallowReactive",2],{"project:/projects/matryoshka-saes":3},{"_path":4,"_dir":5,"_draft":6,"_partial":6,"_locale":7,"title":8,"description":9,"tags":10,"live_url":15,"source_url":16,"image":17,"icon":18,"featured":19,"status":20,"period":21,"role":22,"about":23,"stack":27,"body":29,"_type":36,"_id":37,"_source":38,"_file":39,"_stem":40,"_extension":41},"/projects/matryoshka-saes","projects",false,"","Cross-model Transfer of Matryoshka Sparse Autoencoders","Replicating and extending Matryoshka SAEs across model families",[11,12,13,14],"Mechanistic Interpretability","Sparse Autoencoders","PyTorch","Replication","https://www.lesswrong.com/posts/EpLj8FTBdGvt44TTF/how-matryoshka-sparse-autoencoders-recover-feature","https://github.com/baimamboukar/replicating-matryoshka-sparse-autoencoders-paper","/matryoshka-saes-cover.png","huggingface",true,"Published Writeup (LessWrong)","May 2026 - June 2026","Independent Researcher",[24,25,26],"A replication and extension of Matryoshka Sparse Autoencoders. I reproduced the paper's core result on the synthetic hierarchy and at scale on Gemma-2-2B, confirming that Matryoshka SAEs avoid the feature-absorption failure mode that vanilla SAEs exhibit.","I then transferred the setup to other model families — Llama-3.1-8B and Gemma-3-12B — to test how well the result generalizes beyond the original setting, and documented the full replication in a public writeup on LessWrong.","This project is part of my broader research direction on the generalizability of mechanistic interpretability methods: understanding when results established on one model transfer to others.",[13,12,28],"Gemma-2-2B / Llama-3.1-8B / Gemma-3-12B",{"type":30,"children":31,"toc":32},"root",[],{"title":7,"searchDepth":33,"depth":34,"links":35},2,1,[],"markdown","content:projects:matryoshka-saes.md","content","projects/matryoshka-saes.md","projects/matryoshka-saes","md",1786626763530]