{"as_of":"2026-08-21T15:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:75721f14746e2a1f73716a1259d82204917bbb1118fe55b47cc0af902b5c08ad","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T11:58:37.331025Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:29:34.117929Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-16T12:29:34.117929Z","title":"Sparse autoencoders trained on the same data learn different features","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.12939","last_updated":"2025-04-17T13:37:47Z","snapshot_observed_at":"2026-08-19T04:40:52.835193Z","submitted_at":"2025-04-17T13:37:47Z","title":"Disentangling Polysemantic Channels in Convolutional Neural Networks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T12:29:34.117929Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2504.12939"},"observation_digest":"sha256:96037b9e1d310bb6a637d9db5875ab4505e5514b9138af2f9caa00d39e83836a","observation_id":"5a507f8f-967b-44f5-bff3-5a3cd4c269fa","resolution":{"observed_at":"2026-08-16T12:29:34.117929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-16T04:43:33.510286Z","title":"Sparse autoencoders trained on the same data learn different features","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.00555","last_updated":"2025-05-01T14:30:34Z","snapshot_observed_at":"2026-08-20T02:23:29.242180Z","submitted_at":"2025-05-01T14:30:34Z","title":"On the Mechanistic Interpretability of Neural Networks for Causality in Bio-statistics","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-16T04:43:33.510286Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2505.00555"},"observation_digest":"sha256:5f33577fe63b97c5da579c98771a41ce5b7c77ea78fcb34f2a834232248d1261","observation_id":"a09a1d52-0f7d-41f8-b339-771d5f45fd76","resolution":{"observed_at":"2026-08-16T04:43:33.510286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-07T14:03:02.656998Z","title":"Sparse autoencoders trained on the same data learn different features.arXiv preprint arXiv:2501.16615, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20254","last_updated":"2025-05-26T17:31:36Z","snapshot_observed_at":"2026-08-15T08:10:04.533805Z","submitted_at":"2025-05-26T17:31:36Z","title":"Position: Mechanistic Interpretability Should Prioritize Feature Consistency in SAEs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:03:02.656998Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2505.20254"},"observation_digest":"sha256:286e277759df51d5df5e972d4048e7cef00c9596fc67ea0e447a2a63ac3bf820","observation_id":"0174c757-c952-4a4b-86e9-ca37bda1d101","resolution":{"observed_at":"2026-08-07T14:03:02.656998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-06T23:35:06.700396Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.17673","last_updated":"2025-06-21T10:18:25Z","snapshot_observed_at":"2026-08-17T03:15:54.267854Z","submitted_at":"2025-06-21T10:18:25Z","title":"FaithfulSAE: Towards Capturing Faithful Features with Sparse Autoencoders without External Dataset Dependencies","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T23:35:06.700396Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2506.17673"},"observation_digest":"sha256:f6fab4c4104384ac2e318bae375f65395a421d8004c8432b277348b424a77cfb","observation_id":"06b9e522-c648-4149-9ad9-84b9f48bbed5","resolution":{"observed_at":"2026-08-06T23:35:06.700396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-06T23:03:05.002183Z","title":"Dheeraj Rajagopal, Vidhisha Balachandran, Eduard H Hovy, and Yulia Tsvetkov","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.20040","last_updated":"2026-06-09T21:19:14Z","snapshot_observed_at":"2026-08-14T19:03:44.824832Z","submitted_at":"2025-06-24T22:43:36Z","title":"Cross-Layer Discrete Concept Discovery for Interpreting Language Models","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T23:03:05.002183Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2506.20040"},"observation_digest":"sha256:a0bc7e1f901a84baaa55353c808947a23aa7f31ca85435eab5b0b23b2af0df38","observation_id":"dc209aea-5f78-4784-8b2b-721e8c69e9c1","resolution":{"observed_at":"2026-08-06T23:03:05.002183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-06T21:25:08.563819Z","title":"Sparse autoencoders trained on the same data learn different features","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.00163","last_updated":"2025-07-04T17:10:04Z","snapshot_observed_at":"2026-08-20T16:48:57.739149Z","submitted_at":"2025-06-30T18:11:25Z","title":"Prompting as Scientific Inquiry","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:25:08.563819Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2507.00163"},"observation_digest":"sha256:69c7301669159f558c639e28479fb6204fccc63bd5f2b90e3ea0f3e6ab867c07","observation_id":"ff41e61b-766f-4041-a24f-b844fb5c9680","resolution":{"observed_at":"2026-08-06T21:25:08.563819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-06T15:24:45.771905Z","title":"Sparse autoencoders trained on the same data learn different features, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15977","last_updated":"2025-07-21T18:17:18Z","snapshot_observed_at":"2026-08-18T18:29:40.360758Z","submitted_at":"2025-07-21T18:17:18Z","title":"On the transferability of Sparse Autoencoders for interpreting compressed models","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T15:24:45.771905Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2507.15977"},"observation_digest":"sha256:4d238483da734b00c10bc8c9c55e0c7141a13c23d55e734aa77920d7df0d95c2","observation_id":"737c8818-a4d5-49f2-adf8-d23dba927c0c","resolution":{"observed_at":"2026-08-06T15:24:45.771905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T14:27:25.446156Z","title":"Sparse autoencoders trained on the same data learn different features","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.21324","last_updated":"2025-08-29T04:42:17Z","snapshot_observed_at":"2026-08-15T00:00:17.865061Z","submitted_at":"2025-08-29T04:42:17Z","title":"Distribution-Aware Feature Selection for SAEs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T14:27:25.446156Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2508.21324"},"observation_digest":"sha256:f45d8951a86d69e40830cdf3b64e880ce640e7bcd15845d3dad06e2ef17fb56e","observation_id":"96cfe9ce-a37d-41f3-9def-752e842cb6ed","resolution":{"observed_at":"2026-08-05T14:27:25.446156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2509.09708","last_updated":"2026-04-28T03:29:39Z","snapshot_observed_at":"2026-08-10T21:52:19.966437Z","submitted_at":"2025-09-07T02:29:07Z","title":"Beyond I'm Sorry, I Can't: Dissecting Large Language Model Refusal","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-18T18:56:13.680353Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2509.09708"},"observation_digest":"sha256:0892fd7144b14ff4c149fef6a35325f3c439513edde1b34a2f4268af3e34d7f9","observation_id":"604047c3-fb81-49ed-a662-0554daae9874","resolution":{"observed_at":"2026-05-18T18:56:45.869896Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-15T15:47:51.634104Z","title":"Sparse autoencoders trained on the same data learn different features.arXiv preprint arXiv:2501.16615,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.22015","last_updated":"2026-06-04T09:33:37Z","snapshot_observed_at":"2026-08-17T11:09:43.003460Z","submitted_at":"2025-09-26T07:51:03Z","title":"Concept-SAE: A Controllable and Invertible Concept Interface for Sparse Autoencoders","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T15:47:51.634104Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2509.22015"},"observation_digest":"sha256:16f55d12d8954001f98a35dd5440f6c3c216a4226a02cbcc21d1b0da1dfa53db","observation_id":"87e2e879-2d3f-405e-bc86-af6306f63ba6","resolution":{"observed_at":"2026-08-15T15:47:51.634104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2512.06655","last_updated":"2026-05-15T09:56:17Z","snapshot_observed_at":"2026-08-16T12:48:19.183589Z","submitted_at":"2025-12-07T04:46:30Z","title":"Graph-Regularized Sparse Autoencoders for LLM Safety Steering","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T18:39:02.687301Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2512.06655"},"observation_digest":"sha256:50c7e15accbae78434ec44d98df69330bdd789cac9d1cc4f5b4f880ba1597ffa","observation_id":"7ef256f4-6bf1-4861-8920-fb80ba4b947c","resolution":{"observed_at":"2026-05-21T18:40:28.874914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-02T18:56:19.219712Z","title":"Subhash Kantamneni, Joshua Engels, Senthooran Rajamanoharan, Max Tegmark, and Neel Nanda","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.04198","last_updated":"2026-06-16T16:00:25Z","snapshot_observed_at":"2026-08-13T05:33:13.764056Z","submitted_at":"2026-03-04T15:46:23Z","title":"Stable and Steerable Sparse Autoencoders with Weight Regularization","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T18:56:19.219712Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2603.04198"},"observation_digest":"sha256:80c44d8ad4239231fb2f4d66024eff6649975cc840542902fcd28e05bbb3e517","observation_id":"3c5d7cb2-6517-445b-ae97-d837dff27bbc","resolution":{"observed_at":"2026-08-02T18:56:19.219712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.04072","last_updated":"2026-04-13T12:08:32Z","snapshot_observed_at":"2026-08-13T03:59:30.896577Z","submitted_at":"2026-04-13T12:08:32Z","title":"Sparse Autoencoder Decomposition of Clinical Sequence Model Representations: Feature Complexity, Task Specialisation, and Mortality Prediction","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-10T15:06:12.006883Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.04072"},"observation_digest":"sha256:a9a6b313da54a9d6bcf3350a4e15e3d2468c68dc41c13a6443554cfc16f20854","observation_id":"6253079f-de55-414a-8388-6abd2c21f1dd","resolution":{"observed_at":"2026-05-11T11:11:06.192503Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.12770","last_updated":"2026-05-19T21:54:38Z","snapshot_observed_at":"2026-07-06T23:24:23.402304Z","submitted_at":"2026-05-12T21:32:45Z","title":"WriteSAE: Sparse Autoencoders for Recurrent State","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-21T07:46:41.159688Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.12770"},"observation_digest":"sha256:e9421a83db15c1051409dd9788b746f473ae29141aa4b0e662c7864eaf94d467","observation_id":"8f529f43-7303-4ee8-b6f7-cda07b545ca3","resolution":{"observed_at":"2026-05-21T07:49:50.201457Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.12874","last_updated":"2026-05-13T01:41:38Z","snapshot_observed_at":"2026-08-14T18:27:17.550876Z","submitted_at":"2026-05-13T01:41:38Z","title":"Descriptive Collision in Sparse Autoencoder Auto-Interpretability: When One Explanation Describes Many Features","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-14T20:27:38.363693Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.12874"},"observation_digest":"sha256:984517ca0f95293b260e8bf310a665e153e337c15997f6b1d0039b8872034c97","observation_id":"5e5456da-86ce-4424-ac82-6214d1e99445","resolution":{"observed_at":"2026-05-14T20:29:27.952393Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.12991","last_updated":"2026-08-08T15:53:51Z","snapshot_observed_at":"2026-08-13T23:16:28.421431Z","submitted_at":"2026-05-13T04:45:08Z","title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-14T20:04:57.638215Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.12991"},"observation_digest":"sha256:3d88ef7f8d09fe774cab08008fdc090f85238eeb3616776f34f8a6e9396367d6","observation_id":"870c7a5c-33b1-4e92-b8ed-6c240305a6e8","resolution":{"observed_at":"2026-05-14T20:07:53.668806Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.12991","last_updated":"2026-08-08T15:53:51Z","snapshot_observed_at":"2026-08-13T23:16:28.421431Z","submitted_at":"2026-05-13T04:45:08Z","title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T21:30:30.384184Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.12991"},"observation_digest":"sha256:c820b39d5054ee2aa5818058b1ef6fdfc499f9c11731032e9c75722fe6440deb","observation_id":"920ff42f-aa84-4540-9e20-dbc511288f18","resolution":{"observed_at":"2026-05-20T21:33:46.448245Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.14347","last_updated":"2026-05-15T22:27:22Z","snapshot_observed_at":"2026-08-18T21:24:26.159694Z","submitted_at":"2026-05-14T04:15:30Z","title":"Exemplar Partitioning for Mechanistic Interpretability","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-15T01:35:42.336550Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.14347"},"observation_digest":"sha256:9856d3df8d2d57a71604498fef8a93b10e9be600c75da94c17c8bca19e50f820","observation_id":"c10ecae9-98b8-4af9-996a-8b372b587cfa","resolution":{"observed_at":"2026-05-15T01:38:27.619659Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.14347","last_updated":"2026-05-15T22:27:22Z","snapshot_observed_at":"2026-08-18T21:24:26.159694Z","submitted_at":"2026-05-14T04:15:30Z","title":"Exemplar Partitioning for Mechanistic Interpretability","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-20T20:51:38.699191Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.14347"},"observation_digest":"sha256:a0f2f4d2ee5118c6dc473ba8f537940249eea200e7a3c789a56150f030710311","observation_id":"e017d5f4-b1cf-44c1-9126-5ecb13237715","resolution":{"observed_at":"2026-05-20T20:53:43.657427Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.18229","last_updated":"2026-05-18T11:20:57Z","snapshot_observed_at":"2026-08-15T13:05:39.630785Z","submitted_at":"2026-05-18T11:20:57Z","title":"Are Sparse Autoencoder Benchmarks Reliable?","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T12:43:13.014365Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.18229"},"observation_digest":"sha256:c85ac180eb3468758d53e4d32f2cd6f686773f1b4650f5767b57dd0edc20622d","observation_id":"3623622e-bbb4-4e41-9730-bde521d597c8","resolution":{"observed_at":"2026-05-20T12:43:16.769836Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2605.28149","last_updated":"2026-08-04T06:19:42Z","snapshot_observed_at":"2026-08-07T23:09:28.260788Z","submitted_at":"2026-05-27T08:31:43Z","title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T14:16:44.232080Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.28149"},"observation_digest":"sha256:0aac5fc67a76dc652b5ffb320bb4259196e7820b4268858ff1ff61fe915e29f3","observation_id":"7a86f942-362b-4781-9267-52c324587e07","resolution":{"observed_at":"2026-06-29T14:23:30.747731Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-04T05:02:51.333339Z","title":"Sparse autoencoders trained on the same data learn different features","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28149","last_updated":"2026-08-04T06:19:42Z","snapshot_observed_at":"2026-08-07T23:09:28.260788Z","submitted_at":"2026-05-27T08:31:43Z","title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T05:02:51.333339Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2605.28149"},"observation_digest":"sha256:23ce6e2fb7d68f0960cb32c67266c1c64bad2ba5810c9384e66bdf7936ab15e4","observation_id":"7a1642d3-caee-46c9-b097-d03ebf4a24e4","resolution":{"observed_at":"2026-08-04T05:02:51.333339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2606.03002","last_updated":"2026-06-04T19:39:22Z","snapshot_observed_at":"2026-08-16T01:59:20.179381Z","submitted_at":"2026-06-02T01:17:05Z","title":"Perplexity Can Miss SAE Feature Damage Under Quantization","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T11:41:18.460538Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2606.03002"},"observation_digest":"sha256:5778f8183707602c16182052eddf7469fa935ade6f94886936c4efd6e8e1e15a","observation_id":"3c0265fb-1f6a-4a1a-9e76-8882165393e5","resolution":{"observed_at":"2026-07-02T01:36:25.811675Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2606.12138","last_updated":"2026-06-10T14:32:57Z","snapshot_observed_at":"2026-08-13T03:58:36.881249Z","submitted_at":"2026-06-10T14:32:57Z","title":"Unstable Features, Reproducible Subspaces: Understanding Seed Dependence in Sparse Autoencoders","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-27T10:39:51.615710Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2606.12138"},"observation_digest":"sha256:a5e23ee5af2f962ce0064a6f539072543d95b3e9b0d6e41f2a6e0b81ceb6f281","observation_id":"aa9ccb36-7ef4-4339-aec7-4877f0238e81","resolution":{"observed_at":"2026-06-27T10:40:49.834013Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":"2501.16615","doi":"10.48550/arxiv.2501.16615","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, and Alexander Turner","venue":"ArXiv.org","work_id":"bb37f39f-cdbf-466e-a548-ffbe29964093","year":2025},"citing_paper":{"arxiv_id":"2606.26396","last_updated":"2026-06-24T21:26:43Z","snapshot_observed_at":"2026-08-02T16:34:16.438122Z","submitted_at":"2026-06-24T21:26:43Z","title":"At the Edge of Understanding: Sparse Autoencoders Trace The Limits of Transformer Generalization","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-06-26T01:27:39.812228Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2606.26396"},"observation_digest":"sha256:30f12e90116835e936cf6d8ec0afeb704684608c0342719e63b9c6f37ba1d7fd","observation_id":"6a79cf32-e615-43a7-ba6a-a42e6b690332","resolution":{"observed_at":"2026-06-26T01:28:50.521664Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-07-12T03:00:02.282468Z","title":"Sparse autoencoders trained on the same data learn different features.arXiv preprint arXiv:2501.16615,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03365","last_updated":"2026-07-03T14:22:05Z","snapshot_observed_at":"2026-08-17T12:45:43.057484Z","submitted_at":"2026-07-03T14:22:05Z","title":"Brand-as-Memory: Vision-Language Models Encode Causal, Mechanistically Localizable Credibility Priors for News Sources","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-12T03:00:02.282468Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2607.03365"},"observation_digest":"sha256:618c675c98964fa8d5e3ee1c50bc2d734ff47ca69ad1b3480f32f2e00320dfb2","observation_id":"7f3991a4-2cfc-4c71-ba73-4a9b1db256e0","resolution":{"observed_at":"2026-07-12T03:00:02.282468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-01T10:03:58.308413Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20596","last_updated":"2026-07-22T17:33:16Z","snapshot_observed_at":"2026-08-15T09:33:27.353760Z","submitted_at":"2026-07-22T17:33:16Z","title":"Are Single-Token Sparse Autoencoder Features Causally Necessary? Layer-Depth and SAE-Family Effects","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-01T10:03:58.308413Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2607.20596"},"observation_digest":"sha256:4d59d73b65e5a79935f545cac87cd3ac12bf197607ad60a59d3c14bd1b5361c7","observation_id":"196abc9d-4b93-424c-a922-f9d8a4d1c5cc","resolution":{"observed_at":"2026-08-01T10:03:58.308413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-07-30T19:59:23.290640Z","title":"arXiv preprint arXiv:2501.16615 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26825","last_updated":"2026-07-30T15:41:13Z","snapshot_observed_at":"2026-08-18T19:21:12.868419Z","submitted_at":"2026-07-29T12:18:36Z","title":"From Found to Designed: Concepts as a Design Axis for Large Language Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-07-30T19:59:23.290640Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2607.26825"},"observation_digest":"sha256:40fea2678a9bfbe1de0c81b8ad9420b0068d30e5f5e7537993d91f5225804e8b","observation_id":"cbe84711-72a2-4c18-be10-956fcee445b3","resolution":{"observed_at":"2026-07-30T19:59:23.290640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-01T10:48:50.463361Z","title":"arXiv preprint arXiv:2501.16615 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26825","last_updated":"2026-07-30T15:41:13Z","snapshot_observed_at":"2026-08-18T19:21:12.868419Z","submitted_at":"2026-07-29T12:18:36Z","title":"From Found to Designed: Concepts as a Design Axis for Large Language Models","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-01T10:48:50.463361Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2607.26825"},"observation_digest":"sha256:0031637d09adfe0eb89f11644d35927f0efb1aa062e55444fe8fabf7fb43fc3a","observation_id":"8a26eaee-def6-4030-a281-a4d2db4d5cb9","resolution":{"observed_at":"2026-08-01T10:48:50.463361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16615","snapshot_observed_at":"2026-08-14T13:29:00.224169Z","title":"Sparse autoencoders trained on the same data learn different features","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13337","last_updated":"2026-08-13T15:06:56Z","snapshot_observed_at":"2026-08-19T08:41:50.325589Z","submitted_at":"2026-08-13T15:06:56Z","title":"Where You Measure Decides What You Measure: Position Selection in Ablation-Based SAE Evaluation","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-14T13:29:00.224169Z"},"links":{"cited_paper":"/paper/2501.16615","citing_paper":"/paper/2608.13337"},"observation_digest":"sha256:7210510574ecb8aa8600740e29c8aef5f2bfefa58c5a1fa390401df6cb267ea4","observation_id":"fb5c0c8f-57be-4e01-8c52-d49b6e36df09","resolution":{"observed_at":"2026-08-14T13:29:00.224169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.16615/citation-record","integrity":"/paper/2501.16615/integrity","json":"/paper/2501.16615/citation-record.json","paper":"/paper/2501.16615"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:36.899951Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:36.899951Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:1e88a235ea737fb4700f4a1f05e3fecc9e6f51cb474effa50c4f845236acbe05","observation_id":"4ec5f24a-3453-4ded-bb68-8820c0bcd3eb","resolution":{"observed_at":"2026-08-10T11:58:36.899951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:36.936293Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:36.936293Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:5d1fd78f3ad28163ceeb3e3c163e9879b91e0f073b29204ed840e207739b4a4e","observation_id":"fa931dab-10f2-4acf-96aa-bf59dd4bc20d","resolution":{"observed_at":"2026-08-10T11:58:36.936293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.808789Z","title":"Git re-basin: Merging models modulo permutation symmetries","venue":null,"work_id":"205d352a-9218-46e4-9fec-69ce02fcf8a5","year":2023},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.002478Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:91813f2fbb02611c8681f1752192aab3f11a2ac74e939dc887faf03f96d3a44e","observation_id":"eca188c6-ca03-493b-97ad-9168a85834a0","resolution":{"observed_at":"2026-08-10T11:58:37.811691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.800109Z","title":"Sparse autoencoders do not find canonical units of analysis","venue":null,"work_id":"99d406c4-d812-4a30-be0c-d52f4bb5c34c","year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.010255Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:447c5e614fe196f6acf67b35694f6209f1c6c14c35c4b6433bb8261fe258ff22","observation_id":"6b025c49-dd4c-4b70-b9f1-cdbb6e481606","resolution":{"observed_at":"2026-08-10T11:58:37.803757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.020986Z","title":"Linear algebraic structure of word senses, with applications to polysemy","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.020986Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:300cfe05b0d37a3f01bb5b6128516684727865584cb146787b5a990b27f5e128","observation_id":"afd0de54-4380-40cc-98b4-661fbb1fa857","resolution":{"observed_at":"2026-08-10T11:58:37.020986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11179","last_updated":"2024-10-15T01:38:03Z","snapshot_observed_at":"2026-08-16T13:09:22.337092Z","submitted_at":"2024-10-15T01:38:03Z","title":"Interpretability as Compression: Reconsidering SAE Explanations of Neural Activations with MDL-SAEs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11179","snapshot_observed_at":"2026-08-10T11:58:37.031801Z","title":"T., and Sharkey, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.031801Z"},"links":{"cited_paper":"/paper/2410.11179","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:f60ccea1541ceb3b5c384b2e5b2018f00ac79c3f948a778e0a8d580acbefb996","observation_id":"6ebebd8f-f61d-4f70-9ddb-fa83f7df0cff","resolution":{"observed_at":"2026-08-10T11:58:37.031801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07656","last_updated":"2025-03-01T07:16:47Z","snapshot_observed_at":"2026-08-16T13:10:49.939427Z","submitted_at":"2024-10-10T06:55:38Z","title":"Mechanistic Permutability: Match Features Across Layers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07656","snapshot_observed_at":"2026-08-10T11:58:37.035849Z","title":"Mechanistic permutability: Match features across layers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.035849Z"},"links":{"cited_paper":"/paper/2410.07656","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:9e185d0c38d3dc2ce40158a53f9a6bdb054e6375328ca38f4ad9f1c8c736fc77","observation_id":"ff3d2ade-f299-44de-ad91-82db002abc82","resolution":{"observed_at":"2026-08-10T11:58:37.035849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.787577Z","title":"Sae repository","venue":null,"work_id":"da2586c4-d313-4b96-8cda-a0037be23737","year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.038142Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:88a4efad6faae3aca30eb4bb45fb74f1d0cdf0db375d2cdfce36c43b5c87a988","observation_id":"5bf976c2-bfac-45ce-88f3-e2153da84431","resolution":{"observed_at":"2026-08-10T11:58:37.790457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.041431Z","title":"G., Bradley, H., O’Brien, K., Hallahan, E., Khan, M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.041431Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:27467d48b1837e78476d4f555ab7e6119fc319492c8836bcdc8e2e8fcbbaabb8","observation_id":"2397883a-6002-4f83-afae-c767ced785a7","resolution":{"observed_at":"2026-08-10T11:58:37.041431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12241","last_updated":"2024-05-24T13:16:32Z","snapshot_observed_at":"2026-08-19T06:03:34.593389Z","submitted_at":"2024-05-17T17:03:46Z","title":"Identifying Functionally Important Features with End-to-End Sparse Dictionary Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12241","snapshot_observed_at":"2026-08-10T11:58:37.044442Z","title":"Identifying functionally important features with end-to-end sparse dictionary learning, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.044442Z"},"links":{"cited_paper":"/paper/2405.12241","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:d8156649154fabcd180eb1855c7cb976131607ebde0cf493274fa7a5e765168c","observation_id":"02bf38a3-4166-4844-9a19-177480580936","resolution":{"observed_at":"2026-08-10T11:58:37.044442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.774542Z","title":"L., Anil, C., Denison, C., Askell, A., Lasenby, R., Wu, Y., Kravec, S., Schiefer, N., Maxwell, T., Joseph, N., Tamkin, A., Nguyen, K., McLean, B., Burke, J","venue":null,"work_id":"510409f2-cd1a-4562-80d7-00ca6157dd96","year":2023},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.047845Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:68ff45f38d9d9f43294ae362621fe7990aa6fcf833abf2decdca6fa7fb6da052","observation_id":"91109b86-a059-4d91-8816-b545eaad387b","resolution":{"observed_at":"2026-08-10T11:58:37.778055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.050861Z","title":"A is for absorption: Studying feature splitting and absorption in sparse autoencoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.050861Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:f4d2d8328b9bb3909a22905490c4dd24273f083e3479895acc0a29fe2206c153","observation_id":"8911dae6-91d5-499a-a5cc-36fe7d4080ee","resolution":{"observed_at":"2026-08-10T11:58:37.050861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08600","last_updated":"2023-10-04T13:17:38Z","snapshot_observed_at":"2026-08-16T13:37:26.527260Z","submitted_at":"2023-09-15T17:56:55Z","title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08600","snapshot_observed_at":"2026-08-10T11:58:37.053885Z","title":"Sparse autoencoders find highly interpretable features in language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.053885Z"},"links":{"cited_paper":"/paper/2309.08600","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:32432b3cabfc4a94359f5a604d27ea7c1076b550969b31b96490e3b3f6fd8fad","observation_id":"6fef1f70-6c20-4d81-9308-db0ea5683aa6","resolution":{"observed_at":"2026-08-10T11:58:37.053885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-10T11:58:37.056854Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.056854Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:064f927d4fc191842e575f0f0741909543764a69f866cb86c5b5df6f4d5f1c5b","observation_id":"3cabaa19-cb66-4db0-89af-17c70301a7ac","resolution":{"observed_at":"2026-08-10T11:58:37.056854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.10652","last_updated":"2022-09-21T20:49:26Z","snapshot_observed_at":"2026-08-16T21:36:28.067615Z","submitted_at":"2022-09-21T20:49:26Z","title":"Toy Models of Superposition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.10652","snapshot_observed_at":"2026-08-10T11:58:37.060199Z","title":"Toy models of superposition","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.060199Z"},"links":{"cited_paper":"/paper/2209.10652","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:59a765b243716fe542dc8feefc0de68cd4d603b3d96ef2c4e2b33e37bd571b57","observation_id":"2e47b8e9-422b-4a69-a7c0-2e077293fadc","resolution":{"observed_at":"2026-08-10T11:58:37.060199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14860","last_updated":"2025-02-27T03:03:59Z","snapshot_observed_at":"2026-08-19T12:35:11.476497Z","submitted_at":"2024-05-23T17:59:04Z","title":"Not All Language Model Features Are One-Dimensionally Linear","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14860","snapshot_observed_at":"2026-08-10T11:58:37.064303Z","title":"J., Liao, I., Gurnee, W., and Tegmark, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.064303Z"},"links":{"cited_paper":"/paper/2405.14860","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:90902724809dd2156668b925c9466db48a3d6b253bbffde9f74a0ad85d0e4e4a","observation_id":"f1ef7e34-c91d-4568-bb8a-66a4bfb724d0","resolution":{"observed_at":"2026-08-10T11:58:37.064303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00027","last_updated":"2020-12-31T19:00:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-12-31T19:00:10Z","title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00027","snapshot_observed_at":"2026-08-10T11:58:37.066893Z","title":"The pile: An 800gb dataset of diverse text for language modeling","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.066893Z"},"links":{"cited_paper":"/paper/2101.00027","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:a44decc8904bdcfa0ccb8e192015bd28adf8d7f34772d47b9f011adf6fe3a03c","observation_id":"b2d23deb-61b9-4bf0-b8a8-6ca19d5d3463","resolution":{"observed_at":"2026-08-10T11:58:37.066893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04093","last_updated":"2024-06-06T14:10:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-06T14:10:12Z","title":"Scaling and evaluating sparse autoencoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04093","snapshot_observed_at":"2026-08-10T11:58:37.070778Z","title":"D., Tillman, H., Goh, G., Troll, R., Radford, A., Sutskever, I., Leike, J., and Wu, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.070778Z"},"links":{"cited_paper":"/paper/2406.04093","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:5e2b3f4998aa3adc90f775989f0b1bfbfdddcdbcdf6bed6f8014422365951aaf","observation_id":"455e172a-946d-407c-b4cd-e54d5308e024","resolution":{"observed_at":"2026-08-10T11:58:37.070778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.074327Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.074327Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:60462182678de7e0341026bd7f3341a14ae7160065311fcbcd2bab854dd1f460","observation_id":"ba99b6ad-223a-498a-a90a-b8aaa4cb18bb","resolution":{"observed_at":"2026-08-10T11:58:37.074327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05147","last_updated":"2024-08-19T07:51:05Z","snapshot_observed_at":"2026-08-15T15:50:24.074149Z","submitted_at":"2024-08-09T16:06:42Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05147","snapshot_observed_at":"2026-08-10T11:58:37.077293Z","title":"Gemma scope: Open sparse autoencoders everywhere all at once on gemma 2, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.077293Z"},"links":{"cited_paper":"/paper/2408.05147","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:3c83dbbc4274aa9501d112789b60b8e62e902a5ab5365a291c45528ba56545db","observation_id":"d6d8fbe5-0347-4fd7-b6d7-14cc6fe30d10","resolution":{"observed_at":"2026-08-10T11:58:37.077293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.01220","last_updated":"2024-11-06T08:42:09Z","snapshot_observed_at":"2026-08-20T06:46:17.051631Z","submitted_at":"2024-11-02T11:42:23Z","title":"Enhancing Neural Network Interpretability with Feature-Aligned Sparse Autoencoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.01220","snapshot_observed_at":"2026-08-10T11:58:37.080957Z","title":"Enhancing neural network interpretability with feature-aligned sparse autoencoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.080957Z"},"links":{"cited_paper":"/paper/2411.01220","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:aa755b34c592b4548a7330a3d3ae7f78e0bacb6cc351604a04e5252864981f0e","observation_id":"063649c0-073b-4168-9ff4-7c7b39ebdca7","resolution":{"observed_at":"2026-08-10T11:58:37.080957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.761400Z","title":"Interpretability dreams","venue":null,"work_id":"68bd64b6-4ba1-4888-aa06-04ca41fa2ffe","year":2023},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.106214Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:f59a3bf2e8b389691cc07d9e8bff6348a3f31445e629c7e7ec54fff311d45acc","observation_id":"609ea754-a9c9-4872-bdf8-5198463299df","resolution":{"observed_at":"2026-08-10T11:58:37.764140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13928","last_updated":"2025-08-06T13:47:10Z","snapshot_observed_at":"2026-08-16T13:08:17.757544Z","submitted_at":"2024-10-17T17:56:01Z","title":"Automatically Interpreting Millions of Features in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13928","snapshot_observed_at":"2026-08-10T11:58:37.143219Z","title":"Automatically interpreting millions of features in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.143219Z"},"links":{"cited_paper":"/paper/2410.13928","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:8d70b60f82e26a2145020ca9b9ed21845f1b0fc884f915bd1416143d6ab40d4d","observation_id":"5b5e109b-dc60-434f-b6d1-9e38a1c46758","resolution":{"observed_at":"2026-08-10T11:58:37.143219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16014","last_updated":"2024-04-30T17:54:04Z","snapshot_observed_at":"2026-08-02T10:59:21.418230Z","submitted_at":"2024-04-24T17:47:22Z","title":"Improving Dictionary Learning with Gated Sparse Autoencoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16014","snapshot_observed_at":"2026-08-10T11:58:37.188840Z","title":"Improving dictionary learning with gated sparse autoencoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.188840Z"},"links":{"cited_paper":"/paper/2404.16014","citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:660d77d7dd9b899a7e9d0b13b742a8654ec842b88e66a635d7e65aa295a5470f","observation_id":"b0a12e3f-ebc3-4ecd-8753-f07a8c7dcc1d","resolution":{"observed_at":"2026-08-10T11:58:37.188840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.751920Z","title":"The strong feature hypothesis could be wrong, 2024","venue":null,"work_id":"e71b5ba4-b03d-42cf-b8f9-aa5aec7aa091","year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.230682Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:e0b1d1bb184b12ccc9cd1ab6027238b561c4d324cb0ad041351fc66e71aab0ea","observation_id":"c16a6c4c-1da4-4041-85aa-d6abac9e8b0f","resolution":{"observed_at":"2026-08-10T11:58:37.756045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T11:58:37.331025Z","title":"L., McDougall, C., MacDiarmid, M., Freeman, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T11:58:37.331025Z"},"links":{"citing_paper":"/paper/2501.16615"},"observation_digest":"sha256:970de4ce9f69ef32306125d5f107491f8faabf23128edd6926e1e30e55fb94f4","observation_id":"76947b42-accf-4ad1-a014-ae26b5c502f0","resolution":{"observed_at":"2026-08-10T11:58:37.331025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.16615","last_updated":"2025-01-29T19:18:45Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-18T03:32:04.306379Z","submitted_at":"2025-01-28T01:24:16Z","title":"Sparse Autoencoders Trained on the Same Data Learn Different Features"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 30 inbound Pith citation observations for arXiv:2501.16615."}