{"as_of":"2026-08-07T19:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:972e178887544704cdac49cd9551ce0f030418c5229ef5c2ee2a7f7c416a3499","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:35:25.837701Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:29:45.394907Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-07T12:35:25.837701Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24473","last_updated":"2025-06-05T11:42:24Z","snapshot_observed_at":"2026-08-07T12:19:04.215675Z","submitted_at":"2025-05-30T11:20:44Z","title":"Train One Sparse Autoencoder Across Multiple Sparsity Budgets to Preserve Interpretability and Accuracy","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:25.837701Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2505.24473"},"observation_digest":"sha256:54da3d4775f19ad8753916b49637b576f6f6a0cd1ec0bbc46da256e1d78ddc1c","observation_id":"82a4d355-2fc2-4888-a10d-f5e945759d3e","resolution":{"observed_at":"2026-08-07T12:35:25.837701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-07T11:56:15.179663Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01197","last_updated":"2025-06-01T22:20:07Z","snapshot_observed_at":"2026-08-07T11:47:50.287969Z","submitted_at":"2025-06-01T22:20:07Z","title":"Incorporating Hierarchical Semantics in Sparse Autoencoder Architectures","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:15.179663Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2506.01197"},"observation_digest":"sha256:fd5c7946a42bf83d750dad7fffb2afbb1c476c6f2ddbcc607f954d212656696e","observation_id":"19903c6d-936d-43b0-9afe-7f2362cf3100","resolution":{"observed_at":"2026-08-07T11:56:15.179663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-07T04:45:42.719642Z","title":"Woosuk Kwon, Zhuohan Li, Siyuan Zhuang, Ying Sheng, Lianmin Zheng, Cody Hao Yu, Joseph E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09967","last_updated":"2025-06-13T18:01:49Z","snapshot_observed_at":"2026-08-07T04:34:08.063320Z","submitted_at":"2025-06-11T17:44:01Z","title":"Resa: Transparent Reasoning Models via SAEs","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:42.719642Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2506.09967"},"observation_digest":"sha256:55622bf729eb507a1f7e1b9d1eb5fdd2e3c151cabc5dbc78ac7ab6905a5fcda5","observation_id":"620053fa-a4fe-44e0-a804-5c35444fc6ec","resolution":{"observed_at":"2026-08-07T04:45:42.719642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-06T18:24:51.180244Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08473","last_updated":"2025-07-11T10:31:53Z","snapshot_observed_at":"2026-08-06T18:15:09.954967Z","submitted_at":"2025-07-11T10:31:53Z","title":"Evaluating SAE interpretability without explanations","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:24:51.180244Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2507.08473"},"observation_digest":"sha256:431d2ca71ac118fe2d213c82555353b9156d8e018008fcfbc24990f72297c428","observation_id":"0a50bec3-c959-4b47-af7f-34a36a929314","resolution":{"observed_at":"2026-08-06T18:24:51.180244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-06T15:24:45.744009Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability, 2025 a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15977","last_updated":"2025-07-21T18:17:18Z","snapshot_observed_at":"2026-08-06T23:33:13.768980Z","submitted_at":"2025-07-21T18:17:18Z","title":"On the transferability of Sparse Autoencoders for interpreting compressed models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T15:24:45.744009Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2507.15977"},"observation_digest":"sha256:d24ec2274ce1c61afa8fc8407cde162b25f85355a9256ac58514aa9e62be3e91","observation_id":"4112d7ac-bf41-414e-88b3-355f5663fc65","resolution":{"observed_at":"2026-08-06T15:24:45.744009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-05T14:27:25.439860Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.21324","last_updated":"2025-08-29T04:42:17Z","snapshot_observed_at":"2026-08-05T14:27:24.567504Z","submitted_at":"2025-08-29T04:42:17Z","title":"Distribution-Aware Feature Selection for SAEs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T14:27:25.439860Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2508.21324"},"observation_digest":"sha256:007cba68b41da11f74dc90f7091969d21d77fb2b6bc33499371acfde8ff7e9a0","observation_id":"5c41650a-07f1-43c6-bec8-c9c5004543f8","resolution":{"observed_at":"2026-08-05T14:27:25.439860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2509.18127","last_updated":"2026-04-14T10:00:44Z","snapshot_observed_at":"2026-08-02T17:25:00.233548Z","submitted_at":"2025-09-11T11:22:43Z","title":"Safe-SAIL: Towards a Fine-grained Safety Landscape of Large Language Models via Sparse Autoencoder Interpretation Framework","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-18T18:13:01.662828Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2509.18127"},"observation_digest":"sha256:9595fe15cede44b4486ab69bb7e578e44c5a571ce6a44bf27aafb0701b88a175","observation_id":"00bc17a8-ec58-433b-9b25-7291e4291f8c","resolution":{"observed_at":"2026-05-18T18:16:43.748207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2601.14004","last_updated":"2026-04-14T03:49:06Z","snapshot_observed_at":"2026-07-31T09:24:49.481740Z","submitted_at":"2026-01-20T14:23:23Z","title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","version":4},"reference_index":148,"source":"pdf_text","source_observed_at":"2026-05-16T12:39:57.398423Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2601.14004"},"observation_digest":"sha256:5c62db85ae7aa77042cb41af6e7171e8e12224e6a70fad45494d76542bc71ff5","observation_id":"eb0d2f34-a1c1-4baa-98c7-71091dd817b0","resolution":{"observed_at":"2026-05-16T12:40:54.605878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-02T18:56:20.000242Z","title":"Leo Gao, Tom Dupré la Tour, Henk Tillman, Gabriel Goh, Rajan Troll, Alec Radford, Ilya Sutskever, Jan Leike, and Jeffrey Wu","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.04198","last_updated":"2026-06-16T16:00:25Z","snapshot_observed_at":"2026-08-06T05:44:40.584997Z","submitted_at":"2026-03-04T15:46:23Z","title":"Stable and Steerable Sparse Autoencoders with Weight Regularization","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T18:56:20.000242Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2603.04198"},"observation_digest":"sha256:1d7b2f2c9d236e5828c825b85d80e54b721faef7653e169db5e5ae7c4fc46c7d","observation_id":"1e7376d5-7454-4840-8813-48c7a2539620","resolution":{"observed_at":"2026-08-02T18:56:20.000242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2604.08846","last_updated":"2026-04-10T01:01:56Z","snapshot_observed_at":"2026-08-02T14:54:38.934183Z","submitted_at":"2026-04-10T01:01:56Z","title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T18:04:05.157103Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2604.08846"},"observation_digest":"sha256:18a103420c88deb8262ff5be3f12a667da47d330e61fff8eba427686e464d4db","observation_id":"f372237d-5f3a-434a-9a9c-253ec13b55e8","resolution":{"observed_at":"2026-05-11T05:35:57.627704Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.05223","last_updated":"2026-04-18T05:53:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-18T05:53:31Z","title":"Structural Instability of Feature Composition","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T06:40:00.507484Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.05223"},"observation_digest":"sha256:b4f7d33fe322909f9a0c350bd0aa80ce002a8dca11f656677417cf117cffc442","observation_id":"41ce46e6-8ce3-437d-9d3d-838688b499cf","resolution":{"observed_at":"2026-05-10T06:41:36.519790Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.06494","last_updated":"2026-05-07T16:15:16Z","snapshot_observed_at":"2026-07-06T23:18:55.724335Z","submitted_at":"2026-05-07T16:15:16Z","title":"From Token Lists to Graph Motifs: Weisfeiler-Lehman Analysis of Sparse Autoencoder Features","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-08T09:41:01.775116Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.06494"},"observation_digest":"sha256:2acd942a7da4ae231e64ac3b22d3c0c973195f4aca7b14583cdfeb3fe63bf003","observation_id":"fc8f724e-4403-4fbb-96fc-6f84e4825e25","resolution":{"observed_at":"2026-05-11T20:16:11.060148Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.07922","last_updated":"2026-05-11T02:18:14Z","snapshot_observed_at":"2026-07-06T23:20:15.696237Z","submitted_at":"2026-05-08T15:57:37Z","title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T03:13:58.543525Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.07922"},"observation_digest":"sha256:63b270cdcfdb98b8060f5c5e2052becf0f355f4464477c844a76517040a45386","observation_id":"12da451f-70cb-4dca-a9f9-1ce293485063","resolution":{"observed_at":"2026-05-11T03:15:54.303210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.07922","last_updated":"2026-05-11T02:18:14Z","snapshot_observed_at":"2026-07-06T23:20:15.696237Z","submitted_at":"2026-05-08T15:57:37Z","title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T03:35:50.776347Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.07922"},"observation_digest":"sha256:482b59cba9b5aba92708b357f9f2ee3c72c749a4e7510625f7ef59a127ae28be","observation_id":"793d19d5-c2a0-4e38-a5d5-813223894e8a","resolution":{"observed_at":"2026-05-12T07:16:25.445401Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.10536","last_updated":"2026-05-11T13:19:45Z","snapshot_observed_at":"2026-07-06T23:22:28.318343Z","submitted_at":"2026-05-11T13:19:45Z","title":"HH-SAE: Discovering and Steering Hierarchical Knowledge of Complex Manifolds","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T03:13:53.559096Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.10536"},"observation_digest":"sha256:d496dcfe732b45dd1ba56f48f9e6cade1726a54dc27901d272438f5bdcbf0364","observation_id":"47bdda45-8ee3-4658-8620-e9586d64adf1","resolution":{"observed_at":"2026-05-12T03:16:19.139898Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.28149","last_updated":"2026-08-04T06:19:42Z","snapshot_observed_at":"2026-08-07T19:09:39.391624Z","submitted_at":"2026-05-27T08:31:43Z","title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T14:16:44.232080Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.28149"},"observation_digest":"sha256:374cf2b58d2f41f53dbd0c249f15c096164abf4cfb489de01b9da72418165139","observation_id":"6ec363f6-219e-438e-a626-a06e5bd5fef3","resolution":{"observed_at":"2026-06-29T14:23:30.730641Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-04T05:02:51.380352Z","title":"SAEBench: A comprehensive benchmark for sparse autoencoders in language model interpretability.arXiv preprint arXiv:2503.09532, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28149","last_updated":"2026-08-04T06:19:42Z","snapshot_observed_at":"2026-08-07T19:09:39.391624Z","submitted_at":"2026-05-27T08:31:43Z","title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T05:02:51.380352Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.28149"},"observation_digest":"sha256:33ec2d2667bba7d57ab15038fd812bf11515cfb3d8701570bc56658c2690e4e7","observation_id":"a84392a0-db3d-4784-95e6-381b677202d3","resolution":{"observed_at":"2026-08-04T05:02:51.380352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2606.09653","last_updated":"2026-06-08T15:42:39Z","snapshot_observed_at":"2026-08-07T13:47:05.114508Z","submitted_at":"2026-06-08T15:42:39Z","title":"A Unifying Framework for Concept-Based Representational Similarity","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T17:10:36.855674Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2606.09653"},"observation_digest":"sha256:6cdba0e8ab8ec609f61080aba7965f3ec7bb6c917f12ccb764acd24dc578455b","observation_id":"4091bcb3-81ce-41b3-8e70-c13ca7a8f2df","resolution":{"observed_at":"2026-07-03T00:27:29.987745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2606.22994","last_updated":"2026-06-22T08:12:34Z","snapshot_observed_at":"2026-08-07T09:18:25.424495Z","submitted_at":"2026-06-22T08:12:34Z","title":"Do Sparse Autoencoders Learn Meaningful Concept Hierarchies?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-26T08:46:48.220801Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2606.22994"},"observation_digest":"sha256:61e87fead7e7966079ddac8263fd7d04cb95c7f5fdc66c65abe1fc8b1c0bf639","observation_id":"2e6d96a4-bc02-44ab-b873-98d8a489405f","resolution":{"observed_at":"2026-07-04T10:29:45.396552Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T19:02:29.759226Z","title":"doi:10.48550/arXiv.2503.09532 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17117","last_updated":"2026-07-19T08:07:32Z","snapshot_observed_at":"2026-08-06T16:41:28.805563Z","submitted_at":"2026-07-19T08:07:32Z","title":"Persistent Sparse Autoencoders: Learning Feature Timescales in Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-01T19:02:29.759226Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.17117"},"observation_digest":"sha256:abd04c09530652e456f30380f7129aacef0974317171e59de349c3c555f1723e","observation_id":"054b1070-4bea-4917-a1ad-4f0284afe438","resolution":{"observed_at":"2026-08-01T19:02:29.759226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T18:04:18.007688Z","title":"2503.09532 , archiveprefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17425","last_updated":"2026-07-21T01:20:50Z","snapshot_observed_at":"2026-08-01T21:51:08.284446Z","submitted_at":"2026-07-19T22:19:17Z","title":"Decoder-Preserving Sparse Autoencoders: Which Readouts Survive Sparse Compression?","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-01T18:04:18.007688Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.17425"},"observation_digest":"sha256:63534c5ee0539eed28e73b79c5429677088afae67e377692231532e7b08945ff","observation_id":"c7c1800e-f602-48f6-ab03-8968566103d0","resolution":{"observed_at":"2026-08-01T18:04:18.007688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T10:03:56.301338Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20596","last_updated":"2026-07-22T17:33:16Z","snapshot_observed_at":"2026-08-04T06:09:05.828651Z","submitted_at":"2026-07-22T17:33:16Z","title":"Are Single-Token Sparse Autoencoder Features Causally Necessary? Layer-Depth and SAE-Family Effects","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-01T10:03:56.301338Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.20596"},"observation_digest":"sha256:c38b4b879bb300178fb3654c11375affea3157b78eb1714a2a1ee89e09449db9","observation_id":"b87c7a6c-e658-41fd-ac48-2b0833781747","resolution":{"observed_at":"2026-08-01T10:03:56.301338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T07:58:32.246258Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.27404","last_updated":"2026-07-29T19:20:50Z","snapshot_observed_at":"2026-08-07T16:37:14.209504Z","submitted_at":"2026-07-29T19:20:50Z","title":"ECG-InterpBench: Benchmarking the Interpretability of ECG Foundation Models with Matched-Scale Sparse Autoencoders","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T07:58:32.246258Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.27404"},"observation_digest":"sha256:bc54aad3c0f8a9ab88a849f3cd3a1cd5053a83be72d070deb0f63405f78226a4","observation_id":"959c6a7f-12e1-46eb-8ee7-0a5a4c151dd9","resolution":{"observed_at":"2026-08-01T07:58:32.246258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.09532/citation-record","integrity":"/paper/2503.09532/integrity","json":"/paper/2503.09532/citation-record.json","paper":"/paper/2503.09532"},"outbound":[],"paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","latest_version":4,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T17:09:38.024567Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2503.09532."}