{"as_of":"2026-08-08T02:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:71ef08b01d53198437b1799eb2eb91b4e724be0ed467d6b05d0415b371f3dc96","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:44:34.878071Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T22:20:52.947124Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"cited_work":{"arxiv_id":"2506.07406","doi":"10.48550/arxiv.2506.07406","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07406","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rob James may refer to:\\n\\nRob James (singer) (","venue":"ArXiv.org","work_id":"c1e2dbba-2c7d-4b7e-b558-995d5446d0cb","year":2025},"citing_paper":{"arxiv_id":"2511.06571","last_updated":"2026-05-08T15:19:37Z","snapshot_observed_at":"2026-07-06T22:35:18.393401Z","submitted_at":"2025-11-09T23:18:36Z","title":"Rep2Text: Decoding Full Text from a Single LLM Token Representation","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T23:15:45.828498Z"},"links":{"cited_paper":"/paper/2506.07406","citing_paper":"/paper/2511.06571"},"observation_digest":"sha256:9d6d13c474b71ab6b550a01276cd0f55cf7c2e51caa59487dd3ee1947c47bafc","observation_id":"fb51bfd6-9f2e-41b7-b344-ebd22b0de95f","resolution":{"observed_at":"2026-07-07T03:18:08.538842Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07406","snapshot_observed_at":"2026-08-02T22:20:52.947124Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.17229","last_updated":"2026-07-17T10:02:50Z","snapshot_observed_at":"2026-08-06T15:44:57.387623Z","submitted_at":"2026-02-19T10:19:04Z","title":"Mechanistic Interpretability of Cognitive Complexity in LLMs via Linear Probing using Bloom's Taxonomy","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-02T22:20:52.947124Z"},"links":{"cited_paper":"/paper/2506.07406","citing_paper":"/paper/2602.17229"},"observation_digest":"sha256:3e4b5163496cb7c6f0d1e38c555ff2770e8f300e71fdb9f28a4b7d37c595e49e","observation_id":"6fc72081-ca3a-4f5c-bb10-c4f8d34b83fb","resolution":{"observed_at":"2026-08-02T22:20:52.947124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"cited_work":{"arxiv_id":"2506.07406","doi":"10.48550/arxiv.2506.07406","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07406","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rob James may refer to:\\n\\nRob James (singer) (","venue":"ArXiv.org","work_id":"c1e2dbba-2c7d-4b7e-b558-995d5446d0cb","year":2025},"citing_paper":{"arxiv_id":"2606.09563","last_updated":"2026-06-08T14:37:46Z","snapshot_observed_at":"2026-08-01T00:37:11.487032Z","submitted_at":"2026-06-08T14:37:46Z","title":"PRISM: Recovering Instruction Sets from Language Model Activations","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-27T16:52:02.948457Z"},"links":{"cited_paper":"/paper/2506.07406","citing_paper":"/paper/2606.09563"},"observation_digest":"sha256:5955dd9912a6fe47bb204dc852cc0fe02784de90539e05fbeeb6b9eaaedd9ece","observation_id":"d512d461-2c4e-43ce-bc65-3833159c71c0","resolution":{"observed_at":"2026-07-07T03:18:08.538842Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.07406/citation-record","integrity":"/paper/2506.07406/integrity","json":"/paper/2506.07406/citation-record.json","paper":"/paper/2506.07406"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1610.01644","last_updated":"2018-11-22T23:40:00Z","snapshot_observed_at":"2026-07-06T05:13:30.860932Z","submitted_at":"2016-10-05T20:59:01Z","title":"Understanding intermediate layers using linear classifier probes","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1610.01644","snapshot_observed_at":"2026-08-07T05:44:34.811248Z","title":"Understanding intermediate layers using linear classifier probes.ArXiv, abs/1610.01644,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.811248Z"},"links":{"cited_paper":"/paper/1610.01644","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:00b68a217951484d752547be19c1945767b345bffab5e6ad4946523beb8dea91","observation_id":"7de31c2a-3748-41af-83e5-97b57e0084c2","resolution":{"observed_at":"2026-08-07T05:44:34.811248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08600","last_updated":"2023-10-04T13:17:38Z","snapshot_observed_at":"2026-07-06T16:19:05.495349Z","submitted_at":"2023-09-15T17:56:55Z","title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08600","snapshot_observed_at":"2026-08-07T05:44:34.824559Z","title":"Sparse autoen- coders find highly interpretable features in language models.arXiv preprint arXiv:2309.08600,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.824559Z"},"links":{"cited_paper":"/paper/2309.08600","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:ee4860ffbe749f585f17ababb5e77bfe0e82586ef29221958ed1be2f3bf27848","observation_id":"884b5900-8b90-48f0-9d6c-44d7f2449931","resolution":{"observed_at":"2026-08-07T05:44:34.824559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15916","last_updated":"2023-10-24T15:17:14Z","snapshot_observed_at":"2026-08-07T03:42:41.198792Z","submitted_at":"2023-10-24T15:17:14Z","title":"In-Context Learning Creates Task Vectors","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15916","snapshot_observed_at":"2026-08-07T05:44:34.837497Z","title":"Roee Hendel, Mor Geva, and Amir Globerson","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.837497Z"},"links":{"cited_paper":"/paper/2310.15916","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:0a50eb400a6308a64a609e523a14b54cad74598dea8a7d9d0ca1b8f35d28d3b0","observation_id":"e292e361-3379-4728-9338-230cabb9bbb4","resolution":{"observed_at":"2026-08-07T05:44:34.837497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17700","last_updated":"2024-08-26T19:26:06Z","snapshot_observed_at":"2026-08-05T04:17:23.403839Z","submitted_at":"2024-02-27T17:25:37Z","title":"RAVEL: Evaluating Interpretability Methods on Disentangling Language Model Representations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.17700","snapshot_observed_at":"2026-08-07T05:44:34.840406Z","title":"Jing Huang, Zhengxuan Wu, Christopher Potts, Mor Geva, and Atticus Geiger","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.840406Z"},"links":{"cited_paper":"/paper/2402.17700","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:d703f66ac1b025b57c9a0561f78cf218aee262772a657794f9db103adc96b68f","observation_id":"2e71ec9b-eeab-4355-9b5f-4fca8bccb89f","resolution":{"observed_at":"2026-08-07T05:44:34.840406Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05147","last_updated":"2024-08-19T07:51:05Z","snapshot_observed_at":"2026-08-04T11:44:14.524984Z","submitted_at":"2024-08-09T16:06:42Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05147","snapshot_observed_at":"2026-08-07T05:44:34.843022Z","title":"Gemma scope: Open sparse autoencoders everywhere all at once on gemma 2.arXiv preprint arXiv:2408.05147,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.843022Z"},"links":{"cited_paper":"/paper/2408.05147","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:2c26217a5aef9283b7464a40d2aa48f8f96fcf5a48703d153ab8b6b9e466ca21","observation_id":"99ca24ad-20bb-4290-94dc-230070a0d686","resolution":{"observed_at":"2026-08-07T05:44:34.843022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:44:35.099334Z","title":"Understanding deep image representations by inverting them.2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp","venue":null,"work_id":"32d77f29-8f35-4291-a090-f614f6144c90","year":2015},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.846181Z"},"links":{"citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:1efab2622cafbb994fdba26265ac4acd6a53dd4db9d67c4c443dc247b0749b5f","observation_id":"96bf3527-f6c5-476c-89b8-a478013b0911","resolution":{"observed_at":"2026-08-07T05:44:35.102039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1602.03616","last_updated":"2016-05-07T06:30:51Z","snapshot_observed_at":"2026-07-06T04:45:49.467632Z","submitted_at":"2016-02-11T05:10:42Z","title":"Multifaceted Feature Visualization: Uncovering the Different Types of Features Learned By Each Neuron in Deep Neural Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1602.03616","snapshot_observed_at":"2026-08-07T05:44:34.851804Z","title":"Multifaceted feature visualization: Uncov- ering the different types of features learned by each neuron in deep neural networks.ArXiv, abs/1602.03616,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.851804Z"},"links":{"cited_paper":"/paper/1602.03616","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:eb6129da7c9602eb5217b6bb2e0fa921977325769882dcf5ae57395db98235f2","observation_id":"7465a8b7-834b-4ccf-a356-ba4da15f6817","resolution":{"observed_at":"2026-08-07T05:44:34.851804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03658","last_updated":"2024-07-17T22:24:27Z","snapshot_observed_at":"2026-07-06T16:43:58.947915Z","submitted_at":"2023-11-07T01:59:11Z","title":"The Linear Representation Hypothesis and the Geometry of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03658","snapshot_observed_at":"2026-08-07T05:44:34.857742Z","title":"The linear representation hypothesis and the geometry of large language models.ArXiv, abs/2311.03658,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.857742Z"},"links":{"cited_paper":"/paper/2311.03658","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:cbed1701bda839b129f9b68605fde8428af3fc07b4675637ee231206eee6e7c6","observation_id":"3f4976f0-6dab-4d11-b9ec-2840a7e3155b","resolution":{"observed_at":"2026-08-07T05:44:34.857742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-07T05:44:34.862975Z","title":"Accessed: 2025-05-15","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.862975Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:c7dae5bf40d95a151d625a22fc732f19895660f067714f467c5ac1dd12e8a948","observation_id":"8a525b8f-a385-4852-9f8b-4e0c8b325bc5","resolution":{"observed_at":"2026-08-07T05:44:34.862975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T05:44:34.865749Z","title":"Llama 2: Open founda- tion and fine-tuned chat models.arXiv preprint arXiv:2307.09288,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.865749Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:691b5b536eb5d480074b52eb3bec19d91af7d116d6469c579affe10a0e16d8b9","observation_id":"5f3ca20a-8fa8-4ff4-817c-02d1df676dd5","resolution":{"observed_at":"2026-08-07T05:44:34.865749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.00593","last_updated":"2022-11-01T17:08:44Z","snapshot_observed_at":"2026-08-05T06:04:21.031867Z","submitted_at":"2022-11-01T17:08:44Z","title":"Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.00593","snapshot_observed_at":"2026-08-07T05:44:34.868186Z","title":"In- terpretability in the wild: a circuit for indirect object identification in gpt-2 small.ArXiv, abs/2211.00593,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.868186Z"},"links":{"cited_paper":"/paper/2211.00593","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:a023ab77b8aedbf0612e76d3b4bd7244b46b1401de8a8da52531ab1b5da16b47","observation_id":"f2b18c6f-035b-4e61-bfb7-531a1aa1829f","resolution":{"observed_at":"2026-08-07T05:44:34.868186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:44:35.091271Z","title":"the indirect object name in the prompt","venue":null,"work_id":"5cadd95e-a948-48e4-859d-8d469e3a7ae5","year":2026},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.870694Z"},"links":{"citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:7d74a1cc61736bbe321d05f23f2532b61e5c431afc991f31a056afa6e376db27","observation_id":"841875b3-b3fd-4295-abc1-cd2f2616e098","resolution":{"observed_at":"2026-08-07T05:44:35.094101Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:44:35.075025Z","title":"Training is performed for 100,000 steps with a batch size of 2048 using the AdamW optimizer","venue":null,"work_id":"f4dece0e-2a47-4d39-b789-c4690b366971","year":2024},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.875703Z"},"links":{"citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:1e9987cdcae5c346260c2c152a153a3a11e319a51c153768a9eda7b0cd70f7d6","observation_id":"b2bb17f5-66e4-40da-9e8f-24620bb3cd3e","resolution":{"observed_at":"2026-08-07T05:44:35.078733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:44:35.066128Z","title":"[City] is a city in the country of","venue":null,"work_id":"f24aabd9-62e8-4f7f-832f-a1bccd419681","year":2023},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.878071Z"},"links":{"citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:73ed1f7cd92ccf51f00b873728c97e57c09f0045f033938a571796dfb313f485","observation_id":"2e9583d3-d141-45cc-a038-8bc837b5cbc8","resolution":{"observed_at":"2026-08-07T05:44:35.069695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:44:35.083468Z","title":"Then, [B] and [A] went to the [PLACE]. [B] gave a [OBJECT] to","venue":null,"work_id":"24ee9b22-47a2-4a43-9eef-8fb24529ab53","year":2023},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.873298Z"},"links":{"citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:aa237c0ba934d8c649785b3c1dc2cb141c824b2c1c60adc7f451c4bb1e382195","observation_id":"767522fa-2384-4c3c-9976-8dde24a8ac93","resolution":{"observed_at":"2026-08-07T05:44:35.086347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04093","last_updated":"2024-06-06T14:10:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-06T14:10:12Z","title":"Scaling and evaluating sparse autoencoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04093","snapshot_observed_at":"2026-08-07T05:44:34.827442Z","title":"Scaling and evaluating sparse autoencoders.arXiv preprint arXiv:2406.04093,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2009,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.827442Z"},"links":{"cited_paper":"/paper/2406.04093","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:4828e0ba8c69350451c36da610874b62793c9248fcbb7ccb455fa941623330f3","observation_id":"46645982-f0df-4415-955e-f13c988cb566","resolution":{"observed_at":"2026-08-07T05:44:34.827442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14082","last_updated":"2024-08-23T23:02:28Z","snapshot_observed_at":"2026-07-06T18:03:38.397804Z","submitted_at":"2024-04-22T11:01:51Z","title":"Mechanistic Interpretability for AI Safety -- A Review","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14082","snapshot_observed_at":"2026-08-07T05:44:34.814735Z","title":"Mechanistic interpretability for ai safety–a review.arXiv preprint arXiv:2404.14082,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.814735Z"},"links":{"cited_paper":"/paper/2404.14082","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:ef2d3a0be3249f92ce02765b572c8a7176ed35675c3fd51d71691ca95d2e4dff","observation_id":"54d2460e-b0c6-4b0b-a2e0-f976a85526ea","resolution":{"observed_at":"2026-08-07T05:44:34.814735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08366","last_updated":"2024-05-20T17:46:14Z","snapshot_observed_at":"2026-08-02T17:29:39.715223Z","submitted_at":"2024-05-14T07:07:13Z","title":"Towards Principled Evaluations of Sparse Autoencoders for Interpretability and Control","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08366","snapshot_observed_at":"2026-08-07T05:44:34.849126Z","title":"Aleksandar Makelov, George Lange, and Neel Nanda","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.849126Z"},"links":{"cited_paper":"/paper/2405.08366","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:d14e9669d7c32aa70201500383ddeccbbb5596747a01d8b3a4d44ab016c27d52","observation_id":"441ed5b6-b3e9-411b-81aa-d2ea158f8d60","resolution":{"observed_at":"2026-08-07T05:44:34.849126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:44:34.854358Z","title":"Alexander Pan, Lijie Chen, and Jacob Steinhardt","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.854358Z"},"links":{"citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:38d36ba3f1b449ad5f8fa5ab4c590c2af5611397e9bc20dfcc9921ac7a07398c","observation_id":"4cc61af8-21d7-4fe6-abc9-ff6dbf5e32d0","resolution":{"observed_at":"2026-08-07T05:44:34.854358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16496","last_updated":"2025-01-27T20:57:18Z","snapshot_observed_at":"2026-07-06T20:27:04.664873Z","submitted_at":"2025-01-27T20:57:18Z","title":"Open Problems in Mechanistic Interpretability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16496","snapshot_observed_at":"2026-08-07T05:44:34.860460Z","title":"Open problems in mechanistic interpretability.arXiv preprint arXiv:2501.16496,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.860460Z"},"links":{"cited_paper":"/paper/2501.16496","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:dec2337a0a901bd376f95991251979a8b797d25046ed1a5e5dd8958b6163cb9c","observation_id":"47020588-307b-4faa-a301-657d98bd9649","resolution":{"observed_at":"2026-08-07T05:44:34.860460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10949","last_updated":"2024-03-26T01:15:09Z","snapshot_observed_at":"2026-07-06T17:45:43.726336Z","submitted_at":"2024-03-16T15:30:34Z","title":"SelfIE: Self-Interpretation of Large Language Model Embeddings","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.10949","snapshot_observed_at":"2026-08-07T05:44:34.821849Z","title":"Selfie: Self-interpretation of large language model embeddings.arXiv preprint arXiv:2403.10949,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.821849Z"},"links":{"cited_paper":"/paper/2403.10949","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:6c101c25e497eb9d071564b6dd1dc6a5f2373dea86de0b3b1cf6059b4811cc24","observation_id":"989dcab7-d626-48ea-9f7b-930feae4c211","resolution":{"observed_at":"2026-08-07T05:44:34.821849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06102","last_updated":"2024-06-06T22:59:58Z","snapshot_observed_at":"2026-08-08T01:23:42.996552Z","submitted_at":"2024-01-11T18:33:48Z","title":"Patchscopes: A Unifying Framework for Inspecting Hidden Representations of Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06102","snapshot_observed_at":"2026-08-07T05:44:34.834001Z","title":"Patch- scopes: A unifying framework for inspecting hidden representations of language mod- els.ArXiv, abs/2401.06102,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.834001Z"},"links":{"cited_paper":"/paper/2401.06102","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:d8dfd147f0b0eba818f7b286291d6750534c46f93da8b1ec4019ef1a323cc80c","observation_id":"15a5bb06-6583-43b2-9dbb-38e809088d8f","resolution":{"observed_at":"2026-08-07T05:44:34.834001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:44:35.107109Z","title":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al","venue":null,"work_id":"80d991dc-90d2-4190-9c2e-698dbbeeac89","year":2023},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.818439Z"},"links":{"citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:9d346bd6f361f032dbcdadf047266691a608539ea5a1905a0506a959c6e6162c","observation_id":"905a8033-9350-41ff-9817-2c76ad7eb2e4","resolution":{"observed_at":"2026-08-07T05:44:35.110407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.14680","last_updated":"2022-10-12T18:01:23Z","snapshot_observed_at":"2026-08-03T05:22:23.916601Z","submitted_at":"2022-03-28T12:26:00Z","title":"Transformer Feed-Forward Layers Build Predictions by Promoting Concepts in the Vocabulary Space","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.14680","snapshot_observed_at":"2026-08-07T05:44:34.830593Z","title":"Transformer feed-forward layers build predictions by promoting concepts in the vocabulary space.arXiv preprint arXiv:2203.14680,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models","version":3},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:34.830593Z"},"links":{"cited_paper":"/paper/2203.14680","citing_paper":"/paper/2506.07406"},"observation_digest":"sha256:a492d769db8008cbeb3b01f9b3f5b3666939f766009b0732309d97955861d73e","observation_id":"2a069249-8b0a-4bd1-96b1-37b4a6ef079c","resolution":{"observed_at":"2026-08-07T05:44:34.830593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.07406","last_updated":"2026-07-06T04:18:12Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T05:32:22.683624Z","submitted_at":"2025-06-09T03:59:28Z","title":"InverseScope: Scalable Activation Inversion for Interpreting Large Language Models"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 3 inbound Pith citation observations for arXiv:2506.07406."}