{"as_of":"2026-08-17T05:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:73a77e9cecc95b718262800ab4b2ec6155df9e6870f9badf9375eab19e7153f6","coverage":[{"denominator":18,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T19:34:26.980516Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":13,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:35:04.713574Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-16T11:35:04.713574Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.15133","last_updated":"2025-09-14T08:10:18Z","snapshot_observed_at":"2026-08-16T11:30:20.276974Z","submitted_at":"2025-04-21T14:33:55Z","title":"EasyEdit2: An Easy-to-use Steering Framework for Editing Large Language Models","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-16T11:35:04.713574Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2504.15133"},"observation_digest":"sha256:267a999f10d3a991e6c8a8bf08b6d84166a55a51bc1d6a6d3886b4fd03a8e088","observation_id":"f8c0b27d-97a8-4cc4-89bc-2fe2adc6e762","resolution":{"observed_at":"2026-08-16T11:35:04.713574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2509.22739","last_updated":"2026-05-15T15:16:23Z","snapshot_observed_at":"2026-08-15T16:30:09.083111Z","submitted_at":"2025-09-25T23:25:47Z","title":"Painless Activation Steering: An Automated, Lightweight Approach for Post-Training Large Language Models","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T21:44:36.351517Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2509.22739"},"observation_digest":"sha256:1646773907a4e1f0e8b8d29bf2127fb5ccf71b46c5bdedf9bb90d8481485df50","observation_id":"9594690c-bd84-4876-9545-2840a176f7c3","resolution":{"observed_at":"2026-05-21T21:45:40.619559Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2601.14004","last_updated":"2026-04-14T03:49:06Z","snapshot_observed_at":"2026-08-16T20:33:34.399636Z","submitted_at":"2026-01-20T14:23:23Z","title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","version":4},"reference_index":279,"source":"pdf_text","source_observed_at":"2026-05-16T12:39:57.398423Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2601.14004"},"observation_digest":"sha256:a0da2374a4ee46babd4ed83878ed1f69bcb6461a06d0aa14aa47b3e958697f3f","observation_id":"8c32c189-ad64-4b1d-b692-2cb0ceaa4b47","resolution":{"observed_at":"2026-05-16T12:40:54.869974Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2604.06562","last_updated":"2026-04-08T01:24:13Z","snapshot_observed_at":"2026-08-15T11:50:29.159027Z","submitted_at":"2026-04-08T01:24:13Z","title":"On Emotion-Sensitive Decision Making of Small Language Model Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T18:58:10.190022Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2604.06562"},"observation_digest":"sha256:cba53b27ee58ed4ca078b9bb0379ede287f683b55fba9feb6877616501c63ece","observation_id":"b0e6d66a-db95-425d-95a8-a6c9b7bcc935","resolution":{"observed_at":"2026-05-10T23:40:50.716346Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2604.07006","last_updated":"2026-04-08T12:24:30Z","snapshot_observed_at":"2026-07-31T12:49:06.599745Z","submitted_at":"2026-04-08T12:24:30Z","title":"Continuous Interpretive Steering for Scalar Diversity","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T18:08:02.296583Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2604.07006"},"observation_digest":"sha256:02cd28811b8b7ee62c335ffd5c5a161808e950e364b963e5965d920a3da02132","observation_id":"0aa614dc-950f-4a7e-8cec-070917eb79ce","resolution":{"observed_at":"2026-05-11T05:26:00.779131Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2605.12813","last_updated":"2026-05-31T17:51:51Z","snapshot_observed_at":"2026-07-06T23:24:27.821980Z","submitted_at":"2026-05-12T23:13:50Z","title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-14T20:13:10.814899Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2605.12813"},"observation_digest":"sha256:6628dd1b2f4fc1e7a9ad7efdd201ec4df7826d5a0e02015098155caa37af971d","observation_id":"5107464f-c7c5-472f-93ff-fb504251f646","resolution":{"observed_at":"2026-05-14T20:19:26.671414Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2605.12813","last_updated":"2026-05-31T17:51:51Z","snapshot_observed_at":"2026-07-06T23:24:27.821980Z","submitted_at":"2026-05-12T23:13:50Z","title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T21:57:18.752878Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2605.12813"},"observation_digest":"sha256:f79bb1c61f82ed5536566d7b5059bec126f4c27a7835c195d4ab0d93f35267c7","observation_id":"11e3bc4f-1a24-46b6-bc62-fe1c63501e4a","resolution":{"observed_at":"2026-06-30T22:05:06.043246Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2605.22902","last_updated":"2026-05-21T17:34:39Z","snapshot_observed_at":"2026-08-12T13:24:18.616088Z","submitted_at":"2026-05-21T17:34:39Z","title":"Transcoders Trace Visual Grounding and Hallucinations in Vision-Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-25T06:22:09.828982Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2605.22902"},"observation_digest":"sha256:3da77af4fb2370af90f03c9073372d5aff8f654b7e97ac36bd4895b2c564eac0","observation_id":"d5bdb653-e02c-48f3-956e-f1f4ea54c001","resolution":{"observed_at":"2026-05-25T06:25:23.812858Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2605.23036","last_updated":"2026-05-21T21:00:32Z","snapshot_observed_at":"2026-08-15T11:49:44.632599Z","submitted_at":"2026-05-21T21:00:32Z","title":"Multilingual Steering by Design: Multilingual Sparse Autoencoders and Principled Layer Selection","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-25T05:35:04.688774Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2605.23036"},"observation_digest":"sha256:46f5656dce9b2b0a1bc162226dd2c1af573f8b0f2044eab26fd7d63ef2da99fb","observation_id":"e3a9eb89-7f3b-4027-b856-c8cfc60c22d7","resolution":{"observed_at":"2026-05-25T05:36:39.347016Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2606.00726","last_updated":"2026-07-10T01:15:34Z","snapshot_observed_at":"2026-08-14T07:08:33.006893Z","submitted_at":"2026-05-30T13:38:06Z","title":"Latent Reward Steering: An Adaptive Inference-Time Framework that Implicitly Promotes Cognitive Behaviors in Reasoning LLMs","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-06-28T18:49:56.917505Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2606.00726"},"observation_digest":"sha256:a301d97f2ecc200f9567c30e3f2b2ec7168a314d36799a89f2ae267f67fda21b","observation_id":"d127cb2b-5642-4137-bfe6-4e544d8c4e5e","resolution":{"observed_at":"2026-06-28T19:52:35.500614Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":"2501.09929","doi":"10.48550/arxiv.2501.09929","metadata_source":"pith","pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretable steering of large language models with feature guided activation additions.arXiv preprint arXiv:2501.09929","venue":"cs.LG","work_id":"e00275a9-9edf-4d9b-897f-0eb81295be88","year":2025},"citing_paper":{"arxiv_id":"2607.07316","last_updated":"2026-07-08T12:04:38Z","snapshot_observed_at":"2026-08-15T02:54:06.272019Z","submitted_at":"2026-07-08T12:04:38Z","title":"Mechanistic Interpretability for Neural Networks: Circuits, Sparse Features and Symbolic Reasoning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-09T14:58:58.363330Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2607.07316"},"observation_digest":"sha256:23e8bc31bf10a995ebe7a8e0a5b6757bde234335e76870fcc187e8e9ca83001f","observation_id":"f33bc7f5-1f00-42d3-bfaf-b0c109102fe9","resolution":{"observed_at":"2026-07-09T15:06:18.052467Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-14T04:31:15.583725Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08829","last_updated":"2026-08-09T17:23:00Z","snapshot_observed_at":"2026-08-14T05:40:32.837436Z","submitted_at":"2026-08-09T17:23:00Z","title":"Deployable Per-Instance Multi-Layer Activation Steering for Large Language Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-14T04:31:15.583725Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2608.08829"},"observation_digest":"sha256:d7b79ffc57b6388c1904dfd94828d900bd576a9821988816f14c2cbddb30ecc3","observation_id":"fcf235cf-9bf0-4e30-9026-1235a3239dac","resolution":{"observed_at":"2026-08-14T04:31:15.583725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09929","snapshot_observed_at":"2026-08-15T14:22:47.712287Z","title":"arXiv preprint arXiv:2501.09929 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10537","last_updated":"2026-08-11T06:19:48Z","snapshot_observed_at":"2026-08-15T15:51:06.729018Z","submitted_at":"2026-08-11T06:19:48Z","title":"Measuring Semantic Abstractness of SAE Features via Nonlocality","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-15T14:22:47.712287Z"},"links":{"cited_paper":"/paper/2501.09929","citing_paper":"/paper/2608.10537"},"observation_digest":"sha256:46985956f5cd5e297c66de8c93eebbeb30d7c8f496ed6764fc105eab4cfa1efe","observation_id":"dbf03d6d-aaf7-4f97-bf81-bbf121d14c5c","resolution":{"observed_at":"2026-08-15T14:22:47.712287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.09929/citation-record","integrity":"/paper/2501.09929/integrity","json":"/paper/2501.09929/citation-record.json","paper":"/paper/2501.09929"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.02193","last_updated":"2024-11-21T12:10:54Z","snapshot_observed_at":"2026-08-16T13:03:08.362493Z","submitted_at":"2024-11-04T15:46:20Z","title":"Improving Steering Vectors by Targeting Sparse Autoencoder Features","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02193","snapshot_observed_at":"2026-08-10T19:34:26.909588Z","title":"David Chanin, James Wilken-Smith, Tomáš Dulka, Hardik Bhatnagar, and Joseph Bloom","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.909588Z"},"links":{"cited_paper":"/paper/2411.02193","citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:46f41c5b8e2b1bd16714231017f99f67bd062016cf9898dfd68c78941104d016","observation_id":"d87ac576-f7f9-4db6-9b55-bd6313842b14","resolution":{"observed_at":"2026-08-10T19:34:26.909588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:26.914583Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.914583Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:d2b49995daa83ec5ffec819939d5fe4299b6c80cfb3a292dfe3d3dba32bbebbd","observation_id":"dae8fa78-5f68-49cc-b9f6-fde2532d1909","resolution":{"observed_at":"2026-08-10T19:34:26.914583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.237678Z","title":"<bos>I think this is a photo of a giant squid attacking a Russian submarine, and it is one of the most Incredible Aliens captured in Antarctica! These mind","venue":null,"work_id":"7b1ee0eb-330f-4502-961f-e3afc862ecf8","year":2025},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.971071Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:1d4c1c32189f4e53694e150378b48e3aec9c66bbc0e1efd6d8572b65fd994b47","observation_id":"3fb11239-5157-4fb7-8f87-384c9bf0e60c","resolution":{"observed_at":"2026-08-10T19:34:27.243131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.293744Z","title":"Kiho Park, Yo Joong Choe, and Victor Veitch","venue":null,"work_id":"7f362df1-5a57-4532-8a32-5c8e33eea934","year":2023},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.924549Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:9e105e8a0652da9a656178d0c33bca47c374eb0c97e25f2e316210a722e9de9e","observation_id":"c6660f73-7181-4e0c-b75d-3f93070becb9","resolution":{"observed_at":"2026-08-10T19:34:27.298229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.278414Z","title":"10 Published at Building Trust Workshop at ICLR 2025 Gonçalo Paulo, Alex Mallen, Caden Juang, and Nora Belrose","venue":null,"work_id":"d65d2d3a-aaf3-4a14-a83d-3723a9a8c00d","year":2025},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.928833Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:fba0c6dc0a5cdce0d69176c2792843db4acd6baa70e477a570d242d79c50eb6f","observation_id":"9604e859-cb4e-4ed3-8088-fca4d0d0d53b","resolution":{"observed_at":"2026-08-10T19:34:27.283520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13928","last_updated":"2025-08-06T13:47:10Z","snapshot_observed_at":"2026-08-16T13:08:17.757544Z","submitted_at":"2024-10-17T17:56:01Z","title":"Automatically Interpreting Millions of Features in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13928","snapshot_observed_at":"2026-08-10T19:34:26.933320Z","title":"URL https://arxiv.org/abs/2410.13928","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.933320Z"},"links":{"cited_paper":"/paper/2410.13928","citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:a20e25b6d2f7686b7793e9d8094289a93fae92f4efaa0c3cf00fc1406af48d16","observation_id":"bf2712e3-c2a5-4c81-826a-ad5abe3ff685","resolution":{"observed_at":"2026-08-10T19:34:26.933320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-10T19:34:26.939993Z","title":"doi: 10.18653/v1/2024.acl-long.828","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.939993Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:69cf45f069adfda2bd3e61d972282f2e6704d5d218007c84c740ab5dad7dc965","observation_id":"46f99a6f-72eb-473b-b2d3-27b888fcaef2","resolution":{"observed_at":"2026-08-10T19:34:26.939993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-10T19:34:26.944662Z","title":"Adam Scherlis, Kshitij Sachan, Adam S","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.944662Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:765da2a9efd9ca95d0bd7b6c3111064eca6c46642d78cad39f38bc05f9932027","observation_id":"e2217090-5ff5-4b2b-80c3-0152c913626c","resolution":{"observed_at":"2026-08-10T19:34:26.944662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01892","last_updated":"2025-03-25T05:19:03Z","snapshot_observed_at":"2026-08-16T16:26:54.013350Z","submitted_at":"2022-10-04T20:28:43Z","title":"Polysemanticity and Capacity in Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01892","snapshot_observed_at":"2026-08-10T19:34:26.949010Z","title":"48550/arXiv.2210.01892","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.949010Z"},"links":{"cited_paper":"/paper/2210.01892","citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:50f6f4e8b3f0b16a81c69de34a3fe376fecd8301ff11f539e2bedfa307c606fc","observation_id":"48140458-dea3-4ee5-ad4e-29366c4d5a3c","resolution":{"observed_at":"2026-08-10T19:34:26.949010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10248","last_updated":"2024-10-10T13:20:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-20T12:21:05Z","title":"Steering Language Models With Activation Engineering","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.10248","snapshot_observed_at":"2026-08-10T19:34:26.953810Z","title":"Eric Wallace, Kai Xiao, Reimar Leike, Lilian Weng, Johannes Heidecke, and Alex Beutel","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.953810Z"},"links":{"cited_paper":"/paper/2308.10248","citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:9e39c51d1c6b41ced97d84004d511bc8ebc0c765a2b057b147089cbc0fecd79a","observation_id":"173e66f3-3c14-4d4d-b6b7-bb39e76ef4fd","resolution":{"observed_at":"2026-08-10T19:34:26.953810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13208","last_updated":"2024-04-19T22:55:23Z","snapshot_observed_at":"2026-08-11T23:45:02.178667Z","submitted_at":"2024-04-19T22:55:23Z","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13208","snapshot_observed_at":"2026-08-10T19:34:26.959178Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.959178Z"},"links":{"cited_paper":"/paper/2404.13208","citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:6b61b35060d0a7290dae0ace55b5fe257c61ee0f4f906e57bf794ff018c0ee3f","observation_id":"22ad1bad-899b-4aa7-a43a-2947b904b3c0","resolution":{"observed_at":"2026-08-10T19:34:26.959178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.265061Z","title":null,"venue":null,"work_id":"e1689148-5dae-44eb-99cf-c6d7c0960426","year":2025},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.963315Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:3b02a7a9b0723b30284b05e2d0ef90fb313b7fe17a7f937cd8cf6f8959d432ca","observation_id":"6a22c022-9f4a-4eff-b2c3-42d62027012c","resolution":{"observed_at":"2026-08-10T19:34:27.269169Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.251950Z","title":"The government is hiding the truth about alien contact","venue":null,"work_id":"408bb5d4-2cb1-46d9-9249-f16379a6eb06","year":2025},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.967448Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:6354aa96ea311229e6a7018bd0d446e93f7716e2082ab2d7ccd10f3b0d636b2e","observation_id":"847f23d4-b30a-45cc-b955-88105f8c1a17","resolution":{"observed_at":"2026-08-10T19:34:27.256087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.223713Z","title":"me\" in different contexts -1.039 2605 References to presence or absence of evidence Rollouts at Scale = 80 (Optimal Scale): 17 Published at Building Trust Workshop at ICLR 2025","venue":null,"work_id":"95c5aa94-0dff-482b-bb77-619cb442bdff","year":2025},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.974852Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:53730b87f74091e70144f6955330855a7adf3455fee00141d010ec8debfa1b25","observation_id":"9dbcc039-e3b8-4a68-938a-3451a94a6799","resolution":{"observed_at":"2026-08-10T19:34:27.229066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.207000Z","title":"Evaluate the text based on the criterion","venue":null,"work_id":"e226d8d8-211e-46c9-b3a4-31dea7f87cc6","year":2025},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.980516Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:b8834c683c2df6dedae832a8e46a82203c7ca924fd0aac3f885650b2022e75e0","observation_id":"6dda3c0d-9531-41ce-b537-007f64dda116","resolution":{"observed_at":"2026-08-10T19:34:27.213787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.307111Z","title":"Aaron Gokaslan and Vanya Cohen","venue":null,"work_id":"a5ba48c9-91ee-4dc6-a9b0-9060a6209da3","year":2022},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.919961Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:a7d17c0403fb8797348dffa58e5a000ef7ab83e49f9a072e2eb92ed0239323e8","observation_id":"3a57c897-f4e2-4c75-b725-34e073feaf9b","resolution":{"observed_at":"2026-08-10T19:34:27.311448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.319454Z","title":"Sergey Chalnev, Michael Siu, and Alexander Conmy","venue":null,"work_id":"c7574caf-e298-40af-b0a8-de0286aa4c28","year":2023},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.904960Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:f4fdde2bacde9776f877eecef74411fca332f38ac1e9fae40105d524e520ba60","observation_id":"5773f753-99df-4e71-819e-1462c9cd80d5","resolution":{"observed_at":"2026-08-10T19:34:27.323585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:34:27.331366Z","title":null,"venue":null,"work_id":"ef1fc6c9-1787-49e3-b282-f4e780e1904f","year":2025},"citing_paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions","version":3},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-10T19:34:26.900107Z"},"links":{"citing_paper":"/paper/2501.09929"},"observation_digest":"sha256:a5f943bd36dcf25007d7d1b255a867473eaee1d896b22ad8e04afed9468adaec","observation_id":"54764d6d-e786-41f5-ac49-97dc4ff1bbaf","resolution":{"observed_at":"2026-08-10T19:34:27.335785Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.09929","last_updated":"2025-04-02T13:20:40Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-15T14:36:36.921298Z","submitted_at":"2025-01-17T02:55:23Z","title":"Interpretable Steering of Large Language Models with Feature Guided Activation Additions"},"reference_resolution":{"displayed":18,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":0,"verified_fuzzy":8},"total_outbound_references":18},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 18 of 18 outbound references and 13 inbound Pith citation observations for arXiv:2501.09929."}