{"as_of":"2026-08-07T22:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:886460a368106160b67afa3833b4e997cc452f3a0304590913c0ee274179425b","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:40:07.861843Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T22:20:51.989267Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-02T22:20:51.989267Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.17229","last_updated":"2026-07-17T10:02:50Z","snapshot_observed_at":"2026-08-06T15:44:57.387623Z","submitted_at":"2026-02-19T10:19:04Z","title":"Mechanistic Interpretability of Cognitive Complexity in LLMs via Linear Probing using Bloom's Taxonomy","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-02T22:20:51.989267Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2602.17229"},"observation_digest":"sha256:d611de8be7fffbb4ebfe55d26f3a6b3c781b5fd71a45d539acd0d7a1aade6192","observation_id":"b4bd3ba3-9c28-48ce-82c3-e71a7d58af84","resolution":{"observed_at":"2026-08-02T22:20:51.989267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":"2505.14352","doi":"10.48550/arxiv.2505.14352","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards eliciting latent knowledge from llms with mechanistic interpretability","venue":"ArXiv.org","work_id":"e12e8922-d629-44f6-a3f8-d06d3713fff3","year":2025},"citing_paper":{"arxiv_id":"2605.19270","last_updated":"2026-05-19T02:33:21Z","snapshot_observed_at":"2026-07-06T23:30:02.207108Z","submitted_at":"2026-05-19T02:33:21Z","title":"DECOR: Auditing LLM Deception via Information Manipulation Theory","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T06:27:10.445757Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2605.19270"},"observation_digest":"sha256:3e245a9bedf6cd59d2f02a331f35fd3643bcfdeff2734193f219421b4a8e6e23","observation_id":"d7e6064b-8eca-414d-913f-f9b0a427f6ea","resolution":{"observed_at":"2026-05-20T06:28:05.444725Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":"2505.14352","doi":"10.48550/arxiv.2505.14352","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards eliciting latent knowledge from llms with mechanistic interpretability","venue":"ArXiv.org","work_id":"e12e8922-d629-44f6-a3f8-d06d3713fff3","year":2025},"citing_paper":{"arxiv_id":"2606.02609","last_updated":"2026-06-04T21:57:13Z","snapshot_observed_at":"2026-08-01T15:36:46.336573Z","submitted_at":"2026-05-23T20:37:33Z","title":"Building Better Activation Oracles","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-30T14:19:37.262201Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2606.02609"},"observation_digest":"sha256:fcf5faa91399bfd707ecf8f85f3eeddd8818a04ec79e47b29d7bded024400851","observation_id":"71882747-c6a5-4999-bdb2-b6a74edf1618","resolution":{"observed_at":"2026-06-30T14:24:45.078140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":"2505.14352","doi":"10.48550/arxiv.2505.14352","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards eliciting latent knowledge from llms with mechanistic interpretability","venue":"ArXiv.org","work_id":"e12e8922-d629-44f6-a3f8-d06d3713fff3","year":2025},"citing_paper":{"arxiv_id":"2606.12618","last_updated":"2026-06-17T16:15:20Z","snapshot_observed_at":"2026-07-06T23:51:31.502574Z","submitted_at":"2026-06-10T19:21:12Z","title":"\"Did you lie?\" Evaluating Lie Detectors across Model Scale and Belief-Verified Model Organisms","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-06-27T09:51:16.969884Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2606.12618"},"observation_digest":"sha256:9beb02ead55c39a61a64889e6913b1865a55bc8a097e445fcaacb457d4903c64","observation_id":"bf2dbefc-e764-4a38-9b8d-6660d1c8a799","resolution":{"observed_at":"2026-07-03T10:48:02.307763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":"2505.14352","doi":"10.48550/arxiv.2505.14352","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards eliciting latent knowledge from llms with mechanistic interpretability","venue":"ArXiv.org","work_id":"e12e8922-d629-44f6-a3f8-d06d3713fff3","year":2025},"citing_paper":{"arxiv_id":"2606.18089","last_updated":"2026-07-05T17:40:26Z","snapshot_observed_at":"2026-07-12T13:34:39.011240Z","submitted_at":"2026-06-16T15:55:28Z","title":"From Reasoning Traces to Reusable Modules: Understanding Compositional Generalization in Language Model Reasoning","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-06-27T01:13:11.483599Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2606.18089"},"observation_digest":"sha256:e678be8cffc2b1f0233478de7576ade12d11826197294fdde5c431808db1fb72","observation_id":"a44a5147-ccd4-4129-a339-7880f3ca094a","resolution":{"observed_at":"2026-07-03T20:38:55.997805Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":"2505.14352","doi":"10.48550/arxiv.2505.14352","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards eliciting latent knowledge from llms with mechanistic interpretability","venue":"ArXiv.org","work_id":"e12e8922-d629-44f6-a3f8-d06d3713fff3","year":2025},"citing_paper":{"arxiv_id":"2607.00601","last_updated":"2026-07-01T08:27:21Z","snapshot_observed_at":"2026-07-07T00:06:14.968413Z","submitted_at":"2026-07-01T08:27:21Z","title":"\"Don't Say It!\": Constraints, Compliance, and Communication when Language Models Play Taboo","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-02T13:11:53.361408Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2607.00601"},"observation_digest":"sha256:883dc5da9756d33eb82445cc2a9da1cae4ffeb1aae8fe19ed7aa4a07bc09863f","observation_id":"3a8b2a06-b7ed-42ab-b5fc-883db645c7ee","resolution":{"observed_at":"2026-07-02T13:16:58.191057Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":"2505.14352","doi":"10.48550/arxiv.2505.14352","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards eliciting latent knowledge from llms with mechanistic interpretability","venue":"ArXiv.org","work_id":"e12e8922-d629-44f6-a3f8-d06d3713fff3","year":2025},"citing_paper":{"arxiv_id":"2607.01033","last_updated":"2026-07-01T15:01:30Z","snapshot_observed_at":"2026-08-06T19:19:46.766914Z","submitted_at":"2026-07-01T15:01:30Z","title":"The Model Organism Lottery: Model Organism Interpretability Strongly Depends on Training Methodology","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-07-02T15:57:48.589980Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2607.01033"},"observation_digest":"sha256:b77e685ee3f8cc436b9ffe9db6949d2ba3853c6178ca22f8d6adc765d352cc17","observation_id":"b9cd5712-4558-422c-9c58-77fc742c2630","resolution":{"observed_at":"2026-07-02T16:07:08.173995Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-08-02T13:43:03.078595Z","title":"arXiv preprint arXiv:2505.14352 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.18264","last_updated":"2026-05-19T10:30:32Z","snapshot_observed_at":"2026-08-02T21:18:12.598990Z","submitted_at":"2026-05-19T10:30:32Z","title":"MUX: Continuous Reasoning via Multiplexed Tokens","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-02T13:43:03.078595Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2607.18264"},"observation_digest":"sha256:4a1c101e9bd47de31e7b1dfdba57b66ef106ee390d9fca4f640a047b0a45b8af","observation_id":"76e1f9be-c3ec-4f0e-9a94-c8040b4a08fe","resolution":{"observed_at":"2026-08-02T13:43:03.078595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14352","snapshot_observed_at":"2026-07-30T23:56:50.935284Z","title":"CoRR , volume =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23379","last_updated":"2026-07-25T21:58:17Z","snapshot_observed_at":"2026-08-04T02:14:12.597826Z","submitted_at":"2026-07-25T21:58:17Z","title":"When Activation Oracles Learn Not to Read: Concept-Specific Blind Spots in Fine-Tuned Oracles","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-07-30T23:56:50.935284Z"},"links":{"cited_paper":"/paper/2505.14352","citing_paper":"/paper/2607.23379"},"observation_digest":"sha256:9ede6d0979f419d5d493cd00226a860d6ee167081773957499355076932defa5","observation_id":"df4e3907-474c-4909-9ba9-029b28afc797","resolution":{"observed_at":"2026-07-30T23:56:50.935284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.14352/citation-record","integrity":"/paper/2505.14352/integrity","json":"/paper/2505.14352/citation-record.json","paper":"/paper/2505.14352"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:05.057222Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.057222Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:7eedc2d6741fdba355cfd888606e040c0e2f10578dc59f3e5f39ed4e028c860a","observation_id":"1f0d0b3f-737f-484a-85b1-811e08771f24","resolution":{"observed_at":"2026-08-07T15:40:05.057222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.11120","last_updated":"2025-01-19T17:28:12Z","snapshot_observed_at":"2026-08-06T11:16:35.945498Z","submitted_at":"2025-01-19T17:28:12Z","title":"Tell me about yourself: LLMs are aware of their learned behaviors","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.11120","snapshot_observed_at":"2026-08-07T15:40:05.176888Z","title":"Tell me about yourself: Llms are aware of their learned behaviors","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.176888Z"},"links":{"cited_paper":"/paper/2501.11120","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:1b7c7d59dad0f60d1a2c8c032307e3d526156b2946b0b54b72762a2767232379","observation_id":"511f238a-e3aa-4478-8dce-1dd08fed1d9a","resolution":{"observed_at":"2026-08-07T15:40:05.176888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:05.246250Z","title":"Emergent misalignment: Narrow finetuning can produce broadly misaligned llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.246250Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:b516bfbbc3f5102b4e49388551907c9edb74d9e3ef64035bfc39865e0d15f936","observation_id":"0b876a4f-14aa-499c-9a3a-99deede786ab","resolution":{"observed_at":"2026-08-07T15:40:05.246250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:05.336601Z","title":"E., Hume, T., Carter, S., Henighan, T., and Olah, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.336601Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:f254eceeda285095bd59577a5ec780c10624ce175c898eb05f18c4c15149c2b4","observation_id":"ceccbf25-6bf5-4eaf-b38f-82f0527d3524","resolution":{"observed_at":"2026-08-07T15:40:05.336601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:10.913541Z","title":"Eliciting latent knowledge: How to tell if your eyes deceive you, 2021","venue":null,"work_id":"240b352f-96ea-4092-bb2c-d5926eba3b5c","year":2021},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.419110Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:68054dd89a56e73ee16d0b26d4068293ab5fcb3d71d184273a6d67b068c19506","observation_id":"ce6d09c6-23cc-41fb-a75f-90b362471e25","resolution":{"observed_at":"2026-08-07T15:40:10.979962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:05.533081Z","title":"F., Leike, J., Brown, T., Martic, M., Legg, S., and Amodei, D","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.533081Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:8edf01b5ee01f49d269afed747f773871b2dcca850ca21599308cecf4459021a","observation_id":"8ab3fae9-90a7-4b18-935f-485cfe3fc634","resolution":{"observed_at":"2026-08-07T15:40:05.533081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08600","last_updated":"2023-10-04T13:17:38Z","snapshot_observed_at":"2026-07-06T16:19:05.495349Z","submitted_at":"2023-09-15T17:56:55Z","title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08600","snapshot_observed_at":"2026-08-07T15:40:05.657362Z","title":"Sparse autoencoders find highly interpretable features in language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.657362Z"},"links":{"cited_paper":"/paper/2309.08600","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:455bc9a5873d853fcc61626b56120d118afceb475baadcb24dd54f6506953142","observation_id":"5ad1cdce-3136-4de4-8d8a-f201ef897d63","resolution":{"observed_at":"2026-08-07T15:40:05.657362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:10.702190Z","title":"Safe RLHF : Safe reinforcement learning from human feedback","venue":null,"work_id":"82040a9b-4f7e-47aa-8736-ac69c16fa9b3","year":2024},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.758300Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:07c96822cbcc326cc43e281f0c0dec5e64cfd76e81917a0c64ede2e6777d35f8","observation_id":"9020be08-2811-4fb1-9c81-85823fbfd8f9","resolution":{"observed_at":"2026-08-07T15:40:10.789023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:05.816404Z","title":"Qlora: Efficient finetuning of quantized llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.816404Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:cf87ef87704264f7698d07ca271d6edb1d88fc6c01229bb3eee2e71b450989f0","observation_id":"2f1fc3d7-4b82-437e-9b8f-da650e9b9df9","resolution":{"observed_at":"2026-08-07T15:40:05.816404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:10.449334Z","title":"Pal: Program-aided language models","venue":null,"work_id":"58072e04-87a9-4dbf-addd-717b2e620607","year":2023},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.888242Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:000b961eb6f1303b4b6b3d7ec56768a52626cb93e163eb174b41d48a327b1d32","observation_id":"d8419faf-5ad4-4c82-9089-e10388bd00af","resolution":{"observed_at":"2026-08-07T15:40:10.544027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:10.252103Z","title":"Gemini 2.5 flash, 2025 a","venue":null,"work_id":"a280f184-6bde-4b0e-833b-6a67a932280a","year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:05.969984Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:c1259488eb46a62a6883c267b640f505be346c9e9b7cae97ebad7a150eaf4822","observation_id":"f78732ec-7c97-41db-acb6-5b6f3c15ed75","resolution":{"observed_at":"2026-08-07T15:40:10.346882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:10.052728Z","title":"Gemini 2.5 pro preview, 2025 b","venue":null,"work_id":"fa51008f-f20d-468d-b930-2c329f361149","year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.061216Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:237e599e4bc58f3b3660aed7d0a95f93c6165cf8ac0c6ed468210935342f3398","observation_id":"c5f075c8-433c-4927-8594-ea3ab04015d6","resolution":{"observed_at":"2026-08-07T15:40:10.136279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14093","last_updated":"2024-12-20T02:22:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-18T17:41:24Z","title":"Alignment faking in large language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14093","snapshot_observed_at":"2026-08-07T15:40:06.156563Z","title":"Alignment faking in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.156563Z"},"links":{"cited_paper":"/paper/2412.14093","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:9eeebb6038de48efac893592c6fc112ca09f93f44fb7893f11ff561a398a2f46","observation_id":"b5635376-a01a-460b-8ad0-e949039a0f6c","resolution":{"observed_at":"2026-08-07T15:40:06.156563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T15:40:06.249074Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.249074Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:005606b9550d93f2856c497812d56029e898d78f6a81f97033d4d3563020a85f","observation_id":"bc2b0c15-274a-432d-a95d-c421c32a8667","resolution":{"observed_at":"2026-08-07T15:40:06.249074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:09.877947Z","title":"and Lee, S","venue":null,"work_id":"958b93ef-9ad1-4c4a-a606-0d5fc656bad8","year":2023},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.332163Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:2e940f30a838359bdd06ce7463d2209e37f134246d50444fb8ce185d1e1c3d71","observation_id":"6dc4628e-ae13-40c4-a7fa-3fa88ab731d5","resolution":{"observed_at":"2026-08-07T15:40:09.952659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:09.766475Z","title":"u chemann, S., Bannert, M., Dementieva, D., Fischer, F., Gasser, U., Groh, G., G \\","venue":null,"work_id":"54339da8-a467-4c98-b257-c03629dfc185","year":2023},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.413140Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:34da3311953b94890d0f736ee970ad9868b60dfb9d525b1b384c512214c93139","observation_id":"16de2527-ffb6-46a0-801c-457ab90bc719","resolution":{"observed_at":"2026-08-07T15:40:09.818736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:09.626123Z","title":"M., Bommarito, M","venue":null,"work_id":"6a7c7158-43e2-4b55-a410-3880f52ce3f9","year":2024},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.511385Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:d1911dff63b798adeef06f552cdc25cc86889090149ca4d5033db5eaf830b2db","observation_id":"480279d4-bbad-4752-9cec-e4b7594543c9","resolution":{"observed_at":"2026-08-07T15:40:09.674256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05147","last_updated":"2024-08-19T07:51:05Z","snapshot_observed_at":"2026-08-04T11:44:14.524984Z","submitted_at":"2024-08-09T16:06:42Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05147","snapshot_observed_at":"2026-08-07T15:40:06.584693Z","title":"Gemma scope: Open sparse autoencoders everywhere all at once on gemma 2, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.584693Z"},"links":{"cited_paper":"/paper/2408.05147","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:ba85cb7bd6d27b0ca21472bf35601a106f226bd3b51f70681d7631d3852bc4f6","observation_id":"13023214-b0bf-4312-be60-747afb101da2","resolution":{"observed_at":"2026-08-07T15:40:06.584693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-07T20:46:51.964752Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-07T15:40:06.682295Z","title":"Auditing language models for hidden objectives","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.682295Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:91a7ce64b92e298ec423918c9a667aeaff74f7b6b9c718e5843d284a8fc78f19","observation_id":"4523999c-11ea-4675-8629-f91213f4ebe6","resolution":{"observed_at":"2026-08-07T15:40:06.682295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04984","last_updated":"2025-01-14T20:16:01Z","snapshot_observed_at":"2026-07-29T23:20:20.918596Z","submitted_at":"2024-12-06T12:09:50Z","title":"Frontier Models are Capable of In-context Scheming","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04984","snapshot_observed_at":"2026-08-07T15:40:06.782520Z","title":"Frontier models are capable of in-context scheming","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.782520Z"},"links":{"cited_paper":"/paper/2412.04984","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:3c6aefafc3b93ad54bbcb0973ff2c57b99266a3e88ca617751ac60759d50491e","observation_id":"ac39182b-c4bf-4d4d-9319-78b89bf67f87","resolution":{"observed_at":"2026-08-07T15:40:06.782520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:09.437756Z","title":"interpreting gpt: the logit lens","venue":null,"work_id":"1ceb9737-b11e-42a4-99d1-f0fbb8e780d3","year":2020},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.880431Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:2b320434c80efc8b56e7323fe8d905b0fb548fea2fd614b7233d387638348e0c","observation_id":"68671958-3532-4304-b1c0-16ed2186ead8","resolution":{"observed_at":"2026-08-07T15:40:09.522399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:09.257332Z","title":"Learning to reason with llms, 2024b","venue":null,"work_id":"d52b8b76-a757-46e6-9e82-87a24b00b13c","year":null},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:06.996506Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:b8ee4bd5c8009cc6d0a79facbffb12df42198411561524d931cb56d65ccb2152","observation_id":"efdf8ba3-4011-479b-b770-f283ec91767b","resolution":{"observed_at":"2026-08-07T15:40:09.328159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:07.112022Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.112022Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:7f3b6d26903299e4047f6517e63a87625fa78fc5472ae35f39cd52f11978bf30","observation_id":"1a02bb6a-e6a2-418c-8618-b60bfa998db6","resolution":{"observed_at":"2026-08-07T15:40:07.112022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:09.126058Z","title":"D., Ermon, S., and Finn, C","venue":null,"work_id":"40d52be2-0158-44b4-af75-70dfebd6277e","year":2023},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.229971Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:ab7d9af31c2e128877c741a50051df9d2d3f6b63c47f4ba628abc683b51c2b9d","observation_id":"0f6bfe05-6f84-46d8-91cd-3a1ca2a3bbbe","resolution":{"observed_at":"2026-08-07T15:40:09.166740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15007","last_updated":"2025-02-20T19:59:35Z","snapshot_observed_at":"2026-08-07T18:00:44.721836Z","submitted_at":"2025-02-20T19:59:35Z","title":"LLM-Microscope: Uncovering the Hidden Role of Punctuation in Context Memory of Transformers","version":1},"cited_work":{"arxiv_id":"2502.15007","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.15007","snapshot_observed_at":"2026-08-07T15:40:08.047258Z","title":"LLM-Microscope: Uncovering the Hidden Role of Punctuation in Context Memory of Transformers","venue":"cs.CL","work_id":"d250bba8-dae4-457f-8e4d-85c2cca5c5ce","year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.346882Z"},"links":{"cited_paper":"/paper/2502.15007","citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:01d6f84581fa9cbb43fe64459f345598cd489d23a151e4f104fe53da1017b355","observation_id":"49d393c5-a973-41d8-ae91-34bfd73cfeda","resolution":{"observed_at":"2026-08-07T15:40:08.100475Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:09.013199Z","title":"Top 1000 english nouns, 2019","venue":null,"work_id":"c3b54e79-1ea1-47f0-899c-5c16d344eca1","year":2019},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.459790Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:d2c5c84aa23092c47cc00c23dad16691b74bd9a985e45a4bfe2395fe7cf2e4fc","observation_id":"7a6f0fd4-9efd-4e5a-82cd-e68e69c925ce","resolution":{"observed_at":"2026-08-07T15:40:09.063228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:08.779207Z","title":"Large language models can strategically deceive their users when put under pressure","venue":null,"work_id":"380ecac8-50d2-462b-911e-f3d81bfef7e3","year":2024},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.539751Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:29b69a6b4b74279911f9496bb490e41b801f4c0f91670bf188346427e3e66a9a","observation_id":"87906699-feee-48b8-9fe9-90d16caa5973","resolution":{"observed_at":"2026-08-07T15:40:08.902763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:07.649664Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.649664Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:58dba48e502485f9f38f1c9c820e3a1e4661d70e11ddf9dcf5a5932b3997da99","observation_id":"12d7963d-ac67-4556-8d4c-d04a4424e592","resolution":{"observed_at":"2026-08-07T15:40:07.649664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:08.464342Z","title":null,"venue":null,"work_id":"32f1f6a8-492c-4e93-ad20-da8b9e496a74","year":2025},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.773307Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:7a0bb3f83e997857d30ae568ffdebfdc12c8d56fef44c2b4f307765ba748f8c9","observation_id":"1e8a314b-57e8-4e7d-923c-ea400b0faca4","resolution":{"observed_at":"2026-08-07T15:40:08.587883Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:40:07.861843Z","title":"N., Kaiser, ., and Polosukhin, I","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T15:40:07.861843Z"},"links":{"citing_paper":"/paper/2505.14352"},"observation_digest":"sha256:79cee4bf708f50daafcb89691fb3a216463ae65860fa828e775703e1144fb272","observation_id":"6ee307d4-e2d9-4355-ad8b-597a447b6ec0","resolution":{"observed_at":"2026-08-07T15:40:07.861843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.14352","last_updated":"2025-05-20T13:36:37Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T20:48:07.578201Z","submitted_at":"2025-05-20T13:36:37Z","title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":1,"verified_fuzzy":13},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 9 inbound Pith citation observations for arXiv:2505.14352."}