{"as_of":"2026-08-09T09:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8803b419db0b3cda863943900402bf3f32e623ed3d36d01a2b3ef45bd970589d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":6,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":6,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:12:32.706277Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T14:05:47.060027Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.16332","last_updated":"2025-05-27T10:39:39Z","snapshot_observed_at":"2026-08-01T16:40:47.078977Z","submitted_at":"2024-01-29T17:38:14Z","title":"Tradeoffs Between Alignment and Helpfulness in Language Models with Steering Methods","version":5},"cited_work":{"arxiv_id":"2401.16332","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.16332","snapshot_observed_at":"2026-07-01T14:05:47.060027Z","title":"Tradeoffs between alignment and helpfulness in language models","venue":null,"work_id":"b1f3eeb3-1840-407d-8ae8-742e11fc3153","year":2024},"citing_paper":{"arxiv_id":"2406.11717","last_updated":"2024-10-30T18:57:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T16:36:12Z","title":"Refusal in Language Models Is Mediated by a Single Direction","version":3},"reference_index":199,"source":"arxiv_source","source_observed_at":"2026-05-13T10:47:55.934081Z"},"links":{"cited_paper":"/paper/2401.16332","citing_paper":"/paper/2406.11717"},"observation_digest":"sha256:e04bcdf1fbd229011665996b8f52fc989cc9517fb8aedd0ab6e76cfbb927f280","observation_id":"32edb4db-a48c-4551-85f2-5daa3d6165e0","resolution":{"observed_at":"2026-05-13T10:47:56.155376Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16332","last_updated":"2025-05-27T10:39:39Z","snapshot_observed_at":"2026-08-01T16:40:47.078977Z","submitted_at":"2024-01-29T17:38:14Z","title":"Tradeoffs Between Alignment and Helpfulness in Language Models with Steering Methods","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16332","snapshot_observed_at":"2026-08-06T20:12:32.706277Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.03662","last_updated":"2025-07-04T15:36:58Z","snapshot_observed_at":"2026-08-09T02:24:37.760148Z","submitted_at":"2025-07-04T15:36:58Z","title":"Re-Emergent Misalignment: How Narrow Fine-Tuning Erodes Safety Alignment in LLMs","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T20:12:32.706277Z"},"links":{"cited_paper":"/paper/2401.16332","citing_paper":"/paper/2507.03662"},"observation_digest":"sha256:4fd38b7d66688b397722b85ca702c837363ad1008356c7e89b3bace9b2095b07","observation_id":"197cdb0f-11bf-4a8f-b2f3-cdaac7a31d69","resolution":{"observed_at":"2026-08-06T20:12:32.706277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16332","last_updated":"2025-05-27T10:39:39Z","snapshot_observed_at":"2026-08-01T16:40:47.078977Z","submitted_at":"2024-01-29T17:38:14Z","title":"Tradeoffs Between Alignment and Helpfulness in Language Models with Steering Methods","version":5},"cited_work":{"arxiv_id":"2401.16332","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.16332","snapshot_observed_at":"2026-07-01T14:05:47.060027Z","title":"Tradeoffs between alignment and helpfulness in language models","venue":null,"work_id":"b1f3eeb3-1840-407d-8ae8-742e11fc3153","year":2024},"citing_paper":{"arxiv_id":"2605.14746","last_updated":"2026-07-12T07:31:17Z","snapshot_observed_at":"2026-07-29T20:44:23.574180Z","submitted_at":"2026-05-14T12:13:08Z","title":"Selective Safety Steering via Value-Filtered Decoding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-30T21:03:28.381687Z"},"links":{"cited_paper":"/paper/2401.16332","citing_paper":"/paper/2605.14746"},"observation_digest":"sha256:1205f2768ca63d9d98d16436d827f620b2639369731acafb712c663812cf5f76","observation_id":"268c7f87-d909-4ee9-a98d-bf3522936161","resolution":{"observed_at":"2026-06-30T21:05:03.865596Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16332","last_updated":"2025-05-27T10:39:39Z","snapshot_observed_at":"2026-08-01T16:40:47.078977Z","submitted_at":"2024-01-29T17:38:14Z","title":"Tradeoffs Between Alignment and Helpfulness in Language Models with Steering Methods","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16332","snapshot_observed_at":"2026-07-14T18:59:01.049397Z","title":"Tradeoffs between alignment and helpfulness in language models with steering methods.arXiv preprint arXiv:2401.16332, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.14746","last_updated":"2026-07-12T07:31:17Z","snapshot_observed_at":"2026-07-29T20:44:23.574180Z","submitted_at":"2026-05-14T12:13:08Z","title":"Selective Safety Steering via Value-Filtered Decoding","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-14T18:59:01.049397Z"},"links":{"cited_paper":"/paper/2401.16332","citing_paper":"/paper/2605.14746"},"observation_digest":"sha256:d7f786fd9e7ebb10deccf7ef193617485d99324b37e59914cf86ba656a78d21b","observation_id":"c7f654c8-b951-4a08-a92e-ce281fd065e8","resolution":{"observed_at":"2026-07-14T18:59:01.049397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16332","last_updated":"2025-05-27T10:39:39Z","snapshot_observed_at":"2026-08-01T16:40:47.078977Z","submitted_at":"2024-01-29T17:38:14Z","title":"Tradeoffs Between Alignment and Helpfulness in Language Models with Steering Methods","version":5},"cited_work":{"arxiv_id":"2401.16332","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.16332","snapshot_observed_at":"2026-07-01T14:05:47.060027Z","title":"Tradeoffs between alignment and helpfulness in language models","venue":null,"work_id":"b1f3eeb3-1840-407d-8ae8-742e11fc3153","year":2024},"citing_paper":{"arxiv_id":"2606.05194","last_updated":"2026-07-08T19:54:16Z","snapshot_observed_at":"2026-08-06T06:51:53.907927Z","submitted_at":"2026-05-11T21:09:00Z","title":"Temporal Preference Concepts and their Functions in a Large Language Model","version":1},"reference_index":115,"source":"pdf_text","source_observed_at":"2026-06-30T22:16:47.743387Z"},"links":{"cited_paper":"/paper/2401.16332","citing_paper":"/paper/2606.05194"},"observation_digest":"sha256:e890a4d2cfeb686db1ba97820e1f6b0b57e1b5020a0fbc9b731e29db8e6d242f","observation_id":"1268ed8c-6a33-410f-95e8-367ff9a86958","resolution":{"observed_at":"2026-07-01T14:05:47.061610Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16332","last_updated":"2025-05-27T10:39:39Z","snapshot_observed_at":"2026-08-01T16:40:47.078977Z","submitted_at":"2024-01-29T17:38:14Z","title":"Tradeoffs Between Alignment and Helpfulness in Language Models with Steering Methods","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16332","snapshot_observed_at":"2026-07-12T17:03:44.315006Z","title":"Tradeoffs between alignment and helpfulness in language models with steering methods.arXiv preprint arXiv:2401.16332, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.05194","last_updated":"2026-07-08T19:54:16Z","snapshot_observed_at":"2026-08-06T06:51:53.907927Z","submitted_at":"2026-05-11T21:09:00Z","title":"Temporal Preference Concepts and their Functions in a Large Language Model","version":2},"reference_index":115,"source":"pdf_text","source_observed_at":"2026-07-12T17:03:44.315006Z"},"links":{"cited_paper":"/paper/2401.16332","citing_paper":"/paper/2606.05194"},"observation_digest":"sha256:7a5544f670e1adb8d0e79130c5e3e61c04023f65a146be29aaae9ae213d2b0fc","observation_id":"37a7ca84-8678-42a5-bbd9-b8ee8a9f137d","resolution":{"observed_at":"2026-07-12T17:03:44.315006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2401.16332/citation-record","integrity":"/paper/2401.16332/integrity","json":"/paper/2401.16332/citation-record.json","paper":"/paper/2401.16332"},"outbound":[],"paper":{"arxiv_id":"2401.16332","last_updated":"2025-05-27T10:39:39Z","latest_version":5,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-01T16:40:47.078977Z","submitted_at":"2024-01-29T17:38:14Z","title":"Tradeoffs Between Alignment and Helpfulness in Language Models with Steering Methods"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 6 inbound Pith citation observations for arXiv:2401.16332."}