{"as_of":"2026-08-06T11:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f04ee8d5ab6d46f04ca8db1fe6f70d806dc1928c032706b8ab295809eea27778","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T19:39:10.606079Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T03:29:30.170846Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2505.06708","last_updated":"2025-05-10T17:15:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-10T17:15:49Z","title":"Gated Attention for Large Language Models: Non-linearity, Sparsity, and Attention-Sink-Free","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T09:04:34.807225Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2505.06708"},"observation_digest":"sha256:919080bb0732015b9ea087fcaae749e4c2439d52bfaf4601842e5909e2f94304","observation_id":"24b7b178-4ab5-4b8d-aec8-1aa39c8c63bd","resolution":{"observed_at":"2026-05-12T09:04:34.889061Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2509.22321","last_updated":"2026-04-23T14:46:53Z","snapshot_observed_at":"2026-07-06T22:30:55.313733Z","submitted_at":"2025-09-26T13:20:15Z","title":"Distributed Associative Memory via Online Convex Optimization","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-18T13:22:08.134149Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2509.22321"},"observation_digest":"sha256:209c26448acf4febb73c27f7965b97e21ee96c3ab3e666b558e5458fa803c7bd","observation_id":"c7385187-c710-4120-ac71-4c4b10e2919c","resolution":{"observed_at":"2026-05-18T13:22:37.605998Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2510.26692","last_updated":"2025-11-01T12:05:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-30T16:59:43Z","title":"Kimi Linear: An Expressive, Efficient Attention Architecture","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-13T23:49:10.555255Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2510.26692"},"observation_digest":"sha256:b6879bd1756131bb4cefd6c4dcca690792b0e16c0a50565a33aa8db4ff3a2c8b","observation_id":"4e60121f-f687-41c5-9d8f-0ca7cd8355d7","resolution":{"observed_at":"2026-05-13T23:49:11.201312Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-08-03T19:39:10.606079Z","title":"Forgetting transformer: Softmax attention with a forget gate,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.23347","last_updated":"2026-07-07T19:07:00Z","snapshot_observed_at":"2026-08-03T19:39:05.784784Z","submitted_at":"2025-11-28T16:56:18Z","title":"Distributed Dynamic Associative Memory via Online Convex Optimization","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T19:39:10.606079Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2511.23347"},"observation_digest":"sha256:a1cbd21e45a3b29c252d93b4110d95a224acc88804b28bf5e61d1d5c4f420421","observation_id":"2b16414b-ba7c-46af-918c-f79f2bef3ac9","resolution":{"observed_at":"2026-08-03T19:39:10.606079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2512.07805","last_updated":"2026-05-13T18:33:36Z","snapshot_observed_at":"2026-07-06T22:38:08.821098Z","submitted_at":"2025-12-08T18:39:13Z","title":"Group Representational Position Encoding","version":6},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T00:04:13.707931Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2512.07805"},"observation_digest":"sha256:49d5a37d0ccd19b17667ed3a729bc7db2f792d06ec9dcd77f283994f5900b6bf","observation_id":"ceacd1bb-56fd-4f34-b027-260c158f5d0c","resolution":{"observed_at":"2026-05-17T00:08:43.758245Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2605.08587","last_updated":"2026-05-09T01:07:01Z","snapshot_observed_at":"2026-07-06T23:20:47.880233Z","submitted_at":"2026-05-09T01:07:01Z","title":"Kaczmarz Linear Attention","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-12T01:15:58.330766Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2605.08587"},"observation_digest":"sha256:6baaf9a0a2cecbdf31ae3bac8561bf84e3c6dc8e800317416a34900f2e462dc1","observation_id":"89b5f006-9193-4076-ba99-9fc40fdb6625","resolution":{"observed_at":"2026-05-12T08:21:23.272020Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2605.10359","last_updated":"2026-05-11T11:05:04Z","snapshot_observed_at":"2026-07-06T23:22:18.704329Z","submitted_at":"2026-05-11T11:05:04Z","title":"Learning-Based Spectrum Cartography in Low Earth Orbit Satellite Networks: An Overview","version":1},"reference_index":131,"source":"pdf_text","source_observed_at":"2026-05-12T04:52:12.288356Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2605.10359"},"observation_digest":"sha256:738d74e2e22af0f96d04d7612baa8b96b965f6e6a7103538b82c9eab63af0a4e","observation_id":"e1c12bfe-1201-4094-9175-f8460b689baa","resolution":{"observed_at":"2026-05-12T05:51:26.035674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2605.10414","last_updated":"2026-05-11T11:52:06Z","snapshot_observed_at":"2026-07-06T23:22:23.287792Z","submitted_at":"2026-05-11T11:52:06Z","title":"Remember to Forget: Gated Adaptive Positional Encoding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T04:44:31.449357Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2605.10414"},"observation_digest":"sha256:6b17af96f2f8df3907127218a3f7d55fd5a953f17fa62617300a95c9cb1d02b7","observation_id":"0bc33a37-90bd-42d7-b0cd-e65561bca39a","resolution":{"observed_at":"2026-05-12T05:56:41.213773Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2606.02332","last_updated":"2026-06-02T05:51:37Z","snapshot_observed_at":"2026-07-06T23:42:44.661608Z","submitted_at":"2026-06-01T14:42:06Z","title":"Forget Attention: Importance-Aware Attention Is All You Need","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T14:38:40.948032Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2606.02332"},"observation_digest":"sha256:a72215df51fa36e826d577048f0b40a31fe29e76b779714ffa414f4fd0af0e6d","observation_id":"10d227c2-51af-4b41-b279-c32b85f0681d","resolution":{"observed_at":"2026-07-01T23:06:21.017603Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":"2503.02130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-04T03:29:30.170846Z","title":"Forgetting transformer: Softmax attention with a forget gate","venue":null,"work_id":"cc5197b5-1a1c-48c6-a2b9-498e8ac867ce","year":2025},"citing_paper":{"arxiv_id":"2606.19853","last_updated":"2026-06-18T07:01:14Z","snapshot_observed_at":"2026-08-02T08:41:10.786612Z","submitted_at":"2026-06-18T07:01:14Z","title":"Physics-Informed Neural Network with Squeeze-Excitation-like Attention","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-26T18:02:39.721547Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2606.19853"},"observation_digest":"sha256:d8ada6a700d92d457a3d807d3f3fccd5f5d5b9d63f17faa28f5407e3da3e54e8","observation_id":"8e09ecb3-002e-4367-8504-7bd811cc6ca8","resolution":{"observed_at":"2026-07-04T03:29:30.172882Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-07-14T15:30:23.485228Z","title":"arXiv , author =:2503.02130v2 , primaryclass =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09803","last_updated":"2026-07-09T17:52:16Z","snapshot_observed_at":"2026-08-06T03:03:16.058770Z","submitted_at":"2026-07-09T17:52:16Z","title":"Spectral Origins of the Self-Correction Blind Spot in Autoregressive Generation","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-07-14T15:30:23.485228Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2607.09803"},"observation_digest":"sha256:744950d1e2162a437796bff2b60c94a3df15e4ec594920d74dcaabd0ae34c716","observation_id":"c866d79f-3dee-4e92-afdc-fcb071f05096","resolution":{"observed_at":"2026-07-14T15:30:23.485228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-08-01T02:44:01.453174Z","title":"Forgetting transformer: Softmax attention with a forget gate.arXiv preprint arXiv:2503.02130,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25357","last_updated":"2026-07-28T07:04:32Z","snapshot_observed_at":"2026-08-01T20:09:34.932018Z","submitted_at":"2026-07-28T07:04:32Z","title":"Raven: High-Recall Sequence Modeling with Sparse Memory Routing","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T02:44:01.453174Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2607.25357"},"observation_digest":"sha256:73e88212d8e1476ff62af9a3f03c28c54891fbb78632f765014d1360c52f665b","observation_id":"f8b427f3-4cd4-4da9-acd9-720214a8eb06","resolution":{"observed_at":"2026-08-01T02:44:01.453174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.02130/citation-record","integrity":"/paper/2503.02130/integrity","json":"/paper/2503.02130/citation-record.json","paper":"/paper/2503.02130"},"outbound":[],"paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T03:12:34.774417Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2503.02130."}