{"as_of":"2026-08-07T17:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8cca3ebfeaf8075fc55772cb221571b20e04bf3230269c22bdf01fc87d1e8b1c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:46:59.381114Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T21:17:24.595834Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2410.10781","last_updated":"2025-03-02T14:37:53Z","snapshot_observed_at":"2026-08-02T20:46:05.666977Z","submitted_at":"2024-10-14T17:50:28Z","title":"When Attention Sink Emerges in Language Models: An Empirical View","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-16T17:41:03.674759Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2410.10781"},"observation_digest":"sha256:3d13e95a54b42e9973bae1c23ed20083319a5d84cfe98caf562c4ea40d2e4a71","observation_id":"4c4a8870-83d9-4669-b8e3-f76ba36b05fd","resolution":{"observed_at":"2026-05-16T17:41:03.735502Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-07T11:46:59.381114Z","title":"Theory, analysis, and best practices for sigmoid self-attention.arXiv preprint arXiv:2409.04431, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01562","last_updated":"2025-06-02T11:38:10Z","snapshot_observed_at":"2026-08-07T11:36:03.615544Z","submitted_at":"2025-06-02T11:38:10Z","title":"Unpacking Softmax: How Temperature Drives Representation Collapse, Compression, and Generalization","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:46:59.381114Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2506.01562"},"observation_digest":"sha256:8c969160d2e7c1f104771f12e4fd0c67b3f960e3c812e067237f0ec584941500","observation_id":"c125d649-43ce-4369-910f-d547675c7ca4","resolution":{"observed_at":"2026-08-07T11:46:59.381114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-07T06:10:19.778678Z","title":"Theory, Analysis , and Best Practices for Sigmoid Self - Attention , September 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.778678Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:a9914fd38a4a43179ded0716840a42129ca7cd2c16112ee213c0bf53a7bca420","observation_id":"2737a320-d6ae-4494-b67f-3d9d3bf4cc1d","resolution":{"observed_at":"2026-08-07T06:10:19.778678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-06T20:00:43.782864Z","title":"Theory, analysis, and best practices for sigmoid self-attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04239","last_updated":"2025-07-06T04:15:34Z","snapshot_observed_at":"2026-08-06T19:50:11.930983Z","submitted_at":"2025-07-06T04:15:34Z","title":"Scaling Context Requires Rethinking Attention","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T20:00:43.782864Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2507.04239"},"observation_digest":"sha256:69355717269e61630f8eb6f7dc2e0a8d2cb05e2ad7b069fa8b23f24c737bd305","observation_id":"ffc2af75-1f8e-4578-b29a-6f39eeb387ac","resolution":{"observed_at":"2026-08-06T20:00:43.782864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-06T13:39:10.426250Z","title":"Theory, analysis, and best practices for sigmoid self-attention.arXiv preprint arXiv:2409.04431, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.20453","last_updated":"2025-09-06T03:33:17Z","snapshot_observed_at":"2026-08-06T13:39:06.901666Z","submitted_at":"2025-07-28T01:07:22Z","title":"Your Attention Matters: to Improve Model Robustness to Noise and Spurious Correlations","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:10.426250Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2507.20453"},"observation_digest":"sha256:b72ca73e422f52d079325e3c55a5af40843a24904ddaf6e177df5ea574407402","observation_id":"3b93fb8e-4c74-4dcb-aca6-d41795f98465","resolution":{"observed_at":"2026-08-06T13:39:10.426250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2604.17324","last_updated":"2026-06-07T04:11:52Z","snapshot_observed_at":"2026-08-01T08:49:51.657800Z","submitted_at":"2026-04-19T08:33:48Z","title":"Capacity-Controlled Global Attention for Graph Transformers","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T06:27:24.557866Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2604.17324"},"observation_digest":"sha256:f6258b8ecbb2410ee329e51dab92c166ef1073cf05933b84070f5cef55e513c7","observation_id":"e38e9aa0-9fda-4b27-abae-cde295d70edc","resolution":{"observed_at":"2026-05-10T06:31:30.767360Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.06501","last_updated":"2026-05-19T06:54:59Z","snapshot_observed_at":"2026-08-02T05:36:24.593005Z","submitted_at":"2026-05-07T16:18:55Z","title":"Cubit: Token Mixer with Kernel Ridge Regression","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-08T12:38:19.925573Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.06501"},"observation_digest":"sha256:58f3ff3b8b9d9f102eb62bb98b03b6adc9597f3616195d6f8e4fdc845b0721e6","observation_id":"f86867b3-92f9-4a80-a47b-29f35ab9a31c","resolution":{"observed_at":"2026-05-11T19:06:10.787346Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.06501","last_updated":"2026-05-19T06:54:59Z","snapshot_observed_at":"2026-08-02T05:36:24.593005Z","submitted_at":"2026-05-07T16:18:55Z","title":"Cubit: Token Mixer with Kernel Ridge Regression","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-20T22:34:36.108826Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.06501"},"observation_digest":"sha256:7fbdd4ba54dd65a95652e4b0567002cf4413ad0f2cae977c9fa22ddca7f371f4","observation_id":"b7950fc4-5026-4447-862f-bced4a513992","resolution":{"observed_at":"2026-05-20T22:39:11.009258Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.06611","last_updated":"2026-05-07T17:28:55Z","snapshot_observed_at":"2026-08-03T12:02:38.374401Z","submitted_at":"2026-05-07T17:28:55Z","title":"The Structural Origin of Attention Sink: Variance Discrepancy, Super Neurons, and Dimension Disparity","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-08T12:11:04.146711Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.06611"},"observation_digest":"sha256:3c1c7b61d710ec4f72ec5a6b5137d470288d24034a25f219cf6da64e3a74b96f","observation_id":"a8f9fc8d-b472-4959-8fea-6820f08824cb","resolution":{"observed_at":"2026-05-11T19:21:08.516991Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.08504","last_updated":"2026-05-12T18:33:07Z","snapshot_observed_at":"2026-07-06T23:20:43.010758Z","submitted_at":"2026-05-08T21:37:27Z","title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T02:29:20.796512Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.08504"},"observation_digest":"sha256:788cf90820c8894509a531a28da3947e80682358468996246acf845bb8df83d6","observation_id":"b1fd8067-78c2-4245-94e1-f0862b4855c7","resolution":{"observed_at":"2026-05-12T07:36:40.797845Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.08504","last_updated":"2026-05-12T18:33:07Z","snapshot_observed_at":"2026-07-06T23:20:43.010758Z","submitted_at":"2026-05-08T21:37:27Z","title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-14T21:03:25.624300Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.08504"},"observation_digest":"sha256:ede6d8362d13b227cb1c5b84d596b9e3145990bcadbe64e611337935240f57a6","observation_id":"2ae133e3-518a-4b55-8959-1c2719b83d1e","resolution":{"observed_at":"2026-05-14T21:19:29.111658Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.10123","last_updated":"2026-07-28T06:51:39Z","snapshot_observed_at":"2026-08-07T09:22:28.887723Z","submitted_at":"2026-05-11T07:38:52Z","title":"Complex-Valued Phase-Coherent Transformer","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-12T03:34:20.893124Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.10123"},"observation_digest":"sha256:0e5c80d277c6d3bb4777c6ea50f34b922512a8749a2bb704b02f8da053874502","observation_id":"d20d30e7-4356-401e-b839-9e29cde7c142","resolution":{"observed_at":"2026-05-12T07:16:28.147412Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-02T14:31:47.302932Z","title":"Code:https://github.com/apple/ml-sigmoid-attention • Saratchandran et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.10123","last_updated":"2026-07-28T06:51:39Z","snapshot_observed_at":"2026-08-07T09:22:28.887723Z","submitted_at":"2026-05-11T07:38:52Z","title":"Complex-Valued Phase-Coherent Transformer","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T14:31:47.302932Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.10123"},"observation_digest":"sha256:3523e09c70f8245f1fc9d296477bef7b2b806cefaa5fc339f4cd4ff51466083f","observation_id":"d057774a-4e16-460f-8f77-6f7cac95eb0f","resolution":{"observed_at":"2026-08-02T14:31:47.302932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.20798","last_updated":"2026-05-20T06:43:34Z","snapshot_observed_at":"2026-08-02T17:01:43.729235Z","submitted_at":"2026-05-20T06:43:34Z","title":"Most Transformer Modifications Still Do Not Transfer at 1-3B: A 2020-2026 Update to Narang et al. (2021) with Downstream Evaluation and a Noise Floor","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-21T06:15:47.451870Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.20798"},"observation_digest":"sha256:3ce94a3acf1981ec541f7ff3dcbd16fba62770ab113876c7e7ea9116bc5d8a5b","observation_id":"16d1e8f8-de0a-4812-82f2-b7527e32b3e9","resolution":{"observed_at":"2026-05-21T06:19:42.072047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.21070","last_updated":"2026-05-20T11:56:15Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T11:56:15Z","title":"Towards Understanding Self-Pretraining for Sequence Classification","version":1},"reference_index":178,"source":"arxiv_source","source_observed_at":"2026-05-21T05:29:58.809024Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.21070"},"observation_digest":"sha256:54c911a3280359daab353e324ffedc2f3023992256af4d49a006b6362fc7e826","observation_id":"830e1fcf-1c26-46f4-8683-29111e1fc631","resolution":{"observed_at":"2026-05-21T05:33:58.908338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2606.02332","last_updated":"2026-06-02T05:51:37Z","snapshot_observed_at":"2026-08-06T14:24:37.941578Z","submitted_at":"2026-06-01T14:42:06Z","title":"Forget Attention: Importance-Aware Attention Is All You Need","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T14:38:40.948032Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2606.02332"},"observation_digest":"sha256:e369487973ed50c1268cb1eac346d3286b856030dd23381ff866a3b58102f959","observation_id":"9c37073b-4edb-41c7-8e1f-4e2442124a49","resolution":{"observed_at":"2026-07-01T23:06:21.024853Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2606.08105","last_updated":"2026-06-06T11:10:54Z","snapshot_observed_at":"2026-08-06T19:43:58.625241Z","submitted_at":"2026-06-06T11:10:54Z","title":"A Unifying View of Attention Sinks: Two Algorithms, Two Solutions","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T19:51:41.329591Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2606.08105"},"observation_digest":"sha256:0dc9d630d45335b5503b7ed23a78e31c840c3eaef83e5c70421448c473a8f5b7","observation_id":"a2f19bb8-9c0a-4305-87cd-56ab4aabf127","resolution":{"observed_at":"2026-07-02T21:17:24.597286Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-11T20:06:55.102341Z","title":"Theory, analysis, and best practices for sigmoid self-attention.arXiv preprint arXiv:2409.04431,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04319","last_updated":"2026-07-05T14:08:59Z","snapshot_observed_at":"2026-08-06T22:32:49.321254Z","submitted_at":"2026-07-05T14:08:59Z","title":"Legible-by-Construction: Attention and End-to-End Transformers","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-11T20:06:55.102341Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2607.04319"},"observation_digest":"sha256:2383716948d1a386a0ab00d9d4bf976a1e0a9187bf7d4065ee2c3844368e2c31","observation_id":"b4ea3e01-b25e-48db-861b-8dc620cab92f","resolution":{"observed_at":"2026-07-11T20:06:55.102341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2409.04431/citation-record","integrity":"/paper/2409.04431/integrity","json":"/paper/2409.04431/citation-record.json","paper":"/paper/2409.04431"},"outbound":[],"paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 18 inbound Pith citation observations for arXiv:2409.04431."}