{"as_of":"2026-08-16T04:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5ce2252199409f05c797733a3cd925ec06202b2e679f89bc657f112aab8dadfd","coverage":[{"denominator":56,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T21:33:59.310259Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.07963/citation-record","integrity":"/paper/2509.07963/integrity","json":"/paper/2509.07963/citation-record.json","paper":"/paper/2509.07963"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:33:54.845746Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:54.845746Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:1ac0c4fe13ad2b330babae786058933d13565eee316d6ff60067f8f1f3d260d8","observation_id":"5854e48f-d28f-4313-a9c7-d527029ef4f4","resolution":{"observed_at":"2026-08-04T21:33:54.845746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-08-15T06:28:09.529747Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-04T21:33:54.962502Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:54.962502Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:625593a49d03abf5bb96c272f6b25a6af20c5d5b515e8f7d893f9220d52e2946","observation_id":"8956a352-369f-4f91-a430-1d8cdc3d8532","resolution":{"observed_at":"2026-08-04T21:33:54.962502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16153","last_updated":"2024-07-23T03:40:24Z","snapshot_observed_at":"2026-08-16T01:56:56.989657Z","submitted_at":"2024-07-23T03:40:24Z","title":"On the Benefits of Rank in Attention Layers","version":1},"cited_work":{"arxiv_id":"2407.16153","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.16153","snapshot_observed_at":"2026-08-04T21:34:00.697919Z","title":"On the Benefits of Rank in Attention Layers","venue":"cs.LG","work_id":"3a005d13-76f4-4a9f-ad5e-4945a66aa322","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.059514Z"},"links":{"cited_paper":"/paper/2407.16153","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:4d9c7bb984b0dac568360ffeb89bdc0cd15782b0115737a4ab3060da494490ea","observation_id":"6784ab5d-1b74-42ca-a359-03d658962c58","resolution":{"observed_at":"2026-08-04T21:34:00.775847Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07815","last_updated":"2024-11-04T17:42:45Z","snapshot_observed_at":"2026-08-14T11:34:28.022352Z","submitted_at":"2024-03-12T16:53:54Z","title":"Chronos: Learning the Language of Time Series","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07815","snapshot_observed_at":"2026-08-04T21:33:55.131488Z","title":"F., Stella, L., Turkmen, C., Zhang, X., Mercado, P., Shen, H., Shchur, O., Rangapuram, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.131488Z"},"links":{"cited_paper":"/paper/2403.07815","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:2b71f7adb8cfe1ab88c6798eeb227117a58093e400286dfbbe40175da2e02531","observation_id":"f5772a6b-cb04-48f3-b3f9-150f1c575c1f","resolution":{"observed_at":"2026-08-04T21:33:55.131488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18668","last_updated":"2025-03-07T18:57:52Z","snapshot_observed_at":"2026-08-13T04:07:54.136779Z","submitted_at":"2024-02-28T19:28:27Z","title":"Simple linear attention language models balance the recall-throughput tradeoff","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18668","snapshot_observed_at":"2026-08-04T21:33:55.227016Z","title":"Simple linear attention language models balance the recall-throughput tradeoff, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.227016Z"},"links":{"cited_paper":"/paper/2402.18668","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:cd798c98ee66582e21b82ef57fe3719b2a8eec7215f543b1b1ba61449f593794","observation_id":"60014f2b-f5af-4e5c-88ee-f179655adeaa","resolution":{"observed_at":"2026-08-04T21:33:55.227016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1409.0473","last_updated":"2016-05-19T21:53:22Z","snapshot_observed_at":"2026-08-12T12:07:33.202888Z","submitted_at":"2014-09-01T16:33:02Z","title":"Neural Machine Translation by Jointly Learning to Align and Translate","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.0473","snapshot_observed_at":"2026-08-04T21:33:55.310253Z","title":"Neural machine translation by jointly learning to align and translate, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.310253Z"},"links":{"cited_paper":"/paper/1409.0473","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:55d12cc0367b20e11be4fdc9e328bca33d0f18a395ad75ebe2a8b2c593e5e1a6","observation_id":"05fabbd0-b7f5-4520-bf3f-6d0da13df056","resolution":{"observed_at":"2026-08-04T21:33:55.310253Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00663","last_updated":"2024-12-31T22:32:03Z","snapshot_observed_at":"2026-08-07T09:00:33.558814Z","submitted_at":"2024-12-31T22:32:03Z","title":"Titans: Learning to Memorize at Test Time","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00663","snapshot_observed_at":"2026-08-04T21:33:55.477691Z","title":"Titans: Learning to memorize at test time, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.477691Z"},"links":{"cited_paper":"/paper/2501.00663","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:92b71caff0ba0b4db01c7d2c35f56c879fc0cfabab542fc2919d3a4fbe1a78ac","observation_id":"73954db0-52d3-4f4c-8e45-7940f4650f6c","resolution":{"observed_at":"2026-08-04T21:33:55.477691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-08-04T21:33:55.595248Z","title":"E., and Cohan, A","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.595248Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:510f362e16c9a00cee216eab450d5ac7c5a3cb715ec98c5c90b53699316af816","observation_id":"00c0aeae-9924-4893-8f68-9b83c305ef0d","resolution":{"observed_at":"2026-08-04T21:33:55.595248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:03.478842Z","title":"S., Reddi, S., and Kumar, S","venue":null,"work_id":"e860d302-e1f7-47c0-ae42-009e277ca0d7","year":2020},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.656699Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:237b22f3cd41cfc9ae6306da47af6606f81f5c7d4b2534c26946c6e3b5d0a2f6","observation_id":"44ab0c7a-f2d4-4491-90ab-0ac5999978e5","resolution":{"observed_at":"2026-08-04T21:34:03.554091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-04T21:33:55.749150Z","title":"A., Adeli, E., Altman, R., Arora, S., von Arx, S., Bernstein, M","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.749150Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:0ad0a9742fe6cd5a1bca5603ff93df4cfa3f2c923f9cdf100c14305149b3b5d5","observation_id":"0d2cb7e8-cf42-4a65-9013-e909c47597cf","resolution":{"observed_at":"2026-08-04T21:33:55.749150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:03.240474Z","title":"Scatterbrain: Unifying sparse and low-rank attention","venue":null,"work_id":"f4462255-4983-4d57-b381-bc491e721bfe","year":2021},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.857746Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:0fe233324228c09f56732b4e3b5699db7c7dc3cb96c51dc74e4f6a8089686b91","observation_id":"6d2ec2ef-be97-485b-99e2-74039c132533","resolution":{"observed_at":"2026-08-04T21:34:03.385919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.10509","last_updated":"2019-04-23T19:29:47Z","snapshot_observed_at":"2026-08-09T19:46:04.857927Z","submitted_at":"2019-04-23T19:29:47Z","title":"Generating Long Sequences with Sparse Transformers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.10509","snapshot_observed_at":"2026-08-04T21:33:55.921078Z","title":"Generating long sequences with sparse transformers, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:55.921078Z"},"links":{"cited_paper":"/paper/1904.10509","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:21ed14de3fdf3738e2b5299963ca86571b3735ed8a2ee7923858cdefae818a04","observation_id":"d9902483-24b1-435e-b2e2-077c5011259f","resolution":{"observed_at":"2026-08-04T21:33:55.921078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.14794","last_updated":"2022-11-19T12:45:21Z","snapshot_observed_at":"2026-08-12T04:58:34.201421Z","submitted_at":"2020-09-30T17:09:09Z","title":"Rethinking Attention with Performers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.14794","snapshot_observed_at":"2026-08-04T21:33:56.026504Z","title":"Rethinking attention with performers, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.026504Z"},"links":{"cited_paper":"/paper/2009.14794","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:8a3665425e57743d00342d1192e6028d87738b38e41eb8b9f1b6a38feb93fdb5","observation_id":"746228e5-30e1-46e2-847e-e52739af8785","resolution":{"observed_at":"2026-08-04T21:33:56.026504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21060","last_updated":"2024-05-31T17:50:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:50:01Z","title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21060","snapshot_observed_at":"2026-08-04T21:33:56.121764Z","title":"and Gu, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.121764Z"},"links":{"cited_paper":"/paper/2405.21060","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:5af558402163d68abf9b4f627333a9b7f62ff78483aea152eccbea16cc753db5","observation_id":"f0979cbd-5919-4146-988e-676d9fd70eaf","resolution":{"observed_at":"2026-08-04T21:33:56.121764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1903.05895","last_updated":"2020-12-29T03:32:01Z","snapshot_observed_at":"2026-08-14T17:02:25.688285Z","submitted_at":"2019-03-14T10:20:38Z","title":"Learning Fast Algorithms for Linear Transforms Using Butterfly Factorizations","version":2},"cited_work":{"arxiv_id":"1903.05895","doi":null,"metadata_source":"pith","pith_arxiv_id":"1903.05895","snapshot_observed_at":"2026-08-04T21:34:00.459525Z","title":"Learning Fast Algorithms for Linear Transforms Using Butterfly Factorizations","venue":"cs.LG","work_id":"9467655a-3a1b-4fa7-8ecc-4ff9cfa1c647","year":2019},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.219781Z"},"links":{"cited_paper":"/paper/1903.05895","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:2f81c8cd28b6e6ef6daa407b5499419c6e1789e18a41a0956e9414c9e5737e54","observation_id":"3cb542d8-6e13-4161-b270-47667b0e4c3f","resolution":{"observed_at":"2026-08-04T21:34:00.522778Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.00595","last_updated":"2022-04-01T17:37:29Z","snapshot_observed_at":"2026-08-13T16:10:04.604750Z","submitted_at":"2022-04-01T17:37:29Z","title":"Monarch: Expressive Structured Matrices for Efficient and Accurate Training","version":1},"cited_work":{"arxiv_id":"2204.00595","doi":null,"metadata_source":"pith","pith_arxiv_id":"2204.00595","snapshot_observed_at":"2026-08-04T21:34:00.329872Z","title":"Monarch: Expressive Structured Matrices for Efficient and Accurate Training","venue":"cs.LG","work_id":"2ca8efb6-ac6c-4660-a8dd-db3494699bf1","year":2022},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.286090Z"},"links":{"cited_paper":"/paper/2204.00595","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:dbda040fa813ffc76700060425edb280ffd5e760a472d30c334d062bedaae6b2","observation_id":"c37a0a78-bc55-475f-aec6-51f91c26da22","resolution":{"observed_at":"2026-08-04T21:34:00.392545Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.14052","last_updated":"2023-04-29T03:18:40Z","snapshot_observed_at":"2026-08-13T13:13:21.847890Z","submitted_at":"2022-12-28T17:56:03Z","title":"Hungry Hungry Hippos: Towards Language Modeling with State Space Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.14052","snapshot_observed_at":"2026-08-04T21:33:56.414526Z","title":"Y., Dao, T., Saab, K","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.414526Z"},"links":{"cited_paper":"/paper/2212.14052","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:32d4f8e427e17ff42387686bf98833a32c6a6793057d78df78cb77982b463d17","observation_id":"40dd0845-c4a6-454c-96de-f3dd54359bf1","resolution":{"observed_at":"2026-08-04T21:33:56.414526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2208.01066","last_updated":"2023-08-11T19:27:58Z","snapshot_observed_at":"2026-08-14T13:20:01.387334Z","submitted_at":"2022-08-01T18:01:40Z","title":"What Can Transformers Learn In-Context? A Case Study of Simple Function Classes","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.01066","snapshot_observed_at":"2026-08-04T21:33:56.521170Z","title":"S., and Valiant, G","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.521170Z"},"links":{"cited_paper":"/paper/2208.01066","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:3e4da745a6de250c2077f3332d013957ab7d55dbe95e98e482a1a49327dc3550","observation_id":"5decba14-dac5-43e8-8709-73a644e97584","resolution":{"observed_at":"2026-08-04T21:33:56.521170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-04T21:33:56.581715Z","title":"and Dao, T","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.581715Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:fd6e04afeea4194768675e956890a030e06492a64d7d3676c3eb7cce9ecf27df","observation_id":"62a830ed-fc26-4d4b-955a-fbe69e0f2446","resolution":{"observed_at":"2026-08-04T21:33:56.581715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.00396","last_updated":"2022-08-05T17:54:38Z","snapshot_observed_at":"2026-08-14T01:02:41.198730Z","submitted_at":"2021-10-31T03:32:18Z","title":"Efficiently Modeling Long Sequences with Structured State Spaces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.00396","snapshot_observed_at":"2026-08-04T21:33:56.648509Z","title":"Efficiently modeling long sequences with structured state spaces, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.648509Z"},"links":{"cited_paper":"/paper/2111.00396","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:2e0bf8a78f547dc4cbd76f69b42e72f336d09ce72cb6c6babdd674ec97be0055","observation_id":"e9bc9332-5f03-4fd9-b5e6-fdca43ca168c","resolution":{"observed_at":"2026-08-04T21:33:56.648509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:02.889964Z","title":"SLT rain: a sparse plus low rank approach for parameter and memory efficient pretraining","venue":null,"work_id":"448d71d0-8a07-4c31-9717-052fa31161e1","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.741196Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:b44bb661d03819be843fe61644416d78425c7e37b3e6bf23e19407a4631cc615","observation_id":"4decd882-605a-4dbc-8fb2-718693f1c18a","resolution":{"observed_at":"2026-08-04T21:34:03.066582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:02.672480Z","title":"Global context vision transformers","venue":null,"work_id":"c381f220-8de3-41b9-964d-358bf84493c2","year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.818100Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:5f552eac92ed15408eaabdc4b6081acb09157529199c7e82f5885c5ab41c8efc","observation_id":"b5047c79-aa07-4428-a034-848227dae41d","resolution":{"observed_at":"2026-08-04T21:34:02.760121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-11T08:20:29.798517Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-04T21:33:56.907525Z","title":"J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., and Chen, W","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:56.907525Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:316784138c7d3a53de9967dbceb57440d34442ffd1ec6ccd8e32a6e906596941","observation_id":"d46ffa39-a945-4d74-8fdd-32a069223f72","resolution":{"observed_at":"2026-08-04T21:33:56.907525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12351","last_updated":"2024-02-23T19:22:58Z","snapshot_observed_at":"2026-08-15T12:29:09.416930Z","submitted_at":"2023-11-21T04:59:17Z","title":"Advancing Transformer Architecture in Long-Context Large Language Models: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12351","snapshot_observed_at":"2026-08-04T21:33:57.004385Z","title":"Advancing transformer architecture in long-context large language models: A comprehensive survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.004385Z"},"links":{"cited_paper":"/paper/2311.12351","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:61ee59ac61ebcfe32aec1ac0b95b0c9500e6f8a2cb029a97e512d5d098f650f5","observation_id":"944e5b8a-1d67-4ac6-888c-1474a6ddf9e1","resolution":{"observed_at":"2026-08-04T21:33:57.004385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.09941","last_updated":"2024-07-13T16:34:18Z","snapshot_observed_at":"2026-08-12T23:22:48.674322Z","submitted_at":"2024-07-13T16:34:18Z","title":"Hydra: Bidirectional State Space Models Through Generalized Matrix Mixers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.09941","snapshot_observed_at":"2026-08-04T21:33:57.102946Z","title":"Hydra: Bidirectional state space models through generalized matrix mixers, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.102946Z"},"links":{"cited_paper":"/paper/2407.09941","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:ad506b55bb0d07e942847cb10ec5d63cb45e592ef505bb8ba12bbb5919df465c","observation_id":"e85eeb33-3383-455c-9ce0-0c9d6c5ea362","resolution":{"observed_at":"2026-08-04T21:33:57.102946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:02.537394Z","title":"M., and Malach, E","venue":null,"work_id":"3b0115a7-8dcf-4dd0-b0c2-c8bbeb19ce40","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.184572Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:671f28c5c7010396ca1778e2a039540c361474ea25245dc00aef9b93060b63ee","observation_id":"629fc658-6842-43a5-9268-48bfa93d5168","resolution":{"observed_at":"2026-08-04T21:34:02.601855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.16236","last_updated":"2020-08-31T11:09:32Z","snapshot_observed_at":"2026-08-06T23:24:28.908251Z","submitted_at":"2020-06-29T17:55:38Z","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.16236","snapshot_observed_at":"2026-08-04T21:33:57.271806Z","title":"Transformers are rnns: Fast autoregressive transformers with linear attention, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.271806Z"},"links":{"cited_paper":"/paper/2006.16236","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:453c2974392ca52f5385c2a3abca3f456099f4a4b7c92c8dfecc190b8e075a3c","observation_id":"dc923eb5-9062-4833-a75c-873c904a5a86","resolution":{"observed_at":"2026-08-04T21:33:57.271806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05173","last_updated":"2024-05-28T13:41:26Z","snapshot_observed_at":"2026-08-15T13:35:27.237292Z","submitted_at":"2024-02-07T19:00:01Z","title":"Towards Understanding Inductive Bias in Transformers: A View From Infinity","version":2},"cited_work":{"arxiv_id":"2402.05173","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.05173","snapshot_observed_at":"2026-08-04T21:34:00.059715Z","title":"Towards Understanding Inductive Bias in Transformers: A View From Infinity","venue":"cs.LG","work_id":"d8979aba-1c09-4604-86bc-343d3ee9d46a","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.360599Z"},"links":{"cited_paper":"/paper/2402.05173","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:e0264d6148212d59fa920421e0255a3058ccb37c08451f49569b33c69e2564d5","observation_id":"1af3df4b-f13c-49a3-92fa-29353f6f18d7","resolution":{"observed_at":"2026-08-04T21:34:00.099758Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.05118","last_updated":"2018-07-13T15:00:17Z","snapshot_observed_at":"2026-08-14T18:52:15.765738Z","submitted_at":"2018-07-13T15:00:17Z","title":"Tune: A Research Platform for Distributed Model Selection and Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.05118","snapshot_observed_at":"2026-08-04T21:33:57.450837Z","title":"E., and Stoica, I","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.450837Z"},"links":{"cited_paper":"/paper/1807.05118","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:70f498828ac075e70963f4bf2b149f79dd57958f8bb0fd03bcad3d04f15b2b38","observation_id":"889b5a5f-b92e-4b84-9302-b42c2b848916","resolution":{"observed_at":"2026-08-04T21:33:57.450837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-14T20:13:52.872565Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-04T21:33:57.535828Z","title":"and Hutter, F","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.535828Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:c389edc0bc653a521c28201fdfb8aa6536678fc2f7053e7122dbb9e88c9ad4cb","observation_id":"ce50c369-7cc8-48ef-b4aa-788e6002551b","resolution":{"observed_at":"2026-08-04T21:33:57.535828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17173","last_updated":"2024-07-17T15:32:47Z","snapshot_observed_at":"2026-08-13T04:52:42.660133Z","submitted_at":"2023-12-28T17:58:42Z","title":"Non-Vacuous Generalization Bounds for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17173","snapshot_observed_at":"2026-08-04T21:33:57.602425Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.602425Z"},"links":{"cited_paper":"/paper/2312.17173","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:cc84559e85f10857eb4fbc634d4f7142d87954231f08f44c04f70de13706693b","observation_id":"ee85379b-e4ce-4d10-a93b-cb64929f9616","resolution":{"observed_at":"2026-08-04T21:33:57.602425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18158","last_updated":"2024-07-25T16:13:58Z","snapshot_observed_at":"2026-08-12T23:14:44.898751Z","submitted_at":"2024-07-25T16:13:58Z","title":"Unlocking Tokens as Data Points for Generalization Bounds on Larger Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.18158","snapshot_observed_at":"2026-08-04T21:33:57.675730Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.675730Z"},"links":{"cited_paper":"/paper/2407.18158","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:e3b7f1dfb10b1654d1ee5c7eb4127453f01c512458476cb6286636516ada9b0d","observation_id":"a54bfe07-c46f-47e3-9bfd-17eb2eb1cf9c","resolution":{"observed_at":"2026-08-04T21:33:57.675730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:33:57.729530Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.729530Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:5a85f4c0c31bff16fbeeecadf3950d4bee992b500f3513714a5bd36baffdfd2d","observation_id":"64763d31-4fb7-42ac-937a-3a2d2d542b67","resolution":{"observed_at":"2026-08-04T21:33:57.729530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:02.326905Z","title":"G., Challú, C., Garza, A., Canseco, M","venue":null,"work_id":"e6d39104-7295-41b1-a190-6b2fb914a42b","year":2022},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.795160Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:a7a028ad032d33bd16882e6a019ac94c56b7960852120960fe73f3d51664e2b3","observation_id":"4fa68817-cf3a-4cef-87af-3e41a57ad3e6","resolution":{"observed_at":"2026-08-04T21:34:02.399652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2310.19214","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:33:59.863511Z","title":"Factor fitting, rank allocation, and partitioning in multilevel low rank matrices, 2023","venue":null,"work_id":"b4c7d1e4-c797-40b0-a5e3-1a8db19bb15b","year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.857015Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:b5dab56f926d79d71c3e717cd867fc5d02f5b75f630706de09e9a69a7b5f6601","observation_id":"5ff92820-96b1-41c4-94b1-e78b02516a1d","resolution":{"observed_at":"2026-08-04T21:33:59.912263Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12067","last_updated":"2025-08-25T13:02:26Z","snapshot_observed_at":"2026-08-15T03:04:31.180684Z","submitted_at":"2024-09-18T15:39:12Z","title":"Fitting Multilevel Factor Models","version":4},"cited_work":{"arxiv_id":"2409.12067","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.12067","snapshot_observed_at":"2026-08-04T21:33:59.744124Z","title":"Fitting Multilevel Factor Models","venue":"stat.ML","work_id":"f9c98f84-35b2-4e27-bec2-2ee853e52be9","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.919742Z"},"links":{"cited_paper":"/paper/2409.12067","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:1bc1a166815856f021303582b2ea8e750cf65721856b956a5da75824539c14a5","observation_id":"b5020df4-b203-4893-becd-3f0a8d86f37c","resolution":{"observed_at":"2026-08-04T21:33:59.796860Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.10866","last_updated":"2023-04-19T20:08:39Z","snapshot_observed_at":"2026-08-14T14:59:12.124634Z","submitted_at":"2023-02-21T18:29:25Z","title":"Hyena Hierarchy: Towards Larger Convolutional Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.10866","snapshot_observed_at":"2026-08-04T21:33:57.984068Z","title":"Y., Dao, T., Baccus, S., Bengio, Y., Ermon, S., and Ré, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:57.984068Z"},"links":{"cited_paper":"/paper/2302.10866","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:ba4c7d6a206e100131b44006f8321d74558f1fe5c1078dc3097ffef57c6eb77b","observation_id":"d6584aec-8a13-4da8-aba1-d60e3818a700","resolution":{"observed_at":"2026-08-04T21:33:57.984068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02117","last_updated":"2024-10-04T17:47:01Z","snapshot_observed_at":"2026-08-12T22:32:01.986478Z","submitted_at":"2024-10-03T00:44:50Z","title":"Searching for Efficient Linear Layers over a Continuous Space of Structured Matrices","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02117","snapshot_observed_at":"2026-08-04T21:33:58.050154Z","title":"D., and Wilson, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.050154Z"},"links":{"cited_paper":"/paper/2410.02117","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:b96145f421d6bf5ac49b84a442e990035822023cd121a0fabf6f1db987a19d42","observation_id":"a911506b-8bb7-4ea1-9a5d-6eee342fa50e","resolution":{"observed_at":"2026-08-04T21:33:58.050154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06248","last_updated":"2024-06-10T13:25:43Z","snapshot_observed_at":"2026-08-12T23:46:30.005871Z","submitted_at":"2024-06-10T13:25:43Z","title":"Compute Better Spent: Replacing Dense Layers with Structured Matrices","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06248","snapshot_observed_at":"2026-08-04T21:33:58.122029Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.122029Z"},"links":{"cited_paper":"/paper/2406.06248","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:8c025cbcbedbc9b8d7630c1758973cd0990933c82cfbbc4b3d55b1ed146ae754","observation_id":"6a950f7e-474c-4ecb-9dbe-18605f675785","resolution":{"observed_at":"2026-08-04T21:33:58.122029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:01.972286Z","title":"Combiner: full attention transformer with sparse computation cost","venue":null,"work_id":"ec30e287-644c-4c7e-b13d-36831fbf3163","year":2021},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.205239Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:813daec4ed9cf519c957db22cec98e848901a2e56d412af1550c288724c203f6","observation_id":"2fbc8239-db77-4e0f-a48b-2c09d1dfc3e1","resolution":{"observed_at":"2026-08-04T21:34:02.130187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-04T21:33:58.278888Z","title":"G., Hardin, C., Bhupatiraju, S., Hussenot, L., Mesnard, T., Shahriari, B., Ram \\'e , A., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.278888Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:536608c170a9fc6f532720f4a86c59d57bdd452facb82628ab46b3e2c2d3fa53","observation_id":"585156d6-deff-4d3b-937e-b4d7a7f9fb0f","resolution":{"observed_at":"2026-08-04T21:33:58.278888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:01.679918Z","title":"Representational strengths and limitations of transformers","venue":null,"work_id":"97033df3-a7e7-4c22-a8dd-b55a48233134","year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.351859Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:c70af6c8bdf68e2adc787c8bbe5453a249b50e4b21f89499a7178ae6ca1ef14f","observation_id":"cc103613-90e9-416d-bd47-3eae6ba728ea","resolution":{"observed_at":"2026-08-04T21:34:01.790536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:01.495295Z","title":"A., Choromanski, K","venue":null,"work_id":"ec26a14d-39b9-4975-903f-fcc91fc0fcab","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.405629Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:8a9eb99edd05d718dfc3883c5f1c90e01a19b5180aa0f643751d63d06573a8e4","observation_id":"efd06eb8-9e63-402a-9676-3b57f7f06270","resolution":{"observed_at":"2026-08-04T21:34:01.583496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.09864","last_updated":"2023-11-08T13:36:32Z","snapshot_observed_at":"2026-08-14T03:33:46.294739Z","submitted_at":"2021-04-20T09:54:06Z","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.09864","snapshot_observed_at":"2026-08-04T21:33:58.500684Z","title":"Roformer: Enhanced transformer with rotary position embedding, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.500684Z"},"links":{"cited_paper":"/paper/2104.09864","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:791084846d2d461fa2860fa35863940aff5346fc0916695b2bf731c0ed543dd9","observation_id":"994ef739-8483-4f90-9507-67c99078a52f","resolution":{"observed_at":"2026-08-04T21:33:58.500684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:01.321576Z","title":"T., Gu, A., Dao, T., Rudra, A., and R\\' e , C","venue":null,"work_id":"f6163126-925e-4543-820f-b7a4ae0c86c7","year":2018},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.560591Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:0a6a185248599d331a31a5184d289f38460ff4c217a72f98a586e0c990d323a0","observation_id":"d0201419-cf5a-45f1-a02d-64831d8b799a","resolution":{"observed_at":"2026-08-04T21:34:01.409880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:33:58.625945Z","title":"N., Kaiser, L","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.625945Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:fef8826dd7ef194496abf2c99b44dbed4f1d4be3aff94e4e1218813b1ca80bf4","observation_id":"7c3565e0-babb-4dd2-ab73-d4025e91785c","resolution":{"observed_at":"2026-08-04T21:33:58.625945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13663","last_updated":"2024-12-19T06:32:26Z","snapshot_observed_at":"2026-08-14T20:02:35.071920Z","submitted_at":"2024-12-18T09:39:44Z","title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13663","snapshot_observed_at":"2026-08-04T21:33:58.704084Z","title":"Smarter, better, faster, longer: A modern bidirectional encoder for fast, memory efficient, and long context finetuning and inference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.704084Z"},"links":{"cited_paper":"/paper/2412.13663","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:710ff76d71a62a93f0d7eb778ee38ae4196e08c54656027d00afe26d98e916c7","observation_id":"8849f7da-aa3d-44b9-b700-26f4770a10bd","resolution":{"observed_at":"2026-08-04T21:33:58.704084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:01.110350Z","title":"Building on efficient foundations: Effective training of LLM s with structured feedforward layers","venue":null,"work_id":"b85c2e8b-2893-41fe-8a84-433549c71e20","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.758143Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:bb71fa331a855f300b1ad51d74aca54563dc718f23c3fcfe1922259a7cf6a8dd","observation_id":"ff859eef-07fb-4a23-933b-1dca443c838c","resolution":{"observed_at":"2026-08-04T21:34:01.161730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01039","last_updated":"2025-01-02T03:41:32Z","snapshot_observed_at":"2026-08-15T18:03:19.320348Z","submitted_at":"2025-01-02T03:41:32Z","title":"MSWA: Refining Local Attention with Multi-ScaleWindow Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01039","snapshot_observed_at":"2026-08-04T21:33:58.820791Z","title":"Mswa: Refining local attention with multi-scalewindow attention","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.820791Z"},"links":{"cited_paper":"/paper/2501.01039","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:9ae3fc4b38b3e5efd18f6b24c664aa1e3c5daad127dcc5d0f35ec3b4546c9345","observation_id":"3033482e-7b63-494d-a4d2-5bb7b2cec55e","resolution":{"observed_at":"2026-08-04T21:33:58.820791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.03466","last_updated":"2022-03-28T08:12:14Z","snapshot_observed_at":"2026-08-13T23:41:36.523890Z","submitted_at":"2022-03-07T15:37:35Z","title":"Tensor Programs V: Tuning Large Neural Networks via Zero-Shot Hyperparameter Transfer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.03466","snapshot_observed_at":"2026-08-04T21:33:58.882869Z","title":"J., Babuschkin, I., Sidor, S., Liu, X., Farhi, D., Ryder, N., Pachocki, J., Chen, W., and Gao, J","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.882869Z"},"links":{"cited_paper":"/paper/2203.03466","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:eb044b9b0a38f6c6f0b5b5a422fd2e6978fd5383dbb810b148b372adcfccaff1","observation_id":"0af2bd64-eb77-4913-9748-358a415f6665","resolution":{"observed_at":"2026-08-04T21:33:58.882869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17813","last_updated":"2024-05-14T00:10:33Z","snapshot_observed_at":"2026-08-13T05:39:27.729878Z","submitted_at":"2023-10-26T23:17:39Z","title":"A Spectral Condition for Feature Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17813","snapshot_observed_at":"2026-08-04T21:33:58.943265Z","title":"B., and Bernstein, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:58.943265Z"},"links":{"cited_paper":"/paper/2310.17813","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:cf036b53ffba0615ed1fcd4007fb8a2e330ff888536d903381280ef8025e7cdf","observation_id":"430a08ec-d21d-4af3-aa1c-1384aa4b3481","resolution":{"observed_at":"2026-08-04T21:33:58.943265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.04070","last_updated":"2019-11-11T04:31:23Z","snapshot_observed_at":"2026-08-12T04:15:49.925553Z","submitted_at":"2019-11-11T04:31:23Z","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.04070","snapshot_observed_at":"2026-08-04T21:33:59.005872Z","title":"Bp-transformer: Modelling long-range context via binary partitioning","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:59.005872Z"},"links":{"cited_paper":"/paper/1911.04070","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:63c741764d716c6d2c211b1c67e1fbda0dfc5419ea582951c4693ac7d99664c0","observation_id":"203c83a2-4eb1-42a8-aa2b-632ee73ae8cb","resolution":{"observed_at":"2026-08-04T21:33:59.005872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11089","last_updated":"2025-02-27T09:01:21Z","snapshot_observed_at":"2026-08-07T14:17:15.942844Z","submitted_at":"2025-02-16T11:53:44Z","title":"Native Sparse Attention: Hardware-Aligned and Natively Trainable Sparse Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11089","snapshot_observed_at":"2026-08-04T21:33:59.087667Z","title":"X., Wang, L., Xiao, Z., Wang, Y., Ruan, C., Zhang, M., Liang, W., and Zeng, W","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:59.087667Z"},"links":{"cited_paper":"/paper/2502.11089","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:274941b57281ac000377314bda44a23a3626e50e8c78e275e72aa9a6a65ff7f4","observation_id":"1e2b8942-71ca-4d56-bd38-5c4054607990","resolution":{"observed_at":"2026-08-04T21:33:59.087667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:34:00.902463Z","title":"The hedgehog & the porcupine: Expressive linear attentions with softmax mimicry","venue":null,"work_id":"eba52303-a3f2-4646-bf5a-575a33d99df4","year":2024},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:59.154952Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:bf15c36418de6d3a21d85a0089ce804bd65f62e619e72400e7abe170c9db9615","observation_id":"86486629-e5d7-4f79-a397-9d0df38a749e","resolution":{"observed_at":"2026-08-04T21:34:00.978183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.07436","last_updated":"2021-03-28T14:45:04Z","snapshot_observed_at":"2026-08-14T08:42:55.100972Z","submitted_at":"2020-12-14T11:43:09Z","title":"Informer: Beyond Efficient Transformer for Long Sequence Time-Series Forecasting","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.07436","snapshot_observed_at":"2026-08-04T21:33:59.233909Z","title":"Informer: Beyond efficient transformer for long sequence time-series forecasting, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:59.233909Z"},"links":{"cited_paper":"/paper/2012.07436","citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:a8d1c52bdd39bb490de0ad8e4b58b5c65d6d2c77f57a320a0f3267b824e30d6d","observation_id":"1199fb4c-79df-4cf0-873b-bbd118b4dd8e","resolution":{"observed_at":"2026-08-04T21:33:59.233909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.acl-long.294","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Soricut, R","venue":null,"work_id":"887ceedf-b57e-4cf6-b049-0cf0a2edbc0e","year":2021},"citing_paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-04T21:33:59.310259Z"},"links":{"citing_paper":"/paper/2509.07963"},"observation_digest":"sha256:8ad9ec01a00dde06f793788a18894e6e6ea80b41ca1902bfa2200305cf2f8caf","observation_id":"53304a06-7977-4aa9-b6d3-f4fb627e8bed","resolution":{"observed_at":"2026-08-04T21:33:59.482077Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.07963","last_updated":"2026-06-02T19:48:10Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T14:00:09.050132Z","submitted_at":"2025-09-09T17:50:58Z","title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices"},"reference_resolution":{"displayed":56,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":37,"verified_exact":7,"verified_fuzzy":12},"total_outbound_references":56},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 56 of 56 outbound references and 0 inbound Pith citation observations for arXiv:2509.07963."}