{"as_of":"2026-08-11T05:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:99c1eb3aef023586e6890b8b45cb9c209c960c7d2145b073cb798b2bb2f08fa0","coverage":[{"denominator":69,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T16:15:39.595662Z","state":"measured"},{"denominator":69,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":69,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.13428/citation-record","integrity":"/paper/2501.13428/integrity","json":"/paper/2501.13428/citation-record.json","paper":"/paper/2501.13428"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:39.235757Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.235757Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:71cfd33305a7c8d6d1e352c59a44cb5df6ca6bc8bea185ab98052b045a8f1cd2","observation_id":"afc1a57f-7fde-4311-97e3-c3f6e17e04d4","resolution":{"observed_at":"2026-08-10T16:15:39.235757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:39.243176Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.243176Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:8521cc79f6b0c74ac96a3058ba8d8262b38eef71de6be305ff154c5372f1c1ea","observation_id":"f471450c-13cc-47c8-83c1-4da3b35d00f2","resolution":{"observed_at":"2026-08-10T16:15:39.243176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-10T16:15:39.250909Z","title":"L.; Almeida, D.; Altenschmidt, J.; Altman, S.; Anadkat, S.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.250909Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:3e75d14b8e04058959b30650721a349d2d33f5e3d456f6a4d20c68da33377903","observation_id":"a09c29bc-af4c-439e-b92a-f70a8b768a32","resolution":{"observed_at":"2026-08-10T16:15:39.250909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18668","last_updated":"2025-03-07T18:57:52Z","snapshot_observed_at":"2026-08-11T04:39:27.148460Z","submitted_at":"2024-02-28T19:28:27Z","title":"Simple linear attention language models balance the recall-throughput tradeoff","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18668","snapshot_observed_at":"2026-08-10T16:15:39.257674Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.257674Z"},"links":{"cited_paper":"/paper/2402.18668","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:e4232f576bd28614b294d23700ac3acbd9c096cef691bb02a1cb354cc2aea4b9","observation_id":"4978aec1-4348-41bc-b7c2-51c319609a10","resolution":{"observed_at":"2026-08-10T16:15:39.257674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.758976Z","title":"R.; et al","venue":null,"work_id":"805654d0-d814-4a32-933f-45274c31396e","year":2016},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.263919Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:33b50eccb6be89e88ec561f8dea1101545bffddea0e07e3ca9a3c3e69a0b2dd5","observation_id":"a86c47fc-37b1-4ee8-9863-f59cd4058ef7","resolution":{"observed_at":"2026-08-10T16:15:40.764012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.743226Z","title":"R.; and Hinton, G","venue":null,"work_id":"16e649be-6886-499b-8fd7-43f51773ea4c","year":2016},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.269364Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:1e6f43fbbd4e2a347aa375fa111b040e464c39d341b4dff2c71bab55683d1a38","observation_id":"b71ad9f2-08a8-44ca-9b1d-bebffa3614a0","resolution":{"observed_at":"2026-08-10T16:15:40.748117Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.727754Z","title":null,"venue":null,"work_id":"58aae191-d35f-4710-9070-daf7fa56572e","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.275730Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:296e85c6aa87607de1c46d04e993de5e7d90c45f7fce00b53d90d61cf1c99508","observation_id":"1afd27da-cebc-4d8f-a3c4-4519396c1189","resolution":{"observed_at":"2026-08-10T16:15:40.732554Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:39.280860Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.280860Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:994303d7706acdefdcb52abba4f60aa12c7dde3559de29f8e23f68fded2c8d0a","observation_id":"ae958d99-52d4-448c-98ac-c33aaeedbf3b","resolution":{"observed_at":"2026-08-10T16:15:39.280860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.700565Z","title":"by parts","venue":null,"work_id":"28a75531-0703-4d0c-96a7-58f766865f1d","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.287624Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:a62091615e3d833bce2db2f62e10150bddd67bbd31ea539f5c0864aa24f1fb27","observation_id":"673a6c1e-ef29-46ed-907f-15434ede0e09","resolution":{"observed_at":"2026-08-10T16:15:40.705616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.685244Z","title":null,"venue":null,"work_id":"741a1d17-e703-4f38-b7a1-c49fba9f9f2e","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.292682Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:08d87361fcc0c6433fe65105f5f95648a53c557a7b89039aac6005795e5fa747","observation_id":"382ee36f-b036-46f5-9ba6-00b3e67c0bc9","resolution":{"observed_at":"2026-08-10T16:15:40.689382Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.07091","last_updated":"2022-06-06T19:02:30Z","snapshot_observed_at":"2026-08-09T17:40:21.639101Z","submitted_at":"2021-04-14T19:37:40Z","title":"SummScreen: A Dataset for Abstractive Screenplay Summarization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.07091","snapshot_observed_at":"2026-08-10T16:15:39.297757Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.297757Z"},"links":{"cited_paper":"/paper/2104.07091","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:4c6423f89f532b8717cc0cab64ae30a897e3e6373f644f8d04a655dce8e48ef1","observation_id":"d51955f2-bf2a-40e0-b48f-02a45a354d95","resolution":{"observed_at":"2026-08-10T16:15:39.297757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.667381Z","title":null,"venue":null,"work_id":"287ae4f3-c724-490c-9528-49afe8c61ae7","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.302887Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:d1940987bfea3b550ebca7a9b2a114cecd05ebcc002e58109ee5fa9f08833d11","observation_id":"7f37e89d-4baf-49cb-9356-3cf0bbc04390","resolution":{"observed_at":"2026-08-10T16:15:40.674887Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.651566Z","title":"J.; and Rudnicky, A","venue":null,"work_id":"5b2bd244-65e1-4b37-b53e-b265ae549dcf","year":2022},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.307746Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:3d7658d2400e3a6d8e33e02d8b6458b92383dfa65b9b673b8d98961e20d48513","observation_id":"dabe11c9-89fc-4040-a11a-f312d2dad1e9","resolution":{"observed_at":"2026-08-10T16:15:40.656297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.634629Z","title":null,"venue":null,"work_id":"ec97a9b1-0904-47d0-92fc-aae6085431a7","year":2022},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.313524Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:fb1f4f5422677ae0d23e2db57615c41d813caa0027be5c69e00f38931c4c7502","observation_id":"5826975d-b4f9-468b-b469-fdd910f00b41","resolution":{"observed_at":"2026-08-10T16:15:40.639443Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-10T16:15:39.317976Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.317976Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:a7ba354f2af5c82390612200e0fa07740d35c1093170a7445e8579c538d21679","observation_id":"b205bbf1-ebcb-470c-8875-1fd178736052","resolution":{"observed_at":"2026-08-10T16:15:39.317976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-10T16:15:39.322683Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.322683Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:b1a95f9ef6a51af66798f8c4a9e3a0d82f0ba859bf41601e2aa6dc4e7ac4a0cc","observation_id":"0c392cf5-dd08-4b1b-857a-16960221c315","resolution":{"observed_at":"2026-08-10T16:15:39.322683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.613001Z","title":"P.; Caron, M.; Geirhos, R.; Alabdul mohsin, I.; Jenatton, R.; Beyer, L.; Tschannen, M.; Arnab, A.; Wang, X.; Ruiz, C","venue":null,"work_id":"20ac8bd6-a0eb-4542-95cd-0a2910ad193c","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.327691Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:855e3f1ba44d2f34ac3bda2f9dbe85b219eaeefa52b5e2ecd8af78aee141a879","observation_id":"0e3de720-1306-4a10-b820-76242885159e","resolution":{"observed_at":"2026-08-10T16:15:40.618829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-10T16:15:39.332412Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.332412Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:526767ec01ad2b684092e47560967bac8b9f3f5515d582ebfb4c548557f1dd1c","observation_id":"e03e08df-f67e-4e94-ae92-476ee0d61109","resolution":{"observed_at":"2026-08-10T16:15:39.332412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.594687Z","title":null,"venue":null,"work_id":"65823f91-5577-4aec-bcb9-eba20cd30b59","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.337139Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:1e1fed869245459230f19d00907725c8cecd602b22cc674925bfe462127cb3c8","observation_id":"800ccb2a-abe6-40e8-81b4-b775960de5e2","resolution":{"observed_at":"2026-08-10T16:15:40.601272Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.573622Z","title":null,"venue":null,"work_id":"186de0ab-1139-4858-837b-c6f3e0baa72f","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.341642Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:1e5cfbcba721ad512619dbc1f6c71f9c15b42453bdb4e1fa5909f7fd7257b052","observation_id":"d2269333-c4aa-4807-adcd-d2d6d1d9e19e","resolution":{"observed_at":"2026-08-10T16:15:40.580629Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00805","last_updated":"2018-08-21T00:02:44Z","snapshot_observed_at":"2026-08-10T23:57:50.331464Z","submitted_at":"2017-04-03T20:50:29Z","title":"On the Properties of the Softmax Function with Application in Game Theory and Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00805","snapshot_observed_at":"2026-08-10T16:15:39.346407Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.346407Z"},"links":{"cited_paper":"/paper/1704.00805","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:0d5a30cb9dc0124337c0f2b9f5ba82d955d5daf59f03e93825450e0d765590d9","observation_id":"844e1b39-a54c-4e9f-a16f-f8d71f905103","resolution":{"observed_at":"2026-08-10T16:15:39.346407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.553944Z","title":null,"venue":null,"work_id":"4e0bec58-a25a-4261-adaa-d6c03c8d7c50","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.351864Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:83240b8f6893d21a74c785d614a7766b31379ef2801023c5c16c7a9b62ae2174","observation_id":"e48b8e52-3130-4f29-aaa2-c8d1d7664ea8","resolution":{"observed_at":"2026-08-10T16:15:40.559256Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.533549Z","title":null,"venue":null,"work_id":"0e4f1192-9e3c-4c19-8c3c-be17d5f4c3cf","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.356642Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:ae368110fd305b06c65765454a95c71aa74302bf2e22a8ac398d7bb6adc4fd23","observation_id":"82d38187-94bb-43ed-9ab3-8bbcbc4263e4","resolution":{"observed_at":"2026-08-10T16:15:40.538663Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.518538Z","title":null,"venue":null,"work_id":"ef447a01-7272-4f48-b138-7b7acc3fe2e6","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.362009Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:24fffd963efc4b2dee78617639c4ccf3f98050bbfb147db0e2d0a34a9721ec26","observation_id":"09051a19-953d-4690-bf21-d09e8d5804d2","resolution":{"observed_at":"2026-08-10T16:15:40.523345Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.505328Z","title":null,"venue":null,"work_id":"eca349dc-610f-42d6-83ea-69cb76bbb070","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.367337Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:0bd60655b01995d139d9efe60e53b35bc7b5e8909786bfbde33f9040b37eca9c","observation_id":"37daf6d2-b9a3-485e-9cf3-ab480d62ae3b","resolution":{"observed_at":"2026-08-10T16:15:40.509230Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.491452Z","title":null,"venue":null,"work_id":"3ed1b20a-7537-424c-a805-78c0f83fe6c4","year":2020},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.372609Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:3f0811cbcb7062b75f479f58e8269c15d9dfd6c01e879c214d8261997f6ef796","observation_id":"fe2a43d5-5d83-470e-82cc-2bbb58927f7a","resolution":{"observed_at":"2026-08-10T16:15:40.495508Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.08415","last_updated":"2023-06-06T01:53:32Z","snapshot_observed_at":"2026-07-06T05:01:27.910364Z","submitted_at":"2016-06-27T19:20:40Z","title":"Gaussian Error Linear Units (GELUs)","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.08415","snapshot_observed_at":"2026-08-10T16:15:39.378740Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.378740Z"},"links":{"cited_paper":"/paper/1606.08415","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:b4e1d5e4d39b4d785b9d7028b069df81ed992ee8839c20ec193f4371cc901922","observation_id":"6a9b54c1-16d7-401b-9732-b336aab27439","resolution":{"observed_at":"2026-08-10T16:15:39.378740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.476963Z","title":"R.; Pawar, S","venue":null,"work_id":"c27c83f0-2fb4-483f-ab80-e18d601cf922","year":2020},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.384180Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:05cc487d9b1c661c1441a025960c2f5b53d36fb5b4bbcda6ab3bcbd19aa6824a","observation_id":"d4f249c3-6dc9-41c7-867f-f600191971bc","resolution":{"observed_at":"2026-08-10T16:15:40.481540Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.460666Z","title":"G.; Zhu, M.; Chen, B.; Kalenichenko, D.; Wang, W.; Weyand, T.; Andreetto, M.; and Adam, H","venue":null,"work_id":"a6055611-12d0-45af-a0f9-4d5a33682455","year":2017},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.391079Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:4d677240ea908a017c3eda2b7e8bbea1ae09bee40b3fff88de7d26f3c6c0a49b","observation_id":"9676e927-969a-4fe6-abd9-a992f8c35dd1","resolution":{"observed_at":"2026-08-10T16:15:40.467030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.445030Z","title":null,"venue":null,"work_id":"4adcf56d-b0e7-4f13-b337-0d358da85db1","year":2020},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.395736Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:c2f68e7c2b297a7a8ee0c6ccd7be36346db382d44f161ff3b9c9bc6d1a278a77","observation_id":"aa4a593e-282d-459a-ba09-e5cf7e364d50","resolution":{"observed_at":"2026-08-10T16:15:40.450066Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.430165Z","title":null,"venue":null,"work_id":"35353164-768e-44d4-9393-5829ac9049fa","year":2022},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.400342Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:f325c6f5e3f8fd09f89ae17e2eb3e5b38cd9179360bd5fb1a2e04e1058080eca","observation_id":"7b9921e9-7ca0-4c06-9032-8abc3769845b","resolution":{"observed_at":"2026-08-10T16:15:40.434776Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.414088Z","title":null,"venue":null,"work_id":"c4fc0e73-ae91-454b-9a4b-343b1f30fa67","year":2020},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.405056Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:52375fdd4a38080f6faccfa8d4fc8f3a09b8d8d4b30b82ed45e95aa5e945249e","observation_id":"3b7d9315-0db8-4a0a-aaa4-1f3b4f8905ec","resolution":{"observed_at":"2026-08-10T16:15:40.418760Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1611.01144","last_updated":"2017-08-05T22:45:19Z","snapshot_observed_at":"2026-08-01T18:34:23.156273Z","submitted_at":"2016-11-03T19:48:08Z","title":"Categorical Reparameterization with Gumbel-Softmax","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.01144","snapshot_observed_at":"2026-08-10T16:15:39.411057Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.411057Z"},"links":{"cited_paper":"/paper/1611.01144","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:9a8b4f37898a4f82430a8c20fa8de05c7df9760cbe65de4b92e53b56c96a43f8","observation_id":"94bbdf7f-f9a8-4560-b503-b123140d4603","resolution":{"observed_at":"2026-08-10T16:15:39.411057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.399096Z","title":null,"venue":null,"work_id":"d226e919-14a8-4ad4-ae21-645e742f3fe3","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.416622Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:fc87b0a1c1f8cbb940b3ca723013455a24c0d57f194f249c74acdfe57f2406e6","observation_id":"e45d73a1-ef9e-4631-a9b3-ad2892b48fdd","resolution":{"observed_at":"2026-08-10T16:15:40.403912Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.383805Z","title":null,"venue":null,"work_id":"208b87a8-2f79-4329-a147-02daa4bffe67","year":2020},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.421553Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:8a91d29940d81ebdad6b2ed614c1cadbcf3405cfd2fdff2d022591a4a77de554","observation_id":"a3824dba-abe0-476b-b9ee-21885413a561","resolution":{"observed_at":"2026-08-10T16:15:40.388622Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.19466","last_updated":"2023-11-06T19:48:10Z","snapshot_observed_at":"2026-07-06T15:35:41.522547Z","submitted_at":"2023-05-31T00:29:55Z","title":"The Impact of Positional Encoding on Length Generalization in Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.19466","snapshot_observed_at":"2026-08-10T16:15:39.426076Z","title":"N.; Das, P.; and Reddy, S","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.426076Z"},"links":{"cited_paper":"/paper/2305.19466","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:d3721c3ff7b16c53beb6358d2da6b52b98a49e4d3607868314accb57c00530c0","observation_id":"37fcd849-5bf3-4dd4-b3ee-5ca6f8eee27e","resolution":{"observed_at":"2026-08-10T16:15:39.426076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.366165Z","title":null,"venue":null,"work_id":"e5ef1b47-896d-43aa-9aee-12f38f1dc198","year":2021},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.431564Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:41b6ba4f41bf12148e14369f7fe54886b49f7788fba48ee5b693afbd0ef71023","observation_id":"8bb185fb-0a3d-46ab-863c-1741904b819a","resolution":{"observed_at":"2026-08-10T16:15:40.372052Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.09624","last_updated":"2024-08-19T00:56:44Z","snapshot_observed_at":"2026-07-06T19:02:11.267510Z","submitted_at":"2024-08-19T00:56:44Z","title":"Attention is a smoothed cubic spline","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.09624","snapshot_observed_at":"2026-08-10T16:15:39.436829Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.436829Z"},"links":{"cited_paper":"/paper/2408.09624","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:a3d771d4f98b1c64db60a2fd768c4825067811fb3bd49e8fc239b1f19df77a67","observation_id":"091da57e-2e0a-4363-b48f-e25ec83a1946","resolution":{"observed_at":"2026-08-10T16:15:39.436829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.349657Z","title":null,"venue":null,"work_id":"840d3d62-c0ec-4bd7-a007-1b2b8e3d4bce","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.441125Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:6784080115347332705941edbe421e0660b2df7547ad6466dc3bb396fde9dc6f","observation_id":"b6979abf-00b4-4a54-991a-b774ddcfdb54","resolution":{"observed_at":"2026-08-10T16:15:40.355531Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.331330Z","title":null,"venue":null,"work_id":"aa5ba8fd-c331-4e77-9cf0-09c581c54311","year":2022},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.445439Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:1bd353f8f7f0aa2e3c12c984fb80cb19d64b6c20a51e1d09fcfc51839503b173","observation_id":"0e62bf4f-6e0b-4b97-bd1f-38f3b130f740","resolution":{"observed_at":"2026-08-10T16:15:40.337359Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.314349Z","title":null,"venue":null,"work_id":"baea81db-ab51-444a-8b5c-593d2d65a27a","year":2021},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.450786Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:4ba0d3e7ce762d66ba9e232b1bbdf2493055f56cdb2ee1c0b4c8b4c42c2ddbb6","observation_id":"d09781f5-177a-40e5-af82-b7af3197ebd5","resolution":{"observed_at":"2026-08-10T16:15:40.319524Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-11T01:48:59.557045Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-10T16:15:39.456266Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.456266Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:6681a915d7cc41c8110a2d426d2fc2d897eac2a82d9fd19bfef45c5ebf02ff22","observation_id":"e7531074-dbdf-4d2c-a97b-48e9c4b45065","resolution":{"observed_at":"2026-08-10T16:15:39.456266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.298795Z","title":null,"venue":null,"work_id":"0684d6c3-db7e-47fc-9fd4-3be174733975","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.462828Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:acc8c4dcd6534521d355baed3009e6b8d316c706e305fe1602ece88112b43e80","observation_id":"5d256a35-7ce4-4dd2-8bbd-e90f1ae3e65c","resolution":{"observed_at":"2026-08-10T16:15:40.303432Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.283040Z","title":null,"venue":null,"work_id":"e53838cf-2497-4883-8075-3afc5afd5b2f","year":2022},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.469070Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:3e05a6abb8f315d9555f4b5490b015e823c9ac827137f95583649a842ec49bfb","observation_id":"bc332018-3c71-429f-b65d-9b4363ef0ea3","resolution":{"observed_at":"2026-08-10T16:15:40.287457Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.267876Z","title":null,"venue":null,"work_id":"7844e4cd-8c3a-4469-83df-dcff23a28529","year":2021},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.474554Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:7fb82bf21df2f12776d73d05ae06d167a7cfbfcafb290073713a46c2bf379629","observation_id":"84242bf9-f542-4348-a7e6-9b0c5d1f99bb","resolution":{"observed_at":"2026-08-10T16:15:40.272505Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08681","last_updated":"2020-08-13T05:42:12Z","snapshot_observed_at":"2026-07-06T08:16:16.271286Z","submitted_at":"2019-08-23T06:22:06Z","title":"Mish: A Self Regularized Non-Monotonic Activation Function","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08681","snapshot_observed_at":"2026-08-10T16:15:39.481878Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.481878Z"},"links":{"cited_paper":"/paper/1908.08681","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:0ecfafcd8458af5096de8935d2d1dd80cf26bfbe52b4a7f0eda49f003adda074","observation_id":"53c1ade8-1970-4297-8ee3-ef44a3de66a1","resolution":{"observed_at":"2026-08-10T16:15:39.481878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.252720Z","title":null,"venue":null,"work_id":"782ed620-8a3c-4b54-aa5b-62e65573b457","year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.487483Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:e84090746220cc2dd2eb56e06724354188d7ceba3b6b335161788b804ea61877","observation_id":"392fd148-fa66-4dd3-a5a8-1aef7d21abb8","resolution":{"observed_at":"2026-08-10T16:15:40.257654Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17557","last_updated":"2024-10-31T11:37:49Z","snapshot_observed_at":"2026-08-02T15:25:02.551919Z","submitted_at":"2024-06-25T13:50:56Z","title":"The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17557","snapshot_observed_at":"2026-08-10T16:15:39.493338Z","title":"B.; Lozhkov, A.; Mitchell, M.; Raffel, C.; Werra, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.493338Z"},"links":{"cited_paper":"/paper/2406.17557","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:1e02e91aa2a04d000796e1013ab21ed1376501ec376f1f97d289fe9456e3444a","observation_id":"8f76f559-d5a5-412c-9b2a-ccae095ebe93","resolution":{"observed_at":"2026-08-10T16:15:39.493338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.237977Z","title":null,"venue":null,"work_id":"072486b2-7b71-4482-9d7d-108b48430953","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.498424Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:5de8376a15930a45e7a5b20f037ff3726bb1c455a3434eb60cb4d9b06ff53ae5","observation_id":"f0fd2e6d-213d-4269-83b8-127961d1e8cf","resolution":{"observed_at":"2026-08-10T16:15:40.242213Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.220179Z","title":null,"venue":null,"work_id":"613b8b81-505c-4079-b363-a7f10839c0d6","year":2022},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.504422Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:8493362943f57db8e43c0be778ec8af6d85ff312edb8f02d9e9fdbe48de85a36","observation_id":"c9b58937-1bd0-4257-ac27-c5aa32c4ee25","resolution":{"observed_at":"2026-08-10T16:15:40.225928Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:39.508980Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.508980Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:66bda439fb299bbe1dd671c4d5d309f629d5d9c6ffd92bdbbbd4d4f5442c2a7b","observation_id":"70a5e1cf-e20d-4550-8a5a-9b4d2c353557","resolution":{"observed_at":"2026-08-10T16:15:39.508980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-10T16:15:39.513319Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.513319Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:fad9d92cc429805c45a3300c9a8fc154bd978142cf0253515b2f19c45a2b0f5c","observation_id":"696ec69e-2f2b-4cef-a77d-81ec2fb280fa","resolution":{"observed_at":"2026-08-10T16:15:39.513319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.193960Z","title":"E.; Hinton, G","venue":null,"work_id":"99661c49-060f-4713-bd52-0e3fe417c120","year":1986},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.517988Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:cd053a4cc67b4ce782daa46b3afe1133c4a1bd94ecb93e25df67f17f777813e8","observation_id":"cb95c696-13c1-43cf-a826-a14e5ec2c52c","resolution":{"observed_at":"2026-08-10T16:15:40.198967Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08608","last_updated":"2024-07-12T22:15:02Z","snapshot_observed_at":"2026-07-06T18:44:53.587276Z","submitted_at":"2024-07-11T15:44:48Z","title":"FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08608","snapshot_observed_at":"2026-08-10T16:15:39.522551Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.522551Z"},"links":{"cited_paper":"/paper/2407.08608","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:6ee10586ebae1340ff0c730f23a55d93031ef95acd15488647d6e8a7ce490e56","observation_id":"a4ef9322-3f76-428b-af15-57d9263e839c","resolution":{"observed_at":"2026-08-10T16:15:39.522551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.06461","last_updated":"2023-02-13T15:41:20Z","snapshot_observed_at":"2026-08-10T13:53:47.189150Z","submitted_at":"2023-02-13T15:41:20Z","title":"A Study on ReLU and Softmax in Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.06461","snapshot_observed_at":"2026-08-10T16:15:39.526974Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.526974Z"},"links":{"cited_paper":"/paper/2302.06461","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:80994e672221d4f6e95d4c53927f2a8bb6b388e162bc5ad553ae10c853a73a65","observation_id":"65fcb19a-b739-4256-964e-b4716731c418","resolution":{"observed_at":"2026-08-10T16:15:39.526974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.178044Z","title":null,"venue":null,"work_id":"58c976a8-fcc9-49e2-87d8-5f1580a00416","year":2021},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.531608Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:f29c591ae7fd5d62bc2253762881e1f4e670d7d075e25afb176439d548712733","observation_id":"9a078954-c436-435d-9d36-759fab12192b","resolution":{"observed_at":"2026-08-10T16:15:40.183262Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.162184Z","title":null,"venue":null,"work_id":"74f8805a-73ce-43d5-82e9-84bee1d11309","year":2021},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.535474Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:bbec143685f7e5d790774463d8279007756536e71fc6e76561f810d5a8d03d07","observation_id":"362aadb9-e01a-4876-be90-5631cea15675","resolution":{"observed_at":"2026-08-10T16:15:40.167011Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:39.539536Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.539536Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:8fca87b408b6074cc769150f0fe912c01476a634452734dc18d9f4feec14c353","observation_id":"ec11d1d1-90f7-460d-924d-1abb3bfa6b42","resolution":{"observed_at":"2026-08-10T16:15:39.539536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.135098Z","title":"H.; Bai, S.; Yamada, M.; Morency, L.-P.; and Salakhutdinov, R","venue":null,"work_id":"f12eb20b-4e9d-45ea-b663-2426c005c2b1","year":2019},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.543786Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:abc47169c39f77644910505d7e7c6ef2ae4b7f156cd260c93c0c0c116ad4833f","observation_id":"f25cc4e7-1f4f-4a46-a9e6-38e88e72d2d9","resolution":{"observed_at":"2026-08-10T16:15:40.139813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01104","last_updated":"2025-05-30T23:46:05Z","snapshot_observed_at":"2026-08-09T17:03:53.787790Z","submitted_at":"2024-10-01T22:22:35Z","title":"Softmax is not Enough (for Sharp Size Generalisation)","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01104","snapshot_observed_at":"2026-08-10T16:15:39.547856Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.547856Z"},"links":{"cited_paper":"/paper/2410.01104","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:41311fcb915b4032a3aeb82e8e16e5879b54f48f7bed27e78d6915cedd7088e4","observation_id":"4d973ba7-7ece-4f81-99ab-3b644e681b9f","resolution":{"observed_at":"2026-08-10T16:15:39.547856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.118193Z","title":null,"venue":null,"work_id":"f668cccd-9cc8-470e-a985-e590c923a5e7","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.552487Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:3cb6e074007a730013717ef712900d3171fba4a80178735d7d8765d1b3ef7d00","observation_id":"48099b98-8955-4f01-8028-8706e3b35946","resolution":{"observed_at":"2026-08-10T16:15:40.122809Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06209","last_updated":"2017-07-19T17:28:46Z","snapshot_observed_at":"2026-08-06T06:05:27.671303Z","submitted_at":"2017-07-19T17:28:46Z","title":"Crowdsourcing Multiple Choice Science Questions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06209","snapshot_observed_at":"2026-08-10T16:15:39.556856Z","title":"F.; and Gardner, M","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.556856Z"},"links":{"cited_paper":"/paper/1707.06209","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:f27d44f9ea0aa771519f469f644c8a10002117a342e6926c70ce77563356aec6","observation_id":"9d8df6d7-f4f4-4a25-ad5b-9970416fc9c6","resolution":{"observed_at":"2026-08-10T16:15:39.556856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08586","last_updated":"2023-10-17T00:12:20Z","snapshot_observed_at":"2026-08-10T19:17:37.263732Z","submitted_at":"2023-09-15T17:43:40Z","title":"Replacing softmax with ReLU in Vision Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08586","snapshot_observed_at":"2026-08-10T16:15:39.562221Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.562221Z"},"links":{"cited_paper":"/paper/2309.08586","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:579686a6f145af918c8e245adce4a145864e8fc6116dc0c9c9323dfe1e556d10","observation_id":"426eac08-09b7-4efa-9833-416be8e86664","resolution":{"observed_at":"2026-08-10T16:15:39.562221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.05751","last_updated":"2025-06-06T03:35:15Z","snapshot_observed_at":"2026-08-09T03:02:32.383746Z","submitted_at":"2024-05-09T13:15:40Z","title":"Mirage: A Multi-Level Superoptimizer for Tensor Programs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.05751","snapshot_observed_at":"2026-08-10T16:15:39.567767Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.567767Z"},"links":{"cited_paper":"/paper/2405.05751","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:1dd14688b7e2120f533a716576c09c7b43ec09e37589dc93d7218caaf731ac84","observation_id":"2b7c6100-8cdb-4cf2-a2b7-ab235e99f9ba","resolution":{"observed_at":"2026-08-10T16:15:39.567767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-10T16:15:39.574537Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.574537Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:6eba5e0fe307c8508b2e4467d4dc0308873cc31909f3b153c58f894a7f109c5b","observation_id":"00f934c5-9034-46d0-a0cc-594e12894b5d","resolution":{"observed_at":"2026-08-10T16:15:39.574537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.07830","last_updated":"2019-05-19T23:57:23Z","snapshot_observed_at":"2026-08-10T16:18:23.994244Z","submitted_at":"2019-05-19T23:57:23Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.07830","snapshot_observed_at":"2026-08-10T16:15:39.579659Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.579659Z"},"links":{"cited_paper":"/paper/1905.07830","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:0748a1ce0ecf5b13c296be66169bbfb2b18027c855a33c4251bf9f5ef39296cc","observation_id":"f4bbe4df-26fe-40c8-bd68-2bf91108be37","resolution":{"observed_at":"2026-08-10T16:15:39.579659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:39.585595Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.585595Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:dff105a32c2070a1d0e251f0df13fcdfdeb27a479e849ef7cf89339c099ce483","observation_id":"9e64777e-5e70-4e98-ab92-111f161fcd1f","resolution":{"observed_at":"2026-08-10T16:15:39.585595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.103522Z","title":null,"venue":null,"work_id":"b1e9b727-de11-410f-8825-c8887f606c5b","year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.590728Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:9c336b719979c01db7bba31b8f81ce8b3ea5260abd4007eadbf1c938e5c317c3","observation_id":"fdc91a85-4c6d-47ee-b5c1-e42199091126","resolution":{"observed_at":"2026-08-10T16:15:40.107767Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:15:40.087697Z","title":null,"venue":null,"work_id":"1f2ab859-5e6d-42f3-bf76-c8ead0084562","year":2015},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.595662Z"},"links":{"citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:321f03ce4fa7cbca12ae006484560d04445753deba1d35262e218cfc6de5d95a","observation_id":"dbf88782-b4b7-4c47-ad95-f7cf4d43d904","resolution":{"observed_at":"2026-08-10T16:15:40.093346Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","latest_version":6,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models"},"reference_resolution":{"displayed":69,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":60,"verified_exact":0,"verified_fuzzy":9},"total_outbound_references":69},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 69 of 69 outbound references and 0 inbound Pith citation observations for arXiv:2501.13428."}