{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RVUNNI4WSPJRR23AMO2OKJ2ZIH","short_pith_number":"pith:RVUNNI4W","schema_version":"1.0","canonical_sha256":"8d68d6a39693d318eb6063b4e5275941f30443fea65eeb3cd650c6eb0bc26f12","source":{"kind":"arxiv","id":"2502.16792","version":1},"attestation_state":"computed","paper":{"title":"The Role of Sparsity for Length Generalization in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"David Brandfonbrener, Eran Malach, Noah Golowich, Samy Jelassi, Sham M. Kakade","submitted_at":"2025-02-24T03:01:03Z","abstract_excerpt":"Training large language models to predict beyond their training context lengths has drawn much attention in recent years, yet the principles driving such behavior of length generalization remain underexplored. We propose a new theoretical framework to study length generalization for the next-token prediction task, as performed by decoder-only transformers. Conceptually, we show that length generalization occurs as long as each predicted token depends on a small (fixed) number of previous tokens. We formalize such tasks via a notion we call $k$-sparse planted correlation distributions, and show"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.16792","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-24T03:01:03Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"97814faa4398fb20d0d1378afc3a815a61b901715d48faef1ed89cdd26e2066c","abstract_canon_sha256":"b53ee31fb3702fc523b417a7b7926668c82724867a7d93b5486fdd8d661fa4fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:59.671409Z","signature_b64":"9FFJ8KDKTHAYXt1V4S7tRuv29NnHY955XlVAihQ0KQ+vR21uBfvpOC36ZdH+0u5aooSZRJP8oLVyf4LbIrWaBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d68d6a39693d318eb6063b4e5275941f30443fea65eeb3cd650c6eb0bc26f12","last_reissued_at":"2026-07-05T10:18:59.670867Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:59.670867Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Role of Sparsity for Length Generalization in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"David Brandfonbrener, Eran Malach, Noah Golowich, Samy Jelassi, Sham M. Kakade","submitted_at":"2025-02-24T03:01:03Z","abstract_excerpt":"Training large language models to predict beyond their training context lengths has drawn much attention in recent years, yet the principles driving such behavior of length generalization remain underexplored. We propose a new theoretical framework to study length generalization for the next-token prediction task, as performed by decoder-only transformers. Conceptually, we show that length generalization occurs as long as each predicted token depends on a small (fixed) number of previous tokens. We formalize such tasks via a notion we call $k$-sparse planted correlation distributions, and show"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16792","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.16792","created_at":"2026-07-05T10:18:59.670919+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.16792v1","created_at":"2026-07-05T10:18:59.670919+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16792","created_at":"2026-07-05T10:18:59.670919+00:00"},{"alias_kind":"pith_short_12","alias_value":"RVUNNI4WSPJR","created_at":"2026-07-05T10:18:59.670919+00:00"},{"alias_kind":"pith_short_16","alias_value":"RVUNNI4WSPJRR23A","created_at":"2026-07-05T10:18:59.670919+00:00"},{"alias_kind":"pith_short_8","alias_value":"RVUNNI4W","created_at":"2026-07-05T10:18:59.670919+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18089","citing_title":"From Reasoning Traces to Reusable Modules: Understanding Compositional Generalization in Language Model Reasoning","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00183","citing_title":"Agentic Transformers Provably Learn to Search via Reinforcement Learning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01651","citing_title":"On the Spatiotemporal Dynamics of Generalization in Neural Networks","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH","json":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH.json","graph_json":"https://pith.science/api/pith-number/RVUNNI4WSPJRR23AMO2OKJ2ZIH/graph.json","events_json":"https://pith.science/api/pith-number/RVUNNI4WSPJRR23AMO2OKJ2ZIH/events.json","paper":"https://pith.science/paper/RVUNNI4W"},"agent_actions":{"view_html":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH","download_json":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH.json","view_paper":"https://pith.science/paper/RVUNNI4W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.16792&json=true","fetch_graph":"https://pith.science/api/pith-number/RVUNNI4WSPJRR23AMO2OKJ2ZIH/graph.json","fetch_events":"https://pith.science/api/pith-number/RVUNNI4WSPJRR23AMO2OKJ2ZIH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH/action/storage_attestation","attest_author":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH/action/author_attestation","sign_citation":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH/action/citation_signature","submit_replication":"https://pith.science/pith/RVUNNI4WSPJRR23AMO2OKJ2ZIH/action/replication_record"}},"created_at":"2026-07-05T10:18:59.670919+00:00","updated_at":"2026-07-05T10:18:59.670919+00:00"}