{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BIPRLVCHD7QEJ2JCOLDCU3KQYP","short_pith_number":"pith:BIPRLVCH","schema_version":"1.0","canonical_sha256":"0a1f15d4471fe044e92272c62a6d50c3d18445153d834a76f09cdf9c8db4526e","source":{"kind":"arxiv","id":"2402.01032","version":2},"attestation_state":"computed","paper":{"title":"Repeat After Me: Transformers are Better than State Space Models at Copying","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"David Brandfonbrener, Eran Malach, Samy Jelassi, Sham M. Kakade","submitted_at":"2024-02-01T21:44:11Z","abstract_excerpt":"Transformers are the dominant architecture for sequence modeling, but there is growing interest in models that use a fixed-size latent state that does not depend on the sequence length, which we refer to as \"generalized state space models\" (GSSMs). In this paper we show that while GSSMs are promising in terms of inference-time efficiency, they are limited compared to transformer models on tasks that require copying from the input context. We start with a theoretical analysis of the simple task of string copying and prove that a two layer transformer can copy strings of exponential length while"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01032","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-01T21:44:11Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"de2b1a9ee94f526dd1866baebbdd2e28d432c4aac112645543c65364c06146cd","abstract_canon_sha256":"07de9f974a08d09c10f3f897dbb791ac771ddf76364d8ee09a966b734462f0e6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:43.532873Z","signature_b64":"ifWdWYo6n3isRK3S2TMJwwnjWqMozJ7CBXBNlQSg+Dh8xa2MjdRFHDF4nuel2fX+ePMaEKqKM08WLNn+JgM9Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a1f15d4471fe044e92272c62a6d50c3d18445153d834a76f09cdf9c8db4526e","last_reissued_at":"2026-07-05T08:26:43.532393Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:43.532393Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Repeat After Me: Transformers are Better than State Space Models at Copying","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"David Brandfonbrener, Eran Malach, Samy Jelassi, Sham M. Kakade","submitted_at":"2024-02-01T21:44:11Z","abstract_excerpt":"Transformers are the dominant architecture for sequence modeling, but there is growing interest in models that use a fixed-size latent state that does not depend on the sequence length, which we refer to as \"generalized state space models\" (GSSMs). In this paper we show that while GSSMs are promising in terms of inference-time efficiency, they are limited compared to transformer models on tasks that require copying from the input context. We start with a theoretical analysis of the simple task of string copying and prove that a two layer transformer can copy strings of exponential length while"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01032","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01032/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01032","created_at":"2026-07-05T08:26:43.532444+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01032v2","created_at":"2026-07-05T08:26:43.532444+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01032","created_at":"2026-07-05T08:26:43.532444+00:00"},{"alias_kind":"pith_short_12","alias_value":"BIPRLVCHD7QE","created_at":"2026-07-05T08:26:43.532444+00:00"},{"alias_kind":"pith_short_16","alias_value":"BIPRLVCHD7QEJ2JC","created_at":"2026-07-05T08:26:43.532444+00:00"},{"alias_kind":"pith_short_8","alias_value":"BIPRLVCH","created_at":"2026-07-05T08:26:43.532444+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21884","citing_title":"A Verifiable Search Is Not a Learnable Chain-of-Thought","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02303","citing_title":"A Hippocampus for Linear Attention: An Exact Memory for What the Recurrent State Forgets","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11052","citing_title":"Attention Amnesia in Hybrid LLMs: When CoT Fine-Tuning Breaks Long-Range Recall, and How to Fix It","ref_index":112,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00390","citing_title":"Zamba2-VL Technical Report","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26099","citing_title":"Do Language Models Need Sleep? Offline Recurrence for Improved Online Inference","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21016","citing_title":"Gated KalmaNet: A Fading Memory Layer Through Test-Time Ridge Regression","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2512.22471","citing_title":"The Bayesian Geometry of Transformer Attention","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2406.07887","citing_title":"An Empirical Study of Mamba-based Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2509.26645","citing_title":"TTT3R: 3D Reconstruction as Test-Time Training","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19427","citing_title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13473","citing_title":"OSDN: Improving Delta Rule with Provable Online Preconditioning in Linear Attention","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09472","citing_title":"Positional LSH: Binary Block Matrix Approximation for Attention with Linear Biases","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21215","citing_title":"The Recurrent Transformer: Greater Effective Depth and Efficient Decoding","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06683","citing_title":"Toeplitz MLP Mixers are Low Complexity, Information-Rich Sequence Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06997","citing_title":"Echo: KV-Cache-Free Associative Recall with Spectral Koopman Operators","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05923","citing_title":"The UNDO Flip-Flop: A Controlled Probe for Reversible Semantic State Management in State Space Model","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP","json":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP.json","graph_json":"https://pith.science/api/pith-number/BIPRLVCHD7QEJ2JCOLDCU3KQYP/graph.json","events_json":"https://pith.science/api/pith-number/BIPRLVCHD7QEJ2JCOLDCU3KQYP/events.json","paper":"https://pith.science/paper/BIPRLVCH"},"agent_actions":{"view_html":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP","download_json":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP.json","view_paper":"https://pith.science/paper/BIPRLVCH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01032&json=true","fetch_graph":"https://pith.science/api/pith-number/BIPRLVCHD7QEJ2JCOLDCU3KQYP/graph.json","fetch_events":"https://pith.science/api/pith-number/BIPRLVCHD7QEJ2JCOLDCU3KQYP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP/action/storage_attestation","attest_author":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP/action/author_attestation","sign_citation":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP/action/citation_signature","submit_replication":"https://pith.science/pith/BIPRLVCHD7QEJ2JCOLDCU3KQYP/action/replication_record"}},"created_at":"2026-07-05T08:26:43.532444+00:00","updated_at":"2026-07-05T08:26:43.532444+00:00"}