{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NR3H3PGQIR7FSO64VRPLGQQCDL","short_pith_number":"pith:NR3H3PGQ","schema_version":"1.0","canonical_sha256":"6c767dbcd0447e593bdcac5eb342021ae076910b60abd09c8dace6d1ca23aa60","source":{"kind":"arxiv","id":"2407.17686","version":1},"attestation_state":"computed","paper":{"title":"Transformers on Markov Data: Constant Depth Suffices","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashok Vardhan Makkuva, Kannan Ramchandran, Marco Bondaschi, Michael Gastpar, Nived Rajaraman","submitted_at":"2024-07-25T01:07:09Z","abstract_excerpt":"Attention-based transformers have been remarkably successful at modeling generative processes across various domains and modalities. In this paper, we study the behavior of transformers on data drawn from \\kth Markov processes, where the conditional distribution of the next symbol in a sequence depends on the previous $k$ symbols observed. We observe a surprising phenomenon empirically which contradicts previous findings: when trained for sufficiently long, a transformer with a fixed depth and $1$ head per layer is able to achieve low test loss on sequences drawn from \\kth Markov sources, even"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.17686","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-25T01:07:09Z","cross_cats_sorted":["cs.CL","cs.IT","math.IT","stat.ML"],"title_canon_sha256":"37aaf12b2de44b7f0824c41fc929ee94531e17d9f356a2c7cd453e7bc01df874","abstract_canon_sha256":"edafe713c6eb934da58f3efedb5d4fa49dfb8eaf31bd36ecb3403e384bafb791"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:19.773549Z","signature_b64":"p20iAiLoFOFLgdtCPvC3/ltZlHlGldU/m+xwXl7LxIZP+7DGqDVHiFByeqT7TjUPvZVpdF15Bz1PFGSSL/emCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c767dbcd0447e593bdcac5eb342021ae076910b60abd09c8dace6d1ca23aa60","last_reissued_at":"2026-07-05T08:48:19.772994Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:19.772994Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers on Markov Data: Constant Depth Suffices","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashok Vardhan Makkuva, Kannan Ramchandran, Marco Bondaschi, Michael Gastpar, Nived Rajaraman","submitted_at":"2024-07-25T01:07:09Z","abstract_excerpt":"Attention-based transformers have been remarkably successful at modeling generative processes across various domains and modalities. In this paper, we study the behavior of transformers on data drawn from \\kth Markov processes, where the conditional distribution of the next symbol in a sequence depends on the previous $k$ symbols observed. We observe a surprising phenomenon empirically which contradicts previous findings: when trained for sufficiently long, a transformer with a fixed depth and $1$ head per layer is able to achieve low test loss on sequences drawn from \\kth Markov sources, even"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.17686","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.17686/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.17686","created_at":"2026-07-05T08:48:19.773054+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.17686v1","created_at":"2026-07-05T08:48:19.773054+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.17686","created_at":"2026-07-05T08:48:19.773054+00:00"},{"alias_kind":"pith_short_12","alias_value":"NR3H3PGQIR7F","created_at":"2026-07-05T08:48:19.773054+00:00"},{"alias_kind":"pith_short_16","alias_value":"NR3H3PGQIR7FSO64","created_at":"2026-07-05T08:48:19.773054+00:00"},{"alias_kind":"pith_short_8","alias_value":"NR3H3PGQ","created_at":"2026-07-05T08:48:19.773054+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.10574","citing_title":"The LZ78 Source","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07298","citing_title":"Pre-trained Large Language Models Learn Hidden Markov Models In-context","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10848","citing_title":"Transformers Learn Latent Mixture Models In-Context via Mirror Descent","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL","json":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL.json","graph_json":"https://pith.science/api/pith-number/NR3H3PGQIR7FSO64VRPLGQQCDL/graph.json","events_json":"https://pith.science/api/pith-number/NR3H3PGQIR7FSO64VRPLGQQCDL/events.json","paper":"https://pith.science/paper/NR3H3PGQ"},"agent_actions":{"view_html":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL","download_json":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL.json","view_paper":"https://pith.science/paper/NR3H3PGQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.17686&json=true","fetch_graph":"https://pith.science/api/pith-number/NR3H3PGQIR7FSO64VRPLGQQCDL/graph.json","fetch_events":"https://pith.science/api/pith-number/NR3H3PGQIR7FSO64VRPLGQQCDL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL/action/storage_attestation","attest_author":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL/action/author_attestation","sign_citation":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL/action/citation_signature","submit_replication":"https://pith.science/pith/NR3H3PGQIR7FSO64VRPLGQQCDL/action/replication_record"}},"created_at":"2026-07-05T08:48:19.773054+00:00","updated_at":"2026-07-05T08:48:19.773054+00:00"}