{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XS6BJUJCFUOPHWREJGAFQNROEJ","short_pith_number":"pith:XS6BJUJC","schema_version":"1.0","canonical_sha256":"bcbc14d1222d1cf3da24498058362e227c1aa22fc467fe42c330cc5737a53de5","source":{"kind":"arxiv","id":"2306.02896","version":2},"attestation_state":"computed","paper":{"title":"Representational Strengths and Limitations of Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Clayton Sanford, Daniel Hsu, Matus Telgarsky","submitted_at":"2023-06-05T14:05:04Z","abstract_excerpt":"Attention layers, as commonly used in transformers, form the backbone of modern deep learning, yet there is no mathematical description of their benefits and deficiencies as compared with other architectures. In this work we establish both positive and negative results on the representation power of attention layers, with a focus on intrinsic complexity parameters such as width, depth, and embedding dimension. On the positive side, we present a sparse averaging task, where recurrent networks and feedforward networks all have complexity scaling polynomially in the input size, whereas transforme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.02896","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-05T14:05:04Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"dc8901b6c83a3546453feae04730b4f1e114caf4684f97da322ebbdc2deae28b","abstract_canon_sha256":"e964c9ba7bb468eb73828fc069393a28be532e71095fb74659718fc10ed28608"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:13:23.227918Z","signature_b64":"cZSL6wgo6Rqc6jjM3Pw1eWXqwRqbec6JnrZ+IVTtdl0WlasiRxHwd8uIGBX/IqsiGi9pZA0GvRylHKOrSx2TDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bcbc14d1222d1cf3da24498058362e227c1aa22fc467fe42c330cc5737a53de5","last_reissued_at":"2026-07-05T07:13:23.227425Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:13:23.227425Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Representational Strengths and Limitations of Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Clayton Sanford, Daniel Hsu, Matus Telgarsky","submitted_at":"2023-06-05T14:05:04Z","abstract_excerpt":"Attention layers, as commonly used in transformers, form the backbone of modern deep learning, yet there is no mathematical description of their benefits and deficiencies as compared with other architectures. In this work we establish both positive and negative results on the representation power of attention layers, with a focus on intrinsic complexity parameters such as width, depth, and embedding dimension. On the positive side, we present a sparse averaging task, where recurrent networks and feedforward networks all have complexity scaling polynomially in the input size, whereas transforme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02896","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.02896/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.02896","created_at":"2026-07-05T07:13:23.227478+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.02896v2","created_at":"2026-07-05T07:13:23.227478+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02896","created_at":"2026-07-05T07:13:23.227478+00:00"},{"alias_kind":"pith_short_12","alias_value":"XS6BJUJCFUOP","created_at":"2026-07-05T07:13:23.227478+00:00"},{"alias_kind":"pith_short_16","alias_value":"XS6BJUJCFUOPHWRE","created_at":"2026-07-05T07:13:23.227478+00:00"},{"alias_kind":"pith_short_8","alias_value":"XS6BJUJC","created_at":"2026-07-05T07:13:23.227478+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26749","citing_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21848","citing_title":"Keyless Attention: Value-Space Routing and Value-Only Caching for Efficient Transformers","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20988","citing_title":"A Sharper Picture of Generalization in Transformers","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20988","citing_title":"A Sharper Picture of Generalization in Transformers","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17935","citing_title":"How Much Cache Does Reasoning Need? Depth-Cache Tradeoffs in KV-Compressed Transformers","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ","json":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ.json","graph_json":"https://pith.science/api/pith-number/XS6BJUJCFUOPHWREJGAFQNROEJ/graph.json","events_json":"https://pith.science/api/pith-number/XS6BJUJCFUOPHWREJGAFQNROEJ/events.json","paper":"https://pith.science/paper/XS6BJUJC"},"agent_actions":{"view_html":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ","download_json":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ.json","view_paper":"https://pith.science/paper/XS6BJUJC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.02896&json=true","fetch_graph":"https://pith.science/api/pith-number/XS6BJUJCFUOPHWREJGAFQNROEJ/graph.json","fetch_events":"https://pith.science/api/pith-number/XS6BJUJCFUOPHWREJGAFQNROEJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ/action/storage_attestation","attest_author":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ/action/author_attestation","sign_citation":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ/action/citation_signature","submit_replication":"https://pith.science/pith/XS6BJUJCFUOPHWREJGAFQNROEJ/action/replication_record"}},"created_at":"2026-07-05T07:13:23.227478+00:00","updated_at":"2026-07-05T07:13:23.227478+00:00"}