{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:KPT7CPDA45ZPTGNFF65LR72KCT","short_pith_number":"pith:KPT7CPDA","schema_version":"1.0","canonical_sha256":"53e7f13c60e772f999a52fbab8ff4a14d708d34c116a9ab11b665c656448049d","source":{"kind":"arxiv","id":"2006.03274","version":1},"attestation_state":"computed","paper":{"title":"GMAT: Global Memory Augmentation for Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ankit Gupta, Jonathan Berant","submitted_at":"2020-06-05T07:50:40Z","abstract_excerpt":"Transformer-based models have become ubiquitous in natural language processing thanks to their large capacity, innate parallelism and high performance. The contextualizing component of a Transformer block is the $\\textit{pairwise dot-product}$ attention that has a large $\\Omega(L^2)$ memory requirement for length $L$ sequences, limiting its ability to process long documents. This has been the subject of substantial interest recently, where multiple approximations were proposed to reduce the quadratic memory requirement using sparse attention matrices. In this work, we propose to augment sparse"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.03274","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-05T07:50:40Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"b71276324c0a7880487ca13c4c01abfc4e92d4d722fa12d49ed753513425bda2","abstract_canon_sha256":"79b972cc5b10479c8415fbba65e8f42d5051d9d6017d2c9496f454ae88328aaa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:08:16.617065Z","signature_b64":"o0BpLvdZ8dvovRqA98OEjd9qclqdzwq7MezipZ+ArfgE17fTr2zRawEVRNR7xDMROiICboYjTDUpDUuHluNGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53e7f13c60e772f999a52fbab8ff4a14d708d34c116a9ab11b665c656448049d","last_reissued_at":"2026-07-05T01:08:16.616574Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:08:16.616574Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GMAT: Global Memory Augmentation for Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ankit Gupta, Jonathan Berant","submitted_at":"2020-06-05T07:50:40Z","abstract_excerpt":"Transformer-based models have become ubiquitous in natural language processing thanks to their large capacity, innate parallelism and high performance. The contextualizing component of a Transformer block is the $\\textit{pairwise dot-product}$ attention that has a large $\\Omega(L^2)$ memory requirement for length $L$ sequences, limiting its ability to process long documents. This has been the subject of substantial interest recently, where multiple approximations were proposed to reduce the quadratic memory requirement using sparse attention matrices. In this work, we propose to augment sparse"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.03274","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.03274/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.03274","created_at":"2026-07-05T01:08:16.616632+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.03274v1","created_at":"2026-07-05T01:08:16.616632+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.03274","created_at":"2026-07-05T01:08:16.616632+00:00"},{"alias_kind":"pith_short_12","alias_value":"KPT7CPDA45ZP","created_at":"2026-07-05T01:08:16.616632+00:00"},{"alias_kind":"pith_short_16","alias_value":"KPT7CPDA45ZPTGNF","created_at":"2026-07-05T01:08:16.616632+00:00"},{"alias_kind":"pith_short_8","alias_value":"KPT7CPDA","created_at":"2026-07-05T01:08:16.616632+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2308.08089","citing_title":"DragNUWA: Fine-grained Control in Video Generation by Integrating Text, Image, and Trajectory","ref_index":277,"is_internal_anchor":false},{"citing_arxiv_id":"2208.04933","citing_title":"Simplified State Space Layers for Sequence Modeling","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2004.05150","citing_title":"Longformer: The Long-Document Transformer","ref_index":94,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT","json":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT.json","graph_json":"https://pith.science/api/pith-number/KPT7CPDA45ZPTGNFF65LR72KCT/graph.json","events_json":"https://pith.science/api/pith-number/KPT7CPDA45ZPTGNFF65LR72KCT/events.json","paper":"https://pith.science/paper/KPT7CPDA"},"agent_actions":{"view_html":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT","download_json":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT.json","view_paper":"https://pith.science/paper/KPT7CPDA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.03274&json=true","fetch_graph":"https://pith.science/api/pith-number/KPT7CPDA45ZPTGNFF65LR72KCT/graph.json","fetch_events":"https://pith.science/api/pith-number/KPT7CPDA45ZPTGNFF65LR72KCT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT/action/storage_attestation","attest_author":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT/action/author_attestation","sign_citation":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT/action/citation_signature","submit_replication":"https://pith.science/pith/KPT7CPDA45ZPTGNFF65LR72KCT/action/replication_record"}},"created_at":"2026-07-05T01:08:16.616632+00:00","updated_at":"2026-07-05T01:08:16.616632+00:00"}