{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IGWKGQP2ZKOWGHHPCNM74WOH6U","short_pith_number":"pith:IGWKGQP2","schema_version":"1.0","canonical_sha256":"41aca341faca9d631cef1359fe59c7f52fb0de35bc092260b88a28ee94f8528a","source":{"kind":"arxiv","id":"2302.14017","version":1},"attestation_state":"computed","paper":{"title":"Full Stack Optimization of Transformer Inference: a Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Amir Gholami, Coleman Hooper, Grace Dinh, Hasan Genc, Kurt Keutzer, Michael W. Mahoney, Minwoo Kang, Qijing Huang, Ruohan Yan, Sehoon Kim, Thanakul Wattanawong, Yakun Sophia Shao","submitted_at":"2023-02-27T18:18:13Z","abstract_excerpt":"Recent advances in state-of-the-art DNN architecture design have been moving toward Transformer models. These models achieve superior accuracy across a wide range of applications. This trend has been consistent over the past several years since Transformer models were originally introduced. However, the amount of compute and bandwidth required for inference of recent Transformer models is growing at a significant rate, and this has made their deployment in latency-sensitive applications challenging. As such, there has been an increased focus on making Transformer models more efficient, with me"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.14017","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-27T18:18:13Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9630c34ca40a18bd04f16a624ebdc8dc336c1a0c6264427f89021fe3662f2a68","abstract_canon_sha256":"8f1feb637bd3768778af8ff2eba73e9d0f14a52c70d19ec3ded3bbdd9b29ef4d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:43:21.319207Z","signature_b64":"dkqlP1cFz8xu0YpJSxEij/Nhodfl+OvYT2WKURm+88R9BkXtWAAPeNiO2t/WyJlJdc/EFpvI2Uy8/JL5W4JiBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"41aca341faca9d631cef1359fe59c7f52fb0de35bc092260b88a28ee94f8528a","last_reissued_at":"2026-07-05T06:43:21.318793Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:43:21.318793Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Full Stack Optimization of Transformer Inference: a Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Amir Gholami, Coleman Hooper, Grace Dinh, Hasan Genc, Kurt Keutzer, Michael W. Mahoney, Minwoo Kang, Qijing Huang, Ruohan Yan, Sehoon Kim, Thanakul Wattanawong, Yakun Sophia Shao","submitted_at":"2023-02-27T18:18:13Z","abstract_excerpt":"Recent advances in state-of-the-art DNN architecture design have been moving toward Transformer models. These models achieve superior accuracy across a wide range of applications. This trend has been consistent over the past several years since Transformer models were originally introduced. However, the amount of compute and bandwidth required for inference of recent Transformer models is growing at a significant rate, and this has made their deployment in latency-sensitive applications challenging. As such, there has been an increased focus on making Transformer models more efficient, with me"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.14017","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.14017/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.14017","created_at":"2026-07-05T06:43:21.318847+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.14017v1","created_at":"2026-07-05T06:43:21.318847+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.14017","created_at":"2026-07-05T06:43:21.318847+00:00"},{"alias_kind":"pith_short_12","alias_value":"IGWKGQP2ZKOW","created_at":"2026-07-05T06:43:21.318847+00:00"},{"alias_kind":"pith_short_16","alias_value":"IGWKGQP2ZKOWGHHP","created_at":"2026-07-05T06:43:21.318847+00:00"},{"alias_kind":"pith_short_8","alias_value":"IGWKGQP2","created_at":"2026-07-05T06:43:21.318847+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16106","citing_title":"Edge-Inference Governors Need Memory-Clock State","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06252","citing_title":"D-Legion: A Scalable Many-Core Architecture for Accelerating Matrix Multiplication in Quantized LLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11512","citing_title":"EdgeCIM: A Hardware-Software Co-Design for CIM-Based Acceleration of Small Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09048","citing_title":"Watt Counts: Energy-Aware Benchmark for Sustainable LLM Inference on Heterogeneous GPU Architectures","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15944","citing_title":"CIMple: Standard-cell SRAM-based CIM with LUT-based split softmax for attention acceleration","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U","json":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U.json","graph_json":"https://pith.science/api/pith-number/IGWKGQP2ZKOWGHHPCNM74WOH6U/graph.json","events_json":"https://pith.science/api/pith-number/IGWKGQP2ZKOWGHHPCNM74WOH6U/events.json","paper":"https://pith.science/paper/IGWKGQP2"},"agent_actions":{"view_html":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U","download_json":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U.json","view_paper":"https://pith.science/paper/IGWKGQP2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.14017&json=true","fetch_graph":"https://pith.science/api/pith-number/IGWKGQP2ZKOWGHHPCNM74WOH6U/graph.json","fetch_events":"https://pith.science/api/pith-number/IGWKGQP2ZKOWGHHPCNM74WOH6U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U/action/storage_attestation","attest_author":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U/action/author_attestation","sign_citation":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U/action/citation_signature","submit_replication":"https://pith.science/pith/IGWKGQP2ZKOWGHHPCNM74WOH6U/action/replication_record"}},"created_at":"2026-07-05T06:43:21.318847+00:00","updated_at":"2026-07-05T06:43:21.318847+00:00"}