{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TWGTBIHVVEJSTNSBXKIYWCJPIG","short_pith_number":"pith:TWGTBIHV","schema_version":"1.0","canonical_sha256":"9d8d30a0f5a91329b641ba918b092f41b17a0813640f30e2e5971322ed662b1f","source":{"kind":"arxiv","id":"2407.10416","version":1},"attestation_state":"computed","paper":{"title":"SOFA: A Compute-Memory Optimized Sparsity Accelerator via Cross-Stage Coordinated Tiling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Chao Li, Huizheng Wang, Jiahao Fang, Jinxi Li, Qize Yang, Shouyi Yin, Sihan Guan, Xinru Tang, Yang Hu, Yang Wang, Yubin Qin, Zhiheng Yue","submitted_at":"2024-07-15T03:44:33Z","abstract_excerpt":"Benefiting from the self-attention mechanism, Transformer models have attained impressive contextual comprehension capabilities for lengthy texts. The requirements of high-throughput inference arise as the large language models (LLMs) become increasingly prevalent, which calls for large-scale token parallel processing (LTPP). However, existing dynamic sparse accelerators struggle to effectively handle LTPP, as they solely focus on separate stage optimization, and with most efforts confined to computational enhancements. By re-examining the end-to-end flow of dynamic sparse acceleration, we pin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10416","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AR","submitted_at":"2024-07-15T03:44:33Z","cross_cats_sorted":[],"title_canon_sha256":"5f7c61799935504eebd5b82b502b2a9ca59cfde200fd831ceac246cbc20d0c7c","abstract_canon_sha256":"30efed63ac7c2eb32aac98f31cba01457a7a57b43ab28ee9a3dd5c842216c770"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:59.319502Z","signature_b64":"Z8L46MYgCQ4bX/5LFmtfqkH7k6Rl0CJh4vvfj6fdYwjWkmtVQ0Yxgz08pIg+A5OedEfqdCB7cufnSelHlIfjDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d8d30a0f5a91329b641ba918b092f41b17a0813640f30e2e5971322ed662b1f","last_reissued_at":"2026-07-05T08:43:59.319084Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:59.319084Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SOFA: A Compute-Memory Optimized Sparsity Accelerator via Cross-Stage Coordinated Tiling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Chao Li, Huizheng Wang, Jiahao Fang, Jinxi Li, Qize Yang, Shouyi Yin, Sihan Guan, Xinru Tang, Yang Hu, Yang Wang, Yubin Qin, Zhiheng Yue","submitted_at":"2024-07-15T03:44:33Z","abstract_excerpt":"Benefiting from the self-attention mechanism, Transformer models have attained impressive contextual comprehension capabilities for lengthy texts. The requirements of high-throughput inference arise as the large language models (LLMs) become increasingly prevalent, which calls for large-scale token parallel processing (LTPP). However, existing dynamic sparse accelerators struggle to effectively handle LTPP, as they solely focus on separate stage optimization, and with most efforts confined to computational enhancements. By re-examining the end-to-end flow of dynamic sparse acceleration, we pin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10416","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10416/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10416","created_at":"2026-07-05T08:43:59.319141+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10416v1","created_at":"2026-07-05T08:43:59.319141+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10416","created_at":"2026-07-05T08:43:59.319141+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWGTBIHVVEJS","created_at":"2026-07-05T08:43:59.319141+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWGTBIHVVEJSTNSB","created_at":"2026-07-05T08:43:59.319141+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWGTBIHV","created_at":"2026-07-05T08:43:59.319141+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.27396","citing_title":"VitaLLM: A Versatile, Ultra-Compact Ternary LLM Accelerator with Dependency-Aware Scheduling","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG","json":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG.json","graph_json":"https://pith.science/api/pith-number/TWGTBIHVVEJSTNSBXKIYWCJPIG/graph.json","events_json":"https://pith.science/api/pith-number/TWGTBIHVVEJSTNSBXKIYWCJPIG/events.json","paper":"https://pith.science/paper/TWGTBIHV"},"agent_actions":{"view_html":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG","download_json":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG.json","view_paper":"https://pith.science/paper/TWGTBIHV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10416&json=true","fetch_graph":"https://pith.science/api/pith-number/TWGTBIHVVEJSTNSBXKIYWCJPIG/graph.json","fetch_events":"https://pith.science/api/pith-number/TWGTBIHVVEJSTNSBXKIYWCJPIG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG/action/storage_attestation","attest_author":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG/action/author_attestation","sign_citation":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG/action/citation_signature","submit_replication":"https://pith.science/pith/TWGTBIHVVEJSTNSBXKIYWCJPIG/action/replication_record"}},"created_at":"2026-07-05T08:43:59.319141+00:00","updated_at":"2026-07-05T08:43:59.319141+00:00"}