{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LZDVRBAKD6TO6EUK6V4ZXP7RXQ","short_pith_number":"pith:LZDVRBAK","schema_version":"1.0","canonical_sha256":"5e4758840a1fa6ef128af5799bbff1bc2acec4332adba15cc55631f15a682be8","source":{"kind":"arxiv","id":"2104.11471","version":1},"attestation_state":"computed","paper":{"title":"tcFFT: Accelerating Half-Precision FFT through Tensor Cores","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Binrui Li, James Lin, Shenggan Cheng","submitted_at":"2021-04-23T08:42:01Z","abstract_excerpt":"Fast Fourier Transform (FFT) is an essential tool in scientific and engineering computation. The increasing demand for mixed-precision FFT has made it possible to utilize half-precision floating-point (FP16) arithmetic for faster speed and energy saving. Specializing in lower precision, NVIDIA Tensor Cores can deliver extremely high computation performance. However, the fixed computation pattern makes it hard to utilize the computing power of Tensor Cores in FFT. Therefore, we developed tcFFT to accelerate FFT with Tensor Cores. Our tcFFT supports batched 1D and 2D FFT of various sizes and it "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.11471","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2021-04-23T08:42:01Z","cross_cats_sorted":[],"title_canon_sha256":"9b213e48257f80838e59f2bfede00f9ac67dd1ca9b8c5e874422ee6618d1ab1d","abstract_canon_sha256":"f7024dc378eb4475f818f0b3bfe25ce42a19f9289d2919a470c427df47a1119d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:34:32.277514Z","signature_b64":"fQzql4R236nmFN5ZvtufJcvSDFbX8UAfHhLQ0pPBO9uPJcGOMYIayISO+hoZUVHzhT18S6bfUpBi426e0JqaDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e4758840a1fa6ef128af5799bbff1bc2acec4332adba15cc55631f15a682be8","last_reissued_at":"2026-07-05T02:34:32.276706Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:34:32.276706Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"tcFFT: Accelerating Half-Precision FFT through Tensor Cores","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Binrui Li, James Lin, Shenggan Cheng","submitted_at":"2021-04-23T08:42:01Z","abstract_excerpt":"Fast Fourier Transform (FFT) is an essential tool in scientific and engineering computation. The increasing demand for mixed-precision FFT has made it possible to utilize half-precision floating-point (FP16) arithmetic for faster speed and energy saving. Specializing in lower precision, NVIDIA Tensor Cores can deliver extremely high computation performance. However, the fixed computation pattern makes it hard to utilize the computing power of Tensor Cores in FFT. Therefore, we developed tcFFT to accelerate FFT with Tensor Cores. Our tcFFT supports batched 1D and 2D FFT of various sizes and it "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.11471","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.11471/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.11471","created_at":"2026-07-05T02:34:32.276785+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.11471v1","created_at":"2026-07-05T02:34:32.276785+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.11471","created_at":"2026-07-05T02:34:32.276785+00:00"},{"alias_kind":"pith_short_12","alias_value":"LZDVRBAKD6TO","created_at":"2026-07-05T02:34:32.276785+00:00"},{"alias_kind":"pith_short_16","alias_value":"LZDVRBAKD6TO6EUK","created_at":"2026-07-05T02:34:32.276785+00:00"},{"alias_kind":"pith_short_8","alias_value":"LZDVRBAK","created_at":"2026-07-05T02:34:32.276785+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28451","citing_title":"Range, Not Precision: Block-Floating-Point Half-Precision FFT and SAR Imaging on Apple Silicon","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02871","citing_title":"Floating-point consistent cross-verification methodology for reproducible and interoperable DDA solvers with fair benchmarking","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ","json":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ.json","graph_json":"https://pith.science/api/pith-number/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/graph.json","events_json":"https://pith.science/api/pith-number/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/events.json","paper":"https://pith.science/paper/LZDVRBAK"},"agent_actions":{"view_html":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ","download_json":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ.json","view_paper":"https://pith.science/paper/LZDVRBAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.11471&json=true","fetch_graph":"https://pith.science/api/pith-number/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/graph.json","fetch_events":"https://pith.science/api/pith-number/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/action/storage_attestation","attest_author":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/action/author_attestation","sign_citation":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/action/citation_signature","submit_replication":"https://pith.science/pith/LZDVRBAKD6TO6EUK6V4ZXP7RXQ/action/replication_record"}},"created_at":"2026-07-05T02:34:32.276785+00:00","updated_at":"2026-07-05T02:34:32.276785+00:00"}