{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NQGM2LN7ZAW2WP4BCWOHTSUXED","short_pith_number":"pith:NQGM2LN7","schema_version":"1.0","canonical_sha256":"6c0ccd2dbfc82dab3f81159c79ca9720eb74c3b79f64c7ec3d12bce0943e4327","source":{"kind":"arxiv","id":"2303.03932","version":2},"attestation_state":"computed","paper":{"title":"FFT-based Dynamic Token Mixer for Vision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Masato Taki, Yuki Tatsunami","submitted_at":"2023-03-07T14:38:28Z","abstract_excerpt":"Multi-head-self-attention (MHSA)-equipped models have achieved notable performance in computer vision. Their computational complexity is proportional to quadratic numbers of pixels in input feature maps, resulting in slow processing, especially when dealing with high-resolution images. New types of token-mixer are proposed as an alternative to MHSA to circumvent this problem: an FFT-based token-mixer involves global operations similar to MHSA but with lower computational complexity. However, despite its attractive properties, the FFT-based token-mixer has not been carefully examined in terms o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.03932","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-07T14:38:28Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a95856c0ed5c96ec342ed6a1616df30874cec3623c0e67125c4961525de39727","abstract_canon_sha256":"daa5e8a511fa603ef500eeff5bc160ea5b7921017d4fa82a17495dcc18925e66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:24:48.712536Z","signature_b64":"6XjwEZu0vyU+fyGZv5FrPZQQKMQHI5eRnRUUNi5mS6YiKGOheS7m4CTUXJq+fRtGeBkwSmrDvg9Cs7mTUfw8Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c0ccd2dbfc82dab3f81159c79ca9720eb74c3b79f64c7ec3d12bce0943e4327","last_reissued_at":"2026-07-05T07:24:48.711992Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:24:48.711992Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FFT-based Dynamic Token Mixer for Vision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Masato Taki, Yuki Tatsunami","submitted_at":"2023-03-07T14:38:28Z","abstract_excerpt":"Multi-head-self-attention (MHSA)-equipped models have achieved notable performance in computer vision. Their computational complexity is proportional to quadratic numbers of pixels in input feature maps, resulting in slow processing, especially when dealing with high-resolution images. New types of token-mixer are proposed as an alternative to MHSA to circumvent this problem: an FFT-based token-mixer involves global operations similar to MHSA but with lower computational complexity. However, despite its attractive properties, the FFT-based token-mixer has not been carefully examined in terms o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.03932","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.03932/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.03932","created_at":"2026-07-05T07:24:48.712059+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.03932v2","created_at":"2026-07-05T07:24:48.712059+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.03932","created_at":"2026-07-05T07:24:48.712059+00:00"},{"alias_kind":"pith_short_12","alias_value":"NQGM2LN7ZAW2","created_at":"2026-07-05T07:24:48.712059+00:00"},{"alias_kind":"pith_short_16","alias_value":"NQGM2LN7ZAW2WP4B","created_at":"2026-07-05T07:24:48.712059+00:00"},{"alias_kind":"pith_short_8","alias_value":"NQGM2LN7","created_at":"2026-07-05T07:24:48.712059+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.19297","citing_title":"Enhancing Parameter-Efficient Fine-Tuning of Vision Transformers through Frequency-Based Adaptation","ref_index":40,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED","json":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED.json","graph_json":"https://pith.science/api/pith-number/NQGM2LN7ZAW2WP4BCWOHTSUXED/graph.json","events_json":"https://pith.science/api/pith-number/NQGM2LN7ZAW2WP4BCWOHTSUXED/events.json","paper":"https://pith.science/paper/NQGM2LN7"},"agent_actions":{"view_html":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED","download_json":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED.json","view_paper":"https://pith.science/paper/NQGM2LN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.03932&json=true","fetch_graph":"https://pith.science/api/pith-number/NQGM2LN7ZAW2WP4BCWOHTSUXED/graph.json","fetch_events":"https://pith.science/api/pith-number/NQGM2LN7ZAW2WP4BCWOHTSUXED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED/action/storage_attestation","attest_author":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED/action/author_attestation","sign_citation":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED/action/citation_signature","submit_replication":"https://pith.science/pith/NQGM2LN7ZAW2WP4BCWOHTSUXED/action/replication_record"}},"created_at":"2026-07-05T07:24:48.712059+00:00","updated_at":"2026-07-05T07:24:48.712059+00:00"}