{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QT5EOHB6WO5OGSASRDHVX5KWCO","short_pith_number":"pith:QT5EOHB6","schema_version":"1.0","canonical_sha256":"84fa471c3eb3bae3481288cf5bf55613b0fd176fa298831c506a540648553f33","source":{"kind":"arxiv","id":"2503.02891","version":3},"attestation_state":"computed","paper":{"title":"Vision Transformers on the Edge: A Comprehensive Survey of Model Compression and Acceleration Strategies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AR"],"primary_cat":"cs.CV","authors_text":"Lanyu Xu, Shaibal Saha","submitted_at":"2025-02-26T22:34:44Z","abstract_excerpt":"In recent years, vision transformers (ViTs) have emerged as powerful and promising techniques for computer vision tasks such as image classification, object detection, and segmentation. Unlike convolutional neural networks (CNNs), which rely on hierarchical feature extraction, ViTs treat images as sequences of patches and leverage self-attention mechanisms. However, their high computational complexity and memory demands pose significant challenges for deployment on resource-constrained edge devices. To address these limitations, extensive research has focused on model compression techniques an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.02891","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-26T22:34:44Z","cross_cats_sorted":["cs.AR"],"title_canon_sha256":"63952da8e5d95c4f767f94577d840ba8bdcd9f5b17a4255f8834906e8f194c8a","abstract_canon_sha256":"42e24dae3aaa4de59370eb9c914f53d9e28cc0c7223fa72606e52642f9168c06"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:46.817423Z","signature_b64":"KMSrLkQQneQImzril17vS+Xd/BzRWS+k/6uGdZ3GTehQS6YucYCdE1Q5MavTX+5Pwbu48ROGs3Y6GCT/168NBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84fa471c3eb3bae3481288cf5bf55613b0fd176fa298831c506a540648553f33","last_reissued_at":"2026-07-05T11:04:46.816938Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:46.816938Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision Transformers on the Edge: A Comprehensive Survey of Model Compression and Acceleration Strategies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AR"],"primary_cat":"cs.CV","authors_text":"Lanyu Xu, Shaibal Saha","submitted_at":"2025-02-26T22:34:44Z","abstract_excerpt":"In recent years, vision transformers (ViTs) have emerged as powerful and promising techniques for computer vision tasks such as image classification, object detection, and segmentation. Unlike convolutional neural networks (CNNs), which rely on hierarchical feature extraction, ViTs treat images as sequences of patches and leverage self-attention mechanisms. However, their high computational complexity and memory demands pose significant challenges for deployment on resource-constrained edge devices. To address these limitations, extensive research has focused on model compression techniques an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.02891","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.02891/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.02891","created_at":"2026-07-05T11:04:46.817004+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.02891v3","created_at":"2026-07-05T11:04:46.817004+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.02891","created_at":"2026-07-05T11:04:46.817004+00:00"},{"alias_kind":"pith_short_12","alias_value":"QT5EOHB6WO5O","created_at":"2026-07-05T11:04:46.817004+00:00"},{"alias_kind":"pith_short_16","alias_value":"QT5EOHB6WO5OGSAS","created_at":"2026-07-05T11:04:46.817004+00:00"},{"alias_kind":"pith_short_8","alias_value":"QT5EOHB6","created_at":"2026-07-05T11:04:46.817004+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.21364","citing_title":"Evaluating Deep Learning Models for African Wildlife Image Classification: From DenseNet to Vision Transformers","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO","json":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO.json","graph_json":"https://pith.science/api/pith-number/QT5EOHB6WO5OGSASRDHVX5KWCO/graph.json","events_json":"https://pith.science/api/pith-number/QT5EOHB6WO5OGSASRDHVX5KWCO/events.json","paper":"https://pith.science/paper/QT5EOHB6"},"agent_actions":{"view_html":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO","download_json":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO.json","view_paper":"https://pith.science/paper/QT5EOHB6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.02891&json=true","fetch_graph":"https://pith.science/api/pith-number/QT5EOHB6WO5OGSASRDHVX5KWCO/graph.json","fetch_events":"https://pith.science/api/pith-number/QT5EOHB6WO5OGSASRDHVX5KWCO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO/action/storage_attestation","attest_author":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO/action/author_attestation","sign_citation":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO/action/citation_signature","submit_replication":"https://pith.science/pith/QT5EOHB6WO5OGSASRDHVX5KWCO/action/replication_record"}},"created_at":"2026-07-05T11:04:46.817004+00:00","updated_at":"2026-07-05T11:04:46.817004+00:00"}