{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7TMGQAUVBFIRZ5TN4RHXWV4DDC","short_pith_number":"pith:7TMGQAUV","schema_version":"1.0","canonical_sha256":"fcd868029509511cf66de44f7b57831899d89a207e5fb26374e32dc0bcc2427b","source":{"kind":"arxiv","id":"2302.02108","version":2},"attestation_state":"computed","paper":{"title":"Knowledge Distillation in Vision Transformers: A Critical Review","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Brejesh Lall, Gousia Habib, Tausifa Jan Saleem","submitted_at":"2023-02-04T06:30:57Z","abstract_excerpt":"In Natural Language Processing (NLP), Transformers have already revolutionized the field by utilizing an attention-based encoder-decoder model. Recently, some pioneering works have employed Transformer-like architectures in Computer Vision (CV) and they have reported outstanding performance of these architectures in tasks such as image classification, object detection, and semantic segmentation. Vision Transformers (ViTs) have demonstrated impressive performance improvements over Convolutional Neural Networks (CNNs) due to their competitive modelling capabilities. However, these architectures "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.02108","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-02-04T06:30:57Z","cross_cats_sorted":[],"title_canon_sha256":"fb67e09cb9243e1f3bdfcb5d298e3199cca4ac41a31a16f4fcff755c435b81f9","abstract_canon_sha256":"c48c6b01def95a993029b867422e0899ef59b52c9f2efb5a94e52263ee2ce701"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:43:30.260532Z","signature_b64":"LjNwXjM9ILmUJOBsQ0zWfH6ua0vM1m0EYqEWzoIJs6ANGjmNfH9zPzNzNdexT3hdW/x2GG7ztMtPuH60EZiECw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fcd868029509511cf66de44f7b57831899d89a207e5fb26374e32dc0bcc2427b","last_reissued_at":"2026-07-05T07:43:30.259975Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:43:30.259975Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Knowledge Distillation in Vision Transformers: A Critical Review","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Brejesh Lall, Gousia Habib, Tausifa Jan Saleem","submitted_at":"2023-02-04T06:30:57Z","abstract_excerpt":"In Natural Language Processing (NLP), Transformers have already revolutionized the field by utilizing an attention-based encoder-decoder model. Recently, some pioneering works have employed Transformer-like architectures in Computer Vision (CV) and they have reported outstanding performance of these architectures in tasks such as image classification, object detection, and semantic segmentation. Vision Transformers (ViTs) have demonstrated impressive performance improvements over Convolutional Neural Networks (CNNs) due to their competitive modelling capabilities. However, these architectures "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.02108","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.02108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.02108","created_at":"2026-07-05T07:43:30.260048+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.02108v2","created_at":"2026-07-05T07:43:30.260048+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.02108","created_at":"2026-07-05T07:43:30.260048+00:00"},{"alias_kind":"pith_short_12","alias_value":"7TMGQAUVBFIR","created_at":"2026-07-05T07:43:30.260048+00:00"},{"alias_kind":"pith_short_16","alias_value":"7TMGQAUVBFIRZ5TN","created_at":"2026-07-05T07:43:30.260048+00:00"},{"alias_kind":"pith_short_8","alias_value":"7TMGQAUV","created_at":"2026-07-05T07:43:30.260048+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09719","citing_title":"Distilling 3D Spatial Reasoning into a Lightweight Vision-Language Model with CoT","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC","json":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC.json","graph_json":"https://pith.science/api/pith-number/7TMGQAUVBFIRZ5TN4RHXWV4DDC/graph.json","events_json":"https://pith.science/api/pith-number/7TMGQAUVBFIRZ5TN4RHXWV4DDC/events.json","paper":"https://pith.science/paper/7TMGQAUV"},"agent_actions":{"view_html":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC","download_json":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC.json","view_paper":"https://pith.science/paper/7TMGQAUV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.02108&json=true","fetch_graph":"https://pith.science/api/pith-number/7TMGQAUVBFIRZ5TN4RHXWV4DDC/graph.json","fetch_events":"https://pith.science/api/pith-number/7TMGQAUVBFIRZ5TN4RHXWV4DDC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC/action/storage_attestation","attest_author":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC/action/author_attestation","sign_citation":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC/action/citation_signature","submit_replication":"https://pith.science/pith/7TMGQAUVBFIRZ5TN4RHXWV4DDC/action/replication_record"}},"created_at":"2026-07-05T07:43:30.260048+00:00","updated_at":"2026-07-05T07:43:30.260048+00:00"}