{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:P23OU3WU5KNJ3BH6CY4GYEJ5HX","short_pith_number":"pith:P23OU3WU","schema_version":"1.0","canonical_sha256":"7eb6ea6ed4ea9a9d84fe16386c113d3dc850d01d28cd544e0aa20ffa551731c3","source":{"kind":"arxiv","id":"2309.08035","version":2},"attestation_state":"computed","paper":{"title":"Interpretability-Aware Vision Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengyin Li, Dongxiao Zhu, Prashant Khanduri, Yao Qiang","submitted_at":"2023-09-14T21:50:49Z","abstract_excerpt":"Vision Transformers (ViTs) have become prominent models for solving various vision tasks. However, the interpretability of ViTs has not kept pace with their promising performance. While there has been a surge of interest in developing {\\it post hoc} solutions to explain ViTs' outputs, these methods do not generalize to different downstream tasks and various transformer architectures. Furthermore, if ViTs are not properly trained with the given data and do not prioritize the region of interest, the {\\it post hoc} methods would be less effective. Instead of developing another {\\it post hoc} appr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.08035","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-09-14T21:50:49Z","cross_cats_sorted":[],"title_canon_sha256":"8c8a051234f7cdb938d1d81725bc4abd37ce22078107abfb7acdd8aff3cf2ac0","abstract_canon_sha256":"6d3eefed9be4ec93b74573cde4b282a66d2fa3d4f0b312b18ae5de4cec391a85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:19.722874Z","signature_b64":"+7biKkxIrBbF8disPp3E/OIk5C0HQ0qdaZAMbGc8qt23iqso9rZrdBdQ177YDJfjGxNIljyx/jXKjnWmKXKXAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7eb6ea6ed4ea9a9d84fe16386c113d3dc850d01d28cd544e0aa20ffa551731c3","last_reissued_at":"2026-07-05T10:57:19.722368Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:19.722368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpretability-Aware Vision Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengyin Li, Dongxiao Zhu, Prashant Khanduri, Yao Qiang","submitted_at":"2023-09-14T21:50:49Z","abstract_excerpt":"Vision Transformers (ViTs) have become prominent models for solving various vision tasks. However, the interpretability of ViTs has not kept pace with their promising performance. While there has been a surge of interest in developing {\\it post hoc} solutions to explain ViTs' outputs, these methods do not generalize to different downstream tasks and various transformer architectures. Furthermore, if ViTs are not properly trained with the given data and do not prioritize the region of interest, the {\\it post hoc} methods would be less effective. Instead of developing another {\\it post hoc} appr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.08035","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.08035/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.08035","created_at":"2026-07-05T10:57:19.722431+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.08035v2","created_at":"2026-07-05T10:57:19.722431+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.08035","created_at":"2026-07-05T10:57:19.722431+00:00"},{"alias_kind":"pith_short_12","alias_value":"P23OU3WU5KNJ","created_at":"2026-07-05T10:57:19.722431+00:00"},{"alias_kind":"pith_short_16","alias_value":"P23OU3WU5KNJ3BH6","created_at":"2026-07-05T10:57:19.722431+00:00"},{"alias_kind":"pith_short_8","alias_value":"P23OU3WU","created_at":"2026-07-05T10:57:19.722431+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14689","citing_title":"Are Candidate Models Really Needed for Active Learning?","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14925","citing_title":"Improving Sparse Autoencoder with Dynamic Attention","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX","json":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX.json","graph_json":"https://pith.science/api/pith-number/P23OU3WU5KNJ3BH6CY4GYEJ5HX/graph.json","events_json":"https://pith.science/api/pith-number/P23OU3WU5KNJ3BH6CY4GYEJ5HX/events.json","paper":"https://pith.science/paper/P23OU3WU"},"agent_actions":{"view_html":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX","download_json":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX.json","view_paper":"https://pith.science/paper/P23OU3WU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.08035&json=true","fetch_graph":"https://pith.science/api/pith-number/P23OU3WU5KNJ3BH6CY4GYEJ5HX/graph.json","fetch_events":"https://pith.science/api/pith-number/P23OU3WU5KNJ3BH6CY4GYEJ5HX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX/action/storage_attestation","attest_author":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX/action/author_attestation","sign_citation":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX/action/citation_signature","submit_replication":"https://pith.science/pith/P23OU3WU5KNJ3BH6CY4GYEJ5HX/action/replication_record"}},"created_at":"2026-07-05T10:57:19.722431+00:00","updated_at":"2026-07-05T10:57:19.722431+00:00"}