{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IY6IXKJA4HFC75IO2HEWAYMRMG","short_pith_number":"pith:IY6IXKJA","schema_version":"1.0","canonical_sha256":"463c8ba920e1ca2ff50ed1c960619161b39e0db00730ab551657ee6210b082d0","source":{"kind":"arxiv","id":"2309.16108","version":4},"attestation_state":"computed","paper":{"title":"Channel Vision Transformers: An Image Is Worth 1 x 16 x 16 Words","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Srinivasan Sivanandan, Theofanis Karaletsos, Yujia Bao","submitted_at":"2023-09-28T02:20:59Z","abstract_excerpt":"Vision Transformer (ViT) has emerged as a powerful architecture in the realm of modern computer vision. However, its application in certain imaging fields, such as microscopy and satellite imaging, presents unique challenges. In these domains, images often contain multiple channels, each carrying semantically distinct and independent information. Furthermore, the model must demonstrate robustness to sparsity in input channels, as they may not be densely available during training or testing. In this paper, we propose a modification to the ViT architecture that enhances reasoning across the inpu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16108","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-09-28T02:20:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"98ca756a748f725efbb29969bbd6347c5ad6fb8d6bbb5f1dddfe75dbf9615c79","abstract_canon_sha256":"6808132b36d831236d620b786dcb3a0a745853de7e86951980a959dce0ee8080"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:09:44.888200Z","signature_b64":"WHqhbTIjWzeN9OMaWgfJYqgCN5gYhj/XC90Pv75kRXdjmTHiOyvsBPngoILwuvXbsb33X1w6k/CaZnb47aTHBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"463c8ba920e1ca2ff50ed1c960619161b39e0db00730ab551657ee6210b082d0","last_reissued_at":"2026-07-05T08:09:44.887730Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:09:44.887730Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Channel Vision Transformers: An Image Is Worth 1 x 16 x 16 Words","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Srinivasan Sivanandan, Theofanis Karaletsos, Yujia Bao","submitted_at":"2023-09-28T02:20:59Z","abstract_excerpt":"Vision Transformer (ViT) has emerged as a powerful architecture in the realm of modern computer vision. However, its application in certain imaging fields, such as microscopy and satellite imaging, presents unique challenges. In these domains, images often contain multiple channels, each carrying semantically distinct and independent information. Furthermore, the model must demonstrate robustness to sparsity in input channels, as they may not be densely available during training or testing. In this paper, we propose a modification to the ViT architecture that enhances reasoning across the inpu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16108","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16108","created_at":"2026-07-05T08:09:44.887787+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16108v4","created_at":"2026-07-05T08:09:44.887787+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16108","created_at":"2026-07-05T08:09:44.887787+00:00"},{"alias_kind":"pith_short_12","alias_value":"IY6IXKJA4HFC","created_at":"2026-07-05T08:09:44.887787+00:00"},{"alias_kind":"pith_short_16","alias_value":"IY6IXKJA4HFC75IO","created_at":"2026-07-05T08:09:44.887787+00:00"},{"alias_kind":"pith_short_8","alias_value":"IY6IXKJA","created_at":"2026-07-05T08:09:44.887787+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.09803","citing_title":"LatticeVision: Image to Image Networks for Modeling Non-Stationary Spatial Data","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18541","citing_title":"LESSViT: Robust Hyperspectral Representation Learning under Spectral Configuration Shift","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10970","citing_title":"Using Deep Learning Models Pretrained by Self-Supervised Learning for Protein Localization","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG","json":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG.json","graph_json":"https://pith.science/api/pith-number/IY6IXKJA4HFC75IO2HEWAYMRMG/graph.json","events_json":"https://pith.science/api/pith-number/IY6IXKJA4HFC75IO2HEWAYMRMG/events.json","paper":"https://pith.science/paper/IY6IXKJA"},"agent_actions":{"view_html":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG","download_json":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG.json","view_paper":"https://pith.science/paper/IY6IXKJA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16108&json=true","fetch_graph":"https://pith.science/api/pith-number/IY6IXKJA4HFC75IO2HEWAYMRMG/graph.json","fetch_events":"https://pith.science/api/pith-number/IY6IXKJA4HFC75IO2HEWAYMRMG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG/action/storage_attestation","attest_author":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG/action/author_attestation","sign_citation":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG/action/citation_signature","submit_replication":"https://pith.science/pith/IY6IXKJA4HFC75IO2HEWAYMRMG/action/replication_record"}},"created_at":"2026-07-05T08:09:44.887787+00:00","updated_at":"2026-07-05T08:09:44.887787+00:00"}