{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZLZXOOELXRFQDF2QC5RAIKTMGY","short_pith_number":"pith:ZLZXOOEL","schema_version":"1.0","canonical_sha256":"caf377388bbc4b0197501762042a6c3632eabaa9f1b8a2c0f53f5db11ea228aa","source":{"kind":"arxiv","id":"2310.04134","version":2},"attestation_state":"computed","paper":{"title":"TiC: Exploring Vision Transformer in Convolution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoyi Xiong, Jiang Bian, Qingzhong Wang, Song Zhang","submitted_at":"2023-10-06T10:16:26Z","abstract_excerpt":"While models derived from Vision Transformers (ViTs) have been phonemically surging, pre-trained models cannot seamlessly adapt to arbitrary resolution images without altering the architecture and configuration, such as sampling the positional encoding, limiting their flexibility for various vision tasks. For instance, the Segment Anything Model (SAM) based on ViT-Huge requires all input images to be resized to 1024$\\times$1024. To overcome this limitation, we propose the Multi-Head Self-Attention Convolution (MSA-Conv) that incorporates Self-Attention within generalized convolutions, includin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.04134","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-06T10:16:26Z","cross_cats_sorted":[],"title_canon_sha256":"995798c5f06a051d5cdcd790a53a8ebb86caeabe4e19bc20d117a2a9c0caba1b","abstract_canon_sha256":"00f765ceedbba0e4ec749283c9a8973a6db2c1cae34cd42ff06c70c1df8c0c32"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:12.553333Z","signature_b64":"uXMsZ61L/DBL0omUWmUqRGVBXC2r8CosjlCEVaY/JZTjSSt5Ag/uZGil4PF2p3H3EOBzqpx1fdRnX/o0w5jMAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"caf377388bbc4b0197501762042a6c3632eabaa9f1b8a2c0f53f5db11ea228aa","last_reissued_at":"2026-07-05T08:23:12.552878Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:12.552878Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TiC: Exploring Vision Transformer in Convolution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoyi Xiong, Jiang Bian, Qingzhong Wang, Song Zhang","submitted_at":"2023-10-06T10:16:26Z","abstract_excerpt":"While models derived from Vision Transformers (ViTs) have been phonemically surging, pre-trained models cannot seamlessly adapt to arbitrary resolution images without altering the architecture and configuration, such as sampling the positional encoding, limiting their flexibility for various vision tasks. For instance, the Segment Anything Model (SAM) based on ViT-Huge requires all input images to be resized to 1024$\\times$1024. To overcome this limitation, we propose the Multi-Head Self-Attention Convolution (MSA-Conv) that incorporates Self-Attention within generalized convolutions, includin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.04134","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.04134/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.04134","created_at":"2026-07-05T08:23:12.552942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.04134v2","created_at":"2026-07-05T08:23:12.552942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.04134","created_at":"2026-07-05T08:23:12.552942+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZLZXOOELXRFQ","created_at":"2026-07-05T08:23:12.552942+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZLZXOOELXRFQDF2Q","created_at":"2026-07-05T08:23:12.552942+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZLZXOOEL","created_at":"2026-07-05T08:23:12.552942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY","json":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY.json","graph_json":"https://pith.science/api/pith-number/ZLZXOOELXRFQDF2QC5RAIKTMGY/graph.json","events_json":"https://pith.science/api/pith-number/ZLZXOOELXRFQDF2QC5RAIKTMGY/events.json","paper":"https://pith.science/paper/ZLZXOOEL"},"agent_actions":{"view_html":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY","download_json":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY.json","view_paper":"https://pith.science/paper/ZLZXOOEL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.04134&json=true","fetch_graph":"https://pith.science/api/pith-number/ZLZXOOELXRFQDF2QC5RAIKTMGY/graph.json","fetch_events":"https://pith.science/api/pith-number/ZLZXOOELXRFQDF2QC5RAIKTMGY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY/action/storage_attestation","attest_author":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY/action/author_attestation","sign_citation":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY/action/citation_signature","submit_replication":"https://pith.science/pith/ZLZXOOELXRFQDF2QC5RAIKTMGY/action/replication_record"}},"created_at":"2026-07-05T08:23:12.552942+00:00","updated_at":"2026-07-05T08:23:12.552942+00:00"}