{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:27KI5FWOBZIRAFPBZ3YDRKR6UZ","short_pith_number":"pith:27KI5FWO","schema_version":"1.0","canonical_sha256":"d7d48e96ce0e511015e1cef038aa3ea66c05e96595971b95ec27a6d24ab463cf","source":{"kind":"arxiv","id":"2404.05196","version":2},"attestation_state":"computed","paper":{"title":"HSViT: Horizontally Scalable Vision Transformer","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang-Tsun Li, Chee Peng Lim, Chenhao Xu, Douglas Creighton","submitted_at":"2024-04-08T04:53:29Z","abstract_excerpt":"Due to its deficiency in prior knowledge (inductive bias), Vision Transformer (ViT) requires pre-training on large-scale datasets to perform well. Moreover, the growing layers and parameters in ViT models impede their applicability to devices with limited computing resources. To mitigate the aforementioned challenges, this paper introduces a novel horizontally scalable vision transformer (HSViT) scheme. Specifically, a novel image-level feature embedding is introduced to ViT, where the preserved inductive bias allows the model to eliminate the need for pre-training while outperforming on small"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05196","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-08T04:53:29Z","cross_cats_sorted":[],"title_canon_sha256":"1e2db71aa9a93bb1fe0d606cd6cd0e72ce9db46ed474e23ac124bdf599b42fd1","abstract_canon_sha256":"e115ec19c10b0319668e5afc36c7b1fce5f805b35443cdcc18dd89be03ae0dd2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:13.506659Z","signature_b64":"6xkAlQO2kP++9nYqJDTVxsXcGdYXVIy03BioUkq+qsymUWmWmjw+kvPaEFo2GSBIZuuOnDbhY5qZhsw1/zc2DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d7d48e96ce0e511015e1cef038aa3ea66c05e96595971b95ec27a6d24ab463cf","last_reissued_at":"2026-07-05T08:44:13.506237Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:13.506237Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HSViT: Horizontally Scalable Vision Transformer","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang-Tsun Li, Chee Peng Lim, Chenhao Xu, Douglas Creighton","submitted_at":"2024-04-08T04:53:29Z","abstract_excerpt":"Due to its deficiency in prior knowledge (inductive bias), Vision Transformer (ViT) requires pre-training on large-scale datasets to perform well. Moreover, the growing layers and parameters in ViT models impede their applicability to devices with limited computing resources. To mitigate the aforementioned challenges, this paper introduces a novel horizontally scalable vision transformer (HSViT) scheme. Specifically, a novel image-level feature embedding is introduced to ViT, where the preserved inductive bias allows the model to eliminate the need for pre-training while outperforming on small"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05196","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05196/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05196","created_at":"2026-07-05T08:44:13.506291+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05196v2","created_at":"2026-07-05T08:44:13.506291+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05196","created_at":"2026-07-05T08:44:13.506291+00:00"},{"alias_kind":"pith_short_12","alias_value":"27KI5FWOBZIR","created_at":"2026-07-05T08:44:13.506291+00:00"},{"alias_kind":"pith_short_16","alias_value":"27KI5FWOBZIRAFPB","created_at":"2026-07-05T08:44:13.506291+00:00"},{"alias_kind":"pith_short_8","alias_value":"27KI5FWO","created_at":"2026-07-05T08:44:13.506291+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.14787","citing_title":"FOCUS: Fused Observation of Channels for Unveiling Spectra","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ","json":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ.json","graph_json":"https://pith.science/api/pith-number/27KI5FWOBZIRAFPBZ3YDRKR6UZ/graph.json","events_json":"https://pith.science/api/pith-number/27KI5FWOBZIRAFPBZ3YDRKR6UZ/events.json","paper":"https://pith.science/paper/27KI5FWO"},"agent_actions":{"view_html":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ","download_json":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ.json","view_paper":"https://pith.science/paper/27KI5FWO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05196&json=true","fetch_graph":"https://pith.science/api/pith-number/27KI5FWOBZIRAFPBZ3YDRKR6UZ/graph.json","fetch_events":"https://pith.science/api/pith-number/27KI5FWOBZIRAFPBZ3YDRKR6UZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ/action/storage_attestation","attest_author":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ/action/author_attestation","sign_citation":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ/action/citation_signature","submit_replication":"https://pith.science/pith/27KI5FWOBZIRAFPBZ3YDRKR6UZ/action/replication_record"}},"created_at":"2026-07-05T08:44:13.506291+00:00","updated_at":"2026-07-05T08:44:13.506291+00:00"}