{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SSRBH6UKA4BQ3QJVGYFROEDXD5","short_pith_number":"pith:SSRBH6UK","schema_version":"1.0","canonical_sha256":"94a213fa8a07030dc135360b1710771f5b71ad2551fa3dec83a85892648dc3da","source":{"kind":"arxiv","id":"2312.03558","version":1},"attestation_state":"computed","paper":{"title":"When an Image is Worth 1,024 x 1,024 Words: A Case Study in Computational Pathology","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Furu Wei, Hanwen Xu, Hoifung Poon, Jiayu Ding, Naoto Usuyama, Shuming Ma, Wenhui Wang","submitted_at":"2023-12-06T15:40:28Z","abstract_excerpt":"This technical report presents LongViT, a vision Transformer that can process gigapixel images in an end-to-end manner. Specifically, we split the gigapixel image into a sequence of millions of patches and project them linearly into embeddings. LongNet is then employed to model the extremely long sequence, generating representations that capture both short-range and long-range dependencies. The linear computation complexity of LongNet, along with its distributed algorithm, enables us to overcome the constraints of both computation and memory. We apply LongViT in the field of computational path"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.03558","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-06T15:40:28Z","cross_cats_sorted":[],"title_canon_sha256":"a4b5172e821275f2af69d2447de41aa0627d611c45ac23d90a3df83db2b6740b","abstract_canon_sha256":"6a07e0abe810c659a0ffe43e2f61e3da4bb31a81c41f7eef3b7702ed6c396b25"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:21:06.233617Z","signature_b64":"X8uxaO7whp+03z5z3iLRBE5CXzqRCAhdW0V/UwifoI+2HNx77CwWnKnfzSn7sRm5FWmrFLxKbkeYuZt6uDpZBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"94a213fa8a07030dc135360b1710771f5b71ad2551fa3dec83a85892648dc3da","last_reissued_at":"2026-07-05T07:21:06.233210Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:21:06.233210Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When an Image is Worth 1,024 x 1,024 Words: A Case Study in Computational Pathology","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Furu Wei, Hanwen Xu, Hoifung Poon, Jiayu Ding, Naoto Usuyama, Shuming Ma, Wenhui Wang","submitted_at":"2023-12-06T15:40:28Z","abstract_excerpt":"This technical report presents LongViT, a vision Transformer that can process gigapixel images in an end-to-end manner. Specifically, we split the gigapixel image into a sequence of millions of patches and project them linearly into embeddings. LongNet is then employed to model the extremely long sequence, generating representations that capture both short-range and long-range dependencies. The linear computation complexity of LongNet, along with its distributed algorithm, enables us to overcome the constraints of both computation and memory. We apply LongViT in the field of computational path"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03558","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.03558/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.03558","created_at":"2026-07-05T07:21:06.233266+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.03558v1","created_at":"2026-07-05T07:21:06.233266+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03558","created_at":"2026-07-05T07:21:06.233266+00:00"},{"alias_kind":"pith_short_12","alias_value":"SSRBH6UKA4BQ","created_at":"2026-07-05T07:21:06.233266+00:00"},{"alias_kind":"pith_short_16","alias_value":"SSRBH6UKA4BQ3QJV","created_at":"2026-07-05T07:21:06.233266+00:00"},{"alias_kind":"pith_short_8","alias_value":"SSRBH6UK","created_at":"2026-07-05T07:21:06.233266+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2401.09417","citing_title":"Vision Mamba: Efficient Visual Representation Learning with Bidirectional State Space Model","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5","json":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5.json","graph_json":"https://pith.science/api/pith-number/SSRBH6UKA4BQ3QJVGYFROEDXD5/graph.json","events_json":"https://pith.science/api/pith-number/SSRBH6UKA4BQ3QJVGYFROEDXD5/events.json","paper":"https://pith.science/paper/SSRBH6UK"},"agent_actions":{"view_html":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5","download_json":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5.json","view_paper":"https://pith.science/paper/SSRBH6UK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.03558&json=true","fetch_graph":"https://pith.science/api/pith-number/SSRBH6UKA4BQ3QJVGYFROEDXD5/graph.json","fetch_events":"https://pith.science/api/pith-number/SSRBH6UKA4BQ3QJVGYFROEDXD5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5/action/storage_attestation","attest_author":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5/action/author_attestation","sign_citation":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5/action/citation_signature","submit_replication":"https://pith.science/pith/SSRBH6UKA4BQ3QJVGYFROEDXD5/action/replication_record"}},"created_at":"2026-07-05T07:21:06.233266+00:00","updated_at":"2026-07-05T07:21:06.233266+00:00"}