{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:YA3EYVH67HQXGAYP236ZCYFRUO","short_pith_number":"pith:YA3EYVH6","schema_version":"1.0","canonical_sha256":"c0364c54fef9e173030fd6fd9160b1a38e8f5802736783503a333cf8a44b2541","source":{"kind":"arxiv","id":"2106.09681","version":2},"attestation_state":"computed","paper":{"title":"XCiT: Cross-Covariance Image Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Alaaeldin El-Nouby, Armand Joulin, Gabriel Synnaeve, Herv\\'e Jegou, Hugo Touvron, Ivan Laptev, Jakob Verbeek, Mathilde Caron, Matthijs Douze, Natalia Neverova, Piotr Bojanowski","submitted_at":"2021-06-17T17:33:35Z","abstract_excerpt":"Following their success in natural language processing, transformers have recently shown much promise for computer vision. The self-attention operation underlying transformers yields global interactions between all tokens ,i.e. words or image patches, and enables flexible modelling of image data beyond the local interactions of convolutions. This flexibility, however, comes with a quadratic complexity in time and memory, hindering application to long sequences and high-resolution images. We propose a \"transposed\" version of self-attention that operates across feature channels rather than token"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.09681","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-06-17T17:33:35Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2240087c028f6c687ac6689cc9badf7d1515dc0dc5abae64a9fbb7fcc1c63815","abstract_canon_sha256":"5e399c2511cc491f917d6b38567732dbe4175aa0ac3cebcbb508123eaa8852d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:31.068006Z","signature_b64":"Nx2CXOc02YKKWcAdyO5We6cdHcJbW1bVYLvM6ebYYiPRzlJcl8jU+paQ7IZglAJ7hOFJU6YMDOn3UK4CweFeCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0364c54fef9e173030fd6fd9160b1a38e8f5802736783503a333cf8a44b2541","last_reissued_at":"2026-07-05T02:50:31.067628Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:31.067628Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XCiT: Cross-Covariance Image Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Alaaeldin El-Nouby, Armand Joulin, Gabriel Synnaeve, Herv\\'e Jegou, Hugo Touvron, Ivan Laptev, Jakob Verbeek, Mathilde Caron, Matthijs Douze, Natalia Neverova, Piotr Bojanowski","submitted_at":"2021-06-17T17:33:35Z","abstract_excerpt":"Following their success in natural language processing, transformers have recently shown much promise for computer vision. The self-attention operation underlying transformers yields global interactions between all tokens ,i.e. words or image patches, and enables flexible modelling of image data beyond the local interactions of convolutions. This flexibility, however, comes with a quadratic complexity in time and memory, hindering application to long sequences and high-resolution images. We propose a \"transposed\" version of self-attention that operates across feature channels rather than token"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.09681","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.09681/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.09681","created_at":"2026-07-05T02:50:31.067685+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.09681v2","created_at":"2026-07-05T02:50:31.067685+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.09681","created_at":"2026-07-05T02:50:31.067685+00:00"},{"alias_kind":"pith_short_12","alias_value":"YA3EYVH67HQX","created_at":"2026-07-05T02:50:31.067685+00:00"},{"alias_kind":"pith_short_16","alias_value":"YA3EYVH67HQXGAYP","created_at":"2026-07-05T02:50:31.067685+00:00"},{"alias_kind":"pith_short_8","alias_value":"YA3EYVH6","created_at":"2026-07-05T02:50:31.067685+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22098","citing_title":"TextTeacher: What Can Language Teach About Images?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11007","citing_title":"The Transformer as a Polar State Estimator","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09727","citing_title":"One for All: A Non-Linear Transformer can Enable Cross-Domain Generalization for In-Context Reinforcement Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25065","citing_title":"ShapeY: A Principled Framework for Measuring Shape Recognition Capacity via Nearest-Neighbor Matching","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12894","citing_title":"Representing 3D Faces with Learnable B-Spline Volumes","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11993","citing_title":"Ultra-low-light computer vision using trained photon correlations","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO","json":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO.json","graph_json":"https://pith.science/api/pith-number/YA3EYVH67HQXGAYP236ZCYFRUO/graph.json","events_json":"https://pith.science/api/pith-number/YA3EYVH67HQXGAYP236ZCYFRUO/events.json","paper":"https://pith.science/paper/YA3EYVH6"},"agent_actions":{"view_html":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO","download_json":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO.json","view_paper":"https://pith.science/paper/YA3EYVH6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.09681&json=true","fetch_graph":"https://pith.science/api/pith-number/YA3EYVH67HQXGAYP236ZCYFRUO/graph.json","fetch_events":"https://pith.science/api/pith-number/YA3EYVH67HQXGAYP236ZCYFRUO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO/action/storage_attestation","attest_author":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO/action/author_attestation","sign_citation":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO/action/citation_signature","submit_replication":"https://pith.science/pith/YA3EYVH67HQXGAYP236ZCYFRUO/action/replication_record"}},"created_at":"2026-07-05T02:50:31.067685+00:00","updated_at":"2026-07-05T02:50:31.067685+00:00"}