{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:VSE46BWSNH2PDENI4QECW7XMFN","short_pith_number":"pith:VSE46BWS","schema_version":"1.0","canonical_sha256":"ac89cf06d269f4f191a8e4082b7eec2b53e7186daa27e27da8813f5547f1764b","source":{"kind":"arxiv","id":"2103.03206","version":2},"attestation_state":"computed","paper":{"title":"Perceiver: General Perception with Iterative Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Andrew Brock, Andrew Jaegle, Andrew Zisserman, Felix Gimeno, Joao Carreira, Oriol Vinyals","submitted_at":"2021-03-04T18:20:50Z","abstract_excerpt":"Biological systems perceive the world by simultaneously processing high-dimensional inputs from modalities as diverse as vision, audition, touch, proprioception, etc. The perception models used in deep learning on the other hand are designed for individual modalities, often relying on domain-specific assumptions such as the local grid structures exploited by virtually all existing vision models. These priors introduce helpful inductive biases, but also lock models to individual modalities. In this paper we introduce the Perceiver - a model that builds upon Transformers and hence makes few arch"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.03206","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-03-04T18:20:50Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"951049de2e92af81178c3805a531a2820d1dd6b25f497e2e2ab6c6a0881ab979","abstract_canon_sha256":"8b0d57e489f25187dd3262039221c8ca3f2aea321ecbda28a5d9dabf6545374b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:51:39.493734Z","signature_b64":"71cl9jG9TyopmvzMpQq2/HhGFlz1QngwV2uOOSUyhtHtpitW21pXtpefuag0QRGrolqYkEZPFFZfRm290bxcDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac89cf06d269f4f191a8e4082b7eec2b53e7186daa27e27da8813f5547f1764b","last_reissued_at":"2026-07-05T02:51:39.493276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:51:39.493276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Perceiver: General Perception with Iterative Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Andrew Brock, Andrew Jaegle, Andrew Zisserman, Felix Gimeno, Joao Carreira, Oriol Vinyals","submitted_at":"2021-03-04T18:20:50Z","abstract_excerpt":"Biological systems perceive the world by simultaneously processing high-dimensional inputs from modalities as diverse as vision, audition, touch, proprioception, etc. The perception models used in deep learning on the other hand are designed for individual modalities, often relying on domain-specific assumptions such as the local grid structures exploited by virtually all existing vision models. These priors introduce helpful inductive biases, but also lock models to individual modalities. In this paper we introduce the Perceiver - a model that builds upon Transformers and hence makes few arch"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.03206","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.03206/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.03206","created_at":"2026-07-05T02:51:39.493345+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.03206v2","created_at":"2026-07-05T02:51:39.493345+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.03206","created_at":"2026-07-05T02:51:39.493345+00:00"},{"alias_kind":"pith_short_12","alias_value":"VSE46BWSNH2P","created_at":"2026-07-05T02:51:39.493345+00:00"},{"alias_kind":"pith_short_16","alias_value":"VSE46BWSNH2PDENI","created_at":"2026-07-05T02:51:39.493345+00:00"},{"alias_kind":"pith_short_8","alias_value":"VSE46BWS","created_at":"2026-07-05T02:51:39.493345+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26361","citing_title":"Does Aurora Encode Atmospheric Structure? Latent Regime Analysis and Attribution","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12215","citing_title":"MLT-Dedup: Efficient Large-Scale Online Video Deduplication via Multi-Level Representations and Spatial-Temporal Matching","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04857","citing_title":"Rethinking Incompleteness: Formalizing Protocol Divergence and Train-Once Learning for Robust IMVC","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01495","citing_title":"CART: Context-Anchored Recurrent Transformer -- A Parameter-Efficient Architecture with Learned Stability","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26540","citing_title":"Domain-Gated Latent Diffusion: Generative Inverse Design of HMX-Class Energetic Materials with First-Principles Validation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09989","citing_title":"StereoPolicy: Improving Robotic Manipulation Policies via Stereo Perception","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30318","citing_title":"Chronos: A Physics-Informed Full-History Framework for Non-Markovian Long-Horizon Manipulation","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14135","citing_title":"GAF: Gaussian Action Field as a 4D Representation for Dynamic World Modeling in Robotic Manipulation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2504.06176","citing_title":"A Self-Supervised Framework for Space Object Behaviour Characterisation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11229","citing_title":"Latent Generative Solvers for Generalizable Long-Term Physics Simulation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23750","citing_title":"The Override Gap: A Magnitude Account of Knowledge Conflict Failure in Hypernetwork-Based Instant LLM Adaptation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09989","citing_title":"StereoPolicy: Improving Robotic Manipulation Policies via Stereo Perception","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23750","citing_title":"The Override Gap: A Magnitude Account of Knowledge Conflict Failure in Hypernetwork-Based Instant LLM Adaptation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22442","citing_title":"HubRouter: A Pluggable Sub-Quadratic Routing Primitive for Hybrid Sequence Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11095","citing_title":"Bottleneck Tokens for Unified Multimodal Retrieval","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09402","citing_title":"Enhancing event reconstruction for $\\gamma$-ray particle detector arrays using transformers","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02300","citing_title":"A Meta Reinforcement Learning Approach to Goals-Based Wealth Management","ref_index":264,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN","json":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN.json","graph_json":"https://pith.science/api/pith-number/VSE46BWSNH2PDENI4QECW7XMFN/graph.json","events_json":"https://pith.science/api/pith-number/VSE46BWSNH2PDENI4QECW7XMFN/events.json","paper":"https://pith.science/paper/VSE46BWS"},"agent_actions":{"view_html":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN","download_json":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN.json","view_paper":"https://pith.science/paper/VSE46BWS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.03206&json=true","fetch_graph":"https://pith.science/api/pith-number/VSE46BWSNH2PDENI4QECW7XMFN/graph.json","fetch_events":"https://pith.science/api/pith-number/VSE46BWSNH2PDENI4QECW7XMFN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN/action/storage_attestation","attest_author":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN/action/author_attestation","sign_citation":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN/action/citation_signature","submit_replication":"https://pith.science/pith/VSE46BWSNH2PDENI4QECW7XMFN/action/replication_record"}},"created_at":"2026-07-05T02:51:39.493345+00:00","updated_at":"2026-07-05T02:51:39.493345+00:00"}