{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3HA53BKZG6PDAZSBURWZWM5FWP","short_pith_number":"pith:3HA53BKZ","schema_version":"1.0","canonical_sha256":"d9c1dd8559379e306641a46d9b33a5b3fcebbb24f209c1cbfc1da247c4f704b8","source":{"kind":"arxiv","id":"2107.05790","version":2},"attestation_state":"computed","paper":{"title":"Visual Parser: Representing Part-whole Hierarchies with Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Philip Torr, Shuyang Sun, Song Bai, Xiaoyu Yue","submitted_at":"2021-07-13T00:27:01Z","abstract_excerpt":"Human vision is able to capture the part-whole hierarchical information from the entire scene. This paper presents the Visual Parser (ViP) that explicitly constructs such a hierarchy with transformers. ViP divides visual representations into two levels, the part level and the whole level. Information of each part represents a combination of several independent vectors within the whole. To model the representations of the two levels, we first encode the information from the whole into part vectors through an attention mechanism, then decode the global information within the part vectors back in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.05790","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-07-13T00:27:01Z","cross_cats_sorted":[],"title_canon_sha256":"77b3f121c6eab324113f9494a960f807364f877026fe309851c90a72ea34a9b3","abstract_canon_sha256":"cab7d90d93d3d76809bc28903fa36fd7f7d7aa0ac744ace5def34fa980d3a253"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:46:45.149094Z","signature_b64":"OEyvAh1uet5q+kHYkjpO/vSfoQZ+dwrVscBk0Q8un4bYuPWKJXlA3Yeq1HFsiYetH6Opj6oe8n8WVkRTZauFAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d9c1dd8559379e306641a46d9b33a5b3fcebbb24f209c1cbfc1da247c4f704b8","last_reissued_at":"2026-07-05T03:46:45.148629Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:46:45.148629Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Parser: Representing Part-whole Hierarchies with Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Philip Torr, Shuyang Sun, Song Bai, Xiaoyu Yue","submitted_at":"2021-07-13T00:27:01Z","abstract_excerpt":"Human vision is able to capture the part-whole hierarchical information from the entire scene. This paper presents the Visual Parser (ViP) that explicitly constructs such a hierarchy with transformers. ViP divides visual representations into two levels, the part level and the whole level. Information of each part represents a combination of several independent vectors within the whole. To model the representations of the two levels, we first encode the information from the whole into part vectors through an attention mechanism, then decode the global information within the part vectors back in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.05790","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.05790/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.05790","created_at":"2026-07-05T03:46:45.148684+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.05790v2","created_at":"2026-07-05T03:46:45.148684+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.05790","created_at":"2026-07-05T03:46:45.148684+00:00"},{"alias_kind":"pith_short_12","alias_value":"3HA53BKZG6PD","created_at":"2026-07-05T03:46:45.148684+00:00"},{"alias_kind":"pith_short_16","alias_value":"3HA53BKZG6PDAZSB","created_at":"2026-07-05T03:46:45.148684+00:00"},{"alias_kind":"pith_short_8","alias_value":"3HA53BKZ","created_at":"2026-07-05T03:46:45.148684+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18543","citing_title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18549","citing_title":"Advancing Vision Transformer with Enhanced Spatial Priors","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP","json":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP.json","graph_json":"https://pith.science/api/pith-number/3HA53BKZG6PDAZSBURWZWM5FWP/graph.json","events_json":"https://pith.science/api/pith-number/3HA53BKZG6PDAZSBURWZWM5FWP/events.json","paper":"https://pith.science/paper/3HA53BKZ"},"agent_actions":{"view_html":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP","download_json":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP.json","view_paper":"https://pith.science/paper/3HA53BKZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.05790&json=true","fetch_graph":"https://pith.science/api/pith-number/3HA53BKZG6PDAZSBURWZWM5FWP/graph.json","fetch_events":"https://pith.science/api/pith-number/3HA53BKZG6PDAZSBURWZWM5FWP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP/action/storage_attestation","attest_author":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP/action/author_attestation","sign_citation":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP/action/citation_signature","submit_replication":"https://pith.science/pith/3HA53BKZG6PDAZSBURWZWM5FWP/action/replication_record"}},"created_at":"2026-07-05T03:46:45.148684+00:00","updated_at":"2026-07-05T03:46:45.148684+00:00"}