{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:Q356Y5XUOTQIC6LEUOAUOR3PEL","short_pith_number":"pith:Q356Y5XU","schema_version":"1.0","canonical_sha256":"86fbec76f474e0817964a38147476f22c61ef713374198dc3b63f4bf2d691fef","source":{"kind":"arxiv","id":"2205.14401","version":2},"attestation_state":"computed","paper":{"title":"Point-M2AE: Multi-scale Masked Autoencoders for Hierarchical Point Cloud Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Zhao, Dong Wang, Hongsheng Li, Peng Gao, Renrui Zhang, Rongyao Fang, Yu Qiao, Ziyu Guo","submitted_at":"2022-05-28T11:22:53Z","abstract_excerpt":"Masked Autoencoders (MAE) have shown great potentials in self-supervised pre-training for language and 2D image transformers. However, it still remains an open question on how to exploit masked autoencoding for learning 3D representations of irregular point clouds. In this paper, we propose Point-M2AE, a strong Multi-scale MAE pre-training framework for hierarchical self-supervised learning of 3D point clouds. Unlike the standard transformer in MAE, we modify the encoder and decoder into pyramid architectures to progressively model spatial geometries and capture both fine-grained and high-leve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.14401","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-05-28T11:22:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4562a08c8e14fc9861bace3c857538a052ff58db561b7a1662b7266e70750d0a","abstract_canon_sha256":"9eb0207fd308dc252eb2cbf9e159ea9be6fd18035282d0cae40213a2745d207b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:31.781629Z","signature_b64":"CSjJkororfuoDPJ+zxUh67ml6l5/kfrR4epKKW7kPTQuYee0Dvn21gPACpbmsegSDkcpBS6rj/SRErTIMhwVDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86fbec76f474e0817964a38147476f22c61ef713374198dc3b63f4bf2d691fef","last_reissued_at":"2026-07-05T05:06:31.781131Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:31.781131Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Point-M2AE: Multi-scale Masked Autoencoders for Hierarchical Point Cloud Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Zhao, Dong Wang, Hongsheng Li, Peng Gao, Renrui Zhang, Rongyao Fang, Yu Qiao, Ziyu Guo","submitted_at":"2022-05-28T11:22:53Z","abstract_excerpt":"Masked Autoencoders (MAE) have shown great potentials in self-supervised pre-training for language and 2D image transformers. However, it still remains an open question on how to exploit masked autoencoding for learning 3D representations of irregular point clouds. In this paper, we propose Point-M2AE, a strong Multi-scale MAE pre-training framework for hierarchical self-supervised learning of 3D point clouds. Unlike the standard transformer in MAE, we modify the encoder and decoder into pyramid architectures to progressively model spatial geometries and capture both fine-grained and high-leve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.14401","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.14401/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.14401","created_at":"2026-07-05T05:06:31.781188+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.14401v2","created_at":"2026-07-05T05:06:31.781188+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.14401","created_at":"2026-07-05T05:06:31.781188+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q356Y5XUOTQI","created_at":"2026-07-05T05:06:31.781188+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q356Y5XUOTQIC6LE","created_at":"2026-07-05T05:06:31.781188+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q356Y5XU","created_at":"2026-07-05T05:06:31.781188+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2303.16199","citing_title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","ref_index":213,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL","json":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL.json","graph_json":"https://pith.science/api/pith-number/Q356Y5XUOTQIC6LEUOAUOR3PEL/graph.json","events_json":"https://pith.science/api/pith-number/Q356Y5XUOTQIC6LEUOAUOR3PEL/events.json","paper":"https://pith.science/paper/Q356Y5XU"},"agent_actions":{"view_html":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL","download_json":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL.json","view_paper":"https://pith.science/paper/Q356Y5XU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.14401&json=true","fetch_graph":"https://pith.science/api/pith-number/Q356Y5XUOTQIC6LEUOAUOR3PEL/graph.json","fetch_events":"https://pith.science/api/pith-number/Q356Y5XUOTQIC6LEUOAUOR3PEL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL/action/storage_attestation","attest_author":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL/action/author_attestation","sign_citation":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL/action/citation_signature","submit_replication":"https://pith.science/pith/Q356Y5XUOTQIC6LEUOAUOR3PEL/action/replication_record"}},"created_at":"2026-07-05T05:06:31.781188+00:00","updated_at":"2026-07-05T05:06:31.781188+00:00"}