{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:G4YM4LG2AMCYYDBVUCRMZ26WUR","short_pith_number":"pith:G4YM4LG2","schema_version":"1.0","canonical_sha256":"3730ce2cda03058c0c35a0a2ccebd6a45c4410bbb13fa222582c9c575d7af9ed","source":{"kind":"arxiv","id":"2503.08950","version":1},"attestation_state":"computed","paper":{"title":"FP3: A 3D Foundation Policy for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Chuan Wen, Geng Chen, Rujia Yang, Yang Gao","submitted_at":"2025-03-11T23:01:08Z","abstract_excerpt":"Following its success in natural language processing and computer vision, foundation models that are pre-trained on large-scale multi-task datasets have also shown great potential in robotics. However, most existing robot foundation models rely solely on 2D image observations, ignoring 3D geometric information, which is essential for robots to perceive and reason about the 3D world. In this paper, we introduce FP3, a first large-scale 3D foundation policy model for robotic manipulation. FP3 builds on a scalable diffusion transformer architecture and is pre-trained on 60k trajectories with poin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.08950","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-03-11T23:01:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"db0a6eba92e296bb55c72d9e3d27ad5f59ad01dec6e5645e3701b831f8222223","abstract_canon_sha256":"d1d514ee19dee70ecd01161108e5015fa2381e8e4542266c72126a2f2e80aa61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:29:44.721884Z","signature_b64":"+N7pFn1rx+zQZMvh0i6Bv/KrVYqUvuIEYURHn0dT+ui3/iGLBox99eQ4M5X9YX0uiqG5hi+8Scz3hBwxiQUcCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3730ce2cda03058c0c35a0a2ccebd6a45c4410bbb13fa222582c9c575d7af9ed","last_reissued_at":"2026-07-05T10:29:44.720896Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:29:44.720896Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FP3: A 3D Foundation Policy for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Chuan Wen, Geng Chen, Rujia Yang, Yang Gao","submitted_at":"2025-03-11T23:01:08Z","abstract_excerpt":"Following its success in natural language processing and computer vision, foundation models that are pre-trained on large-scale multi-task datasets have also shown great potential in robotics. However, most existing robot foundation models rely solely on 2D image observations, ignoring 3D geometric information, which is essential for robots to perceive and reason about the 3D world. In this paper, we introduce FP3, a first large-scale 3D foundation policy model for robotic manipulation. FP3 builds on a scalable diffusion transformer architecture and is pre-trained on 60k trajectories with poin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.08950","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.08950/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.08950","created_at":"2026-07-05T10:29:44.721010+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.08950v1","created_at":"2026-07-05T10:29:44.721010+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.08950","created_at":"2026-07-05T10:29:44.721010+00:00"},{"alias_kind":"pith_short_12","alias_value":"G4YM4LG2AMCY","created_at":"2026-07-05T10:29:44.721010+00:00"},{"alias_kind":"pith_short_16","alias_value":"G4YM4LG2AMCYYDBV","created_at":"2026-07-05T10:29:44.721010+00:00"},{"alias_kind":"pith_short_8","alias_value":"G4YM4LG2","created_at":"2026-07-05T10:29:44.721010+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.21414","citing_title":"PointACT: Vision-Language-Action Models with Multi-Scale Point-Action Interaction","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17522","citing_title":"RoboFlow4D: A Lightweight Flow World Model Toward Real-Time Flow-Guided Robotic Manipulation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03181","citing_title":"Multi-View Video Diffusion Policy: A 3D Spatio-Temporal-Aware Video Action Model","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05126","citing_title":"ConsisVLA-4D: Advancing Spatiotemporal Consistency in Efficient 3D-Perception and 4D-Reasoning for Robotic Manipulation","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15281","citing_title":"R3D: Revisiting 3D Policy Learning","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR","json":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR.json","graph_json":"https://pith.science/api/pith-number/G4YM4LG2AMCYYDBVUCRMZ26WUR/graph.json","events_json":"https://pith.science/api/pith-number/G4YM4LG2AMCYYDBVUCRMZ26WUR/events.json","paper":"https://pith.science/paper/G4YM4LG2"},"agent_actions":{"view_html":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR","download_json":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR.json","view_paper":"https://pith.science/paper/G4YM4LG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.08950&json=true","fetch_graph":"https://pith.science/api/pith-number/G4YM4LG2AMCYYDBVUCRMZ26WUR/graph.json","fetch_events":"https://pith.science/api/pith-number/G4YM4LG2AMCYYDBVUCRMZ26WUR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR/action/storage_attestation","attest_author":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR/action/author_attestation","sign_citation":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR/action/citation_signature","submit_replication":"https://pith.science/pith/G4YM4LG2AMCYYDBVUCRMZ26WUR/action/replication_record"}},"created_at":"2026-07-05T10:29:44.721010+00:00","updated_at":"2026-07-05T10:29:44.721010+00:00"}