{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GKPDMIICAAOKACYFAPQSHU4ES3","short_pith_number":"pith:GKPDMIIC","schema_version":"1.0","canonical_sha256":"329e362102001ca00b0503e123d38496f40bb50835d336859cbdb9222f095a89","source":{"kind":"arxiv","id":"2503.19510","version":1},"attestation_state":"computed","paper":{"title":"RoboFlamingo-Plus: Fusion of Depth and RGB Perception with Vision-Language Models for Enhanced Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Sheng Wang","submitted_at":"2025-03-25T10:01:57Z","abstract_excerpt":"As robotic technologies advancing towards more complex multimodal interactions and manipulation tasks, the integration of advanced Vision-Language Models (VLMs) has become a key driver in the field. Despite progress with current methods, challenges persist in fusing depth and RGB information within 3D environments and executing tasks guided by linguistic instructions. In response to these challenges, we have enhanced the existing RoboFlamingo framework by introducing RoboFlamingo-Plus, which incorporates depth data into VLMs to significantly improve robotic manipulation performance. Our resear"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.19510","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-03-25T10:01:57Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"29200d27b7751c9cafbceb47e4241bcc4f1744213eeda48282a8b47bf55f3f50","abstract_canon_sha256":"9f3b6a1b577ad0da83c570158a340478241a259b2c8aef797dd5bd5f4dcc4332"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:56.466837Z","signature_b64":"HNChd6Hu8aSlHkGkTHoWosXeZ876RuH8DzS7pVhsvucQI1xM81hvAC8XOQzkyf7/zZFEkhZ+L1bINVHl2CvNDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"329e362102001ca00b0503e123d38496f40bb50835d336859cbdb9222f095a89","last_reissued_at":"2026-07-05T10:38:56.466350Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:56.466350Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RoboFlamingo-Plus: Fusion of Depth and RGB Perception with Vision-Language Models for Enhanced Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Sheng Wang","submitted_at":"2025-03-25T10:01:57Z","abstract_excerpt":"As robotic technologies advancing towards more complex multimodal interactions and manipulation tasks, the integration of advanced Vision-Language Models (VLMs) has become a key driver in the field. Despite progress with current methods, challenges persist in fusing depth and RGB information within 3D environments and executing tasks guided by linguistic instructions. In response to these challenges, we have enhanced the existing RoboFlamingo framework by introducing RoboFlamingo-Plus, which incorporates depth data into VLMs to significantly improve robotic manipulation performance. Our resear"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.19510","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.19510/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.19510","created_at":"2026-07-05T10:38:56.466408+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.19510v1","created_at":"2026-07-05T10:38:56.466408+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.19510","created_at":"2026-07-05T10:38:56.466408+00:00"},{"alias_kind":"pith_short_12","alias_value":"GKPDMIICAAOK","created_at":"2026-07-05T10:38:56.466408+00:00"},{"alias_kind":"pith_short_16","alias_value":"GKPDMIICAAOKACYF","created_at":"2026-07-05T10:38:56.466408+00:00"},{"alias_kind":"pith_short_8","alias_value":"GKPDMIIC","created_at":"2026-07-05T10:38:56.466408+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24890","citing_title":"QuoVLA: Quotient Space for Vision-Language-Action Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29562","citing_title":"VLA-Pro: Cross-Task Procedural Memory Transfer for Vision-Language-Action Models","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3","json":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3.json","graph_json":"https://pith.science/api/pith-number/GKPDMIICAAOKACYFAPQSHU4ES3/graph.json","events_json":"https://pith.science/api/pith-number/GKPDMIICAAOKACYFAPQSHU4ES3/events.json","paper":"https://pith.science/paper/GKPDMIIC"},"agent_actions":{"view_html":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3","download_json":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3.json","view_paper":"https://pith.science/paper/GKPDMIIC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.19510&json=true","fetch_graph":"https://pith.science/api/pith-number/GKPDMIICAAOKACYFAPQSHU4ES3/graph.json","fetch_events":"https://pith.science/api/pith-number/GKPDMIICAAOKACYFAPQSHU4ES3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3/action/storage_attestation","attest_author":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3/action/author_attestation","sign_citation":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3/action/citation_signature","submit_replication":"https://pith.science/pith/GKPDMIICAAOKACYFAPQSHU4ES3/action/replication_record"}},"created_at":"2026-07-05T10:38:56.466408+00:00","updated_at":"2026-07-05T10:38:56.466408+00:00"}