{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WBLBY7DSRLRQLIXJ4WNB27ZJGX","short_pith_number":"pith:WBLBY7DS","schema_version":"1.0","canonical_sha256":"b0561c7c728ae305a2e9e59a1d7f2935d17f7ad46359b6b632db19439e518828","source":{"kind":"arxiv","id":"2503.13446","version":1},"attestation_state":"computed","paper":{"title":"MoManipVLA: Transferring Vision-language-action Models for General Mobile Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Haibin Yan, Xiuwei Xu, Yuheng Zhou, Zhenyu Wu, Ziwei Wang","submitted_at":"2025-03-17T17:59:52Z","abstract_excerpt":"Mobile manipulation is the fundamental challenge for robotics to assist humans with diverse tasks and environments in everyday life. However, conventional mobile manipulation approaches often struggle to generalize across different tasks and environments because of the lack of large-scale training. In contrast, recent advances in vision-language-action (VLA) models have shown impressive generalization capabilities, but these foundation models are developed for fixed-base manipulation tasks. Therefore, we propose an efficient policy adaptation framework named MoManipVLA to transfer pre-trained "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.13446","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-03-17T17:59:52Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"436acab0b63c83bcb7e977dbdf2ee9e56727944d4c7581a9f2321cd1184c0c05","abstract_canon_sha256":"e738aea4dc3feec89ea533f9a8099961840b35f81e2afe6c4c6b083764dc8ebb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:33:00.875382Z","signature_b64":"sOqGPILMHRASf+P+6Wj3DYzKoLwN3H3F8GZnVU3gHPrxtzeWRHe7Ny7gYDdLTQ3FMj/Wwv5ZFTpqaB9vjUaeBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0561c7c728ae305a2e9e59a1d7f2935d17f7ad46359b6b632db19439e518828","last_reissued_at":"2026-07-05T10:33:00.874589Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:33:00.874589Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoManipVLA: Transferring Vision-language-action Models for General Mobile Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Haibin Yan, Xiuwei Xu, Yuheng Zhou, Zhenyu Wu, Ziwei Wang","submitted_at":"2025-03-17T17:59:52Z","abstract_excerpt":"Mobile manipulation is the fundamental challenge for robotics to assist humans with diverse tasks and environments in everyday life. However, conventional mobile manipulation approaches often struggle to generalize across different tasks and environments because of the lack of large-scale training. In contrast, recent advances in vision-language-action (VLA) models have shown impressive generalization capabilities, but these foundation models are developed for fixed-base manipulation tasks. Therefore, we propose an efficient policy adaptation framework named MoManipVLA to transfer pre-trained "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13446","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.13446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.13446","created_at":"2026-07-05T10:33:00.874672+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.13446v1","created_at":"2026-07-05T10:33:00.874672+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13446","created_at":"2026-07-05T10:33:00.874672+00:00"},{"alias_kind":"pith_short_12","alias_value":"WBLBY7DSRLRQ","created_at":"2026-07-05T10:33:00.874672+00:00"},{"alias_kind":"pith_short_16","alias_value":"WBLBY7DSRLRQLIXJ","created_at":"2026-07-05T10:33:00.874672+00:00"},{"alias_kind":"pith_short_8","alias_value":"WBLBY7DS","created_at":"2026-07-05T10:33:00.874672+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00253","citing_title":"Per-Group Error, Not Total MSE: Fine-Tuning Vision-Language-Action Models for 11-DoF Mobile Manipulation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14148","citing_title":"AsyncVLA: Asynchronous Flow Matching for Vision-Language-Action Models","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01925","citing_title":"A Survey on Vision-Language-Action Models: An Action Tokenization Perspective","ref_index":272,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02487","citing_title":"Visibility-Aware Mobile Grasping in Dynamic Environments","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02487","citing_title":"Visibility-Aware Mobile Grasping in Dynamic Environments","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX","json":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX.json","graph_json":"https://pith.science/api/pith-number/WBLBY7DSRLRQLIXJ4WNB27ZJGX/graph.json","events_json":"https://pith.science/api/pith-number/WBLBY7DSRLRQLIXJ4WNB27ZJGX/events.json","paper":"https://pith.science/paper/WBLBY7DS"},"agent_actions":{"view_html":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX","download_json":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX.json","view_paper":"https://pith.science/paper/WBLBY7DS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.13446&json=true","fetch_graph":"https://pith.science/api/pith-number/WBLBY7DSRLRQLIXJ4WNB27ZJGX/graph.json","fetch_events":"https://pith.science/api/pith-number/WBLBY7DSRLRQLIXJ4WNB27ZJGX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX/action/storage_attestation","attest_author":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX/action/author_attestation","sign_citation":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX/action/citation_signature","submit_replication":"https://pith.science/pith/WBLBY7DSRLRQLIXJ4WNB27ZJGX/action/replication_record"}},"created_at":"2026-07-05T10:33:00.874672+00:00","updated_at":"2026-07-05T10:33:00.874672+00:00"}