{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LULZF3BOFV2YNDJNTS3RE6WRLD","short_pith_number":"pith:LULZF3BO","schema_version":"1.0","canonical_sha256":"5d1792ec2e2d75868d2d9cb7127ad158deb7b3aadc650de77d7c3b74bd4fbc0a","source":{"kind":"arxiv","id":"2503.20384","version":2},"attestation_state":"computed","paper":{"title":"MoLe-VLA: Dynamic Layer-skipping Vision Language Action Model via Mixture-of-Layers for Efficient Robot Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Gaole Dai, Liang Heng, Li Du, Menghang Dong, Rongyu Zhang, Shanghang Zhang, Xiaowei Chi, Yuan Du, Yuan Zhang","submitted_at":"2025-03-26T10:05:38Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) excel in understanding complex language and visual data, enabling generalist robotic systems to interpret instructions and perform embodied tasks. Nevertheless, their real-world deployment is hindered by substantial computational and storage demands. Recent insights into the homogeneous patterns in the LLM layer have inspired sparsification techniques to address these challenges, such as early exit and token pruning. However, these methods often neglect the critical role of the final layers that encode the semantic information most relevant to downstrea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20384","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-03-26T10:05:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6defd9c4706a00d3939558a569c7025438b413fa751a13307b34ec0885bed91a","abstract_canon_sha256":"cbf54a9e2c32fc25e5c649de04d6ba8f4eb4f762d9d87dc19fde8929fbd1a811"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:28.020575Z","signature_b64":"jhDwnXiiRV+2LniT8BUcvE9/kcuC8WETOf/XTaqGA3byPG5qyW85koTUo93v2v4G9/mUKuKtoS4z0j+ov7xXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d1792ec2e2d75868d2d9cb7127ad158deb7b3aadc650de77d7c3b74bd4fbc0a","last_reissued_at":"2026-07-05T10:48:28.020087Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:28.020087Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoLe-VLA: Dynamic Layer-skipping Vision Language Action Model via Mixture-of-Layers for Efficient Robot Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Gaole Dai, Liang Heng, Li Du, Menghang Dong, Rongyu Zhang, Shanghang Zhang, Xiaowei Chi, Yuan Du, Yuan Zhang","submitted_at":"2025-03-26T10:05:38Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) excel in understanding complex language and visual data, enabling generalist robotic systems to interpret instructions and perform embodied tasks. Nevertheless, their real-world deployment is hindered by substantial computational and storage demands. Recent insights into the homogeneous patterns in the LLM layer have inspired sparsification techniques to address these challenges, such as early exit and token pruning. However, these methods often neglect the critical role of the final layers that encode the semantic information most relevant to downstrea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20384","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20384/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20384","created_at":"2026-07-05T10:48:28.020145+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20384v2","created_at":"2026-07-05T10:48:28.020145+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20384","created_at":"2026-07-05T10:48:28.020145+00:00"},{"alias_kind":"pith_short_12","alias_value":"LULZF3BOFV2Y","created_at":"2026-07-05T10:48:28.020145+00:00"},{"alias_kind":"pith_short_16","alias_value":"LULZF3BOFV2YNDJN","created_at":"2026-07-05T10:48:28.020145+00:00"},{"alias_kind":"pith_short_8","alias_value":"LULZF3BO","created_at":"2026-07-05T10:48:28.020145+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12403","citing_title":"World Pilot: Steering Vision-Language-Action Models with World-Action Priors","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31382","citing_title":"Revisiting Parameter Redundancy in Vision-Language-Action Models: Insights from VLM-to-VLA Adaptation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29662","citing_title":"SAFE-Pruner: Semantic Attention-Guided Future-Aware Token Pruning for Efficient Vision-Language-Action Manipulation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29438","citing_title":"ElegantVLA: Learning When to Think for Efficient Vision-Language-Action Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14371","citing_title":"OxyGen: Unified KV Cache Management for VLA Inference under Multi-Task Parallelism","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13073","citing_title":"Large VLM-based Vision-Language-Action Models for Robotic Manipulation: A Survey","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18082","citing_title":"ActDistill: General Action-Guided Self-Derived Distillation for Efficient Vision-Language-Action Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18085","citing_title":"Continually Evolving Skill Knowledge in Vision Language Action Model","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20309","citing_title":"QuantVLA: Scale-Calibrated Post-Training Quantization for Vision-Language-Action Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01581","citing_title":"KERV: Kinematic-Rectified Speculative Decoding for Embodied VLA Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17573","citing_title":"HeiSD: Hybrid Speculative Decoding for Embodied Vision-Language-Action Models with Kinematic Awareness","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20711","citing_title":"RoboECC: Multi-Factor-Aware Edge-Cloud Collaborative Deployment for VLA Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05126","citing_title":"ConsisVLA-4D: Advancing Spatiotemporal Consistency in Efficient 3D-Perception and 4D-Reasoning for Robotic Manipulation","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05672","citing_title":"A1: A Fully Transparent Open-Source, Adaptive and Efficient Truncated Vision-Language-Action Model","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD","json":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD.json","graph_json":"https://pith.science/api/pith-number/LULZF3BOFV2YNDJNTS3RE6WRLD/graph.json","events_json":"https://pith.science/api/pith-number/LULZF3BOFV2YNDJNTS3RE6WRLD/events.json","paper":"https://pith.science/paper/LULZF3BO"},"agent_actions":{"view_html":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD","download_json":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD.json","view_paper":"https://pith.science/paper/LULZF3BO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20384&json=true","fetch_graph":"https://pith.science/api/pith-number/LULZF3BOFV2YNDJNTS3RE6WRLD/graph.json","fetch_events":"https://pith.science/api/pith-number/LULZF3BOFV2YNDJNTS3RE6WRLD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD/action/storage_attestation","attest_author":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD/action/author_attestation","sign_citation":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD/action/citation_signature","submit_replication":"https://pith.science/pith/LULZF3BOFV2YNDJNTS3RE6WRLD/action/replication_record"}},"created_at":"2026-07-05T10:48:28.020145+00:00","updated_at":"2026-07-05T10:48:28.020145+00:00"}