{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FNUUT2P3U2QIECV4EGJT72JVIV","short_pith_number":"pith:FNUUT2P3","schema_version":"1.0","canonical_sha256":"2b6949e9fba6a0820abc21933fe9354573f94639e0508f00b2e53bb6d070ee6d","source":{"kind":"arxiv","id":"2508.10333","version":1},"attestation_state":"computed","paper":{"title":"ReconVLA: Reconstructive Vision-Language-Action Model as Effective Robot Perceiver","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Donglin Wang, Feilong Tang, Han Zhao, Haoang Li, Haodong Yan, Jiayi Chen, Pengxiang Ding, Wenxuan Song, Yuxin Huang, Ziyang Zhou","submitted_at":"2025-08-14T04:20:19Z","abstract_excerpt":"Recent advances in Vision-Language-Action (VLA) models have enabled robotic agents to integrate multimodal understanding with action execution. However, our empirical analysis reveals that current VLAs struggle to allocate visual attention to target regions. Instead, visual attention is always dispersed. To guide the visual attention grounding on the correct target, we propose ReconVLA, a reconstructive VLA model with an implicit grounding paradigm. Conditioned on the model's visual outputs, a diffusion transformer aims to reconstruct the gaze region of the image, which corresponds to the targ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.10333","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-08-14T04:20:19Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"06295dee55473ecc04a4c5d87ad7d02e0cf4229d4652e6ca5f8d789763c3628a","abstract_canon_sha256":"55deea8b25f9f4b492fda3e95869ed5dee28cddfce82039ac0c88dc2cc094d2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:54.161472Z","signature_b64":"jgRP77WEUYOXcpgACsNPO9Mpj79adQvbHi71mZrbdNnnem5FAc4xA7bKVuBgLuhUcpQ2WLJWNEA0SNWm6G3zAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b6949e9fba6a0820abc21933fe9354573f94639e0508f00b2e53bb6d070ee6d","last_reissued_at":"2026-07-05T11:53:54.160988Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:54.160988Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReconVLA: Reconstructive Vision-Language-Action Model as Effective Robot Perceiver","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Donglin Wang, Feilong Tang, Han Zhao, Haoang Li, Haodong Yan, Jiayi Chen, Pengxiang Ding, Wenxuan Song, Yuxin Huang, Ziyang Zhou","submitted_at":"2025-08-14T04:20:19Z","abstract_excerpt":"Recent advances in Vision-Language-Action (VLA) models have enabled robotic agents to integrate multimodal understanding with action execution. However, our empirical analysis reveals that current VLAs struggle to allocate visual attention to target regions. Instead, visual attention is always dispersed. To guide the visual attention grounding on the correct target, we propose ReconVLA, a reconstructive VLA model with an implicit grounding paradigm. Conditioned on the model's visual outputs, a diffusion transformer aims to reconstruct the gaze region of the image, which corresponds to the targ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.10333","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.10333/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.10333","created_at":"2026-07-05T11:53:54.161045+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.10333v1","created_at":"2026-07-05T11:53:54.161045+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.10333","created_at":"2026-07-05T11:53:54.161045+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNUUT2P3U2QI","created_at":"2026-07-05T11:53:54.161045+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNUUT2P3U2QIECV4","created_at":"2026-07-05T11:53:54.161045+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNUUT2P3","created_at":"2026-07-05T11:53:54.161045+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06155","citing_title":"AffordanceVLA: A Vision-Language-Action Model Empowering Action Generation through Affordance-Aware Understanding","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01241","citing_title":"OneVLA: A Unified Framework for Embodied Tasks","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22671","citing_title":"From Abstraction to Instantiation: Learning Behavioral Representation for Vision-Language-Action Model","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28548","citing_title":"GEM: Generative Supervision Helps Embodied Intelligence","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31116","citing_title":"NTR: Neural Token Reconstruction for Scene Token Bottleneck in End-to-End Driving","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22671","citing_title":"From Abstraction to Instantiation: Learning Behavioral Representation for Vision-Language-Action Model","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20200","citing_title":"Global Prior Meets Local Consistency: Dual-Memory Augmented Vision-Language-Action Model for Efficient Robotic Manipulation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18960","citing_title":"AVA-VLA: Improving Vision-Language-Action models with Active Visual Attention","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15620","citing_title":"Towards Generalizable Robotic Manipulation in Dynamic Environments","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12160","citing_title":"Premover: Fast Vision-Language-Action Control by Acting Before Instructions Are Complete","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14125","citing_title":"HiVLA: A Visual-Grounded-Centric Hierarchical Embodied Manipulation System","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10903","citing_title":"CapVector: Learning Transferable Capability Vectors in Parametric Space for Vision-Language-Action Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24182","citing_title":"$M^2$-VLA: Boosting Vision-Language Models for Generalizable Manipulation via Layer Mixture and Meta-Skills","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21241","citing_title":"CorridorVLA: Explicit Spatial Constraints for Generative Action Heads via Sparse Anchors","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19683","citing_title":"Mask World Model: Predicting What Matters for Robust Robot Policy Learning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11751","citing_title":"Grounded World Model for Semantically Generalizable Planning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14125","citing_title":"HiVLA: A Visual-Grounded-Centric Hierarchical Embodied Manipulation System","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV","json":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV.json","graph_json":"https://pith.science/api/pith-number/FNUUT2P3U2QIECV4EGJT72JVIV/graph.json","events_json":"https://pith.science/api/pith-number/FNUUT2P3U2QIECV4EGJT72JVIV/events.json","paper":"https://pith.science/paper/FNUUT2P3"},"agent_actions":{"view_html":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV","download_json":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV.json","view_paper":"https://pith.science/paper/FNUUT2P3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.10333&json=true","fetch_graph":"https://pith.science/api/pith-number/FNUUT2P3U2QIECV4EGJT72JVIV/graph.json","fetch_events":"https://pith.science/api/pith-number/FNUUT2P3U2QIECV4EGJT72JVIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV/action/storage_attestation","attest_author":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV/action/author_attestation","sign_citation":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV/action/citation_signature","submit_replication":"https://pith.science/pith/FNUUT2P3U2QIECV4EGJT72JVIV/action/replication_record"}},"created_at":"2026-07-05T11:53:54.161045+00:00","updated_at":"2026-07-05T11:53:54.161045+00:00"}