{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D4FDP6BITI57OCVSULEREV6GWQ","short_pith_number":"pith:D4FDP6BI","schema_version":"1.0","canonical_sha256":"1f0a37f8289a3bf70ab2a2c91257c6b429c651ac2f90a299b0fe44efcf86ce25","source":{"kind":"arxiv","id":"2405.17418","version":2},"attestation_state":"computed","paper":{"title":"A Self-Correcting Vision-Language-Action Model for Fast and Slow System Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenxuan Li, Chuyan Xiong, Guanqun Wang, Jiaming Liu, Jiaxin Ge, Kaichen Zhou, Liang Heng, Renrui Zhang, Shanghang Zhang, Sixiang Chen, Xiaoqi Li","submitted_at":"2024-05-27T17:58:48Z","abstract_excerpt":"Recently, some studies have integrated Multimodal Large Language Models into robotic manipulation, constructing vision-language-action models (VLAs) to interpret multimodal information and predict SE(3) poses. While VLAs have shown promising progress, they may suffer from failures when faced with novel and complex tasks. To emulate human-like reasoning for more robust manipulation, we propose the self-corrected (SC-)VLA framework, which integrates fast system for directly predicting actions and slow system for reflecting on failed actions within a single VLA policy. For the fast system, we inc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17418","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-27T17:58:48Z","cross_cats_sorted":[],"title_canon_sha256":"7d94128136195e55f43ce625d9717e957ac92db02d38afade04c0ce082ca9be5","abstract_canon_sha256":"6964983d1663a973717bf4739a0f61b79665b2f2ebf6bf2c0c02262d290638cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:34:07.916969Z","signature_b64":"6sDeas0Cp8mhPqjmQxIGQBV58dRdHz2AB22kyaxDucV8gKfSofZ1ySnLJ3UGcEZw/tgHj4zzwLrJAkhuDXZoDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f0a37f8289a3bf70ab2a2c91257c6b429c651ac2f90a299b0fe44efcf86ce25","last_reissued_at":"2026-07-05T10:34:07.916404Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:34:07.916404Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Self-Correcting Vision-Language-Action Model for Fast and Slow System Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenxuan Li, Chuyan Xiong, Guanqun Wang, Jiaming Liu, Jiaxin Ge, Kaichen Zhou, Liang Heng, Renrui Zhang, Shanghang Zhang, Sixiang Chen, Xiaoqi Li","submitted_at":"2024-05-27T17:58:48Z","abstract_excerpt":"Recently, some studies have integrated Multimodal Large Language Models into robotic manipulation, constructing vision-language-action models (VLAs) to interpret multimodal information and predict SE(3) poses. While VLAs have shown promising progress, they may suffer from failures when faced with novel and complex tasks. To emulate human-like reasoning for more robust manipulation, we propose the self-corrected (SC-)VLA framework, which integrates fast system for directly predicting actions and slow system for reflecting on failed actions within a single VLA policy. For the fast system, we inc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17418","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17418","created_at":"2026-07-05T10:34:07.916473+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17418v2","created_at":"2026-07-05T10:34:07.916473+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17418","created_at":"2026-07-05T10:34:07.916473+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4FDP6BITI57","created_at":"2026-07-05T10:34:07.916473+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4FDP6BITI57OCVS","created_at":"2026-07-05T10:34:07.916473+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4FDP6BI","created_at":"2026-07-05T10:34:07.916473+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01191","citing_title":"Sentinel-VLA: A Metacognitive VLA Model with Active Status Monitoring for Dynamic Reasoning and Error Recovery","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":198,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10408","citing_title":"VISOR: A Vision-Language Model-based Test Oracle for Testing Robots","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14148","citing_title":"AsyncVLA: Asynchronous Flow Matching for Vision-Language-Action Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2503.10631","citing_title":"HybridVLA: Collaborative Diffusion and Autoregression in a Unified Vision-Language-Action Model","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10118","citing_title":"Plan in Sandbox, Navigate in Open Worlds: Learning Physics-Grounded Abstracted Experience for Embodied Navigation","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10408","citing_title":"VISOR: A Vision-Language Model-based Test Oracle for Testing Robots","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01191","citing_title":"Sentinel-VLA: A Metacognitive VLA Model with Active Status Monitoring for Dynamic Reasoning and Error Recovery","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ","json":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ.json","graph_json":"https://pith.science/api/pith-number/D4FDP6BITI57OCVSULEREV6GWQ/graph.json","events_json":"https://pith.science/api/pith-number/D4FDP6BITI57OCVSULEREV6GWQ/events.json","paper":"https://pith.science/paper/D4FDP6BI"},"agent_actions":{"view_html":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ","download_json":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ.json","view_paper":"https://pith.science/paper/D4FDP6BI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17418&json=true","fetch_graph":"https://pith.science/api/pith-number/D4FDP6BITI57OCVSULEREV6GWQ/graph.json","fetch_events":"https://pith.science/api/pith-number/D4FDP6BITI57OCVSULEREV6GWQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ/action/storage_attestation","attest_author":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ/action/author_attestation","sign_citation":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ/action/citation_signature","submit_replication":"https://pith.science/pith/D4FDP6BITI57OCVSULEREV6GWQ/action/replication_record"}},"created_at":"2026-07-05T10:34:07.916473+00:00","updated_at":"2026-07-05T10:34:07.916473+00:00"}