{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DBBMOISHGZTPSZJMXRSUH5KDIB","short_pith_number":"pith:DBBMOISH","schema_version":"1.0","canonical_sha256":"1842c722473666f9652cbc6543f5434058f3145a9cb3889dfef1acca7559b0e4","source":{"kind":"arxiv","id":"2502.09268","version":2},"attestation_state":"computed","paper":{"title":"GEVRM: Goal-Expressive Video Generation Model For Robust Visual Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Donglin Wang, Hongyin Zhang, Pengxiang Ding, Shangke Lyu, Ying Peng","submitted_at":"2025-02-13T12:29:50Z","abstract_excerpt":"With the rapid development of embodied artificial intelligence, significant progress has been made in vision-language-action (VLA) models for general robot decision-making. However, the majority of existing VLAs fail to account for the inevitable external perturbations encountered during deployment. These perturbations introduce unforeseen state information to the VLA, resulting in inaccurate actions and consequently, a significant decline in generalization performance. The classic internal model control (IMC) principle demonstrates that a closed-loop system with an internal model that include"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.09268","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-02-13T12:29:50Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e81ebbe86921c6828e9bb81f1d26265a6814ce98e9d59220015a0913fcafffd2","abstract_canon_sha256":"fed788f2a7dedcdc100428ca9688c45980cd7df7057c415bac9ea4e874d427df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:09.285238Z","signature_b64":"M5vCMYDOWUAOnlaEQXIltF8SfCQ3RAlWkY8zx2HeSU8HFU3IEWKroKd2AB2EYGxNuNM6mPluGOKg3qZhCVksCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1842c722473666f9652cbc6543f5434058f3145a9cb3889dfef1acca7559b0e4","last_reissued_at":"2026-07-05T10:14:09.283672Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:09.283672Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GEVRM: Goal-Expressive Video Generation Model For Robust Visual Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Donglin Wang, Hongyin Zhang, Pengxiang Ding, Shangke Lyu, Ying Peng","submitted_at":"2025-02-13T12:29:50Z","abstract_excerpt":"With the rapid development of embodied artificial intelligence, significant progress has been made in vision-language-action (VLA) models for general robot decision-making. However, the majority of existing VLAs fail to account for the inevitable external perturbations encountered during deployment. These perturbations introduce unforeseen state information to the VLA, resulting in inaccurate actions and consequently, a significant decline in generalization performance. The classic internal model control (IMC) principle demonstrates that a closed-loop system with an internal model that include"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.09268","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.09268/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.09268","created_at":"2026-07-05T10:14:09.283746+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.09268v2","created_at":"2026-07-05T10:14:09.283746+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.09268","created_at":"2026-07-05T10:14:09.283746+00:00"},{"alias_kind":"pith_short_12","alias_value":"DBBMOISHGZTP","created_at":"2026-07-05T10:14:09.283746+00:00"},{"alias_kind":"pith_short_16","alias_value":"DBBMOISHGZTPSZJM","created_at":"2026-07-05T10:14:09.283746+00:00"},{"alias_kind":"pith_short_8","alias_value":"DBBMOISH","created_at":"2026-07-05T10:14:09.283746+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01222","citing_title":"Ink3D: Sculpting 3D Assets with Extremely Complex Textures via Video Generative Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10055","citing_title":"STRONG-VLA: Decoupled Robustness Learning for Vision-Language-Action Models under Multimodal Perturbations","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08168","citing_title":"ViVa: A Video-Generative Value Model for Robot Reinforcement Learning","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14732","citing_title":"World-Value-Action Model: Implicit Planning for Vision-Language-Action Systems","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB","json":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB.json","graph_json":"https://pith.science/api/pith-number/DBBMOISHGZTPSZJMXRSUH5KDIB/graph.json","events_json":"https://pith.science/api/pith-number/DBBMOISHGZTPSZJMXRSUH5KDIB/events.json","paper":"https://pith.science/paper/DBBMOISH"},"agent_actions":{"view_html":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB","download_json":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB.json","view_paper":"https://pith.science/paper/DBBMOISH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.09268&json=true","fetch_graph":"https://pith.science/api/pith-number/DBBMOISHGZTPSZJMXRSUH5KDIB/graph.json","fetch_events":"https://pith.science/api/pith-number/DBBMOISHGZTPSZJMXRSUH5KDIB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB/action/storage_attestation","attest_author":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB/action/author_attestation","sign_citation":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB/action/citation_signature","submit_replication":"https://pith.science/pith/DBBMOISHGZTPSZJMXRSUH5KDIB/action/replication_record"}},"created_at":"2026-07-05T10:14:09.283746+00:00","updated_at":"2026-07-05T10:14:09.283746+00:00"}