{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:56LSGEQSAXRC2E4PINDB37MIYQ","short_pith_number":"pith:56LSGEQS","schema_version":"1.0","canonical_sha256":"ef9723121205e22d138f43461dfd88c40e79412bb4fdd1f7b3a039fe9e3039f8","source":{"kind":"arxiv","id":"2411.19309","version":2},"attestation_state":"computed","paper":{"title":"GRAPE: Generalizing Robot Policy via Preference Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Chaoqi Wang, Dieter Fox, Huaxiu Yao, Joel Jang, Kaiyuan Zheng, Mingyu Ding, Siwei Han, Yi Li, Zhaorun Chen, Zijian Zhang","submitted_at":"2024-11-28T18:30:10Z","abstract_excerpt":"Despite the recent advancements of vision-language-action (VLA) models on a variety of robotics tasks, they suffer from critical issues such as poor generalizability to unseen tasks, due to their reliance on behavior cloning exclusively from successful rollouts. Furthermore, they are typically fine-tuned to replicate demonstrations collected by experts under different settings, thus introducing distribution bias and limiting their adaptability to diverse manipulation objectives, such as efficiency, safety, and task completion. To bridge this gap, we introduce GRAPE: Generalizing Robot Policy v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.19309","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-11-28T18:30:10Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"fe56c42b8697056c713a62f12b4cd547e35af3f2e25bacce511b2099250f2373","abstract_canon_sha256":"641c04141b5cdb3d8aee3478b90d601ac99b33b24bbd5ac663025fc5ddf73444"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:12.501984Z","signature_b64":"XltzbOhSDTmuToUDEHo3PVZm19Pza7SNOJfBnzQpVwDcPdDBM04rdz49EJ62bNZdcrDRJ627AeOvr96yJBLmDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef9723121205e22d138f43461dfd88c40e79412bb4fdd1f7b3a039fe9e3039f8","last_reissued_at":"2026-07-05T10:09:12.501553Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:12.501553Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GRAPE: Generalizing Robot Policy via Preference Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Chaoqi Wang, Dieter Fox, Huaxiu Yao, Joel Jang, Kaiyuan Zheng, Mingyu Ding, Siwei Han, Yi Li, Zhaorun Chen, Zijian Zhang","submitted_at":"2024-11-28T18:30:10Z","abstract_excerpt":"Despite the recent advancements of vision-language-action (VLA) models on a variety of robotics tasks, they suffer from critical issues such as poor generalizability to unseen tasks, due to their reliance on behavior cloning exclusively from successful rollouts. Furthermore, they are typically fine-tuned to replicate demonstrations collected by experts under different settings, thus introducing distribution bias and limiting their adaptability to diverse manipulation objectives, such as efficiency, safety, and task completion. To bridge this gap, we introduce GRAPE: Generalizing Robot Policy v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.19309","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.19309/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.19309","created_at":"2026-07-05T10:09:12.501605+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.19309v2","created_at":"2026-07-05T10:09:12.501605+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.19309","created_at":"2026-07-05T10:09:12.501605+00:00"},{"alias_kind":"pith_short_12","alias_value":"56LSGEQSAXRC","created_at":"2026-07-05T10:09:12.501605+00:00"},{"alias_kind":"pith_short_16","alias_value":"56LSGEQSAXRC2E4P","created_at":"2026-07-05T10:09:12.501605+00:00"},{"alias_kind":"pith_short_8","alias_value":"56LSGEQS","created_at":"2026-07-05T10:09:12.501605+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":33,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25800","citing_title":"ROAD-VLA: Robust Online Adaptation via Self-Distillation for Vision-Language-Action Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22860","citing_title":"HiL-ResRL: A Model-Agnostic Finetuning Adapter via Human-in-the-loop Residual Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20698","citing_title":"SafeDojo: Safe Reinforcement Learning for VLA via Interactive World Model","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12550","citing_title":"Foresight: Iterative Reasoning About Clues that Matter for Navigation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12372","citing_title":"UniIntervene: Agentic Intervention for Efficient Real-World Reinforcement Learning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05468","citing_title":"FlowPRO: Reward-Free Reinforced Fine-Tuning of Flow-Matching VLAs via Proximalized Preference Optimization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32027","citing_title":"Freeform Preference Learning for Robotic Manipulation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00416","citing_title":"Learning While Deploying: Fleet-Scale Reinforcement Learning for Generalist Robot Policies","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31157","citing_title":"Rethinking Foundation Model Collaboration: Enhancing Specialized Models through Proxy Task Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29892","citing_title":"Trust Your Instincts: Confidence-Driven Test-Time RL for Vision-Language-Action Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01036","citing_title":"Position: Good Embodied Reward Models Need Bad Behavior Data","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20698","citing_title":"SafeDojo: Safe Reinforcement Learning for VLA via Interactive World Model","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2503.03480","citing_title":"SafeVLA: Towards Safety Alignment of Vision-Language-Action Model via Constrained Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09023","citing_title":"TwinRL: Digital Twin-Driven Reinforcement Learning for Real-World Robotic Manipulation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17486","citing_title":"DyGRO-VLA: Cross-Task Scaling of Vision-Language-Action Models via Dynamic Grouped Residual Optimization","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19580","citing_title":"PAPO-VLA: Planning-Aware Policy Optimization for Vision-Language-Action Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12710","citing_title":"Reflection-Based Task Adaptation for Self-Improving VLA","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17640","citing_title":"RESample: A Robust Data Augmentation Framework via Exploratory Sampling for Robotic Manipulation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2511.15669","citing_title":"DeepThinkVLA: Enhancing Reasoning Capability of Vision-Language-Action Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13073","citing_title":"Large VLM-based Vision-Language-Action Models for Robotic Manipulation: A Survey","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18719","citing_title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09674","citing_title":"SimpleVLA-RL: Scaling VLA Training via Reinforcement Learning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05855","citing_title":"DexVLA: Vision-Language Model with Plug-In Diffusion Expert for General Robot Control","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12167","citing_title":"From Imagined Futures to Executable Actions: Mixture of Latent Actions for Robot Manipulation","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ","json":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ.json","graph_json":"https://pith.science/api/pith-number/56LSGEQSAXRC2E4PINDB37MIYQ/graph.json","events_json":"https://pith.science/api/pith-number/56LSGEQSAXRC2E4PINDB37MIYQ/events.json","paper":"https://pith.science/paper/56LSGEQS"},"agent_actions":{"view_html":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ","download_json":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ.json","view_paper":"https://pith.science/paper/56LSGEQS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.19309&json=true","fetch_graph":"https://pith.science/api/pith-number/56LSGEQSAXRC2E4PINDB37MIYQ/graph.json","fetch_events":"https://pith.science/api/pith-number/56LSGEQSAXRC2E4PINDB37MIYQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ/action/storage_attestation","attest_author":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ/action/author_attestation","sign_citation":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ/action/citation_signature","submit_replication":"https://pith.science/pith/56LSGEQSAXRC2E4PINDB37MIYQ/action/replication_record"}},"created_at":"2026-07-05T10:09:12.501605+00:00","updated_at":"2026-07-05T10:09:12.501605+00:00"}