{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OAXHWHKX4UEFWSRJC47SI74SXK","short_pith_number":"pith:OAXHWHKX","schema_version":"1.0","canonical_sha256":"702e7b1d57e5085b4a29173f247f92baad88ce202f1e28b14b44a236083208ab","source":{"kind":"arxiv","id":"2410.13816","version":2},"attestation_state":"computed","paper":{"title":"Steering Your Generalists: Improving Robotic Foundation Models via Value Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Aviral Kumar, Mitsuhiko Nakamoto, Oier Mees, Sergey Levine","submitted_at":"2024-10-17T17:46:26Z","abstract_excerpt":"Large, general-purpose robotic policies trained on diverse demonstration datasets have been shown to be remarkably effective both for controlling a variety of robots in a range of different scenes, and for acquiring broad repertoires of manipulation skills. However, the data that such policies are trained on is generally of mixed quality -- not only are human-collected demonstrations unlikely to perform the task perfectly, but the larger the dataset is, the harder it is to curate only the highest quality examples. It also remains unclear how optimal data from one embodiment is for training on "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13816","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-10-17T17:46:26Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1b3eadb43d187a765a48d675cf9ce2b37d336e72ddfcee076e264a04a1dfc860","abstract_canon_sha256":"9cf07d0090d2cb1d2fb954cda60aa2cb146376a9ec91226f865174b6eff93f05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:29.856282Z","signature_b64":"ev7mCmBm/lpD4iXPiy4CGRXe26saRerQ+OFH13Ko3NjZOghO42SXy12lR3JOeDHLZRX3we1bCHlv6CMzL5dPBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"702e7b1d57e5085b4a29173f247f92baad88ce202f1e28b14b44a236083208ab","last_reissued_at":"2026-07-05T10:19:29.855789Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:29.855789Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Steering Your Generalists: Improving Robotic Foundation Models via Value Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Aviral Kumar, Mitsuhiko Nakamoto, Oier Mees, Sergey Levine","submitted_at":"2024-10-17T17:46:26Z","abstract_excerpt":"Large, general-purpose robotic policies trained on diverse demonstration datasets have been shown to be remarkably effective both for controlling a variety of robots in a range of different scenes, and for acquiring broad repertoires of manipulation skills. However, the data that such policies are trained on is generally of mixed quality -- not only are human-collected demonstrations unlikely to perform the task perfectly, but the larger the dataset is, the harder it is to curate only the highest quality examples. It also remains unclear how optimal data from one embodiment is for training on "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13816","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13816/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13816","created_at":"2026-07-05T10:19:29.855845+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13816v2","created_at":"2026-07-05T10:19:29.855845+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13816","created_at":"2026-07-05T10:19:29.855845+00:00"},{"alias_kind":"pith_short_12","alias_value":"OAXHWHKX4UEF","created_at":"2026-07-05T10:19:29.855845+00:00"},{"alias_kind":"pith_short_16","alias_value":"OAXHWHKX4UEFWSRJ","created_at":"2026-07-05T10:19:29.855845+00:00"},{"alias_kind":"pith_short_8","alias_value":"OAXHWHKX","created_at":"2026-07-05T10:19:29.855845+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05391","citing_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","ref_index":86,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26588","citing_title":"Inference-Time Robot Behavior Steering through Physically-Aware Reconfiguration of Task-Structure","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22860","citing_title":"HiL-ResRL: A Model-Agnostic Finetuning Adapter via Human-in-the-loop Residual Reinforcement Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21572","citing_title":"Robot Critics that Sweat the Small Stuff","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19656","citing_title":"DF-ExpEnse: Diffusion Filtered Exploration for Sample Efficient Finetuning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10568","citing_title":"VeriSpace: Spatially Grounded Action Verification for Vision-Language-Action Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11087","citing_title":"Test-Time Gradient Guidance of Flow Policies in Reinforcement Learning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31958","citing_title":"Adapting Generalist Robot Policies with Semantic Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29892","citing_title":"Trust Your Instincts: Confidence-Driven Test-Time RL for Vision-Language-Action Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30660","citing_title":"BOKBO (Best of K Bad Options): Calibrated Abstention for VLA Policies","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15757","citing_title":"You've Got a Golden Ticket: Improving Generative Robot Policies With A Single Noise Vector","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11479","citing_title":"Offline Policy Evaluation for Manipulation Policies via Discounted Liveness Formulation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24661","citing_title":"Agent-Centric Observation Adaptation for Robust Visual Control under Dynamic Perturbations","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23121","citing_title":"Breaking Lock-In: Preserving Steerability under Low-Data VLA Post-Training","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01194","citing_title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08508","citing_title":"Sumo: Dynamic and Generalizable Whole-Body Loco-Manipulation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24661","citing_title":"Agent-Centric Observation Adaptation for Robust Visual Control under Dynamic Perturbations","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06168","citing_title":"Action Images: End-to-End Policy Learning via Multiview Video Generation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19730","citing_title":"FASTER: Value-Guided Sampling for Fast RL","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK","json":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK.json","graph_json":"https://pith.science/api/pith-number/OAXHWHKX4UEFWSRJC47SI74SXK/graph.json","events_json":"https://pith.science/api/pith-number/OAXHWHKX4UEFWSRJC47SI74SXK/events.json","paper":"https://pith.science/paper/OAXHWHKX"},"agent_actions":{"view_html":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK","download_json":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK.json","view_paper":"https://pith.science/paper/OAXHWHKX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13816&json=true","fetch_graph":"https://pith.science/api/pith-number/OAXHWHKX4UEFWSRJC47SI74SXK/graph.json","fetch_events":"https://pith.science/api/pith-number/OAXHWHKX4UEFWSRJC47SI74SXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK/action/storage_attestation","attest_author":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK/action/author_attestation","sign_citation":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK/action/citation_signature","submit_replication":"https://pith.science/pith/OAXHWHKX4UEFWSRJC47SI74SXK/action/replication_record"}},"created_at":"2026-07-05T10:19:29.855845+00:00","updated_at":"2026-07-05T10:19:29.855845+00:00"}