{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MHJA455FLC6PWZYWWIZ54BDXGX","short_pith_number":"pith:MHJA455F","schema_version":"1.0","canonical_sha256":"61d20e77a558bcfb6716b233de047735ff2df3fa1b9a4844d39bd994885c2abd","source":{"kind":"arxiv","id":"2506.17811","version":2},"attestation_state":"computed","paper":{"title":"RoboMonkey: Scaling Test-Time Sampling and Verification for Vision-Language-Action Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Azalia Mirhoseini, Christopher Agia, Ion Stoica, Jacky Kwok, Marco Pavone, Matt Foutter, Rohan Sinha, Shulu Li","submitted_at":"2025-06-21T20:56:17Z","abstract_excerpt":"Vision-Language-Action (VLA) models have demonstrated remarkable capabilities in visuomotor control, yet ensuring their robustness in unstructured real-world environments remains a persistent challenge. In this paper, we investigate test-time scaling through the lens of sampling and verification as means to enhance the robustness and generalization of VLAs. We first demonstrate that the relationship between action error and the number of generated samples follows an exponentiated power law across a range of VLAs, indicating the existence of inference-time scaling laws. Building on these insigh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.17811","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-06-21T20:56:17Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"title_canon_sha256":"1f871150ec1abd721492e21034ce1d6ce5bc237bbdbe126f1ce503518e9c054b","abstract_canon_sha256":"49748365c0496885133837797989bb68826355867faa85edf2b8d86879535443"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:31.840256Z","signature_b64":"jN8VOOyPqWJDSTbNUxayLtSgjOYPToZKjl/e7O9st7MlJTQ752rLouRC0OU8h7azxdRO9ruOyBxPgHiojLc+Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61d20e77a558bcfb6716b233de047735ff2df3fa1b9a4844d39bd994885c2abd","last_reissued_at":"2026-07-05T11:32:31.839753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:31.839753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RoboMonkey: Scaling Test-Time Sampling and Verification for Vision-Language-Action Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Azalia Mirhoseini, Christopher Agia, Ion Stoica, Jacky Kwok, Marco Pavone, Matt Foutter, Rohan Sinha, Shulu Li","submitted_at":"2025-06-21T20:56:17Z","abstract_excerpt":"Vision-Language-Action (VLA) models have demonstrated remarkable capabilities in visuomotor control, yet ensuring their robustness in unstructured real-world environments remains a persistent challenge. In this paper, we investigate test-time scaling through the lens of sampling and verification as means to enhance the robustness and generalization of VLAs. We first demonstrate that the relationship between action error and the number of generated samples follows an exponentiated power law across a range of VLAs, indicating the existence of inference-time scaling laws. Building on these insigh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.17811","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.17811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.17811","created_at":"2026-07-05T11:32:31.839814+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.17811v2","created_at":"2026-07-05T11:32:31.839814+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.17811","created_at":"2026-07-05T11:32:31.839814+00:00"},{"alias_kind":"pith_short_12","alias_value":"MHJA455FLC6P","created_at":"2026-07-05T11:32:31.839814+00:00"},{"alias_kind":"pith_short_16","alias_value":"MHJA455FLC6PWZYW","created_at":"2026-07-05T11:32:31.839814+00:00"},{"alias_kind":"pith_short_8","alias_value":"MHJA455F","created_at":"2026-07-05T11:32:31.839814+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26080","citing_title":"Neglected Free Lunch from Post-training: Progress Advantage for LLM Agents","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27268","citing_title":"E-TTS: A New Embodied Test-Time Scaling Framework for Robotic Manipulation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22998","citing_title":"TEXEDO : Test Time Scaling for Controller-aware Language-conditioned Humanoid Motion Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18589","citing_title":"DREAM-Chunk: Reactive Action Chunking with Latent World Model","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13675","citing_title":"Improving Robotic Generalist Policies via Flow Reversal Steering","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10568","citing_title":"VeriSpace: Spatially Grounded Action Verification for Vision-Language-Action Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08231","citing_title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08015","citing_title":"Q-VGM: Q-Value-Gradient Matching for Off-Policy Reinforcement Learning of Flow-Matching VLA","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31132","citing_title":"ELASTIC: Efficiently Learning to Adaptively Scale Test-Time Compute for Generative Control Policies","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30613","citing_title":"Sequential Planning via Anchored Robotic Keypoints","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30660","citing_title":"BOKBO (Best of K Bad Options): Calibrated Abstention for VLA Policies","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01036","citing_title":"Position: Good Embodied Reward Models Need Bad Behavior Data","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26006","citing_title":"FORCE: Efficient VLA Reinforcement Fine-Tuning via Value-Calibrated Warm-up and Self-Distillation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14093","citing_title":"A Survey on Vision-Language-Action Models for Embodied AI","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13193","citing_title":"Steerable Vision-Language-Action Policies for Embodied Reasoning and Hierarchical Control","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2602.22474","citing_title":"When to Act, Ask, or Learn: Uncertainty-Aware Policy Steering","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12620","citing_title":"Think Twice, Act Once: Verifier-Guided Action Selection For Embodied Agents","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10094","citing_title":"Retrieve-then-Steer: Online Success Memory for Test-Time Adaptation of Generative VLAs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10094","citing_title":"Retrieve-then-Steer: Online Success Memory for Test-Time Adaptation of Generative VLAs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01194","citing_title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05672","citing_title":"A1: A Fully Transparent Open-Source, Adaptive and Efficient Truncated Vision-Language-Action Model","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18107","citing_title":"Test-Time Perturbation Learning with Delayed Feedback for Vision-Language-Action Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19730","citing_title":"FASTER: Value-Guided Sampling for Fast RL","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX","json":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX.json","graph_json":"https://pith.science/api/pith-number/MHJA455FLC6PWZYWWIZ54BDXGX/graph.json","events_json":"https://pith.science/api/pith-number/MHJA455FLC6PWZYWWIZ54BDXGX/events.json","paper":"https://pith.science/paper/MHJA455F"},"agent_actions":{"view_html":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX","download_json":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX.json","view_paper":"https://pith.science/paper/MHJA455F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.17811&json=true","fetch_graph":"https://pith.science/api/pith-number/MHJA455FLC6PWZYWWIZ54BDXGX/graph.json","fetch_events":"https://pith.science/api/pith-number/MHJA455FLC6PWZYWWIZ54BDXGX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX/action/storage_attestation","attest_author":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX/action/author_attestation","sign_citation":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX/action/citation_signature","submit_replication":"https://pith.science/pith/MHJA455FLC6PWZYWWIZ54BDXGX/action/replication_record"}},"created_at":"2026-07-05T11:32:31.839814+00:00","updated_at":"2026-07-05T11:32:31.839814+00:00"}