{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CFJUE25W7JNTPDM6YDJBK5FQAR","short_pith_number":"pith:CFJUE25W","schema_version":"1.0","canonical_sha256":"1153426bb6fa5b378d9ec0d21574b004459fef7112521e7ab2c0316e4c75e623","source":{"kind":"arxiv","id":"2411.05821","version":2},"attestation_state":"computed","paper":{"title":"Benchmarking Vision, Language, & Action Models on Robotic Learning Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Harshvardhan Sikka, Jaewoo Song, Paul Pu Liang, Pranav Guruprasad, Yangyue Wang","submitted_at":"2024-11-04T18:01:34Z","abstract_excerpt":"Vision-language-action (VLA) models represent a promising direction for developing general-purpose robotic systems, demonstrating the ability to combine visual understanding, language comprehension, and action generation. However, systematic evaluation of these models across diverse robotic tasks remains limited. In this work, we present a comprehensive evaluation framework and benchmark suite for assessing VLA models. We profile three state-of-the-art VLM and VLAs - GPT-4o, OpenVLA, and JAT - across 20 diverse datasets from the Open-X-Embodiment collection, evaluating their performance on var"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.05821","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-11-04T18:01:34Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"6f049699e2ddd50c46f23c937646594883852c4ba1526f48fadf5c445f0dff1e","abstract_canon_sha256":"401b11369c278685693f24ed5e563ab046dae83003d1ae625b1daa29232aaa6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:57.532182Z","signature_b64":"4SDXcSnJ4mg3gLVcR8T5eYRLF2XUZb7rU0wIfhuLgRMJPEIfn+SwSIWRHI1ucYgN0DdI9XIIEedns0AzNKEAAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1153426bb6fa5b378d9ec0d21574b004459fef7112521e7ab2c0316e4c75e623","last_reissued_at":"2026-07-05T09:45:57.531644Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:57.531644Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Vision, Language, & Action Models on Robotic Learning Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Harshvardhan Sikka, Jaewoo Song, Paul Pu Liang, Pranav Guruprasad, Yangyue Wang","submitted_at":"2024-11-04T18:01:34Z","abstract_excerpt":"Vision-language-action (VLA) models represent a promising direction for developing general-purpose robotic systems, demonstrating the ability to combine visual understanding, language comprehension, and action generation. However, systematic evaluation of these models across diverse robotic tasks remains limited. In this work, we present a comprehensive evaluation framework and benchmark suite for assessing VLA models. We profile three state-of-the-art VLM and VLAs - GPT-4o, OpenVLA, and JAT - across 20 diverse datasets from the Open-X-Embodiment collection, evaluating their performance on var"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.05821","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.05821/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.05821","created_at":"2026-07-05T09:45:57.531705+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.05821v2","created_at":"2026-07-05T09:45:57.531705+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.05821","created_at":"2026-07-05T09:45:57.531705+00:00"},{"alias_kind":"pith_short_12","alias_value":"CFJUE25W7JNT","created_at":"2026-07-05T09:45:57.531705+00:00"},{"alias_kind":"pith_short_16","alias_value":"CFJUE25W7JNTPDM6","created_at":"2026-07-05T09:45:57.531705+00:00"},{"alias_kind":"pith_short_8","alias_value":"CFJUE25W","created_at":"2026-07-05T09:45:57.531705+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10267","citing_title":"What Matters in Orchestrating Robot Policies: A Systematic Study of Hierarchical VLA Agents","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02387","citing_title":"VS-Bench: Evaluating VLMs for Strategic Abilities in Multi-Agent Environments","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03827","citing_title":"LIBERO-PRO: Towards Robust and Fair Evaluation of Vision-Language-Action Models Beyond Memorization","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR","json":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR.json","graph_json":"https://pith.science/api/pith-number/CFJUE25W7JNTPDM6YDJBK5FQAR/graph.json","events_json":"https://pith.science/api/pith-number/CFJUE25W7JNTPDM6YDJBK5FQAR/events.json","paper":"https://pith.science/paper/CFJUE25W"},"agent_actions":{"view_html":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR","download_json":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR.json","view_paper":"https://pith.science/paper/CFJUE25W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.05821&json=true","fetch_graph":"https://pith.science/api/pith-number/CFJUE25W7JNTPDM6YDJBK5FQAR/graph.json","fetch_events":"https://pith.science/api/pith-number/CFJUE25W7JNTPDM6YDJBK5FQAR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR/action/storage_attestation","attest_author":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR/action/author_attestation","sign_citation":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR/action/citation_signature","submit_replication":"https://pith.science/pith/CFJUE25W7JNTPDM6YDJBK5FQAR/action/replication_record"}},"created_at":"2026-07-05T09:45:57.531705+00:00","updated_at":"2026-07-05T09:45:57.531705+00:00"}