{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JWJNCFNBCVQ44J7OQOOEUMMBUB","short_pith_number":"pith:JWJNCFNB","schema_version":"1.0","canonical_sha256":"4d92d115a11561ce27ee839c4a3181a040e148216bbe1f7a98e2b900a711d746","source":{"kind":"arxiv","id":"2507.17049","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Uncertainty and Quality of Visual Language Action-enabled Robots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.SE","authors_text":"Aitor Arrieta, Chengjie Lu, Pablo Valle, Shaukat Ali","submitted_at":"2025-07-22T22:15:59Z","abstract_excerpt":"Visual Language Action (VLA) models are a multi-modal class of Artificial Intelligence (AI) systems that integrate visual perception, natural language understanding, and action planning to enable agents to interpret their environment, comprehend instructions, and perform embodied tasks autonomously. Recently, significant progress has been made to advance this field. These kinds of models are typically evaluated through task success rates, which fail to capture the quality of task execution and the mode's confidence in its decisions. In this paper, we propose eight uncertainty metrics and five "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.17049","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-07-22T22:15:59Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"f47fc2950610ecc65417bf9f9a53de75059c0850b67d4a88a3180eff485e7b05","abstract_canon_sha256":"e2fba265d507ca425f32fc3a08cccff5eacf5dd97364da0c858d72c571b19411"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:45.297760Z","signature_b64":"pU6xOVFksg8jRNlTzgpcCuqCwbkUOp1Z1Hv4wmrB2ZG6oMmI0Sn9yuO2DTa6/tiVEjf5AVhpeYiBB/+MsPtWCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d92d115a11561ce27ee839c4a3181a040e148216bbe1f7a98e2b900a711d746","last_reissued_at":"2026-07-05T11:46:45.297268Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:45.297268Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Uncertainty and Quality of Visual Language Action-enabled Robots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.SE","authors_text":"Aitor Arrieta, Chengjie Lu, Pablo Valle, Shaukat Ali","submitted_at":"2025-07-22T22:15:59Z","abstract_excerpt":"Visual Language Action (VLA) models are a multi-modal class of Artificial Intelligence (AI) systems that integrate visual perception, natural language understanding, and action planning to enable agents to interpret their environment, comprehend instructions, and perform embodied tasks autonomously. Recently, significant progress has been made to advance this field. These kinds of models are typically evaluated through task success rates, which fail to capture the quality of task execution and the mode's confidence in its decisions. In this paper, we propose eight uncertainty metrics and five "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.17049","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.17049/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.17049","created_at":"2026-07-05T11:46:45.297338+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.17049v2","created_at":"2026-07-05T11:46:45.297338+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.17049","created_at":"2026-07-05T11:46:45.297338+00:00"},{"alias_kind":"pith_short_12","alias_value":"JWJNCFNBCVQ4","created_at":"2026-07-05T11:46:45.297338+00:00"},{"alias_kind":"pith_short_16","alias_value":"JWJNCFNBCVQ44J7O","created_at":"2026-07-05T11:46:45.297338+00:00"},{"alias_kind":"pith_short_8","alias_value":"JWJNCFNB","created_at":"2026-07-05T11:46:45.297338+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":10,"sample":[{"citing_arxiv_id":"2606.24815","citing_title":"MANGO: Automated Multi-Agent Test Oracle Generation for Vision-Language-Action Models","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02307","citing_title":"FATE-VLA:Failue-aware test generation for vision-language-action models","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.00053","citing_title":"VLAMotor: Test-Guided Enhancement of Vision-Language-Action Models via Agent-BasedData Synthesis","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.10408","citing_title":"VISOR: A Vision-Language Model-based Test Oracle for Testing Robots","ref_index":55,"is_internal_anchor":true},{"citing_arxiv_id":"2510.03827","citing_title":"LIBERO-PRO: Towards Robust and Fair Evaluation of Vision-Language-Action Models Beyond Memorization","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2602.22474","citing_title":"When to Act, Ask, or Learn: Uncertainty-Aware Policy Steering","ref_index":26,"is_internal_anchor":true},{"citing_arxiv_id":"2605.10408","citing_title":"VISOR: A Vision-Language Model-based Test Oracle for Testing Robots","ref_index":55,"is_internal_anchor":true},{"citing_arxiv_id":"2604.25161","citing_title":"Where Did It Go Wrong? Capability-Oriented Failure Attribution for Vision-and-Language Navigation Agents","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2604.23775","citing_title":"Vision-Language-Action Safety: Threats, Challenges, Evaluations, and Mechanisms","ref_index":67,"is_internal_anchor":true},{"citing_arxiv_id":"2604.16677","citing_title":"ReconVLA: An Uncertainty-Guided and Failure-Aware Vision-Language-Action Framework for Robotic Control","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB","json":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB.json","graph_json":"https://pith.science/api/pith-number/JWJNCFNBCVQ44J7OQOOEUMMBUB/graph.json","events_json":"https://pith.science/api/pith-number/JWJNCFNBCVQ44J7OQOOEUMMBUB/events.json","paper":"https://pith.science/paper/JWJNCFNB"},"agent_actions":{"view_html":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB","download_json":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB.json","view_paper":"https://pith.science/paper/JWJNCFNB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.17049&json=true","fetch_graph":"https://pith.science/api/pith-number/JWJNCFNBCVQ44J7OQOOEUMMBUB/graph.json","fetch_events":"https://pith.science/api/pith-number/JWJNCFNBCVQ44J7OQOOEUMMBUB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB/action/storage_attestation","attest_author":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB/action/author_attestation","sign_citation":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB/action/citation_signature","submit_replication":"https://pith.science/pith/JWJNCFNBCVQ44J7OQOOEUMMBUB/action/replication_record"}},"created_at":"2026-07-05T11:46:45.297338+00:00","updated_at":"2026-07-05T11:46:45.297338+00:00"}