{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:X4SOMCBVXGIVGYZ5NDSSYTZ6Q4","short_pith_number":"pith:X4SOMCBV","canonical_record":{"source":{"id":"2410.06355","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-08T20:46:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0a2d20d9248a52d1c7bc49a626a88f8d863b908ec6da9247fbc9d0b384a90930","abstract_canon_sha256":"5e4711413644e1bddabaf5d238176f0fc4cd00fc8f0f1388438caad692a33142"},"schema_version":"1.0"},"canonical_sha256":"bf24e60835b99153633d68e52c4f3e87141e32946b0638f56d86e994d5e1ab40","source":{"kind":"arxiv","id":"2410.06355","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.06355","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"arxiv_version","alias_value":"2410.06355v3","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.06355","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"pith_short_12","alias_value":"X4SOMCBVXGIV","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"pith_short_16","alias_value":"X4SOMCBVXGIVGYZ5","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"pith_short_8","alias_value":"X4SOMCBV","created_at":"2026-06-30T01:17:20Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:X4SOMCBVXGIVGYZ5NDSSYTZ6Q4","target":"record","payload":{"canonical_record":{"source":{"id":"2410.06355","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-08T20:46:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0a2d20d9248a52d1c7bc49a626a88f8d863b908ec6da9247fbc9d0b384a90930","abstract_canon_sha256":"5e4711413644e1bddabaf5d238176f0fc4cd00fc8f0f1388438caad692a33142"},"schema_version":"1.0"},"canonical_sha256":"bf24e60835b99153633d68e52c4f3e87141e32946b0638f56d86e994d5e1ab40","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-30T01:17:20.735729Z","signature_b64":"MX43gla3w+CE6ijGk8LJaxqd/CHnVcg0q+Jp/vVdgzWWgCtyTLKSHLo30nPQc6UeAiDEtkBg8W9rjHM0Xu64DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf24e60835b99153633d68e52c4f3e87141e32946b0638f56d86e994d5e1ab40","last_reissued_at":"2026-06-30T01:17:20.735059Z","signature_status":"signed_v1","first_computed_at":"2026-06-30T01:17:20.735059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.06355","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-30T01:17:20Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sDvFILyQRG62aNg8WR0DQGaIkhtgImlo/BW+6ERRXz1Pf/5xaXQJsX+iMWYWLOHlUwO2gLPKhjWWMXOEi9A+Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T02:44:30.491524Z"},"content_sha256":"8b4d58f223e718d6d46e8b35f984decd82faa0c38c6e0619d908292d633c13bb","schema_version":"1.0","event_id":"sha256:8b4d58f223e718d6d46e8b35f984decd82faa0c38c6e0619d908292d633c13bb"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:X4SOMCBVXGIVGYZ5NDSSYTZ6Q4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"UNCOM: Zero-shot Context-Aware Command Understanding for Tabletop Scenarios","license":"http://creativecommons.org/licenses/by/4.0/","headline":"A modular system fuses speech, gestures, and scene context to understand natural commands for robots without task-specific training.","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Antonio Galiza Cerdeira Gonzalez, Bipin Indurkhya, Pawe{\\l} Gajewski","submitted_at":"2024-10-08T20:46:39Z","abstract_excerpt":"This paper presents UNCOM, a novel hybrid framework for interpreting natural human commands in tabletop scenarios. The system integrates multiple sources of information -- speech, gestures, and scene context -- to extract structured, actionable instructions for robots. Addressing the need for general-purpose human-robot interaction in domestic environments, UNCOM is designed for zero-shot operation, without reliance on predefined object models or training data specific to a given task. Using foundational and task-specific deep learning models, it allows out-of-the-box speech recognition, natur"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"The system achieves an 82.39% success rate over our benchmark data set, highlighting the robustness of the system to diversity, noise, and communication ambiguity.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"Foundational deep learning models for speech recognition, natural language understanding, gesture detection, and object segmentation can be applied directly out-of-the-box to tabletop scenarios without task-specific fine-tuning or adaptation.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"UNCOM integrates off-the-shelf multimodal AI models into a modular zero-shot system that parses commands into object-action-target representations and achieves 82.39% success on a real-world tabletop HRI benchmark with public code and data.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A modular system fuses speech, gestures, and scene context to understand natural commands for robots without task-specific training.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"dd02992a15b335a2bd6bfef63ab4c25c0fc02d864af31e95b7b386ba8dc1459d"},"source":{"id":"2410.06355","kind":"arxiv","version":3},"verdict":{"id":"afaec499-3800-4060-b3ee-bbfc44918e8e","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-23T19:15:09.763172Z","strongest_claim":"The system achieves an 82.39% success rate over our benchmark data set, highlighting the robustness of the system to diversity, noise, and communication ambiguity.","one_line_summary":"UNCOM integrates off-the-shelf multimodal AI models into a modular zero-shot system that parses commands into object-action-target representations and achieves 82.39% success on a real-world tabletop HRI benchmark with public code and data.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"Foundational deep learning models for speech recognition, natural language understanding, gesture detection, and object segmentation can be applied directly out-of-the-box to tabletop scenarios without task-specific fine-tuning or adaptation.","pith_extraction_headline":"A modular system fuses speech, gestures, and scene context to understand natural commands for robots without task-specific training."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.06355/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":28,"sample":[{"doi":"","year":2024,"title":"OpenVLA: An Open-Source Vision-Language-Action Model","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","ref_index":1,"cited_arxiv_id":"2406.09246","is_internal_anchor":true},{"doi":"10.1080/01691864.2016.1277554","year":2017,"title":"A review of spatial reasoning and interaction for real-world robotics","work_id":"13c297f5-13c3-4576-8ab8-f79ad6201efc","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2019,"title":"Adapting everyday manipulation skills to varied scenarios","work_id":"a8991af6-d9f2-4bd3-a808-a8e159915b74","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2022,"title":"An approach to task representation based on object features and affordances","work_id":"4a072611-dd24-4df5-b23a-8a1f7221e4ba","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2010,"title":"Cram—a cognitive robot abstract machine for everyday manipulation in human environments","work_id":"1f34e4c9-453e-4589-aaf0-bc0365504167","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":28,"snapshot_sha256":"f7e2f3e1e60ece4bf1daf73e86c7fe8ea2b8c0954d968ad0b031e0947306934c","internal_anchors":6},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"afaec499-3800-4060-b3ee-bbfc44918e8e"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-30T01:17:20Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"znCu3Thutbh3bWEzDVbJqoeNrTEQl+t0aN9vJ4RMYF3a2oCqRac/xR02zC5onWBsPGdXBEC9hFXYLoBSy/O5Bg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T02:44:30.492361Z"},"content_sha256":"32dd4f23cb97bb480269da75be635daee0847b4914bd9978c377b2d15bc34875","schema_version":"1.0","event_id":"sha256:32dd4f23cb97bb480269da75be635daee0847b4914bd9978c377b2d15bc34875"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/X4SOMCBVXGIVGYZ5NDSSYTZ6Q4/bundle.json","state_url":"https://pith.science/pith/X4SOMCBVXGIVGYZ5NDSSYTZ6Q4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/X4SOMCBVXGIVGYZ5NDSSYTZ6Q4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T02:44:30Z","links":{"resolver":"https://pith.science/pith/X4SOMCBVXGIVGYZ5NDSSYTZ6Q4","bundle":"https://pith.science/pith/X4SOMCBVXGIVGYZ5NDSSYTZ6Q4/bundle.json","state":"https://pith.science/pith/X4SOMCBVXGIVGYZ5NDSSYTZ6Q4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/X4SOMCBVXGIVGYZ5NDSSYTZ6Q4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:X4SOMCBVXGIVGYZ5NDSSYTZ6Q4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5e4711413644e1bddabaf5d238176f0fc4cd00fc8f0f1388438caad692a33142","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-08T20:46:39Z","title_canon_sha256":"0a2d20d9248a52d1c7bc49a626a88f8d863b908ec6da9247fbc9d0b384a90930"},"schema_version":"1.0","source":{"id":"2410.06355","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.06355","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"arxiv_version","alias_value":"2410.06355v3","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.06355","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"pith_short_12","alias_value":"X4SOMCBVXGIV","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"pith_short_16","alias_value":"X4SOMCBVXGIVGYZ5","created_at":"2026-06-30T01:17:20Z"},{"alias_kind":"pith_short_8","alias_value":"X4SOMCBV","created_at":"2026-06-30T01:17:20Z"}],"graph_snapshots":[{"event_id":"sha256:32dd4f23cb97bb480269da75be635daee0847b4914bd9978c377b2d15bc34875","target":"graph","created_at":"2026-06-30T01:17:20Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"The system achieves an 82.39% success rate over our benchmark data set, highlighting the robustness of the system to diversity, noise, and communication ambiguity."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"Foundational deep learning models for speech recognition, natural language understanding, gesture detection, and object segmentation can be applied directly out-of-the-box to tabletop scenarios without task-specific fine-tuning or adaptation."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"UNCOM integrates off-the-shelf multimodal AI models into a modular zero-shot system that parses commands into object-action-target representations and achieves 82.39% success on a real-world tabletop HRI benchmark with public code and data."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"A modular system fuses speech, gestures, and scene context to understand natural commands for robots without task-specific training."}],"snapshot_sha256":"dd02992a15b335a2bd6bfef63ab4c25c0fc02d864af31e95b7b386ba8dc1459d"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.06355/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper presents UNCOM, a novel hybrid framework for interpreting natural human commands in tabletop scenarios. The system integrates multiple sources of information -- speech, gestures, and scene context -- to extract structured, actionable instructions for robots. Addressing the need for general-purpose human-robot interaction in domestic environments, UNCOM is designed for zero-shot operation, without reliance on predefined object models or training data specific to a given task. Using foundational and task-specific deep learning models, it allows out-of-the-box speech recognition, natur","authors_text":"Antonio Galiza Cerdeira Gonzalez, Bipin Indurkhya, Pawe{\\l} Gajewski","cross_cats":["cs.AI"],"headline":"A modular system fuses speech, gestures, and scene context to understand natural commands for robots without task-specific training.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-08T20:46:39Z","title":"UNCOM: Zero-shot Context-Aware Command Understanding for Tabletop Scenarios"},"references":{"count":28,"internal_anchors":6,"resolved_work":28,"sample":[{"cited_arxiv_id":"2406.09246","doi":"","is_internal_anchor":true,"ref_index":1,"title":"OpenVLA: An Open-Source Vision-Language-Action Model","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","year":2024},{"cited_arxiv_id":"","doi":"10.1080/01691864.2016.1277554","is_internal_anchor":false,"ref_index":2,"title":"A review of spatial reasoning and interaction for real-world robotics","work_id":"13c297f5-13c3-4576-8ab8-f79ad6201efc","year":2017},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Adapting everyday manipulation skills to varied scenarios","work_id":"a8991af6-d9f2-4bd3-a808-a8e159915b74","year":2019},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":4,"title":"An approach to task representation based on object features and affordances","work_id":"4a072611-dd24-4df5-b23a-8a1f7221e4ba","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Cram—a cognitive robot abstract machine for everyday manipulation in human environments","work_id":"1f34e4c9-453e-4589-aaf0-bc0365504167","year":2010}],"snapshot_sha256":"f7e2f3e1e60ece4bf1daf73e86c7fe8ea2b8c0954d968ad0b031e0947306934c"},"source":{"id":"2410.06355","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-23T19:15:09.763172Z","id":"afaec499-3800-4060-b3ee-bbfc44918e8e","model_set":{"reader":"grok-4.3"},"one_line_summary":"UNCOM integrates off-the-shelf multimodal AI models into a modular zero-shot system that parses commands into object-action-target representations and achieves 82.39% success on a real-world tabletop HRI benchmark with public code and data.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"A modular system fuses speech, gestures, and scene context to understand natural commands for robots without task-specific training.","strongest_claim":"The system achieves an 82.39% success rate over our benchmark data set, highlighting the robustness of the system to diversity, noise, and communication ambiguity.","weakest_assumption":"Foundational deep learning models for speech recognition, natural language understanding, gesture detection, and object segmentation can be applied directly out-of-the-box to tabletop scenarios without task-specific fine-tuning or adaptation."}},"verdict_id":"afaec499-3800-4060-b3ee-bbfc44918e8e"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8b4d58f223e718d6d46e8b35f984decd82faa0c38c6e0619d908292d633c13bb","target":"record","created_at":"2026-06-30T01:17:20Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5e4711413644e1bddabaf5d238176f0fc4cd00fc8f0f1388438caad692a33142","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-08T20:46:39Z","title_canon_sha256":"0a2d20d9248a52d1c7bc49a626a88f8d863b908ec6da9247fbc9d0b384a90930"},"schema_version":"1.0","source":{"id":"2410.06355","kind":"arxiv","version":3}},"canonical_sha256":"bf24e60835b99153633d68e52c4f3e87141e32946b0638f56d86e994d5e1ab40","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"bf24e60835b99153633d68e52c4f3e87141e32946b0638f56d86e994d5e1ab40","first_computed_at":"2026-06-30T01:17:20.735059Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-30T01:17:20.735059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"MX43gla3w+CE6ijGk8LJaxqd/CHnVcg0q+Jp/vVdgzWWgCtyTLKSHLo30nPQc6UeAiDEtkBg8W9rjHM0Xu64DQ==","signature_status":"signed_v1","signed_at":"2026-06-30T01:17:20.735729Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.06355","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8b4d58f223e718d6d46e8b35f984decd82faa0c38c6e0619d908292d633c13bb","sha256:32dd4f23cb97bb480269da75be635daee0847b4914bd9978c377b2d15bc34875"],"state_sha256":"adcd18ce8a5e143c033eccf40c1c12388875307777972130e4f89545791c08e1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FWjX5qf047CeDn6hXRolnWv6boYVWxkJ6szoAfDTSDPVzL9ZrZdlW5348bh/3wsz50Zouf0fxE9W+KVSxBDPDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T02:44:30.496538Z","bundle_sha256":"3df950f045ced5a2a273558c58cda9318ba07d35f152a10dd121b4378bbf8898"}}