{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SRP5VTLZWIAMXNOQZ3EK5D6QGR","short_pith_number":"pith:SRP5VTLZ","schema_version":"1.0","canonical_sha256":"945fdacd79b200cbb5d0cec8ae8fd0346829f0b419238911131c3d54d227bc9a","source":{"kind":"arxiv","id":"2304.04321","version":2},"attestation_state":"computed","paper":{"title":"ARNOLD: A Benchmark for Language-Grounded Task Learning With Continuous States in Realistic 3D Scenes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.RO"],"primary_cat":"cs.AI","authors_text":"Baoxiong Jia, Demetri Terzopoulos, Haoran Geng, Jiangyong Huang, Qingyang Wu, Ran Gong, Siyuan Huang, Song-Chun Zhu, Wensi Ai, Xiaofeng Gao, Yizhou Zhao, Ziheng Zhou","submitted_at":"2023-04-09T21:42:57Z","abstract_excerpt":"Understanding the continuous states of objects is essential for task learning and planning in the real world. However, most existing task learning benchmarks assume discrete (e.g., binary) object goal states, which poses challenges for the learning of complex tasks and transferring learned policy from simulated environments to the real world. Furthermore, state discretization limits a robot's ability to follow human instructions based on the grounding of actions and states. To tackle these challenges, we present ARNOLD, a benchmark that evaluates language-grounded task learning with continuous"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.04321","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-04-09T21:42:57Z","cross_cats_sorted":["cs.CL","cs.CV","cs.RO"],"title_canon_sha256":"e50a2877b28c5c2e4b012d336da0d55ac5952b31f80299fc5cc30aa38223f790","abstract_canon_sha256":"79a6faeab5abc18d41bb5c702357c4086c3d88c32c217fe733fb3c03623ec55e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:21.273416Z","signature_b64":"laJEXMn6qsgtNiExNa+AA4bXkS6J0UpXc1EJ7AyQihFcr2USk7Zvfk17ZCtNO7OBkxSsv7/o6uXyfETOFMYdAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"945fdacd79b200cbb5d0cec8ae8fd0346829f0b419238911131c3d54d227bc9a","last_reissued_at":"2026-07-05T06:49:21.272920Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:21.272920Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ARNOLD: A Benchmark for Language-Grounded Task Learning With Continuous States in Realistic 3D Scenes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.RO"],"primary_cat":"cs.AI","authors_text":"Baoxiong Jia, Demetri Terzopoulos, Haoran Geng, Jiangyong Huang, Qingyang Wu, Ran Gong, Siyuan Huang, Song-Chun Zhu, Wensi Ai, Xiaofeng Gao, Yizhou Zhao, Ziheng Zhou","submitted_at":"2023-04-09T21:42:57Z","abstract_excerpt":"Understanding the continuous states of objects is essential for task learning and planning in the real world. However, most existing task learning benchmarks assume discrete (e.g., binary) object goal states, which poses challenges for the learning of complex tasks and transferring learned policy from simulated environments to the real world. Furthermore, state discretization limits a robot's ability to follow human instructions based on the grounding of actions and states. To tackle these challenges, we present ARNOLD, a benchmark that evaluates language-grounded task learning with continuous"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.04321","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.04321/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.04321","created_at":"2026-07-05T06:49:21.272970+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.04321v2","created_at":"2026-07-05T06:49:21.272970+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.04321","created_at":"2026-07-05T06:49:21.272970+00:00"},{"alias_kind":"pith_short_12","alias_value":"SRP5VTLZWIAM","created_at":"2026-07-05T06:49:21.272970+00:00"},{"alias_kind":"pith_short_16","alias_value":"SRP5VTLZWIAMXNOQ","created_at":"2026-07-05T06:49:21.272970+00:00"},{"alias_kind":"pith_short_8","alias_value":"SRP5VTLZ","created_at":"2026-07-05T06:49:21.272970+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10366","citing_title":"A Practical Recipe Towards Improving Sim-and-Real Correlation for VLA Evaluation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16886","citing_title":"Chain Of Interaction Benchmark (COIN): When Reasoning meets Embodied Interaction","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR","json":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR.json","graph_json":"https://pith.science/api/pith-number/SRP5VTLZWIAMXNOQZ3EK5D6QGR/graph.json","events_json":"https://pith.science/api/pith-number/SRP5VTLZWIAMXNOQZ3EK5D6QGR/events.json","paper":"https://pith.science/paper/SRP5VTLZ"},"agent_actions":{"view_html":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR","download_json":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR.json","view_paper":"https://pith.science/paper/SRP5VTLZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.04321&json=true","fetch_graph":"https://pith.science/api/pith-number/SRP5VTLZWIAMXNOQZ3EK5D6QGR/graph.json","fetch_events":"https://pith.science/api/pith-number/SRP5VTLZWIAMXNOQZ3EK5D6QGR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR/action/storage_attestation","attest_author":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR/action/author_attestation","sign_citation":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR/action/citation_signature","submit_replication":"https://pith.science/pith/SRP5VTLZWIAMXNOQZ3EK5D6QGR/action/replication_record"}},"created_at":"2026-07-05T06:49:21.272970+00:00","updated_at":"2026-07-05T06:49:21.272970+00:00"}