{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KEAEJJMTG7MIKLNBFNIJSM4IP3","short_pith_number":"pith:KEAEJJMT","schema_version":"1.0","canonical_sha256":"510044a59337d8852da12b509933887ed95c0d1bc669bc4a4a0c9347fc26c2ce","source":{"kind":"arxiv","id":"2411.03540","version":1},"attestation_state":"computed","paper":{"title":"VLA-3D: A Dataset for 3D Semantic Scene Understanding and Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Haochen Zhang, Ji Zhang, Nader Zantout, Pujith Kachana, Wenshan Wang, Zongyuan Wu","submitted_at":"2024-11-05T22:42:41Z","abstract_excerpt":"With the recent rise of Large Language Models (LLMs), Vision-Language Models (VLMs), and other general foundation models, there is growing potential for multimodal, multi-task embodied agents that can operate in diverse environments given only natural language as input. One such application area is indoor navigation using natural language instructions. However, despite recent progress, this problem remains challenging due to the spatial reasoning and semantic understanding required, particularly in arbitrary scenes that may contain many objects belonging to fine-grained classes. To address thi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.03540","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-11-05T22:42:41Z","cross_cats_sorted":[],"title_canon_sha256":"1352e37a1ce28b904f87348057d6d0da3d7389cecadadaf1629d8ad0c00f5790","abstract_canon_sha256":"ab17a488c006ce395591679d9d667945a0186f08800f4fdea6ea2e783ac715d2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:34.181820Z","signature_b64":"I02A7UkNoIeNoyKIcp2IeD+dBzQ4hCaborj+8VdvZID8arJP+9zeHkH2N0uGfp9LJeL3IJPhPDE36hi7qg7QCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"510044a59337d8852da12b509933887ed95c0d1bc669bc4a4a0c9347fc26c2ce","last_reissued_at":"2026-07-05T09:31:34.181331Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:34.181331Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VLA-3D: A Dataset for 3D Semantic Scene Understanding and Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Haochen Zhang, Ji Zhang, Nader Zantout, Pujith Kachana, Wenshan Wang, Zongyuan Wu","submitted_at":"2024-11-05T22:42:41Z","abstract_excerpt":"With the recent rise of Large Language Models (LLMs), Vision-Language Models (VLMs), and other general foundation models, there is growing potential for multimodal, multi-task embodied agents that can operate in diverse environments given only natural language as input. One such application area is indoor navigation using natural language instructions. However, despite recent progress, this problem remains challenging due to the spatial reasoning and semantic understanding required, particularly in arbitrary scenes that may contain many objects belonging to fine-grained classes. To address thi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.03540","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.03540/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.03540","created_at":"2026-07-05T09:31:34.181393+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.03540v1","created_at":"2026-07-05T09:31:34.181393+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.03540","created_at":"2026-07-05T09:31:34.181393+00:00"},{"alias_kind":"pith_short_12","alias_value":"KEAEJJMTG7MI","created_at":"2026-07-05T09:31:34.181393+00:00"},{"alias_kind":"pith_short_16","alias_value":"KEAEJJMTG7MIKLNB","created_at":"2026-07-05T09:31:34.181393+00:00"},{"alias_kind":"pith_short_8","alias_value":"KEAEJJMT","created_at":"2026-07-05T09:31:34.181393+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06556","citing_title":"Robots Need More than VLA and World Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31144","citing_title":"A Modular Vision-Language-Action Robotics Framework for Indoor Environments","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10484","citing_title":"OpenSGA: Efficient 3D Scene Graph Alignment in the Open World","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17800","citing_title":"ReFineVLA: Multimodal Reasoning-Aware Generalist Robotic Policies via Teacher-Guided Fine-Tuning","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3","json":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3.json","graph_json":"https://pith.science/api/pith-number/KEAEJJMTG7MIKLNBFNIJSM4IP3/graph.json","events_json":"https://pith.science/api/pith-number/KEAEJJMTG7MIKLNBFNIJSM4IP3/events.json","paper":"https://pith.science/paper/KEAEJJMT"},"agent_actions":{"view_html":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3","download_json":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3.json","view_paper":"https://pith.science/paper/KEAEJJMT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.03540&json=true","fetch_graph":"https://pith.science/api/pith-number/KEAEJJMTG7MIKLNBFNIJSM4IP3/graph.json","fetch_events":"https://pith.science/api/pith-number/KEAEJJMTG7MIKLNBFNIJSM4IP3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3/action/storage_attestation","attest_author":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3/action/author_attestation","sign_citation":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3/action/citation_signature","submit_replication":"https://pith.science/pith/KEAEJJMTG7MIKLNBFNIJSM4IP3/action/replication_record"}},"created_at":"2026-07-05T09:31:34.181393+00:00","updated_at":"2026-07-05T09:31:34.181393+00:00"}