{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RRTYS6VWDBKKWY7B42C5CTL5SZ","short_pith_number":"pith:RRTYS6VW","schema_version":"1.0","canonical_sha256":"8c67897ab61854ab63e1e685d14d7d967a6da6887ae80e625cdeb629cc01ae7c","source":{"kind":"arxiv","id":"2501.10074","version":3},"attestation_state":"computed","paper":{"title":"SpatialCoT: Advancing Spatial Reasoning through Coordinate Alignment and Chain-of-Thought for Embodied Task Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Dafeng Chi, Guangjian Tian, Guowei Huang, Helong Huang, Jianye Hao, Lingfeng Zhang, Shiguang Wu, Shuang Wu, Tongtong Cao, Weichao Qiu, Xingyue Quan, Yaochen Hu, Yingxue Zhang, Yuecheng Liu, Yuzheng Zhuang, Zhanguang Zhang","submitted_at":"2025-01-17T09:46:27Z","abstract_excerpt":"Spatial reasoning is an essential problem in embodied AI research. Efforts to enhance spatial reasoning abilities through supplementary spatial data and fine-tuning have proven limited and ineffective when addressing complex embodied tasks, largely due to their dependence on language-based outputs. While some approaches have introduced a point-based action space to mitigate this issue, they fall short in managing more intricate tasks within complex environments. This deficiency arises from their failure to fully exploit the inherent thinking and reasoning capabilities that are fundamental stre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.10074","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-01-17T09:46:27Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"7379637419e70b01432f36ac45acfda237a2abbf5869a113d5cda6cb0d7cce98","abstract_canon_sha256":"cf995b88ed64a17eda177d4cb669579068d20a8d4d829a764da00343788a5e30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:22.486613Z","signature_b64":"exmX+rB9tqKOy0Lqj4aJ6cQ5ayvrFyEc8Qy+fz/tpzF2MiCrBWBKdhT99EvZonRZq7Av7vpM8EPkoDB3VvDmDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c67897ab61854ab63e1e685d14d7d967a6da6887ae80e625cdeb629cc01ae7c","last_reissued_at":"2026-07-05T10:04:22.486113Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:22.486113Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpatialCoT: Advancing Spatial Reasoning through Coordinate Alignment and Chain-of-Thought for Embodied Task Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Dafeng Chi, Guangjian Tian, Guowei Huang, Helong Huang, Jianye Hao, Lingfeng Zhang, Shiguang Wu, Shuang Wu, Tongtong Cao, Weichao Qiu, Xingyue Quan, Yaochen Hu, Yingxue Zhang, Yuecheng Liu, Yuzheng Zhuang, Zhanguang Zhang","submitted_at":"2025-01-17T09:46:27Z","abstract_excerpt":"Spatial reasoning is an essential problem in embodied AI research. Efforts to enhance spatial reasoning abilities through supplementary spatial data and fine-tuning have proven limited and ineffective when addressing complex embodied tasks, largely due to their dependence on language-based outputs. While some approaches have introduced a point-based action space to mitigate this issue, they fall short in managing more intricate tasks within complex environments. This deficiency arises from their failure to fully exploit the inherent thinking and reasoning capabilities that are fundamental stre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.10074","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.10074/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.10074","created_at":"2026-07-05T10:04:22.486168+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.10074v3","created_at":"2026-07-05T10:04:22.486168+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.10074","created_at":"2026-07-05T10:04:22.486168+00:00"},{"alias_kind":"pith_short_12","alias_value":"RRTYS6VWDBKK","created_at":"2026-07-05T10:04:22.486168+00:00"},{"alias_kind":"pith_short_16","alias_value":"RRTYS6VWDBKKWY7B","created_at":"2026-07-05T10:04:22.486168+00:00"},{"alias_kind":"pith_short_8","alias_value":"RRTYS6VW","created_at":"2026-07-05T10:04:22.486168+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31285","citing_title":"Spatial Reasoning via Modality Switching Between Language and Symbolic Representation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17539","citing_title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01784","citing_title":"SpaceEra++: A Unified Framework Towards 3D Spatial Reasoning in Video","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09669","citing_title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00881","citing_title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18746","citing_title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31285","citing_title":"Spatial Reasoning via Modality Switching Between Language and Symbolic Representation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23176","citing_title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30307","citing_title":"Grounded 3D-Aware Spatial Vision-Language Modeling","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05997","citing_title":"4DThinker: Thinking with 4D Imagery for Dynamic Spatial Understanding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23176","citing_title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20733","citing_title":"Sketch2MinSurf: Vision-Language Guided Generation of Editable Minimal Surfaces from Hand-Drawn Sketches","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18746","citing_title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.15669","citing_title":"DeepThinkVLA: Enhancing Reasoning Capability of Vision-Language-Action Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03944","citing_title":"SCP: Spatial Causal Prediction in Video","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02870","citing_title":"Token Warping Helps MLLMs Look from Nearby Viewpoints","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11462","citing_title":"SpatialForge: Bootstrapping 3D-Aware Spatial Reasoning from Open-World 2D Images","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05997","citing_title":"4DThinker: Thinking with 4D Imagery for Dynamic Spatial Understanding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07592","citing_title":"Spatio-Temporal Grounding of Large Language Models from Perception Streams","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08064","citing_title":"Proxy3D: Efficient 3D Representations for Vision-Language Models via Semantic Clustering and Alignment","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ","json":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ.json","graph_json":"https://pith.science/api/pith-number/RRTYS6VWDBKKWY7B42C5CTL5SZ/graph.json","events_json":"https://pith.science/api/pith-number/RRTYS6VWDBKKWY7B42C5CTL5SZ/events.json","paper":"https://pith.science/paper/RRTYS6VW"},"agent_actions":{"view_html":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ","download_json":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ.json","view_paper":"https://pith.science/paper/RRTYS6VW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.10074&json=true","fetch_graph":"https://pith.science/api/pith-number/RRTYS6VWDBKKWY7B42C5CTL5SZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RRTYS6VWDBKKWY7B42C5CTL5SZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ/action/storage_attestation","attest_author":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ/action/author_attestation","sign_citation":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ/action/citation_signature","submit_replication":"https://pith.science/pith/RRTYS6VWDBKKWY7B42C5CTL5SZ/action/replication_record"}},"created_at":"2026-07-05T10:04:22.486168+00:00","updated_at":"2026-07-05T10:04:22.486168+00:00"}