{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XF4ZG5HZYDPNLMYJMWL56ZN623","short_pith_number":"pith:XF4ZG5HZ","schema_version":"1.0","canonical_sha256":"b9799374f9c0ded5b3096597df65bed6f28f231b705a8018f413f9e7eab36196","source":{"kind":"arxiv","id":"2312.01054","version":1},"attestation_state":"computed","paper":{"title":"Exploring and Improving the Spatial Reasoning Abilities of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.RO","authors_text":"Manasi Sharma","submitted_at":"2023-12-02T07:41:46Z","abstract_excerpt":"Large Language Models (LLMs) represent formidable tools for sequence modeling, boasting an innate capacity for general pattern recognition. Nevertheless, their broader spatial reasoning capabilities, especially applied to numerical trajectory data, remain insufficiently explored. In this paper, we investigate the out-of-the-box performance of ChatGPT-3.5, ChatGPT-4 and Llama 2 7B models when confronted with 3D robotic trajectory data from the CALVIN baseline and associated tasks, including 2D directional and shape labeling. Additionally, we introduce a novel prefix-based prompting mechanism, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.01054","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-12-02T07:41:46Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"57d7492004caccffc5ca2a4ea10f3b0952c187704a1ba2156477396a1946a98d","abstract_canon_sha256":"48b54483a4cd849e5d9ed148ceef56a8f7f633743eef89d1dbaed869684d9a9f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:19:28.442263Z","signature_b64":"VGWtp1I5xKNXiJAckoKJjjxQtnj7um+3LdhkKHtrkfzV1lm+gijAdMlLUeahXUcPDyxL3z8ymkvzVSAS06KRCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9799374f9c0ded5b3096597df65bed6f28f231b705a8018f413f9e7eab36196","last_reissued_at":"2026-07-05T07:19:28.441253Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:19:28.441253Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring and Improving the Spatial Reasoning Abilities of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.RO","authors_text":"Manasi Sharma","submitted_at":"2023-12-02T07:41:46Z","abstract_excerpt":"Large Language Models (LLMs) represent formidable tools for sequence modeling, boasting an innate capacity for general pattern recognition. Nevertheless, their broader spatial reasoning capabilities, especially applied to numerical trajectory data, remain insufficiently explored. In this paper, we investigate the out-of-the-box performance of ChatGPT-3.5, ChatGPT-4 and Llama 2 7B models when confronted with 3D robotic trajectory data from the CALVIN baseline and associated tasks, including 2D directional and shape labeling. Additionally, we introduce a novel prefix-based prompting mechanism, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.01054","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.01054/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.01054","created_at":"2026-07-05T07:19:28.441311+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.01054v1","created_at":"2026-07-05T07:19:28.441311+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.01054","created_at":"2026-07-05T07:19:28.441311+00:00"},{"alias_kind":"pith_short_12","alias_value":"XF4ZG5HZYDPN","created_at":"2026-07-05T07:19:28.441311+00:00"},{"alias_kind":"pith_short_16","alias_value":"XF4ZG5HZYDPNLMYJ","created_at":"2026-07-05T07:19:28.441311+00:00"},{"alias_kind":"pith_short_8","alias_value":"XF4ZG5HZ","created_at":"2026-07-05T07:19:28.441311+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.10428","citing_title":"SC2Arena and StarEvolve: Benchmark and Self-Improvement Framework for LLMs in Complex Decision-Making Tasks","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623","json":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623.json","graph_json":"https://pith.science/api/pith-number/XF4ZG5HZYDPNLMYJMWL56ZN623/graph.json","events_json":"https://pith.science/api/pith-number/XF4ZG5HZYDPNLMYJMWL56ZN623/events.json","paper":"https://pith.science/paper/XF4ZG5HZ"},"agent_actions":{"view_html":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623","download_json":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623.json","view_paper":"https://pith.science/paper/XF4ZG5HZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.01054&json=true","fetch_graph":"https://pith.science/api/pith-number/XF4ZG5HZYDPNLMYJMWL56ZN623/graph.json","fetch_events":"https://pith.science/api/pith-number/XF4ZG5HZYDPNLMYJMWL56ZN623/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623/action/storage_attestation","attest_author":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623/action/author_attestation","sign_citation":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623/action/citation_signature","submit_replication":"https://pith.science/pith/XF4ZG5HZYDPNLMYJMWL56ZN623/action/replication_record"}},"created_at":"2026-07-05T07:19:28.441311+00:00","updated_at":"2026-07-05T07:19:28.441311+00:00"}