{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VMFD4SW7EXDC3YPOFLC2GI65ZI","short_pith_number":"pith:VMFD4SW7","schema_version":"1.0","canonical_sha256":"ab0a3e4adf25c62de1ee2ac5a323ddca09f48ca451dbf06bf466fd0ec64ee086","source":{"kind":"arxiv","id":"2407.01892","version":2},"attestation_state":"computed","paper":{"title":"GRASP: A Grid-Based Benchmark for Evaluating Commonsense Spatial Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Mayank Kejriwal, Zhisheng Tang","submitted_at":"2024-07-02T02:27:46Z","abstract_excerpt":"Spatial reasoning, an important faculty of human cognition with many practical applications, is one of the core commonsense skills that is not purely language-based and, for satisfying (as opposed to optimal) solutions, requires some minimum degree of planning. Existing benchmarks of Commonsense Spatial Reasoning (CSR) tend to evaluate how Large Language Models (LLMs) interpret text-based spatial $\\textit{descriptions}$ rather than directly evaluate a plan produced by the LLM in response to a $\\textit{specific}$ spatial reasoning problem. In this paper, we construct a large-scale benchmark cal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.01892","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-07-02T02:27:46Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f37c389f4b71664c7c5ed01b88b7eb772ee82c74ec329ce4718ea1748dcf12c2","abstract_canon_sha256":"1feb246491160e2a2a6a716fd1e84271bd5b4fed7c8e4cc6aa76a31b34e65300"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:04.806658Z","signature_b64":"vJ/RhTE3ZAOTJE0bhAA4X5sRDHzzmNuohdasq4F4d6e0U1OVpqYzBidqtxl1juw0NQZV2GXAHIkd9iN3M4ImCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab0a3e4adf25c62de1ee2ac5a323ddca09f48ca451dbf06bf466fd0ec64ee086","last_reissued_at":"2026-07-05T10:02:04.806213Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:04.806213Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GRASP: A Grid-Based Benchmark for Evaluating Commonsense Spatial Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Mayank Kejriwal, Zhisheng Tang","submitted_at":"2024-07-02T02:27:46Z","abstract_excerpt":"Spatial reasoning, an important faculty of human cognition with many practical applications, is one of the core commonsense skills that is not purely language-based and, for satisfying (as opposed to optimal) solutions, requires some minimum degree of planning. Existing benchmarks of Commonsense Spatial Reasoning (CSR) tend to evaluate how Large Language Models (LLMs) interpret text-based spatial $\\textit{descriptions}$ rather than directly evaluate a plan produced by the LLM in response to a $\\textit{specific}$ spatial reasoning problem. In this paper, we construct a large-scale benchmark cal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.01892","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.01892/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.01892","created_at":"2026-07-05T10:02:04.806270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.01892v2","created_at":"2026-07-05T10:02:04.806270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.01892","created_at":"2026-07-05T10:02:04.806270+00:00"},{"alias_kind":"pith_short_12","alias_value":"VMFD4SW7EXDC","created_at":"2026-07-05T10:02:04.806270+00:00"},{"alias_kind":"pith_short_16","alias_value":"VMFD4SW7EXDC3YPO","created_at":"2026-07-05T10:02:04.806270+00:00"},{"alias_kind":"pith_short_8","alias_value":"VMFD4SW7","created_at":"2026-07-05T10:02:04.806270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09883","citing_title":"The Cartesian Shortcut: Re-evaluate Vision Reasoning in Polar Coordinate Space","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09883","citing_title":"The Cartesian Shortcut: Re-evaluate Vision Reasoning in Polar Coordinate Space","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07592","citing_title":"Spatio-Temporal Grounding of Large Language Models from Perception Streams","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18539","citing_title":"Transition-Matrix Regularization for Next Dialogue Act Prediction in Counselling Conversations","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI","json":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI.json","graph_json":"https://pith.science/api/pith-number/VMFD4SW7EXDC3YPOFLC2GI65ZI/graph.json","events_json":"https://pith.science/api/pith-number/VMFD4SW7EXDC3YPOFLC2GI65ZI/events.json","paper":"https://pith.science/paper/VMFD4SW7"},"agent_actions":{"view_html":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI","download_json":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI.json","view_paper":"https://pith.science/paper/VMFD4SW7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.01892&json=true","fetch_graph":"https://pith.science/api/pith-number/VMFD4SW7EXDC3YPOFLC2GI65ZI/graph.json","fetch_events":"https://pith.science/api/pith-number/VMFD4SW7EXDC3YPOFLC2GI65ZI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI/action/storage_attestation","attest_author":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI/action/author_attestation","sign_citation":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI/action/citation_signature","submit_replication":"https://pith.science/pith/VMFD4SW7EXDC3YPOFLC2GI65ZI/action/replication_record"}},"created_at":"2026-07-05T10:02:04.806270+00:00","updated_at":"2026-07-05T10:02:04.806270+00:00"}