{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M7YL5Y54DLBUTUOLAZH4PL2XOT","short_pith_number":"pith:M7YL5Y54","schema_version":"1.0","canonical_sha256":"67f0bee3bc1ac349d1cb064fc7af5774cd9e441a3f26d118a307c5bc5e81fb3e","source":{"kind":"arxiv","id":"2505.16579","version":1},"attestation_state":"computed","paper":{"title":"Bridging the Dynamic Perception Gap: Training-Free Draft Chain-of-Thought for Dynamic Multimodal Spatial Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Chuan Xuan, Hongcheng Liu, Pingjie Wang, Siqu Ou, Yanfeng Wang, Yusheng Liao, Yu Wang","submitted_at":"2025-05-22T12:14:23Z","abstract_excerpt":"While chains-of-thought (CoT) have advanced complex reasoning in multimodal large language models (MLLMs), existing methods remain confined to text or static visual domains, often faltering in dynamic spatial reasoning tasks. To bridge this gap, we present GRASSLAND, a novel maze navigation benchmark designed to evaluate dynamic spatial reasoning. Our experiments show that augmenting textual reasoning chains with dynamic visual drafts, overlaid on input images, significantly outperforms conventional approaches, offering new insights into spatial reasoning in evolving environments. To generaliz"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16579","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-22T12:14:23Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"f7fda7ff980b9bdeee46f78b35f67ac064bd9c23fb3c1414bdb21cf65386301e","abstract_canon_sha256":"63e8b7fe633af44a83408a25f9bfc1bfcacab146b993658cdf908fafbd6f9841"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:37.386778Z","signature_b64":"OX5Q975m04460geD1kzqP/CJjLNs1IAVcvEwx5ax9L7YfNM+kaDnwXFA1max7CwDfucqLWaR7lrLfYp4F84CCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"67f0bee3bc1ac349d1cb064fc7af5774cd9e441a3f26d118a307c5bc5e81fb3e","last_reissued_at":"2026-07-05T11:07:37.386258Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:37.386258Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bridging the Dynamic Perception Gap: Training-Free Draft Chain-of-Thought for Dynamic Multimodal Spatial Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Chuan Xuan, Hongcheng Liu, Pingjie Wang, Siqu Ou, Yanfeng Wang, Yusheng Liao, Yu Wang","submitted_at":"2025-05-22T12:14:23Z","abstract_excerpt":"While chains-of-thought (CoT) have advanced complex reasoning in multimodal large language models (MLLMs), existing methods remain confined to text or static visual domains, often faltering in dynamic spatial reasoning tasks. To bridge this gap, we present GRASSLAND, a novel maze navigation benchmark designed to evaluate dynamic spatial reasoning. Our experiments show that augmenting textual reasoning chains with dynamic visual drafts, overlaid on input images, significantly outperforms conventional approaches, offering new insights into spatial reasoning in evolving environments. To generaliz"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16579","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16579","created_at":"2026-07-05T11:07:37.386322+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16579v1","created_at":"2026-07-05T11:07:37.386322+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16579","created_at":"2026-07-05T11:07:37.386322+00:00"},{"alias_kind":"pith_short_12","alias_value":"M7YL5Y54DLBU","created_at":"2026-07-05T11:07:37.386322+00:00"},{"alias_kind":"pith_short_16","alias_value":"M7YL5Y54DLBUTUOL","created_at":"2026-07-05T11:07:37.386322+00:00"},{"alias_kind":"pith_short_8","alias_value":"M7YL5Y54","created_at":"2026-07-05T11:07:37.386322+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22875","citing_title":"SketchVLM: Vision language models can annotate images to explain thoughts and guide users","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT","json":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT.json","graph_json":"https://pith.science/api/pith-number/M7YL5Y54DLBUTUOLAZH4PL2XOT/graph.json","events_json":"https://pith.science/api/pith-number/M7YL5Y54DLBUTUOLAZH4PL2XOT/events.json","paper":"https://pith.science/paper/M7YL5Y54"},"agent_actions":{"view_html":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT","download_json":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT.json","view_paper":"https://pith.science/paper/M7YL5Y54","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16579&json=true","fetch_graph":"https://pith.science/api/pith-number/M7YL5Y54DLBUTUOLAZH4PL2XOT/graph.json","fetch_events":"https://pith.science/api/pith-number/M7YL5Y54DLBUTUOLAZH4PL2XOT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT/action/storage_attestation","attest_author":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT/action/author_attestation","sign_citation":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT/action/citation_signature","submit_replication":"https://pith.science/pith/M7YL5Y54DLBUTUOLAZH4PL2XOT/action/replication_record"}},"created_at":"2026-07-05T11:07:37.386322+00:00","updated_at":"2026-07-05T11:07:37.386322+00:00"}