{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:GYYL47OIHFUFQCLNWZQSJK2GOL","short_pith_number":"pith:GYYL47OI","canonical_record":{"source":{"id":"2605.10376","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-11T11:20:14Z","cross_cats_sorted":[],"title_canon_sha256":"b6ddadbd04385ec906b24190aa1945607d616e5c1c317b1d282f45333469dc5f","abstract_canon_sha256":"462d37585af47e39fe83cc2e988fdc03b7bb2def9004050509c4ff88a20c53cb"},"schema_version":"1.0"},"canonical_sha256":"3630be7dc8396858096db66124ab4672c33f7214e38d46f15968c2c3cffc74db","source":{"kind":"arxiv","id":"2605.10376","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.10376","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"arxiv_version","alias_value":"2605.10376v2","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.10376","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"pith_short_12","alias_value":"GYYL47OIHFUF","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"pith_short_16","alias_value":"GYYL47OIHFUFQCLN","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"pith_short_8","alias_value":"GYYL47OI","created_at":"2026-06-09T01:05:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:GYYL47OIHFUFQCLNWZQSJK2GOL","target":"record","payload":{"canonical_record":{"source":{"id":"2605.10376","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-11T11:20:14Z","cross_cats_sorted":[],"title_canon_sha256":"b6ddadbd04385ec906b24190aa1945607d616e5c1c317b1d282f45333469dc5f","abstract_canon_sha256":"462d37585af47e39fe83cc2e988fdc03b7bb2def9004050509c4ff88a20c53cb"},"schema_version":"1.0"},"canonical_sha256":"3630be7dc8396858096db66124ab4672c33f7214e38d46f15968c2c3cffc74db","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-09T01:05:18.975951Z","signature_b64":"wJBf7D6AnIKHdbdMyr0ngbKgl6SC850e5pXShGU5TZHlSdIIO5tLf/BQNZ8fPsN39IFFUBe5rSAl3sttH+DTAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3630be7dc8396858096db66124ab4672c33f7214e38d46f15968c2c3cffc74db","last_reissued_at":"2026-06-09T01:05:18.975532Z","signature_status":"signed_v1","first_computed_at":"2026-06-09T01:05:18.975532Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.10376","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-09T01:05:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"i9tDgUnDyvJ2W90n4UN6lAMZgBlEsXL5L2yIuQdgz1OGFg6k02Zbz4fIXvnGjCt+a7qKTLcjgnFHTViKyIMICg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T14:53:20.398573Z"},"content_sha256":"8a7eb6995114a237232849069daac6697ec802e348bf274fb46418186562bd83","schema_version":"1.0","event_id":"sha256:8a7eb6995114a237232849069daac6697ec802e348bf274fb46418186562bd83"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:GYYL47OIHFUFQCLNWZQSJK2GOL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"SleepWalk: A Three-Tier Benchmark for Stress-Testing Instruction-Guided Vision-Language Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"SleepWalk benchmark shows current vision-language models systematically fail at grounded spatial reasoning for instructions in 3D scenes.","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aman Chadha, Amitava Das, Niyati Rawal, Saksham Jain, Shah Alam Abir, Suranjana Trivedy, Sushant Ravva, Vinija Jain","submitted_at":"2026-05-11T11:20:14Z","abstract_excerpt":"Vision-Language Models (VLMs) have advanced rapidly in multimodal perception and language understanding, yet it remains unclear whether they can reliably ground language into spatially coherent, plausibly executable actions in 3D digital environments. We introduce SleepWalk, a benchmark for evaluating instruction-grounded trajectory prediction in single-scene 3D worlds generated from textual scene descriptions and filtered for navigability. Unlike prior navigation benchmarks centered on long-range exploration across rooms, SleepWalk targets localized, interaction-centric embodied reasoning: gi"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Results reveal systematic failures in grounded spatial reasoning, especially under occlusion, interaction constraints, and multi-step instructions: performance drops as the difficulty level of the tasks increase. In general, current VLMs can somewhat produce trajectories that are simultaneously spatially coherent, plausibly executable, and aligned with intended actions.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that scenes generated from textual descriptions, after filtering for navigability, provide a faithful and unbiased testbed for real-world spatial grounding, and that the pointwise judge-based protocol accurately captures instruction alignment without introducing its own biases.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"SleepWalk is a new benchmark exposing that frontier VLMs struggle with spatially grounded trajectory prediction in 3D environments, with performance declining sharply as task difficulty increases across three tiers.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"SleepWalk benchmark shows current vision-language models systematically fail at grounded spatial reasoning for instructions in 3D scenes.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"1e5a0c299370fe1ebe4718fcb3d5fc18df8c89adaa1afa808737bfd3854ab207"},"source":{"id":"2605.10376","kind":"arxiv","version":2},"verdict":{"id":"8172a8eb-62cf-4782-b5cd-a61e4ee0be24","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-12T04:23:01.320191Z","strongest_claim":"Results reveal systematic failures in grounded spatial reasoning, especially under occlusion, interaction constraints, and multi-step instructions: performance drops as the difficulty level of the tasks increase. In general, current VLMs can somewhat produce trajectories that are simultaneously spatially coherent, plausibly executable, and aligned with intended actions.","one_line_summary":"SleepWalk is a new benchmark exposing that frontier VLMs struggle with spatially grounded trajectory prediction in 3D environments, with performance declining sharply as task difficulty increases across three tiers.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that scenes generated from textual descriptions, after filtering for navigability, provide a faithful and unbiased testbed for real-world spatial grounding, and that the pointwise judge-based protocol accurately captures instruction alignment without introducing its own biases.","pith_extraction_headline":"SleepWalk benchmark shows current vision-language models systematically fail at grounded spatial reasoning for instructions in 3D scenes."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.10376/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"claim_evidence","ran_at":"2026-05-20T06:02:01.081732Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"ai_meta_artifact","ran_at":"2026-05-19T15:34:41.891726Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_title_agreement","ran_at":"2026-05-19T11:31:18.407060Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T09:24:25.021467Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"3fd62b3bd8b60011cf5124e8eecf6ffbb785b5dc2a4b7abb39f6e68fc5eab7dc"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":3,"snapshot_sha256":"d959eacf402c973598a2d03d3a1b92009f080bd315130dd9a73e1cdae26394bb"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"8172a8eb-62cf-4782-b5cd-a61e4ee0be24"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-09T01:05:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"EyPC+NHXjOmj8m0LI9UCUgiTk20Og3HK2KjPEFwJQ17wJBdPKh47JGAkUkuGo03D7hg/ycuG6UvQxk0IsIfxCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T14:53:20.399295Z"},"content_sha256":"c846affe833dea095a165f7e162518a71bffbf876ba1cd55fc3a38bb875f6b5c","schema_version":"1.0","event_id":"sha256:c846affe833dea095a165f7e162518a71bffbf876ba1cd55fc3a38bb875f6b5c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GYYL47OIHFUFQCLNWZQSJK2GOL/bundle.json","state_url":"https://pith.science/pith/GYYL47OIHFUFQCLNWZQSJK2GOL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GYYL47OIHFUFQCLNWZQSJK2GOL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T14:53:20Z","links":{"resolver":"https://pith.science/pith/GYYL47OIHFUFQCLNWZQSJK2GOL","bundle":"https://pith.science/pith/GYYL47OIHFUFQCLNWZQSJK2GOL/bundle.json","state":"https://pith.science/pith/GYYL47OIHFUFQCLNWZQSJK2GOL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GYYL47OIHFUFQCLNWZQSJK2GOL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:GYYL47OIHFUFQCLNWZQSJK2GOL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"462d37585af47e39fe83cc2e988fdc03b7bb2def9004050509c4ff88a20c53cb","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-11T11:20:14Z","title_canon_sha256":"b6ddadbd04385ec906b24190aa1945607d616e5c1c317b1d282f45333469dc5f"},"schema_version":"1.0","source":{"id":"2605.10376","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.10376","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"arxiv_version","alias_value":"2605.10376v2","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.10376","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"pith_short_12","alias_value":"GYYL47OIHFUF","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"pith_short_16","alias_value":"GYYL47OIHFUFQCLN","created_at":"2026-06-09T01:05:18Z"},{"alias_kind":"pith_short_8","alias_value":"GYYL47OI","created_at":"2026-06-09T01:05:18Z"}],"graph_snapshots":[{"event_id":"sha256:c846affe833dea095a165f7e162518a71bffbf876ba1cd55fc3a38bb875f6b5c","target":"graph","created_at":"2026-06-09T01:05:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Results reveal systematic failures in grounded spatial reasoning, especially under occlusion, interaction constraints, and multi-step instructions: performance drops as the difficulty level of the tasks increase. In general, current VLMs can somewhat produce trajectories that are simultaneously spatially coherent, plausibly executable, and aligned with intended actions."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The assumption that scenes generated from textual descriptions, after filtering for navigability, provide a faithful and unbiased testbed for real-world spatial grounding, and that the pointwise judge-based protocol accurately captures instruction alignment without introducing its own biases."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"SleepWalk is a new benchmark exposing that frontier VLMs struggle with spatially grounded trajectory prediction in 3D environments, with performance declining sharply as task difficulty increases across three tiers."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"SleepWalk benchmark shows current vision-language models systematically fail at grounded spatial reasoning for instructions in 3D scenes."}],"snapshot_sha256":"1e5a0c299370fe1ebe4718fcb3d5fc18df8c89adaa1afa808737bfd3854ab207"},"formal_canon":{"evidence_count":3,"snapshot_sha256":"d959eacf402c973598a2d03d3a1b92009f080bd315130dd9a73e1cdae26394bb"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"claim_evidence","ran_at":"2026-05-20T06:02:01.081732Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-19T15:34:41.891726Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_title_agreement","ran_at":"2026-05-19T11:31:18.407060Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T09:24:25.021467Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2605.10376/integrity.json","findings":[],"snapshot_sha256":"3fd62b3bd8b60011cf5124e8eecf6ffbb785b5dc2a4b7abb39f6e68fc5eab7dc","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Vision-Language Models (VLMs) have advanced rapidly in multimodal perception and language understanding, yet it remains unclear whether they can reliably ground language into spatially coherent, plausibly executable actions in 3D digital environments. We introduce SleepWalk, a benchmark for evaluating instruction-grounded trajectory prediction in single-scene 3D worlds generated from textual scene descriptions and filtered for navigability. Unlike prior navigation benchmarks centered on long-range exploration across rooms, SleepWalk targets localized, interaction-centric embodied reasoning: gi","authors_text":"Aman Chadha, Amitava Das, Niyati Rawal, Saksham Jain, Shah Alam Abir, Suranjana Trivedy, Sushant Ravva, Vinija Jain","cross_cats":[],"headline":"SleepWalk benchmark shows current vision-language models systematically fail at grounded spatial reasoning for instructions in 3D scenes.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-11T11:20:14Z","title":"SleepWalk: A Three-Tier Benchmark for Stress-Testing Instruction-Guided Vision-Language Navigation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.10376","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-12T04:23:01.320191Z","id":"8172a8eb-62cf-4782-b5cd-a61e4ee0be24","model_set":{"reader":"grok-4.3"},"one_line_summary":"SleepWalk is a new benchmark exposing that frontier VLMs struggle with spatially grounded trajectory prediction in 3D environments, with performance declining sharply as task difficulty increases across three tiers.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"SleepWalk benchmark shows current vision-language models systematically fail at grounded spatial reasoning for instructions in 3D scenes.","strongest_claim":"Results reveal systematic failures in grounded spatial reasoning, especially under occlusion, interaction constraints, and multi-step instructions: performance drops as the difficulty level of the tasks increase. In general, current VLMs can somewhat produce trajectories that are simultaneously spatially coherent, plausibly executable, and aligned with intended actions.","weakest_assumption":"The assumption that scenes generated from textual descriptions, after filtering for navigability, provide a faithful and unbiased testbed for real-world spatial grounding, and that the pointwise judge-based protocol accurately captures instruction alignment without introducing its own biases."}},"verdict_id":"8172a8eb-62cf-4782-b5cd-a61e4ee0be24"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8a7eb6995114a237232849069daac6697ec802e348bf274fb46418186562bd83","target":"record","created_at":"2026-06-09T01:05:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"462d37585af47e39fe83cc2e988fdc03b7bb2def9004050509c4ff88a20c53cb","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-11T11:20:14Z","title_canon_sha256":"b6ddadbd04385ec906b24190aa1945607d616e5c1c317b1d282f45333469dc5f"},"schema_version":"1.0","source":{"id":"2605.10376","kind":"arxiv","version":2}},"canonical_sha256":"3630be7dc8396858096db66124ab4672c33f7214e38d46f15968c2c3cffc74db","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3630be7dc8396858096db66124ab4672c33f7214e38d46f15968c2c3cffc74db","first_computed_at":"2026-06-09T01:05:18.975532Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-09T01:05:18.975532Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"wJBf7D6AnIKHdbdMyr0ngbKgl6SC850e5pXShGU5TZHlSdIIO5tLf/BQNZ8fPsN39IFFUBe5rSAl3sttH+DTAg==","signature_status":"signed_v1","signed_at":"2026-06-09T01:05:18.975951Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.10376","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8a7eb6995114a237232849069daac6697ec802e348bf274fb46418186562bd83","sha256:c846affe833dea095a165f7e162518a71bffbf876ba1cd55fc3a38bb875f6b5c"],"state_sha256":"a6397f6b79d444b1eb6fe5bdc65b13ba8073064ae22b6f41f61e5759e33b446f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IxzYd1IsrunSoTSsEsauEE5opL+8VqUXJcDlTVGiVg/yQIAJJtmODIBpE7Q5imb2MHX8aXfXVFFT9fEvYWVkBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T14:53:20.403483Z","bundle_sha256":"c969a4044eb60065fa2f2a4d4b73faca0787ce39aeaef22f6167e196d069a3cd"}}