{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JUV5PMHXFUYWY42RUQBYWSTSUB","short_pith_number":"pith:JUV5PMHX","schema_version":"1.0","canonical_sha256":"4d2bd7b0f72d316c7351a4038b4a72a07deccd2cdeb37ae1d8ad0567f76e7f71","source":{"kind":"arxiv","id":"2505.08444","version":2},"attestation_state":"computed","paper":{"title":"Extracting Visual Plans from Unlabeled Videos via Symbolic Guidance","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Ahmet Tikna, Joni Pajarinen, Luigi Palopoli, Marco Roveri, Wenyan Yang, Yi Zhao, Yuying Zhang","submitted_at":"2025-05-13T11:13:00Z","abstract_excerpt":"Visual planning, by offering a sequence of intermediate visual subgoals to a goal-conditioned low-level policy, achieves promising performance on long-horizon manipulation tasks. To obtain the subgoals, existing methods typically resort to video generation models but suffer from model hallucination and computational cost. We present Vis2Plan, an efficient, explainable and white-box visual planning framework powered by symbolic guidance. From raw, unlabeled play data, Vis2Plan harnesses vision foundation models to automatically extract a compact set of task symbols, which allows building a high"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08444","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2025-05-13T11:13:00Z","cross_cats_sorted":[],"title_canon_sha256":"aabb17c7474b14c61f662feb3e6bf0158d80dc10ad84197ed2b06ff2a13c6d47","abstract_canon_sha256":"8d491c6e278826314d60448fe7640fd9d781f26e9bfcf2b12b5f96f5e532381e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:54.561271Z","signature_b64":"B2ZcgNMn4/dWJqyS2bOyIYDjgweG+OitJ1bflcpXpFJ5vBo4NHX2HxQgrFTC7vsG/kNBz3cuqPmZ8dYEeyEsBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d2bd7b0f72d316c7351a4038b4a72a07deccd2cdeb37ae1d8ad0567f76e7f71","last_reissued_at":"2026-07-05T11:49:54.560793Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:54.560793Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Extracting Visual Plans from Unlabeled Videos via Symbolic Guidance","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Ahmet Tikna, Joni Pajarinen, Luigi Palopoli, Marco Roveri, Wenyan Yang, Yi Zhao, Yuying Zhang","submitted_at":"2025-05-13T11:13:00Z","abstract_excerpt":"Visual planning, by offering a sequence of intermediate visual subgoals to a goal-conditioned low-level policy, achieves promising performance on long-horizon manipulation tasks. To obtain the subgoals, existing methods typically resort to video generation models but suffer from model hallucination and computational cost. We present Vis2Plan, an efficient, explainable and white-box visual planning framework powered by symbolic guidance. From raw, unlabeled play data, Vis2Plan harnesses vision foundation models to automatically extract a compact set of task symbols, which allows building a high"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08444","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08444/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08444","created_at":"2026-07-05T11:49:54.560850+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08444v2","created_at":"2026-07-05T11:49:54.560850+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08444","created_at":"2026-07-05T11:49:54.560850+00:00"},{"alias_kind":"pith_short_12","alias_value":"JUV5PMHXFUYW","created_at":"2026-07-05T11:49:54.560850+00:00"},{"alias_kind":"pith_short_16","alias_value":"JUV5PMHXFUYWY42R","created_at":"2026-07-05T11:49:54.560850+00:00"},{"alias_kind":"pith_short_8","alias_value":"JUV5PMHX","created_at":"2026-07-05T11:49:54.560850+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.04447","citing_title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB","json":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB.json","graph_json":"https://pith.science/api/pith-number/JUV5PMHXFUYWY42RUQBYWSTSUB/graph.json","events_json":"https://pith.science/api/pith-number/JUV5PMHXFUYWY42RUQBYWSTSUB/events.json","paper":"https://pith.science/paper/JUV5PMHX"},"agent_actions":{"view_html":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB","download_json":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB.json","view_paper":"https://pith.science/paper/JUV5PMHX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08444&json=true","fetch_graph":"https://pith.science/api/pith-number/JUV5PMHXFUYWY42RUQBYWSTSUB/graph.json","fetch_events":"https://pith.science/api/pith-number/JUV5PMHXFUYWY42RUQBYWSTSUB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB/action/storage_attestation","attest_author":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB/action/author_attestation","sign_citation":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB/action/citation_signature","submit_replication":"https://pith.science/pith/JUV5PMHXFUYWY42RUQBYWSTSUB/action/replication_record"}},"created_at":"2026-07-05T11:49:54.560850+00:00","updated_at":"2026-07-05T11:49:54.560850+00:00"}