{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:Q6VOUKF5IMURQAE52BQD6NJKHI","short_pith_number":"pith:Q6VOUKF5","schema_version":"1.0","canonical_sha256":"87aaea28bd432918009dd0603f352a3a17114cd45e94fd450ba4ca8af9542538","source":{"kind":"arxiv","id":"2210.03825","version":1},"attestation_state":"computed","paper":{"title":"See, Plan, Predict: Language-guided Cognitive Planning with Video Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Advaya Gupta, Animesh Garg, Igor Gilitschenski, Maria Attarian, Wei Yu, Ziyi Zhou","submitted_at":"2022-10-07T21:27:16Z","abstract_excerpt":"Cognitive planning is the structural decomposition of complex tasks into a sequence of future behaviors. In the computational setting, performing cognitive planning entails grounding plans and concepts in one or more modalities in order to leverage them for low level control. Since real-world tasks are often described in natural language, we devise a cognitive planning algorithm via language-guided video prediction. Current video prediction models do not support conditioning on natural language instructions. Therefore, we propose a new video prediction architecture which leverages the power of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.03825","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2022-10-07T21:27:16Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"54ab2b85c3f7827abc2d5bc25acf0367ca2b0959c989a09ac35dc85b7ce1139a","abstract_canon_sha256":"5149af4e0b99b9d0b08bffd917ff789d8451ff742273f5db293899f28cc94290"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:32.755496Z","signature_b64":"Am4OVOkrGCm4VKZH2WMLYZDydks82qu7TWkdwvzEENPDKzO7PYzOs/u4jHlSoMjcn4tPqQPbxCh3VxjwhD0lCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87aaea28bd432918009dd0603f352a3a17114cd45e94fd450ba4ca8af9542538","last_reissued_at":"2026-07-05T05:04:32.755004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:32.755004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"See, Plan, Predict: Language-guided Cognitive Planning with Video Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Advaya Gupta, Animesh Garg, Igor Gilitschenski, Maria Attarian, Wei Yu, Ziyi Zhou","submitted_at":"2022-10-07T21:27:16Z","abstract_excerpt":"Cognitive planning is the structural decomposition of complex tasks into a sequence of future behaviors. In the computational setting, performing cognitive planning entails grounding plans and concepts in one or more modalities in order to leverage them for low level control. Since real-world tasks are often described in natural language, we devise a cognitive planning algorithm via language-guided video prediction. Current video prediction models do not support conditioning on natural language instructions. Therefore, we propose a new video prediction architecture which leverages the power of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.03825","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.03825/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.03825","created_at":"2026-07-05T05:04:32.755061+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.03825v1","created_at":"2026-07-05T05:04:32.755061+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.03825","created_at":"2026-07-05T05:04:32.755061+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q6VOUKF5IMUR","created_at":"2026-07-05T05:04:32.755061+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q6VOUKF5IMURQAE5","created_at":"2026-07-05T05:04:32.755061+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q6VOUKF5","created_at":"2026-07-05T05:04:32.755061+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.05454","citing_title":"Temporal Representation Alignment: Successor Features Enable Emergent Compositionality in Robot Instruction Following","ref_index":2017,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI","json":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI.json","graph_json":"https://pith.science/api/pith-number/Q6VOUKF5IMURQAE52BQD6NJKHI/graph.json","events_json":"https://pith.science/api/pith-number/Q6VOUKF5IMURQAE52BQD6NJKHI/events.json","paper":"https://pith.science/paper/Q6VOUKF5"},"agent_actions":{"view_html":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI","download_json":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI.json","view_paper":"https://pith.science/paper/Q6VOUKF5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.03825&json=true","fetch_graph":"https://pith.science/api/pith-number/Q6VOUKF5IMURQAE52BQD6NJKHI/graph.json","fetch_events":"https://pith.science/api/pith-number/Q6VOUKF5IMURQAE52BQD6NJKHI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI/action/storage_attestation","attest_author":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI/action/author_attestation","sign_citation":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI/action/citation_signature","submit_replication":"https://pith.science/pith/Q6VOUKF5IMURQAE52BQD6NJKHI/action/replication_record"}},"created_at":"2026-07-05T05:04:32.755061+00:00","updated_at":"2026-07-05T05:04:32.755061+00:00"}