{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CDM76VI4TPKNBRFJV4LYOTIZSA","short_pith_number":"pith:CDM76VI4","schema_version":"1.0","canonical_sha256":"10d9ff551c9bd4d0c4a9af17874d199023554ef48db9a094b0f7bcc05df37d8f","source":{"kind":"arxiv","id":"2305.01795","version":1},"attestation_state":"computed","paper":{"title":"Multimodal Procedural Planning via Dual Text-Image Prompting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Pan Lu, Wanrong Zhu, William Yang Wang, Xin Eric Wang, Yujie Lu, Zhiyu Chen","submitted_at":"2023-05-02T21:46:44Z","abstract_excerpt":"Embodied agents have achieved prominent performance in following human instructions to complete tasks. However, the potential of providing instructions informed by texts and images to assist humans in completing tasks remains underexplored. To uncover this capability, we present the multimodal procedural planning (MPP) task, in which models are given a high-level goal and generate plans of paired text-image steps, providing more complementary and informative guidance than unimodal plans. The key challenges of MPP are to ensure the informativeness, temporal coherence,and accuracy of plans acros"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.01795","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-02T21:46:44Z","cross_cats_sorted":[],"title_canon_sha256":"0d71dc102b90f7f759b845c7dd9157a5f597505112ddfef367d7c27ee34466aa","abstract_canon_sha256":"81fd5d09b4a5c52dffba8ab16390121d81986c7e2c6d420ec3dea22cc259f925"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:06:32.380031Z","signature_b64":"i/YjE6ZrmiE9zQIHX8UFFzFHp9Cf5Q1mX/a4NG4TUcBrnb8TllKULjCiRfdeWZTMUhGpHWAAwZ7Eld7wfBBxBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"10d9ff551c9bd4d0c4a9af17874d199023554ef48db9a094b0f7bcc05df37d8f","last_reissued_at":"2026-07-05T06:06:32.379532Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:06:32.379532Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Procedural Planning via Dual Text-Image Prompting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Pan Lu, Wanrong Zhu, William Yang Wang, Xin Eric Wang, Yujie Lu, Zhiyu Chen","submitted_at":"2023-05-02T21:46:44Z","abstract_excerpt":"Embodied agents have achieved prominent performance in following human instructions to complete tasks. However, the potential of providing instructions informed by texts and images to assist humans in completing tasks remains underexplored. To uncover this capability, we present the multimodal procedural planning (MPP) task, in which models are given a high-level goal and generate plans of paired text-image steps, providing more complementary and informative guidance than unimodal plans. The key challenges of MPP are to ensure the informativeness, temporal coherence,and accuracy of plans acros"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.01795","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.01795/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.01795","created_at":"2026-07-05T06:06:32.379596+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.01795v1","created_at":"2026-07-05T06:06:32.379596+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.01795","created_at":"2026-07-05T06:06:32.379596+00:00"},{"alias_kind":"pith_short_12","alias_value":"CDM76VI4TPKN","created_at":"2026-07-05T06:06:32.379596+00:00"},{"alias_kind":"pith_short_16","alias_value":"CDM76VI4TPKNBRFJ","created_at":"2026-07-05T06:06:32.379596+00:00"},{"alias_kind":"pith_short_8","alias_value":"CDM76VI4","created_at":"2026-07-05T06:06:32.379596+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25851","citing_title":"RePlan-Bot: Multi-Level Replanning for Embodied Instruction Following","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2307.05973","citing_title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA","json":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA.json","graph_json":"https://pith.science/api/pith-number/CDM76VI4TPKNBRFJV4LYOTIZSA/graph.json","events_json":"https://pith.science/api/pith-number/CDM76VI4TPKNBRFJV4LYOTIZSA/events.json","paper":"https://pith.science/paper/CDM76VI4"},"agent_actions":{"view_html":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA","download_json":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA.json","view_paper":"https://pith.science/paper/CDM76VI4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.01795&json=true","fetch_graph":"https://pith.science/api/pith-number/CDM76VI4TPKNBRFJV4LYOTIZSA/graph.json","fetch_events":"https://pith.science/api/pith-number/CDM76VI4TPKNBRFJV4LYOTIZSA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA/action/storage_attestation","attest_author":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA/action/author_attestation","sign_citation":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA/action/citation_signature","submit_replication":"https://pith.science/pith/CDM76VI4TPKNBRFJV4LYOTIZSA/action/replication_record"}},"created_at":"2026-07-05T06:06:32.379596+00:00","updated_at":"2026-07-05T06:06:32.379596+00:00"}