{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H3BVYG4OAOFOEQ676ENOFEBX6V","short_pith_number":"pith:H3BVYG4O","schema_version":"1.0","canonical_sha256":"3ec35c1b8e038ae243dff11ae29037f56cafab209c08158899309b42066d5dd7","source":{"kind":"arxiv","id":"2405.13951","version":1},"attestation_state":"computed","paper":{"title":"Text Prompting for Multi-Concept Video Customization by Autoregressive Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dinesh Manocha, Divya Kothandaraman, Kihyuk Sohn, Mohammad Babaeizadeh, Paul Voigtlaender, Ruben Villegas","submitted_at":"2024-05-22T19:35:00Z","abstract_excerpt":"We present a method for multi-concept customization of pretrained text-to-video (T2V) models. Intuitively, the multi-concept customized video can be derived from the (non-linear) intersection of the video manifolds of the individual concepts, which is not straightforward to find. We hypothesize that sequential and controlled walking towards the intersection of the video manifolds, directed by text prompting, leads to the solution. To do so, we generate the various concepts and their corresponding interactions, sequentially, in an autoregressive manner. Our method can generate videos of multipl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13951","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-22T19:35:00Z","cross_cats_sorted":[],"title_canon_sha256":"502e31cdefbbcdfd57ffad395844eb4d62c89572311f8fc32ff912724e672914","abstract_canon_sha256":"23fd7b841c5c253bd62669b46d71046c474b58669ac9cb95e46124ceea7ea15c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:08.622813Z","signature_b64":"oli1RC9UROhqG7S9FgxH5H/qMHcHQQZr3IfZExPM2TL1QlwQ8QIxxsjVT5tJv9nARIUYqbrrzXiVfBbdOSd+Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ec35c1b8e038ae243dff11ae29037f56cafab209c08158899309b42066d5dd7","last_reissued_at":"2026-07-05T08:22:08.622396Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:08.622396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Text Prompting for Multi-Concept Video Customization by Autoregressive Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dinesh Manocha, Divya Kothandaraman, Kihyuk Sohn, Mohammad Babaeizadeh, Paul Voigtlaender, Ruben Villegas","submitted_at":"2024-05-22T19:35:00Z","abstract_excerpt":"We present a method for multi-concept customization of pretrained text-to-video (T2V) models. Intuitively, the multi-concept customized video can be derived from the (non-linear) intersection of the video manifolds of the individual concepts, which is not straightforward to find. We hypothesize that sequential and controlled walking towards the intersection of the video manifolds, directed by text prompting, leads to the solution. To do so, we generate the various concepts and their corresponding interactions, sequentially, in an autoregressive manner. Our method can generate videos of multipl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13951","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13951/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13951","created_at":"2026-07-05T08:22:08.622464+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13951v1","created_at":"2026-07-05T08:22:08.622464+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13951","created_at":"2026-07-05T08:22:08.622464+00:00"},{"alias_kind":"pith_short_12","alias_value":"H3BVYG4OAOFO","created_at":"2026-07-05T08:22:08.622464+00:00"},{"alias_kind":"pith_short_16","alias_value":"H3BVYG4OAOFOEQ67","created_at":"2026-07-05T08:22:08.622464+00:00"},{"alias_kind":"pith_short_8","alias_value":"H3BVYG4O","created_at":"2026-07-05T08:22:08.622464+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.07802","citing_title":"Movie Weaver: Tuning-Free Multi-Concept Video Personalization with Anchored Prompts","ref_index":33,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V","json":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V.json","graph_json":"https://pith.science/api/pith-number/H3BVYG4OAOFOEQ676ENOFEBX6V/graph.json","events_json":"https://pith.science/api/pith-number/H3BVYG4OAOFOEQ676ENOFEBX6V/events.json","paper":"https://pith.science/paper/H3BVYG4O"},"agent_actions":{"view_html":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V","download_json":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V.json","view_paper":"https://pith.science/paper/H3BVYG4O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13951&json=true","fetch_graph":"https://pith.science/api/pith-number/H3BVYG4OAOFOEQ676ENOFEBX6V/graph.json","fetch_events":"https://pith.science/api/pith-number/H3BVYG4OAOFOEQ676ENOFEBX6V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V/action/storage_attestation","attest_author":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V/action/author_attestation","sign_citation":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V/action/citation_signature","submit_replication":"https://pith.science/pith/H3BVYG4OAOFOEQ676ENOFEBX6V/action/replication_record"}},"created_at":"2026-07-05T08:22:08.622464+00:00","updated_at":"2026-07-05T08:22:08.622464+00:00"}