{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VOFFQUXKIXKYD66JIH6IRTM4ES","short_pith_number":"pith:VOFFQUXK","schema_version":"1.0","canonical_sha256":"ab8a5852ea45d581fbc941fc88cd9c249386fdd68fe1d8b7336d0ddc10630c78","source":{"kind":"arxiv","id":"2503.10391","version":1},"attestation_state":"computed","paper":{"title":"CINEMA: Coherent Multi-Subject Video Generation via MLLM-Based Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Angtian Wang, Bo Liu, Chongyang Ma, Haibin Huang, Jacob Zhiyuan Fang, Shenghai Yuan, Xun Guo, Yiding Yang, Yizhi Wang, Yufan Deng","submitted_at":"2025-03-13T14:07:58Z","abstract_excerpt":"Video generation has witnessed remarkable progress with the advent of deep generative models, particularly diffusion models. While existing methods excel in generating high-quality videos from text prompts or single images, personalized multi-subject video generation remains a largely unexplored challenge. This task involves synthesizing videos that incorporate multiple distinct subjects, each defined by separate reference images, while ensuring temporal and spatial consistency. Current approaches primarily rely on mapping subject images to keywords in text prompts, which introduces ambiguity "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10391","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-13T14:07:58Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"dd6cdc295be644da2c13073c47fb6200f13434e1f55cb453685ff14e86dfc88b","abstract_canon_sha256":"cd5d20dcaf5a0682678b4e92965a0231bc0eba02a1042e51769b450acbe4f38f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:44.240446Z","signature_b64":"4kbvqOWd5kE/YM9u91+WEXOPCfZTLENgRi41tMIPaBJIP5F1Zq2TtCt3bIHKxlpqxVWxuAAk/VxbLZMq85XqDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab8a5852ea45d581fbc941fc88cd9c249386fdd68fe1d8b7336d0ddc10630c78","last_reissued_at":"2026-07-05T10:30:44.239903Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:44.239903Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CINEMA: Coherent Multi-Subject Video Generation via MLLM-Based Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Angtian Wang, Bo Liu, Chongyang Ma, Haibin Huang, Jacob Zhiyuan Fang, Shenghai Yuan, Xun Guo, Yiding Yang, Yizhi Wang, Yufan Deng","submitted_at":"2025-03-13T14:07:58Z","abstract_excerpt":"Video generation has witnessed remarkable progress with the advent of deep generative models, particularly diffusion models. While existing methods excel in generating high-quality videos from text prompts or single images, personalized multi-subject video generation remains a largely unexplored challenge. This task involves synthesizing videos that incorporate multiple distinct subjects, each defined by separate reference images, while ensuring temporal and spatial consistency. Current approaches primarily rely on mapping subject images to keywords in text prompts, which introduces ambiguity "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10391","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10391/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10391","created_at":"2026-07-05T10:30:44.239963+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10391v1","created_at":"2026-07-05T10:30:44.239963+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10391","created_at":"2026-07-05T10:30:44.239963+00:00"},{"alias_kind":"pith_short_12","alias_value":"VOFFQUXKIXKY","created_at":"2026-07-05T10:30:44.239963+00:00"},{"alias_kind":"pith_short_16","alias_value":"VOFFQUXKIXKYD66J","created_at":"2026-07-05T10:30:44.239963+00:00"},{"alias_kind":"pith_short_8","alias_value":"VOFFQUXK","created_at":"2026-07-05T10:30:44.239963+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26058","citing_title":"DomainShuttle: Freeform Open Domain Subject-driven Text-to-video Generation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22347","citing_title":"Customizing Video Portraits via Identity-ActionDecoupling","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11670","citing_title":"ARGUS: Stacked Multi-View Identity Mosaic Injection for Subject-Preserving Video Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12088","citing_title":"UniCustom: Unified Visual Conditioning for Multi-Reference Image Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12088","citing_title":"UniCustom: Unified Visual Conditioning for Multi-Reference Image Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19473","citing_title":"TS-Attn: Temporal-wise Separable Attention for Multi-Event Video Generation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES","json":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES.json","graph_json":"https://pith.science/api/pith-number/VOFFQUXKIXKYD66JIH6IRTM4ES/graph.json","events_json":"https://pith.science/api/pith-number/VOFFQUXKIXKYD66JIH6IRTM4ES/events.json","paper":"https://pith.science/paper/VOFFQUXK"},"agent_actions":{"view_html":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES","download_json":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES.json","view_paper":"https://pith.science/paper/VOFFQUXK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10391&json=true","fetch_graph":"https://pith.science/api/pith-number/VOFFQUXKIXKYD66JIH6IRTM4ES/graph.json","fetch_events":"https://pith.science/api/pith-number/VOFFQUXKIXKYD66JIH6IRTM4ES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES/action/storage_attestation","attest_author":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES/action/author_attestation","sign_citation":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES/action/citation_signature","submit_replication":"https://pith.science/pith/VOFFQUXKIXKYD66JIH6IRTM4ES/action/replication_record"}},"created_at":"2026-07-05T10:30:44.239963+00:00","updated_at":"2026-07-05T10:30:44.239963+00:00"}