{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RWUHUIUDYZE37WDK2UX345RG5V","short_pith_number":"pith:RWUHUIUD","schema_version":"1.0","canonical_sha256":"8da87a2283c649bfd86ad52fbe7626ed77296e6d825ba03cfd161393ed57ad59","source":{"kind":"arxiv","id":"2506.21416","version":1},"attestation_state":"computed","paper":{"title":"XVerse: Consistent Multi-Subject Control of Identity and Semantic Attributes via DiT Modulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Chen, Haomiao Sun, Kang Du, Li Chen, Mengyi Zhao, Xinglong Wu, Xu Wang","submitted_at":"2025-06-26T16:04:16Z","abstract_excerpt":"Achieving fine-grained control over subject identity and semantic attributes (pose, style, lighting) in text-to-image generation, particularly for multiple subjects, often undermines the editability and coherence of Diffusion Transformers (DiTs). Many approaches introduce artifacts or suffer from attribute entanglement. To overcome these challenges, we propose a novel multi-subject controlled generation model XVerse. By transforming reference images into offsets for token-specific text-stream modulation, XVerse allows for precise and independent control for specific subject without disrupting "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21416","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-26T16:04:16Z","cross_cats_sorted":[],"title_canon_sha256":"1a16a19c1851f33026250313b4485f78395799c15b16176c8232e6f34c2ebe61","abstract_canon_sha256":"b4f01b9613c2a375d0e2d053eeb7fd11dc024cc72a0b950b5d49c2d570467c4f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:45.479646Z","signature_b64":"Nq6ZnXJQ+O7DI3ykxtnE1BIWohwJxbHFQcoyKBaU7xxmYXwnbtEqp+RylczgUqJ624UIF9pe4COhbA0xKL7vAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8da87a2283c649bfd86ad52fbe7626ed77296e6d825ba03cfd161393ed57ad59","last_reissued_at":"2026-07-05T11:27:45.479132Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:45.479132Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XVerse: Consistent Multi-Subject Control of Identity and Semantic Attributes via DiT Modulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Chen, Haomiao Sun, Kang Du, Li Chen, Mengyi Zhao, Xinglong Wu, Xu Wang","submitted_at":"2025-06-26T16:04:16Z","abstract_excerpt":"Achieving fine-grained control over subject identity and semantic attributes (pose, style, lighting) in text-to-image generation, particularly for multiple subjects, often undermines the editability and coherence of Diffusion Transformers (DiTs). Many approaches introduce artifacts or suffer from attribute entanglement. To overcome these challenges, we propose a novel multi-subject controlled generation model XVerse. By transforming reference images into offsets for token-specific text-stream modulation, XVerse allows for precise and independent control for specific subject without disrupting "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21416","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21416/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21416","created_at":"2026-07-05T11:27:45.479198+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21416v1","created_at":"2026-07-05T11:27:45.479198+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21416","created_at":"2026-07-05T11:27:45.479198+00:00"},{"alias_kind":"pith_short_12","alias_value":"RWUHUIUDYZE3","created_at":"2026-07-05T11:27:45.479198+00:00"},{"alias_kind":"pith_short_16","alias_value":"RWUHUIUDYZE37WDK","created_at":"2026-07-05T11:27:45.479198+00:00"},{"alias_kind":"pith_short_8","alias_value":"RWUHUIUD","created_at":"2026-07-05T11:27:45.479198+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26947","citing_title":"Scaling Multi-Reference Image Generation with Dynamic Reward Optimization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01383","citing_title":"MIBE: Multi-subject Interaction Benchmark and Evaluator for Personalized Image Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00351","citing_title":"UniVerse: A Unified Modulation Framework for Segmentation-Free,Disentangled Multi-Concept Personalization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20512","citing_title":"Adversarial Concept Distillation for One-Step Diffusion Personalization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01236","citing_title":"PSR: Scaling Multi-Subject Personalized Image Generation with Pairwise Subject-Consistency Rewards","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12675","citing_title":"Scone: Bridging Composition and Distinction in Subject-Driven Image Generation via Unified Understanding-Generation Modeling","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21788","citing_title":"InstructMoLE: Instruction-Guided Mixture of Low-rank Experts for Multi-Conditional Image Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20725","citing_title":"Premier: Personalized Preference Modulation with Learnable User Embedding in Text-to-Image Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04487","citing_title":"Training-Free Image Editing with Visual Context Integration and Concept Alignment","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V","json":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V.json","graph_json":"https://pith.science/api/pith-number/RWUHUIUDYZE37WDK2UX345RG5V/graph.json","events_json":"https://pith.science/api/pith-number/RWUHUIUDYZE37WDK2UX345RG5V/events.json","paper":"https://pith.science/paper/RWUHUIUD"},"agent_actions":{"view_html":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V","download_json":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V.json","view_paper":"https://pith.science/paper/RWUHUIUD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21416&json=true","fetch_graph":"https://pith.science/api/pith-number/RWUHUIUDYZE37WDK2UX345RG5V/graph.json","fetch_events":"https://pith.science/api/pith-number/RWUHUIUDYZE37WDK2UX345RG5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V/action/storage_attestation","attest_author":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V/action/author_attestation","sign_citation":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V/action/citation_signature","submit_replication":"https://pith.science/pith/RWUHUIUDYZE37WDK2UX345RG5V/action/replication_record"}},"created_at":"2026-07-05T11:27:45.479198+00:00","updated_at":"2026-07-05T11:27:45.479198+00:00"}