{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3L3SIAOR4PVUR2NXRZ7737HINC","short_pith_number":"pith:3L3SIAOR","schema_version":"1.0","canonical_sha256":"daf72401d1e3eb48e9b78e7ffdfce86885cbbeb641a489139abcb81514df3ce8","source":{"kind":"arxiv","id":"2303.09522","version":3},"attestation_state":"computed","paper":{"title":"P+: Extended Textual Conditioning in Text-to-Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.GR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Andrey Voynov, Daniel Cohen-Or, Kfir Aberman, Qinghao Chu","submitted_at":"2023-03-16T17:38:15Z","abstract_excerpt":"We introduce an Extended Textual Conditioning space in text-to-image models, referred to as $P+$. This space consists of multiple textual conditions, derived from per-layer prompts, each corresponding to a layer of the denoising U-net of the diffusion model.\n  We show that the extended space provides greater disentangling and control over image synthesis. We further introduce Extended Textual Inversion (XTI), where the images are inverted into $P+$, and represented by per-layer tokens.\n  We show that XTI is more expressive and precise, and converges faster than the original Textual Inversion ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.09522","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-16T17:38:15Z","cross_cats_sorted":["cs.CL","cs.GR","cs.LG"],"title_canon_sha256":"15610e153326783eb60bd3ce87077452aa3bf27335c0bf644c1b34c9d6fe867a","abstract_canon_sha256":"96c3b28e9fbe728a8b88cad55aafaed4e0504215a7b710e341c7549cb71aaed5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:31:10.223875Z","signature_b64":"2CGDNLZBpqoHqxavge3C3SBJKw+/hmCHMm6K9g5P3A2wb3w5IxR/mhn6pUjHsyZYWKKKH8KZ2f9ukI6DQcJUCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"daf72401d1e3eb48e9b78e7ffdfce86885cbbeb641a489139abcb81514df3ce8","last_reissued_at":"2026-07-05T06:31:10.223443Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:31:10.223443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"P+: Extended Textual Conditioning in Text-to-Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.GR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Andrey Voynov, Daniel Cohen-Or, Kfir Aberman, Qinghao Chu","submitted_at":"2023-03-16T17:38:15Z","abstract_excerpt":"We introduce an Extended Textual Conditioning space in text-to-image models, referred to as $P+$. This space consists of multiple textual conditions, derived from per-layer prompts, each corresponding to a layer of the denoising U-net of the diffusion model.\n  We show that the extended space provides greater disentangling and control over image synthesis. We further introduce Extended Textual Inversion (XTI), where the images are inverted into $P+$, and represented by per-layer tokens.\n  We show that XTI is more expressive and precise, and converges faster than the original Textual Inversion ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.09522","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.09522/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.09522","created_at":"2026-07-05T06:31:10.223499+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.09522v3","created_at":"2026-07-05T06:31:10.223499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.09522","created_at":"2026-07-05T06:31:10.223499+00:00"},{"alias_kind":"pith_short_12","alias_value":"3L3SIAOR4PVU","created_at":"2026-07-05T06:31:10.223499+00:00"},{"alias_kind":"pith_short_16","alias_value":"3L3SIAOR4PVUR2NX","created_at":"2026-07-05T06:31:10.223499+00:00"},{"alias_kind":"pith_short_8","alias_value":"3L3SIAOR","created_at":"2026-07-05T06:31:10.223499+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06813","citing_title":"Breaking the Lock-in: Diversifying Text-to-Image Generation via Representation Modulation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03792","citing_title":"Training-Free Multi-Concept LoRA Composition with Prompt-Aware Weighting","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02129","citing_title":"Equilibrated Diffusion: Frequency-aware Textual Embedding for Equilibrated Image Customization","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29880","citing_title":"IREU: Identity-Related Encoder-Only Unlearning for Customized Portrait Generation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11051","citing_title":"NP-LoRA: Null Space Projection for Subject-Style LoRA Fusion","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2409.08248","citing_title":"TextBoost: Boosting Text Encoder for Personalized Text-to-Image Generation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2412.12242","citing_title":"OmniPrism: Learning Disentangled Visual Concept for Image Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20306","citing_title":"TPGDiff: Hierarchical Triple-Prior Guided Diffusion for Image Restoration","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2603.07561","citing_title":"PureCC: Pure Learning for Text-to-Image Concept Customization","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02198","citing_title":"SlimDiffSR: Toward Lightweight and Efficient Remote Sensing Image Super-Resolution via Diffusion Model Distillation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16990","citing_title":"DreamEdit3D: Personalization of Multi-View Diffusion Models for 3D Editing","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19519","citing_title":"Preserve and Personalize: Personalized Text-to-Image Diffusion Models without Distributional Drift","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23690","citing_title":"SynMotion: Semantic-Visual Adaptation for Motion Customized Video Generation","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18493","citing_title":"ShowFlow: From Robust Single Concept to Condition-Free Multi-Concept Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23323","citing_title":"FA-Seg: A Fast and Accurate Diffusion-Based Method for Open-Vocabulary Segmentation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06027","citing_title":"DreamAudio: Customized Text-to-Audio Generation with Diffusion Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20512","citing_title":"Adversarial Concept Distillation for One-Step Diffusion Personalization","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08090","citing_title":"DSH-Bench: A Difficulty- and Scenario-Aware Benchmark with Hierarchical Subject Taxonomy for Subject-Driven Text-to-Image Generation","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12724","citing_title":"Inline Critic Steers Image Editing","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06061","citing_title":"PromptEvolver: Prompt Inversion through Evolutionary Optimization in Natural-Language Space","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02198","citing_title":"SlimDiffSR: Toward Lightweight and Efficient Remote Sensing Image Super-Resolution via Diffusion Model Distillation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10023","citing_title":"FREE-Switch: Frequency-based Dynamic LoRA Switch for Style Transfer","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08364","citing_title":"MegaStyle: Constructing Diverse and Scalable Style Dataset via Consistent Text-to-Image Style Mapping","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13863","citing_title":"PostureObjectstitch: Anomaly Image Generation Considering Assembly Relationships in Industrial Scenarios","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC","json":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC.json","graph_json":"https://pith.science/api/pith-number/3L3SIAOR4PVUR2NXRZ7737HINC/graph.json","events_json":"https://pith.science/api/pith-number/3L3SIAOR4PVUR2NXRZ7737HINC/events.json","paper":"https://pith.science/paper/3L3SIAOR"},"agent_actions":{"view_html":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC","download_json":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC.json","view_paper":"https://pith.science/paper/3L3SIAOR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.09522&json=true","fetch_graph":"https://pith.science/api/pith-number/3L3SIAOR4PVUR2NXRZ7737HINC/graph.json","fetch_events":"https://pith.science/api/pith-number/3L3SIAOR4PVUR2NXRZ7737HINC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC/action/storage_attestation","attest_author":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC/action/author_attestation","sign_citation":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC/action/citation_signature","submit_replication":"https://pith.science/pith/3L3SIAOR4PVUR2NXRZ7737HINC/action/replication_record"}},"created_at":"2026-07-05T06:31:10.223499+00:00","updated_at":"2026-07-05T06:31:10.223499+00:00"}