{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6XXDE33V6LO4JZGSS6KOTP76DI","short_pith_number":"pith:6XXDE33V","schema_version":"1.0","canonical_sha256":"f5ee326f75f2ddc4e4d29794e9bffe1a2f0cc86c5dfbb4bb849bd21b12c5f145","source":{"kind":"arxiv","id":"2310.03502","version":1},"attestation_state":"computed","paper":{"title":"Kandinsky: an Improved Text-to-Image Synthesis with Image Prior and Latent Diffusion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexander Panchenko, Anastasia Maltseva, Andrey Kuznetsov, Angelina Kuts, Anton Razzhigaev, Arseniy Shakhmatov, Denis Dimitrov, Igor Pavlov, Ilya Ryabov, Vladimir Arkhipkin","submitted_at":"2023-10-05T12:29:41Z","abstract_excerpt":"Text-to-image generation is a significant domain in modern computer vision and has achieved substantial improvements through the evolution of generative architectures. Among these, there are diffusion-based models that have demonstrated essential quality enhancements. These models are generally split into two categories: pixel-level and latent-level approaches. We present Kandinsky1, a novel exploration of latent diffusion architecture, combining the principles of the image prior models with latent diffusion techniques. The image prior model is trained separately to map text embeddings to imag"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03502","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-05T12:29:41Z","cross_cats_sorted":[],"title_canon_sha256":"634cf3b12e0bebdbc6db38b12e629876751d8de4c2b2a04b2fb036a22e402b7f","abstract_canon_sha256":"777989251b2e159bfa54353ded7267ab60021d7e65645940a1aeb269d95bd8e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:57:34.392562Z","signature_b64":"2l+Swb5DuKBJebfml9fvnuVMVfKkwJZOZbsesQPumAWkVmsDUXcW2EVRjAMZlweW4Oj1vJs9NDLfYPorS37cCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5ee326f75f2ddc4e4d29794e9bffe1a2f0cc86c5dfbb4bb849bd21b12c5f145","last_reissued_at":"2026-07-05T06:57:34.392059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:57:34.392059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Kandinsky: an Improved Text-to-Image Synthesis with Image Prior and Latent Diffusion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexander Panchenko, Anastasia Maltseva, Andrey Kuznetsov, Angelina Kuts, Anton Razzhigaev, Arseniy Shakhmatov, Denis Dimitrov, Igor Pavlov, Ilya Ryabov, Vladimir Arkhipkin","submitted_at":"2023-10-05T12:29:41Z","abstract_excerpt":"Text-to-image generation is a significant domain in modern computer vision and has achieved substantial improvements through the evolution of generative architectures. Among these, there are diffusion-based models that have demonstrated essential quality enhancements. These models are generally split into two categories: pixel-level and latent-level approaches. We present Kandinsky1, a novel exploration of latent diffusion architecture, combining the principles of the image prior models with latent diffusion techniques. The image prior model is trained separately to map text embeddings to imag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03502","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03502/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03502","created_at":"2026-07-05T06:57:34.392116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03502v1","created_at":"2026-07-05T06:57:34.392116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03502","created_at":"2026-07-05T06:57:34.392116+00:00"},{"alias_kind":"pith_short_12","alias_value":"6XXDE33V6LO4","created_at":"2026-07-05T06:57:34.392116+00:00"},{"alias_kind":"pith_short_16","alias_value":"6XXDE33V6LO4JZGS","created_at":"2026-07-05T06:57:34.392116+00:00"},{"alias_kind":"pith_short_8","alias_value":"6XXDE33V","created_at":"2026-07-05T06:57:34.392116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10309","citing_title":"Dissect and Prune: Enhancing Robustness in AI-Generated Image Detection","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08634","citing_title":"SSAFE: Simple and Strong AI-Generated Image Detection via Frozen Vision Encoders","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02167","citing_title":"Manifold-Aligned Guided Integrated Gradients for Reliable Feature Attribution","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2603.07119","citing_title":"TIQA: Human-Aligned Perceptual Text Quality Assessment in Generated Images","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16363","citing_title":"CSF: Black-box Fingerprinting via Compositional Semantics for Text-to-Image Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06074","citing_title":"Graph-PiT: Enhancing Structural Coherence in Part-Based Image Synthesis via Graph Priors","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02167","citing_title":"Manifold-Aligned Guided Integrated Gradients for Reliable Feature Attribution","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI","json":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI.json","graph_json":"https://pith.science/api/pith-number/6XXDE33V6LO4JZGSS6KOTP76DI/graph.json","events_json":"https://pith.science/api/pith-number/6XXDE33V6LO4JZGSS6KOTP76DI/events.json","paper":"https://pith.science/paper/6XXDE33V"},"agent_actions":{"view_html":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI","download_json":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI.json","view_paper":"https://pith.science/paper/6XXDE33V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03502&json=true","fetch_graph":"https://pith.science/api/pith-number/6XXDE33V6LO4JZGSS6KOTP76DI/graph.json","fetch_events":"https://pith.science/api/pith-number/6XXDE33V6LO4JZGSS6KOTP76DI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI/action/storage_attestation","attest_author":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI/action/author_attestation","sign_citation":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI/action/citation_signature","submit_replication":"https://pith.science/pith/6XXDE33V6LO4JZGSS6KOTP76DI/action/replication_record"}},"created_at":"2026-07-05T06:57:34.392116+00:00","updated_at":"2026-07-05T06:57:34.392116+00:00"}