{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KMX6ULIBS46LKFZ7YCUPAO5KBL","short_pith_number":"pith:KMX6ULIB","schema_version":"1.0","canonical_sha256":"532fea2d01973cb5173fc0a8f03baa0adc8c740987151716e5d50688eae1167a","source":{"kind":"arxiv","id":"2502.01507","version":1},"attestation_state":"computed","paper":{"title":"End-to-end Training for Text-to-Image Synthesis using Dual-Text Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anurag Mittal, Yeruru Asrar Ahmed","submitted_at":"2025-02-03T16:40:47Z","abstract_excerpt":"Text-to-Image (T2I) synthesis is a challenging task that requires modeling complex interactions between two modalities ( i.e., text and image). A common framework adopted in recent state-of-the-art approaches to achieving such multimodal interactions is to bootstrap the learning process with pre-trained image-aligned text embeddings trained using contrastive loss. Furthermore, these embeddings are typically trained generically and reused across various synthesis models. In contrast, we explore an approach to learning text embeddings specifically tailored to the T2I synthesis network, trained i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01507","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-03T16:40:47Z","cross_cats_sorted":[],"title_canon_sha256":"253824af95505b8c2314dec569a66fd5239225adbc4a79b0f87d9b0862d63ecc","abstract_canon_sha256":"29147e2360873f3f4dd6918436203ae88902b5e3ee8ec7feeb419aca587b91f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:56.055067Z","signature_b64":"XJ/uEMdDL5OAE18RBelcRjNYqLs93xt/jdqzqfnfLOXyD0IL9QADUMDWXqNiVm6pkyzoKRSJtx/T2b/H8/IECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"532fea2d01973cb5173fc0a8f03baa0adc8c740987151716e5d50688eae1167a","last_reissued_at":"2026-07-05T10:08:56.054598Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:56.054598Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-end Training for Text-to-Image Synthesis using Dual-Text Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anurag Mittal, Yeruru Asrar Ahmed","submitted_at":"2025-02-03T16:40:47Z","abstract_excerpt":"Text-to-Image (T2I) synthesis is a challenging task that requires modeling complex interactions between two modalities ( i.e., text and image). A common framework adopted in recent state-of-the-art approaches to achieving such multimodal interactions is to bootstrap the learning process with pre-trained image-aligned text embeddings trained using contrastive loss. Furthermore, these embeddings are typically trained generically and reused across various synthesis models. In contrast, we explore an approach to learning text embeddings specifically tailored to the T2I synthesis network, trained i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01507","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01507/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01507","created_at":"2026-07-05T10:08:56.054657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01507v1","created_at":"2026-07-05T10:08:56.054657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01507","created_at":"2026-07-05T10:08:56.054657+00:00"},{"alias_kind":"pith_short_12","alias_value":"KMX6ULIBS46L","created_at":"2026-07-05T10:08:56.054657+00:00"},{"alias_kind":"pith_short_16","alias_value":"KMX6ULIBS46LKFZ7","created_at":"2026-07-05T10:08:56.054657+00:00"},{"alias_kind":"pith_short_8","alias_value":"KMX6ULIB","created_at":"2026-07-05T10:08:56.054657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL","json":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL.json","graph_json":"https://pith.science/api/pith-number/KMX6ULIBS46LKFZ7YCUPAO5KBL/graph.json","events_json":"https://pith.science/api/pith-number/KMX6ULIBS46LKFZ7YCUPAO5KBL/events.json","paper":"https://pith.science/paper/KMX6ULIB"},"agent_actions":{"view_html":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL","download_json":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL.json","view_paper":"https://pith.science/paper/KMX6ULIB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01507&json=true","fetch_graph":"https://pith.science/api/pith-number/KMX6ULIBS46LKFZ7YCUPAO5KBL/graph.json","fetch_events":"https://pith.science/api/pith-number/KMX6ULIBS46LKFZ7YCUPAO5KBL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL/action/storage_attestation","attest_author":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL/action/author_attestation","sign_citation":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL/action/citation_signature","submit_replication":"https://pith.science/pith/KMX6ULIBS46LKFZ7YCUPAO5KBL/action/replication_record"}},"created_at":"2026-07-05T10:08:56.054657+00:00","updated_at":"2026-07-05T10:08:56.054657+00:00"}