{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NGZWAFFYRAHGXA7EVNTSIT23OK","short_pith_number":"pith:NGZWAFFY","schema_version":"1.0","canonical_sha256":"69b36014b8880e6b83e4ab67244f5b728864a91728c8311332cdb2a34e481678","source":{"kind":"arxiv","id":"2505.01104","version":1},"attestation_state":"computed","paper":{"title":"VSC: Visual Search Compositional Text-to-Image Diffusion Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Do Huu Dat, Nam Hyeonu, Po-Yuan Mao, Tae-Hyun Oh","submitted_at":"2025-05-02T08:31:43Z","abstract_excerpt":"Text-to-image diffusion models have shown impressive capabilities in generating realistic visuals from natural-language prompts, yet they often struggle with accurately binding attributes to corresponding objects, especially in prompts containing multiple attribute-object pairs. This challenge primarily arises from the limitations of commonly used text encoders, such as CLIP, which can fail to encode complex linguistic relationships and modifiers effectively. Existing approaches have attempted to mitigate these issues through attention map control during inference and the use of layout informa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.01104","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-02T08:31:43Z","cross_cats_sorted":[],"title_canon_sha256":"1755d466936c1228ad3e43d2f252805255788aff5fdab0ddf84111ffca2cee90","abstract_canon_sha256":"ae17105bbf05ecde41d43a368ff25b460cca138ccc2e9c557b147b7b23ae699a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:45.186650Z","signature_b64":"+Nd/HOa8LIy+jM1sNgZ2GPbDpCxe6wBs0fE8O6NtXmtCnh7naoM3BKx4wL8371vdDVClw/xXDIVphttOrMuwAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69b36014b8880e6b83e4ab67244f5b728864a91728c8311332cdb2a34e481678","last_reissued_at":"2026-07-05T10:57:45.186148Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:45.186148Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VSC: Visual Search Compositional Text-to-Image Diffusion Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Do Huu Dat, Nam Hyeonu, Po-Yuan Mao, Tae-Hyun Oh","submitted_at":"2025-05-02T08:31:43Z","abstract_excerpt":"Text-to-image diffusion models have shown impressive capabilities in generating realistic visuals from natural-language prompts, yet they often struggle with accurately binding attributes to corresponding objects, especially in prompts containing multiple attribute-object pairs. This challenge primarily arises from the limitations of commonly used text encoders, such as CLIP, which can fail to encode complex linguistic relationships and modifiers effectively. Existing approaches have attempted to mitigate these issues through attention map control during inference and the use of layout informa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.01104","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.01104/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.01104","created_at":"2026-07-05T10:57:45.186215+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.01104v1","created_at":"2026-07-05T10:57:45.186215+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.01104","created_at":"2026-07-05T10:57:45.186215+00:00"},{"alias_kind":"pith_short_12","alias_value":"NGZWAFFYRAHG","created_at":"2026-07-05T10:57:45.186215+00:00"},{"alias_kind":"pith_short_16","alias_value":"NGZWAFFYRAHGXA7E","created_at":"2026-07-05T10:57:45.186215+00:00"},{"alias_kind":"pith_short_8","alias_value":"NGZWAFFY","created_at":"2026-07-05T10:57:45.186215+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK","json":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK.json","graph_json":"https://pith.science/api/pith-number/NGZWAFFYRAHGXA7EVNTSIT23OK/graph.json","events_json":"https://pith.science/api/pith-number/NGZWAFFYRAHGXA7EVNTSIT23OK/events.json","paper":"https://pith.science/paper/NGZWAFFY"},"agent_actions":{"view_html":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK","download_json":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK.json","view_paper":"https://pith.science/paper/NGZWAFFY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.01104&json=true","fetch_graph":"https://pith.science/api/pith-number/NGZWAFFYRAHGXA7EVNTSIT23OK/graph.json","fetch_events":"https://pith.science/api/pith-number/NGZWAFFYRAHGXA7EVNTSIT23OK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK/action/storage_attestation","attest_author":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK/action/author_attestation","sign_citation":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK/action/citation_signature","submit_replication":"https://pith.science/pith/NGZWAFFYRAHGXA7EVNTSIT23OK/action/replication_record"}},"created_at":"2026-07-05T10:57:45.186215+00:00","updated_at":"2026-07-05T10:57:45.186215+00:00"}