{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S3QX2OLIRR6LFOC2UAINFZHLBX","short_pith_number":"pith:S3QX2OLI","schema_version":"1.0","canonical_sha256":"96e17d39688c7cb2b85aa010d2e4eb0de4a46fabcf4c6166ac18ba16545f3d30","source":{"kind":"arxiv","id":"2405.15321","version":1},"attestation_state":"computed","paper":{"title":"SG-Adapter: Enhancing Text-to-Image Generation with Scene Graph Guidance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaozhe Zhang, Guangyong Chen, Guibao Shen, Jiantao Lin, Luozhou Wang, Pengfei Wan, Wenhang Ge, Xin Tao, Yijun Li, Ying-Cong Chen, Yuan Zhang, Zhongyuan Wang","submitted_at":"2024-05-24T08:00:46Z","abstract_excerpt":"Recent advancements in text-to-image generation have been propelled by the development of diffusion models and multi-modality learning. However, since text is typically represented sequentially in these models, it often falls short in providing accurate contextualization and structural control. So the generated images do not consistently align with human expectations, especially in complex scenarios involving multiple objects and relationships. In this paper, we introduce the Scene Graph Adapter(SG-Adapter), leveraging the structured representation of scene graphs to rectify inaccuracies in th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15321","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-24T08:00:46Z","cross_cats_sorted":[],"title_canon_sha256":"970585bbae4f18f637cf2d0ae6f83c9f040c3fdcfc153417d8a3de9bbe22e140","abstract_canon_sha256":"b3fc886beed0de5e4ae5647a0dac9bd54d64c30e7f002a01113640f0767db44c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:47.766767Z","signature_b64":"4MGL1iqMMYaZTpzPRY4wQn2XOe1sTXuXzoel1cKqLzfiA64NUn++PnsXdOku0KXIz0IH9r1QbVHj+24h/DTbAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96e17d39688c7cb2b85aa010d2e4eb0de4a46fabcf4c6166ac18ba16545f3d30","last_reissued_at":"2026-07-05T08:22:47.766273Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:47.766273Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SG-Adapter: Enhancing Text-to-Image Generation with Scene Graph Guidance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaozhe Zhang, Guangyong Chen, Guibao Shen, Jiantao Lin, Luozhou Wang, Pengfei Wan, Wenhang Ge, Xin Tao, Yijun Li, Ying-Cong Chen, Yuan Zhang, Zhongyuan Wang","submitted_at":"2024-05-24T08:00:46Z","abstract_excerpt":"Recent advancements in text-to-image generation have been propelled by the development of diffusion models and multi-modality learning. However, since text is typically represented sequentially in these models, it often falls short in providing accurate contextualization and structural control. So the generated images do not consistently align with human expectations, especially in complex scenarios involving multiple objects and relationships. In this paper, we introduce the Scene Graph Adapter(SG-Adapter), leveraging the structured representation of scene graphs to rectify inaccuracies in th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15321","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15321/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15321","created_at":"2026-07-05T08:22:47.766339+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15321v1","created_at":"2026-07-05T08:22:47.766339+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15321","created_at":"2026-07-05T08:22:47.766339+00:00"},{"alias_kind":"pith_short_12","alias_value":"S3QX2OLIRR6L","created_at":"2026-07-05T08:22:47.766339+00:00"},{"alias_kind":"pith_short_16","alias_value":"S3QX2OLIRR6LFOC2","created_at":"2026-07-05T08:22:47.766339+00:00"},{"alias_kind":"pith_short_8","alias_value":"S3QX2OLI","created_at":"2026-07-05T08:22:47.766339+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17118","citing_title":"Differentiable Optimization Layers for Guaranteed Fairness in Deep Learning","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02210","citing_title":"HiFi-Inpaint: Towards High-Fidelity Reference-Based Inpainting for Generating Detail-Preserving Human-Product Images","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07958","citing_title":"ImVideoEdit: Image-learning Video Editing via 2D Spatial Difference Attention Blocks","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX","json":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX.json","graph_json":"https://pith.science/api/pith-number/S3QX2OLIRR6LFOC2UAINFZHLBX/graph.json","events_json":"https://pith.science/api/pith-number/S3QX2OLIRR6LFOC2UAINFZHLBX/events.json","paper":"https://pith.science/paper/S3QX2OLI"},"agent_actions":{"view_html":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX","download_json":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX.json","view_paper":"https://pith.science/paper/S3QX2OLI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15321&json=true","fetch_graph":"https://pith.science/api/pith-number/S3QX2OLIRR6LFOC2UAINFZHLBX/graph.json","fetch_events":"https://pith.science/api/pith-number/S3QX2OLIRR6LFOC2UAINFZHLBX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX/action/storage_attestation","attest_author":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX/action/author_attestation","sign_citation":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX/action/citation_signature","submit_replication":"https://pith.science/pith/S3QX2OLIRR6LFOC2UAINFZHLBX/action/replication_record"}},"created_at":"2026-07-05T08:22:47.766339+00:00","updated_at":"2026-07-05T08:22:47.766339+00:00"}