{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HUXKD4LJSWLECQYZC5IWUEK3DF","short_pith_number":"pith:HUXKD4LJ","schema_version":"1.0","canonical_sha256":"3d2ea1f169959641431917516a115b19424cfb0db789f68d0f9b8c22b927ec63","source":{"kind":"arxiv","id":"2306.02236","version":1},"attestation_state":"computed","paper":{"title":"Detector Guidance for Multi-Object Text-to-Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Luping Liu, Rongjie Huang, Xiang Yin, Yi Ren, Zhou Zhao, Zijian Zhang","submitted_at":"2023-06-04T02:33:12Z","abstract_excerpt":"Diffusion models have demonstrated impressive performance in text-to-image generation. They utilize a text encoder and cross-attention blocks to infuse textual information into images at a pixel level. However, their capability to generate images with text containing multiple objects is still restricted. Previous works identify the problem of information mixing in the CLIP text encoder and introduce the T5 text encoder or incorporate strong prior knowledge to assist with the alignment. We find that mixing problems also occur on the image side and in the cross-attention blocks. The noisy images"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.02236","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-06-04T02:33:12Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a71f14cc02a021ab4a100d51b6a23d3b2c2a5da37c022be273a6902502f8e2aa","abstract_canon_sha256":"05bd458d9354ed7c6226a649e5c7af88c93cac07d71eae1c8624a8cf69c39ed6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:14.628342Z","signature_b64":"YBA2MsNTdiHeosKlzAHSQeWdqIR7oxaKSEPkC7DMHpfUWlWLyPIX3aMCAyufrSqHG1N3f752quTzU9BQYZ1wBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d2ea1f169959641431917516a115b19424cfb0db789f68d0f9b8c22b927ec63","last_reissued_at":"2026-07-05T06:17:14.627930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:14.627930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Detector Guidance for Multi-Object Text-to-Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Luping Liu, Rongjie Huang, Xiang Yin, Yi Ren, Zhou Zhao, Zijian Zhang","submitted_at":"2023-06-04T02:33:12Z","abstract_excerpt":"Diffusion models have demonstrated impressive performance in text-to-image generation. They utilize a text encoder and cross-attention blocks to infuse textual information into images at a pixel level. However, their capability to generate images with text containing multiple objects is still restricted. Previous works identify the problem of information mixing in the CLIP text encoder and introduce the T5 text encoder or incorporate strong prior knowledge to assist with the alignment. We find that mixing problems also occur on the image side and in the cross-attention blocks. The noisy images"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02236","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.02236/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.02236","created_at":"2026-07-05T06:17:14.627987+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.02236v1","created_at":"2026-07-05T06:17:14.627987+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02236","created_at":"2026-07-05T06:17:14.627987+00:00"},{"alias_kind":"pith_short_12","alias_value":"HUXKD4LJSWLE","created_at":"2026-07-05T06:17:14.627987+00:00"},{"alias_kind":"pith_short_16","alias_value":"HUXKD4LJSWLECQYZ","created_at":"2026-07-05T06:17:14.627987+00:00"},{"alias_kind":"pith_short_8","alias_value":"HUXKD4LJ","created_at":"2026-07-05T06:17:14.627987+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.22980","citing_title":"MOVi: Training-free Text-conditioned Multi-Object Video Generation","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF","json":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF.json","graph_json":"https://pith.science/api/pith-number/HUXKD4LJSWLECQYZC5IWUEK3DF/graph.json","events_json":"https://pith.science/api/pith-number/HUXKD4LJSWLECQYZC5IWUEK3DF/events.json","paper":"https://pith.science/paper/HUXKD4LJ"},"agent_actions":{"view_html":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF","download_json":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF.json","view_paper":"https://pith.science/paper/HUXKD4LJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.02236&json=true","fetch_graph":"https://pith.science/api/pith-number/HUXKD4LJSWLECQYZC5IWUEK3DF/graph.json","fetch_events":"https://pith.science/api/pith-number/HUXKD4LJSWLECQYZC5IWUEK3DF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF/action/storage_attestation","attest_author":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF/action/author_attestation","sign_citation":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF/action/citation_signature","submit_replication":"https://pith.science/pith/HUXKD4LJSWLECQYZC5IWUEK3DF/action/replication_record"}},"created_at":"2026-07-05T06:17:14.627987+00:00","updated_at":"2026-07-05T06:17:14.627987+00:00"}