{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KJJTJDROBXZPXLYLGAA6O5326I","short_pith_number":"pith:KJJTJDRO","schema_version":"1.0","canonical_sha256":"5253348e2e0df2fbaf0b3001e7777af2014bcd1968cf613ff6f12a2cd65300dc","source":{"kind":"arxiv","id":"2401.15688","version":2},"attestation_state":"computed","paper":{"title":"Divide and Conquer: Language Models can Plan and Self-Correct for Compositional Text-to-Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aoxue Li, Enze Xie, Xihui Liu, Zhenguo Li, Zhenyu Wang, Zhongdao Wang","submitted_at":"2024-01-28T16:18:39Z","abstract_excerpt":"Despite significant advancements in text-to-image models for generating high-quality images, these methods still struggle to ensure the controllability of text prompts over images in the context of complex text prompts, especially when it comes to retaining object attributes and relationships. In this paper, we propose CompAgent, a training-free approach for compositional text-to-image generation, with a large language model (LLM) agent as its core. The fundamental idea underlying CompAgent is premised on a divide-and-conquer methodology. Given a complex text prompt containing multiple concept"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.15688","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-28T16:18:39Z","cross_cats_sorted":[],"title_canon_sha256":"01b43af2bc496775480e37125b015760d28b4743865273d5d2f49d4e652abb3d","abstract_canon_sha256":"538e2c80e2427a32c56a02b88705d7bd47f13de046ab5b766ec40c5aae2cb868"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:39:13.900773Z","signature_b64":"N7uZI6fhH9rVTjCYwh5aANSBlcFRmi5ETe5V1DwnW9X+GO8+988CUzutH+JMLCGdn9ZOkCJJC0eHun+8NFhUBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5253348e2e0df2fbaf0b3001e7777af2014bcd1968cf613ff6f12a2cd65300dc","last_reissued_at":"2026-07-05T07:39:13.900286Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:39:13.900286Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Divide and Conquer: Language Models can Plan and Self-Correct for Compositional Text-to-Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aoxue Li, Enze Xie, Xihui Liu, Zhenguo Li, Zhenyu Wang, Zhongdao Wang","submitted_at":"2024-01-28T16:18:39Z","abstract_excerpt":"Despite significant advancements in text-to-image models for generating high-quality images, these methods still struggle to ensure the controllability of text prompts over images in the context of complex text prompts, especially when it comes to retaining object attributes and relationships. In this paper, we propose CompAgent, a training-free approach for compositional text-to-image generation, with a large language model (LLM) agent as its core. The fundamental idea underlying CompAgent is premised on a divide-and-conquer methodology. Given a complex text prompt containing multiple concept"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.15688","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.15688/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.15688","created_at":"2026-07-05T07:39:13.900357+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.15688v2","created_at":"2026-07-05T07:39:13.900357+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.15688","created_at":"2026-07-05T07:39:13.900357+00:00"},{"alias_kind":"pith_short_12","alias_value":"KJJTJDROBXZP","created_at":"2026-07-05T07:39:13.900357+00:00"},{"alias_kind":"pith_short_16","alias_value":"KJJTJDROBXZPXLYL","created_at":"2026-07-05T07:39:13.900357+00:00"},{"alias_kind":"pith_short_8","alias_value":"KJJTJDRO","created_at":"2026-07-05T07:39:13.900357+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05031","citing_title":"MetaPoint: Unlocking Precise Spatial Control in Agentic Visual Generation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04015","citing_title":"GenED-SC: Generative Editing Semantic Communication with Integrated Multi-Modal LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31392","citing_title":"ReGRPO: Reflection-Augmented Policy Optimization for Tool-Using Agents","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2403.05135","citing_title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19339","citing_title":"Divide-and-Conquer Approach to Holistic Cognition in High-Similarity Contexts with Limited Data","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I","json":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I.json","graph_json":"https://pith.science/api/pith-number/KJJTJDROBXZPXLYLGAA6O5326I/graph.json","events_json":"https://pith.science/api/pith-number/KJJTJDROBXZPXLYLGAA6O5326I/events.json","paper":"https://pith.science/paper/KJJTJDRO"},"agent_actions":{"view_html":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I","download_json":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I.json","view_paper":"https://pith.science/paper/KJJTJDRO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.15688&json=true","fetch_graph":"https://pith.science/api/pith-number/KJJTJDROBXZPXLYLGAA6O5326I/graph.json","fetch_events":"https://pith.science/api/pith-number/KJJTJDROBXZPXLYLGAA6O5326I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I/action/storage_attestation","attest_author":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I/action/author_attestation","sign_citation":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I/action/citation_signature","submit_replication":"https://pith.science/pith/KJJTJDROBXZPXLYLGAA6O5326I/action/replication_record"}},"created_at":"2026-07-05T07:39:13.900357+00:00","updated_at":"2026-07-05T07:39:13.900357+00:00"}