{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SWKFJJS72CNRT36N4QJFTRMFQX","short_pith_number":"pith:SWKFJJS7","schema_version":"1.0","canonical_sha256":"959454a65fd09b19efcde41259c58585d757b8b04e001960cfcbb520411ac1ca","source":{"kind":"arxiv","id":"2509.10704","version":1},"attestation_state":"computed","paper":{"title":"Maestro: Self-Improving Text-to-Image Generation via Agent Orchestration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Han Zhou, Hootan Nakhost, Ke Jiang, Rajarishi Sinha, Ruoxi Sun, Sercan \\\"O. Ar{\\i}k, Xingchen Wan","submitted_at":"2025-09-12T21:45:16Z","abstract_excerpt":"Text-to-image (T2I) models, while offering immense creative potential, are highly reliant on human intervention, posing significant usability challenges that often necessitate manual, iterative prompt engineering over often underspecified prompts. This paper introduces Maestro, a novel self-evolving image generation system that enables T2I models to autonomously self-improve generated images through iterative evolution of prompts, using only an initial prompt. Maestro incorporates two key innovations: 1) self-critique, where specialized multimodal LLM (MLLM) agents act as 'critics' to identify"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.10704","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-09-12T21:45:16Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"0c4568a185b6cc7cab6316ba09b5231c40da595c9de112b06cdf63a879455728","abstract_canon_sha256":"4094031a3591beace0454b77dd4ba1ab76b6c954563f6e59103c589876222823"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:26.335329Z","signature_b64":"5w2lmgH5PwvLcmCIM5r6j9E4CnW0TMVcNujcC5LuM/q1XtyHCZC5YWpujed5Yw+o3ZwvtmvP9EBXd3PUpY06DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"959454a65fd09b19efcde41259c58585d757b8b04e001960cfcbb520411ac1ca","last_reissued_at":"2026-07-05T12:11:26.334673Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:26.334673Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Maestro: Self-Improving Text-to-Image Generation via Agent Orchestration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Han Zhou, Hootan Nakhost, Ke Jiang, Rajarishi Sinha, Ruoxi Sun, Sercan \\\"O. Ar{\\i}k, Xingchen Wan","submitted_at":"2025-09-12T21:45:16Z","abstract_excerpt":"Text-to-image (T2I) models, while offering immense creative potential, are highly reliant on human intervention, posing significant usability challenges that often necessitate manual, iterative prompt engineering over often underspecified prompts. This paper introduces Maestro, a novel self-evolving image generation system that enables T2I models to autonomously self-improve generated images through iterative evolution of prompts, using only an initial prompt. Maestro incorporates two key innovations: 1) self-critique, where specialized multimodal LLM (MLLM) agents act as 'critics' to identify"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.10704","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.10704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.10704","created_at":"2026-07-05T12:11:26.334774+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.10704v1","created_at":"2026-07-05T12:11:26.334774+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.10704","created_at":"2026-07-05T12:11:26.334774+00:00"},{"alias_kind":"pith_short_12","alias_value":"SWKFJJS72CNR","created_at":"2026-07-05T12:11:26.334774+00:00"},{"alias_kind":"pith_short_16","alias_value":"SWKFJJS72CNRT36N","created_at":"2026-07-05T12:11:26.334774+00:00"},{"alias_kind":"pith_short_8","alias_value":"SWKFJJS7","created_at":"2026-07-05T12:11:26.334774+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23679","citing_title":"Semantic Browsing: Controllable Diversity for Image Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11459","citing_title":"APEX: Automated Prompt Engineering eXpert with Dynamic Data Selection","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09380","citing_title":"Reasoning Arena: Trace Tournaments When Verifiable Rewards Fall Short","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21605","citing_title":"GenEvolve: Self-Evolving Image Generation Agents via Tool-Orchestrated Visual Experience Distillation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21605","citing_title":"GenEvolve: Self-Evolving Image Generation Agents via Tool-Orchestrated Visual Experience Distillation","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX","json":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX.json","graph_json":"https://pith.science/api/pith-number/SWKFJJS72CNRT36N4QJFTRMFQX/graph.json","events_json":"https://pith.science/api/pith-number/SWKFJJS72CNRT36N4QJFTRMFQX/events.json","paper":"https://pith.science/paper/SWKFJJS7"},"agent_actions":{"view_html":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX","download_json":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX.json","view_paper":"https://pith.science/paper/SWKFJJS7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.10704&json=true","fetch_graph":"https://pith.science/api/pith-number/SWKFJJS72CNRT36N4QJFTRMFQX/graph.json","fetch_events":"https://pith.science/api/pith-number/SWKFJJS72CNRT36N4QJFTRMFQX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX/action/storage_attestation","attest_author":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX/action/author_attestation","sign_citation":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX/action/citation_signature","submit_replication":"https://pith.science/pith/SWKFJJS72CNRT36N4QJFTRMFQX/action/replication_record"}},"created_at":"2026-07-05T12:11:26.334774+00:00","updated_at":"2026-07-05T12:11:26.334774+00:00"}