{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ASNAF3BNU4DKNQ4TT5YKUSIA7O","short_pith_number":"pith:ASNAF3BN","schema_version":"1.0","canonical_sha256":"049a02ec2da706a6c3939f70aa4900fb9b74504f7a4a11894cac2edd592164e5","source":{"kind":"arxiv","id":"2405.16785","version":2},"attestation_state":"computed","paper":{"title":"PromptFix: You Prompt and We Fix the Photo","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Hua, Jianlong Fu, Jiebo Luo, Yongsheng Yu, Ziyun Zeng","submitted_at":"2024-05-27T03:13:28Z","abstract_excerpt":"Diffusion models equipped with language models demonstrate excellent controllability in image generation tasks, allowing image processing to adhere to human instructions. However, the lack of diverse instruction-following data hampers the development of models that effectively recognize and execute user-customized instructions, particularly in low-level tasks. Moreover, the stochastic nature of the diffusion process leads to deficiencies in image generation or editing tasks that require the detailed preservation of the generated images. To address these limitations, we propose PromptFix, a com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16785","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-27T03:13:28Z","cross_cats_sorted":[],"title_canon_sha256":"a5876d11f0deb352a05f756c375e38b0a591d9277a256a353a8a0b724f3c0603","abstract_canon_sha256":"15ba4b5f438d32ccac0c2a84b06b4b1f0df527dcf195f3a2d1f3ff556fd29b43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:26.062570Z","signature_b64":"dlcws1x4wEMKT27SD//m1BLwNCRTf8XAaWsNDzV2gl3QbsUPz0FrL9Kg9JdaAsy5Mr1A4lZRERhpOyO7ygB+CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"049a02ec2da706a6c3939f70aa4900fb9b74504f7a4a11894cac2edd592164e5","last_reissued_at":"2026-07-05T09:18:26.062135Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:26.062135Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PromptFix: You Prompt and We Fix the Photo","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Hua, Jianlong Fu, Jiebo Luo, Yongsheng Yu, Ziyun Zeng","submitted_at":"2024-05-27T03:13:28Z","abstract_excerpt":"Diffusion models equipped with language models demonstrate excellent controllability in image generation tasks, allowing image processing to adhere to human instructions. However, the lack of diverse instruction-following data hampers the development of models that effectively recognize and execute user-customized instructions, particularly in low-level tasks. Moreover, the stochastic nature of the diffusion process leads to deficiencies in image generation or editing tasks that require the detailed preservation of the generated images. To address these limitations, we propose PromptFix, a com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16785","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16785/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16785","created_at":"2026-07-05T09:18:26.062187+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16785v2","created_at":"2026-07-05T09:18:26.062187+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16785","created_at":"2026-07-05T09:18:26.062187+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASNAF3BNU4DK","created_at":"2026-07-05T09:18:26.062187+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASNAF3BNU4DKNQ4T","created_at":"2026-07-05T09:18:26.062187+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASNAF3BN","created_at":"2026-07-05T09:18:26.062187+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18652","citing_title":"MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18871","citing_title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26583","citing_title":"Emu3.5: Native Multimodal Models are World Learners","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20275","citing_title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2504.17761","citing_title":"Step1X-Edit: A Practical Framework for General Image Editing","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O","json":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O.json","graph_json":"https://pith.science/api/pith-number/ASNAF3BNU4DKNQ4TT5YKUSIA7O/graph.json","events_json":"https://pith.science/api/pith-number/ASNAF3BNU4DKNQ4TT5YKUSIA7O/events.json","paper":"https://pith.science/paper/ASNAF3BN"},"agent_actions":{"view_html":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O","download_json":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O.json","view_paper":"https://pith.science/paper/ASNAF3BN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16785&json=true","fetch_graph":"https://pith.science/api/pith-number/ASNAF3BNU4DKNQ4TT5YKUSIA7O/graph.json","fetch_events":"https://pith.science/api/pith-number/ASNAF3BNU4DKNQ4TT5YKUSIA7O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O/action/storage_attestation","attest_author":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O/action/author_attestation","sign_citation":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O/action/citation_signature","submit_replication":"https://pith.science/pith/ASNAF3BNU4DKNQ4TT5YKUSIA7O/action/replication_record"}},"created_at":"2026-07-05T09:18:26.062187+00:00","updated_at":"2026-07-05T09:18:26.062187+00:00"}