{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OMNRNIG2YGDIKCMRXIS4YM252U","short_pith_number":"pith:OMNRNIG2","schema_version":"1.0","canonical_sha256":"731b16a0dac186850991ba25cc335dd520ae1e322c5690e5b967831356a1c1f7","source":{"kind":"arxiv","id":"2505.08622","version":2},"attestation_state":"computed","paper":{"title":"Visually Guided Decoding: Gradient-Free Hard Prompt Inversion with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.AI","authors_text":"Byonghyo Shim, Donghoon Kim, Kyuhong Shim, Minji Bae","submitted_at":"2025-05-13T14:40:22Z","abstract_excerpt":"Text-to-image generative models like DALL-E and Stable Diffusion have revolutionized visual content creation across various applications, including advertising, personalized media, and design prototyping. However, crafting effective textual prompts to guide these models remains challenging, often requiring extensive trial and error. Existing prompt inversion approaches, such as soft and hard prompt techniques, are not so effective due to the limited interpretability and incoherent prompt generation. To address these issues, we propose Visually Guided Decoding (VGD), a gradient-free approach th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08622","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-13T14:40:22Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"402ce694abaa60cda360627d4e4bfe0c4fece4ec317b1df022f1882b8080ae6e","abstract_canon_sha256":"7af0b5660477b3396075905c262321304f5465e642278bc2f4caed2c79a7d651"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:39:53.949736Z","signature_b64":"+B8PMlRLuZXPqxmv+aGcjYiVYkiLwUjIN4KZN9UKjoGuN7t/lNY9hCY5dOIToyrBGqEeCFQXh1k4hWVfeALvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"731b16a0dac186850991ba25cc335dd520ae1e322c5690e5b967831356a1c1f7","last_reissued_at":"2026-07-05T11:39:53.948974Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:39:53.948974Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visually Guided Decoding: Gradient-Free Hard Prompt Inversion with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.AI","authors_text":"Byonghyo Shim, Donghoon Kim, Kyuhong Shim, Minji Bae","submitted_at":"2025-05-13T14:40:22Z","abstract_excerpt":"Text-to-image generative models like DALL-E and Stable Diffusion have revolutionized visual content creation across various applications, including advertising, personalized media, and design prototyping. However, crafting effective textual prompts to guide these models remains challenging, often requiring extensive trial and error. Existing prompt inversion approaches, such as soft and hard prompt techniques, are not so effective due to the limited interpretability and incoherent prompt generation. To address these issues, we propose Visually Guided Decoding (VGD), a gradient-free approach th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08622","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08622/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08622","created_at":"2026-07-05T11:39:53.949045+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08622v2","created_at":"2026-07-05T11:39:53.949045+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08622","created_at":"2026-07-05T11:39:53.949045+00:00"},{"alias_kind":"pith_short_12","alias_value":"OMNRNIG2YGDI","created_at":"2026-07-05T11:39:53.949045+00:00"},{"alias_kind":"pith_short_16","alias_value":"OMNRNIG2YGDIKCMR","created_at":"2026-07-05T11:39:53.949045+00:00"},{"alias_kind":"pith_short_8","alias_value":"OMNRNIG2","created_at":"2026-07-05T11:39:53.949045+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.06061","citing_title":"PromptEvolver: Prompt Inversion through Evolutionary Optimization in Natural-Language Space","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U","json":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U.json","graph_json":"https://pith.science/api/pith-number/OMNRNIG2YGDIKCMRXIS4YM252U/graph.json","events_json":"https://pith.science/api/pith-number/OMNRNIG2YGDIKCMRXIS4YM252U/events.json","paper":"https://pith.science/paper/OMNRNIG2"},"agent_actions":{"view_html":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U","download_json":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U.json","view_paper":"https://pith.science/paper/OMNRNIG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08622&json=true","fetch_graph":"https://pith.science/api/pith-number/OMNRNIG2YGDIKCMRXIS4YM252U/graph.json","fetch_events":"https://pith.science/api/pith-number/OMNRNIG2YGDIKCMRXIS4YM252U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U/action/storage_attestation","attest_author":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U/action/author_attestation","sign_citation":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U/action/citation_signature","submit_replication":"https://pith.science/pith/OMNRNIG2YGDIKCMRXIS4YM252U/action/replication_record"}},"created_at":"2026-07-05T11:39:53.949045+00:00","updated_at":"2026-07-05T11:39:53.949045+00:00"}