{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HJSZSVMQLUPEV4QL6TP22JT25Q","short_pith_number":"pith:HJSZSVMQ","schema_version":"1.0","canonical_sha256":"3a659955905d1e4af20bf4dfad267aec0ae699c0df32c03153aeaeb7be814bfd","source":{"kind":"arxiv","id":"2310.16656","version":1},"attestation_state":"computed","paper":{"title":"A Picture is Worth a Thousand Words: Principled Recaptioning Improves Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dani Valevski, Danny Lumen, Eyal Segalis, Yaniv Leviathan, Yossi Matias","submitted_at":"2023-10-25T14:10:08Z","abstract_excerpt":"Text-to-image diffusion models achieved a remarkable leap in capabilities over the last few years, enabling high-quality and diverse synthesis of images from a textual prompt. However, even the most advanced models often struggle to precisely follow all of the directions in their prompts. The vast majority of these models are trained on datasets consisting of (image, caption) pairs where the images often come from the web, and the captions are their HTML alternate text. A notable example is the LAION dataset, used by Stable Diffusion and other models. In this work we observe that these caption"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.16656","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-25T14:10:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"5624d36a38b7d2cd021f9d1f85f569f0bb350faa4cb40fda28de617092aa61f2","abstract_canon_sha256":"66e5e4f767a4c107f73ffe0628bf46f5c01817650447ad91bee49b4fb0c60a59"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:00.814132Z","signature_b64":"WWg60POPKxm2syk6AcjSdyQK+/i9s771BUW/IimBd5zNubJG/Y9+D5VjHw0BR8u/HzA1cMGWUPSXGp8cOjnJAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a659955905d1e4af20bf4dfad267aec0ae699c0df32c03153aeaeb7be814bfd","last_reissued_at":"2026-07-05T07:05:00.813659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:00.813659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Picture is Worth a Thousand Words: Principled Recaptioning Improves Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dani Valevski, Danny Lumen, Eyal Segalis, Yaniv Leviathan, Yossi Matias","submitted_at":"2023-10-25T14:10:08Z","abstract_excerpt":"Text-to-image diffusion models achieved a remarkable leap in capabilities over the last few years, enabling high-quality and diverse synthesis of images from a textual prompt. However, even the most advanced models often struggle to precisely follow all of the directions in their prompts. The vast majority of these models are trained on datasets consisting of (image, caption) pairs where the images often come from the web, and the captions are their HTML alternate text. A notable example is the LAION dataset, used by Stable Diffusion and other models. In this work we observe that these caption"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16656","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.16656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.16656","created_at":"2026-07-05T07:05:00.813716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.16656v1","created_at":"2026-07-05T07:05:00.813716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16656","created_at":"2026-07-05T07:05:00.813716+00:00"},{"alias_kind":"pith_short_12","alias_value":"HJSZSVMQLUPE","created_at":"2026-07-05T07:05:00.813716+00:00"},{"alias_kind":"pith_short_16","alias_value":"HJSZSVMQLUPEV4QL","created_at":"2026-07-05T07:05:00.813716+00:00"},{"alias_kind":"pith_short_8","alias_value":"HJSZSVMQ","created_at":"2026-07-05T07:05:00.813716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26923","citing_title":"GAVEL: Grounded Caption Error Verification and Localization","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14936","citing_title":"Bridging the Intention-Expression Gap: Aligning Multi-Dimensional Preferences via Hierarchical Relevance Feedback in Text-to-Image Diffusion","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2401.03568","citing_title":"Agent AI: Surveying the Horizons of Multimodal Interaction","ref_index":207,"is_internal_anchor":false},{"citing_arxiv_id":"2511.07756","citing_title":"Determinism of Randomness: Prompt-Residual Seed Shaping for Diffusion Generation","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q","json":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q.json","graph_json":"https://pith.science/api/pith-number/HJSZSVMQLUPEV4QL6TP22JT25Q/graph.json","events_json":"https://pith.science/api/pith-number/HJSZSVMQLUPEV4QL6TP22JT25Q/events.json","paper":"https://pith.science/paper/HJSZSVMQ"},"agent_actions":{"view_html":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q","download_json":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q.json","view_paper":"https://pith.science/paper/HJSZSVMQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.16656&json=true","fetch_graph":"https://pith.science/api/pith-number/HJSZSVMQLUPEV4QL6TP22JT25Q/graph.json","fetch_events":"https://pith.science/api/pith-number/HJSZSVMQLUPEV4QL6TP22JT25Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q/action/storage_attestation","attest_author":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q/action/author_attestation","sign_citation":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q/action/citation_signature","submit_replication":"https://pith.science/pith/HJSZSVMQLUPEV4QL6TP22JT25Q/action/replication_record"}},"created_at":"2026-07-05T07:05:00.813716+00:00","updated_at":"2026-07-05T07:05:00.813716+00:00"}