{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:R4TZFHULWKJHK6VEELLS3KBXN5","short_pith_number":"pith:R4TZFHUL","schema_version":"1.0","canonical_sha256":"8f27929e8bb292757aa422d72da8376f73e55de36513edcdd6de7ee1f0ea0d48","source":{"kind":"arxiv","id":"2211.11694","version":2},"attestation_state":"computed","paper":{"title":"Exploring Discrete Diffusion Models for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gang Hua, Han Hu, Jianfeng Wang, Le Wang, Lijuan Wang, Yixuan Wei, Zhe Gan, Zheng Zhang, Zicheng Liu, Zixin Zhu","submitted_at":"2022-11-21T18:12:53Z","abstract_excerpt":"The image captioning task is typically realized by an auto-regressive method that decodes the text tokens one by one. We present a diffusion-based captioning model, dubbed the name DDCap, to allow more decoding flexibility. Unlike image generation, where the output is continuous and redundant with a fixed length, texts in image captions are categorical and short with varied lengths. Therefore, naively applying the discrete diffusion model to text decoding does not work well, as shown in our experiments. To address the performance gap, we propose several key techniques including best-first infe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.11694","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-11-21T18:12:53Z","cross_cats_sorted":[],"title_canon_sha256":"da175ac6a026702ff1bb7ef61e0f956cb70c60fa775ff629fa55f36964198e68","abstract_canon_sha256":"84380d730502aa467d029702c98543547fd9d7ba3e4b941955f793377b6e76c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:23:47.699425Z","signature_b64":"xzQFiGGLncx3p32c+BMzkXRGjpPrz4kauCv5GuBrpj0CvGT5fLLB/2RFYOOWr+Nedc9bMg2wcrVO/i8alj7CDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f27929e8bb292757aa422d72da8376f73e55de36513edcdd6de7ee1f0ea0d48","last_reissued_at":"2026-07-05T05:23:47.699017Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:23:47.699017Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Discrete Diffusion Models for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gang Hua, Han Hu, Jianfeng Wang, Le Wang, Lijuan Wang, Yixuan Wei, Zhe Gan, Zheng Zhang, Zicheng Liu, Zixin Zhu","submitted_at":"2022-11-21T18:12:53Z","abstract_excerpt":"The image captioning task is typically realized by an auto-regressive method that decodes the text tokens one by one. We present a diffusion-based captioning model, dubbed the name DDCap, to allow more decoding flexibility. Unlike image generation, where the output is continuous and redundant with a fixed length, texts in image captions are categorical and short with varied lengths. Therefore, naively applying the discrete diffusion model to text decoding does not work well, as shown in our experiments. To address the performance gap, we propose several key techniques including best-first infe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.11694","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.11694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.11694","created_at":"2026-07-05T05:23:47.699072+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.11694v2","created_at":"2026-07-05T05:23:47.699072+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.11694","created_at":"2026-07-05T05:23:47.699072+00:00"},{"alias_kind":"pith_short_12","alias_value":"R4TZFHULWKJH","created_at":"2026-07-05T05:23:47.699072+00:00"},{"alias_kind":"pith_short_16","alias_value":"R4TZFHULWKJHK6VE","created_at":"2026-07-05T05:23:47.699072+00:00"},{"alias_kind":"pith_short_8","alias_value":"R4TZFHUL","created_at":"2026-07-05T05:23:47.699072+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.11681","citing_title":"Learning from Multimodal Pseudo-Labels for Robust Open-Vocabulary Instance and Panoptic Segmentation","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5","json":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5.json","graph_json":"https://pith.science/api/pith-number/R4TZFHULWKJHK6VEELLS3KBXN5/graph.json","events_json":"https://pith.science/api/pith-number/R4TZFHULWKJHK6VEELLS3KBXN5/events.json","paper":"https://pith.science/paper/R4TZFHUL"},"agent_actions":{"view_html":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5","download_json":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5.json","view_paper":"https://pith.science/paper/R4TZFHUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.11694&json=true","fetch_graph":"https://pith.science/api/pith-number/R4TZFHULWKJHK6VEELLS3KBXN5/graph.json","fetch_events":"https://pith.science/api/pith-number/R4TZFHULWKJHK6VEELLS3KBXN5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5/action/storage_attestation","attest_author":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5/action/author_attestation","sign_citation":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5/action/citation_signature","submit_replication":"https://pith.science/pith/R4TZFHULWKJHK6VEELLS3KBXN5/action/replication_record"}},"created_at":"2026-07-05T05:23:47.699072+00:00","updated_at":"2026-07-05T05:23:47.699072+00:00"}