{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:GSKMHK62JBFXTTMCFJUICNUIPC","short_pith_number":"pith:GSKMHK62","schema_version":"1.0","canonical_sha256":"3494c3abda484b79cd822a6881368878a7ce4aabf4ed31eacaba6c2b3cc0fa59","source":{"kind":"arxiv","id":"2209.08231","version":2},"attestation_state":"computed","paper":{"title":"Learning Distinct and Representative Styles for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaorui Deng, Qi Chen, Qi Wu","submitted_at":"2022-09-17T03:25:46Z","abstract_excerpt":"Over the years, state-of-the-art (SoTA) image captioning methods have achieved promising results on some evaluation metrics (e.g., CIDEr). However, recent findings show that the captions generated by these methods tend to be biased toward the \"average\" caption that only captures the most general mode (a.k.a, language pattern) in the training corpus, i.e., the so-called mode collapse problem. Affected by it, the generated captions are limited in diversity and usually less informative than natural image descriptions made by humans. In this paper, we seek to avoid this problem by proposing a Disc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.08231","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-09-17T03:25:46Z","cross_cats_sorted":[],"title_canon_sha256":"19af14b95adb5703540274b0102d56776b60969bcadc28abfd02df44e2e99893","abstract_canon_sha256":"e5da55b1f8cbfbb41b0b38406a3c487ef345f99614e4f1b10276be9f77ef0c47"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:41:02.965182Z","signature_b64":"4HriuOLy7aLe4S5N9dIh5G+Y5si3DswJb4bLQNY33laPwn8ArzsYn2xWz68XWRl1zvWcHl3UL/6j3b3Nvo3qDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3494c3abda484b79cd822a6881368878a7ce4aabf4ed31eacaba6c2b3cc0fa59","last_reissued_at":"2026-07-05T06:41:02.964830Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:41:02.964830Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Distinct and Representative Styles for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaorui Deng, Qi Chen, Qi Wu","submitted_at":"2022-09-17T03:25:46Z","abstract_excerpt":"Over the years, state-of-the-art (SoTA) image captioning methods have achieved promising results on some evaluation metrics (e.g., CIDEr). However, recent findings show that the captions generated by these methods tend to be biased toward the \"average\" caption that only captures the most general mode (a.k.a, language pattern) in the training corpus, i.e., the so-called mode collapse problem. Affected by it, the generated captions are limited in diversity and usually less informative than natural image descriptions made by humans. In this paper, we seek to avoid this problem by proposing a Disc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.08231","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.08231/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.08231","created_at":"2026-07-05T06:41:02.964886+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.08231v2","created_at":"2026-07-05T06:41:02.964886+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.08231","created_at":"2026-07-05T06:41:02.964886+00:00"},{"alias_kind":"pith_short_12","alias_value":"GSKMHK62JBFX","created_at":"2026-07-05T06:41:02.964886+00:00"},{"alias_kind":"pith_short_16","alias_value":"GSKMHK62JBFXTTMC","created_at":"2026-07-05T06:41:02.964886+00:00"},{"alias_kind":"pith_short_8","alias_value":"GSKMHK62","created_at":"2026-07-05T06:41:02.964886+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.10118","citing_title":"Image Embedding Sampling Method for Diverse Captioning","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC","json":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC.json","graph_json":"https://pith.science/api/pith-number/GSKMHK62JBFXTTMCFJUICNUIPC/graph.json","events_json":"https://pith.science/api/pith-number/GSKMHK62JBFXTTMCFJUICNUIPC/events.json","paper":"https://pith.science/paper/GSKMHK62"},"agent_actions":{"view_html":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC","download_json":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC.json","view_paper":"https://pith.science/paper/GSKMHK62","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.08231&json=true","fetch_graph":"https://pith.science/api/pith-number/GSKMHK62JBFXTTMCFJUICNUIPC/graph.json","fetch_events":"https://pith.science/api/pith-number/GSKMHK62JBFXTTMCFJUICNUIPC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC/action/storage_attestation","attest_author":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC/action/author_attestation","sign_citation":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC/action/citation_signature","submit_replication":"https://pith.science/pith/GSKMHK62JBFXTTMCFJUICNUIPC/action/replication_record"}},"created_at":"2026-07-05T06:41:02.964886+00:00","updated_at":"2026-07-05T06:41:02.964886+00:00"}