{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XFJBUXP4BJ2V7PYGQ4ABLJBRRJ","short_pith_number":"pith:XFJBUXP4","schema_version":"1.0","canonical_sha256":"b9521a5dfc0a755fbf06870015a4318a764add09bb9f14bd4b6909c395ad9d54","source":{"kind":"arxiv","id":"2402.13936","version":1},"attestation_state":"computed","paper":{"title":"Distinctive Image Captioning: Leveraging Ground Truth Captions in CLIP Guided Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Antoine Chaffin, Ewa Kijak, Vincent Claveau","submitted_at":"2024-02-21T17:05:06Z","abstract_excerpt":"Training image captioning models using teacher forcing results in very generic samples, whereas more distinctive captions can be very useful in retrieval applications or to produce alternative texts describing images for accessibility. Reinforcement Learning (RL) allows to use cross-modal retrieval similarity score between the generated caption and the input image as reward to guide the training, leading to more distinctive captions. Recent studies show that pre-trained cross-modal retrieval models can be used to provide this reward, completely eliminating the need for reference captions. Howe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13936","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T17:05:06Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"15d7b4f2b5cdef80d59284a7e8b3c45e6435896aca6228b0b4f5e8ff7c2f5959","abstract_canon_sha256":"809fe0a32be3055e5acc504479e69d4a1f3472cbe221dca1d9314f06d9a98282"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:47:50.466835Z","signature_b64":"JAME4ATnDIzl83xY2iAgZn6nQjE8yn/njbDVaHoMsYCE3+q2oc6KXnUOJRmmh7ljatI+UKztQr7ucTlHGWdeCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9521a5dfc0a755fbf06870015a4318a764add09bb9f14bd4b6909c395ad9d54","last_reissued_at":"2026-07-05T07:47:50.466235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:47:50.466235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distinctive Image Captioning: Leveraging Ground Truth Captions in CLIP Guided Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Antoine Chaffin, Ewa Kijak, Vincent Claveau","submitted_at":"2024-02-21T17:05:06Z","abstract_excerpt":"Training image captioning models using teacher forcing results in very generic samples, whereas more distinctive captions can be very useful in retrieval applications or to produce alternative texts describing images for accessibility. Reinforcement Learning (RL) allows to use cross-modal retrieval similarity score between the generated caption and the input image as reward to guide the training, leading to more distinctive captions. Recent studies show that pre-trained cross-modal retrieval models can be used to provide this reward, completely eliminating the need for reference captions. Howe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13936","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13936/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13936","created_at":"2026-07-05T07:47:50.466296+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13936v1","created_at":"2026-07-05T07:47:50.466296+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13936","created_at":"2026-07-05T07:47:50.466296+00:00"},{"alias_kind":"pith_short_12","alias_value":"XFJBUXP4BJ2V","created_at":"2026-07-05T07:47:50.466296+00:00"},{"alias_kind":"pith_short_16","alias_value":"XFJBUXP4BJ2V7PYG","created_at":"2026-07-05T07:47:50.466296+00:00"},{"alias_kind":"pith_short_8","alias_value":"XFJBUXP4","created_at":"2026-07-05T07:47:50.466296+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00661","citing_title":"Automatic Identification and Description of Jewelry Through Computer Vision and Neural Networks for Translators and Interpreters","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ","json":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ.json","graph_json":"https://pith.science/api/pith-number/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/graph.json","events_json":"https://pith.science/api/pith-number/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/events.json","paper":"https://pith.science/paper/XFJBUXP4"},"agent_actions":{"view_html":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ","download_json":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ.json","view_paper":"https://pith.science/paper/XFJBUXP4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13936&json=true","fetch_graph":"https://pith.science/api/pith-number/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/graph.json","fetch_events":"https://pith.science/api/pith-number/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/action/storage_attestation","attest_author":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/action/author_attestation","sign_citation":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/action/citation_signature","submit_replication":"https://pith.science/pith/XFJBUXP4BJ2V7PYGQ4ABLJBRRJ/action/replication_record"}},"created_at":"2026-07-05T07:47:50.466296+00:00","updated_at":"2026-07-05T07:47:50.466296+00:00"}