{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5KZBKWEJL3AJIHS5YUDU7EBUXF","short_pith_number":"pith:5KZBKWEJ","schema_version":"1.0","canonical_sha256":"eab21558895ec0941e5dc5074f9034b948d5bc85ef7d7b43e9f36339c66fdc63","source":{"kind":"arxiv","id":"2302.08268","version":1},"attestation_state":"computed","paper":{"title":"Retrieval-augmented Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bruno Martins, Desmond Elliott, Rita Ramos","submitted_at":"2023-02-16T12:54:13Z","abstract_excerpt":"Inspired by retrieval-augmented language generation and pretrained Vision and Language (V&L) encoders, we present a new approach to image captioning that generates sentences given the input image and a set of captions retrieved from a datastore, as opposed to the image alone. The encoder in our model jointly processes the image and retrieved captions using a pretrained V&L BERT, while the decoder attends to the multimodal encoder representations, benefiting from the extra textual evidence from the retrieved captions. Experimental results on the COCO dataset show that image captioning can be ef"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08268","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-02-16T12:54:13Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"981c5fa19af49f0c48e872b2f4f3b1e985edeb177f8f8eb1d78ffb7d208ecef0","abstract_canon_sha256":"25275409cad505860cbf8a507141a74f6d3aaaa9e1c48ea7ccee83ae14135eeb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:42:36.219894Z","signature_b64":"i/0xEoUU9wJ4R18tbWAvb2s77eUYlCYLIjivi4zT/LikHEHR3xAChVIF6QDjblyDVWdpHG6Y2x1+gJYzdvhEDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eab21558895ec0941e5dc5074f9034b948d5bc85ef7d7b43e9f36339c66fdc63","last_reissued_at":"2026-07-05T05:42:36.219464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:42:36.219464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Retrieval-augmented Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bruno Martins, Desmond Elliott, Rita Ramos","submitted_at":"2023-02-16T12:54:13Z","abstract_excerpt":"Inspired by retrieval-augmented language generation and pretrained Vision and Language (V&L) encoders, we present a new approach to image captioning that generates sentences given the input image and a set of captions retrieved from a datastore, as opposed to the image alone. The encoder in our model jointly processes the image and retrieved captions using a pretrained V&L BERT, while the decoder attends to the multimodal encoder representations, benefiting from the extra textual evidence from the retrieved captions. Experimental results on the COCO dataset show that image captioning can be ef"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08268","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08268/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08268","created_at":"2026-07-05T05:42:36.219521+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08268v1","created_at":"2026-07-05T05:42:36.219521+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08268","created_at":"2026-07-05T05:42:36.219521+00:00"},{"alias_kind":"pith_short_12","alias_value":"5KZBKWEJL3AJ","created_at":"2026-07-05T05:42:36.219521+00:00"},{"alias_kind":"pith_short_16","alias_value":"5KZBKWEJL3AJIHS5","created_at":"2026-07-05T05:42:36.219521+00:00"},{"alias_kind":"pith_short_8","alias_value":"5KZBKWEJ","created_at":"2026-07-05T05:42:36.219521+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.09298","citing_title":"Multi-Modal LLM based Image Captioning in ICT: Bridging the Gap Between General and Industry Domain","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF","json":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF.json","graph_json":"https://pith.science/api/pith-number/5KZBKWEJL3AJIHS5YUDU7EBUXF/graph.json","events_json":"https://pith.science/api/pith-number/5KZBKWEJL3AJIHS5YUDU7EBUXF/events.json","paper":"https://pith.science/paper/5KZBKWEJ"},"agent_actions":{"view_html":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF","download_json":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF.json","view_paper":"https://pith.science/paper/5KZBKWEJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08268&json=true","fetch_graph":"https://pith.science/api/pith-number/5KZBKWEJL3AJIHS5YUDU7EBUXF/graph.json","fetch_events":"https://pith.science/api/pith-number/5KZBKWEJL3AJIHS5YUDU7EBUXF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF/action/storage_attestation","attest_author":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF/action/author_attestation","sign_citation":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF/action/citation_signature","submit_replication":"https://pith.science/pith/5KZBKWEJL3AJIHS5YUDU7EBUXF/action/replication_record"}},"created_at":"2026-07-05T05:42:36.219521+00:00","updated_at":"2026-07-05T05:42:36.219521+00:00"}