{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:WVUUAVRJFH6XP3KATNH27QZOEL","short_pith_number":"pith:WVUUAVRJ","schema_version":"1.0","canonical_sha256":"b56940562929fd77ed409b4fafc32e22ed94ad2876063dd103ebeb70c3b2aa4c","source":{"kind":"arxiv","id":"2012.02339","version":3},"attestation_state":"computed","paper":{"title":"Understanding Guided Image Captioning Performance across Domains","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bo Pang, Edwin G. Ng, Piyush Sharma, Radu Soricut","submitted_at":"2020-12-04T00:05:02Z","abstract_excerpt":"Image captioning models generally lack the capability to take into account user interest, and usually default to global descriptions that try to balance readability, informativeness, and information overload. On the other hand, VQA models generally lack the ability to provide long descriptive answers, while expecting the textual question to be quite precise. We present a method to control the concepts that an image caption should focus on, using an additional input called the guiding text that refers to either groundable or ungroundable concepts in the image. Our model consists of a Transforme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.02339","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-12-04T00:05:02Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"982069ed1355c94e2aefd1efc2be27164bfb7f5aac3f67b09b796e746b0b77b3","abstract_canon_sha256":"5580b4dde43876155a2189b2d722143ca0ddbe0e8af7738b2fbbddac318c2979"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:30:50.737866Z","signature_b64":"CINeP93Nt0A02n+kCcUz1WVQxo5CYYkv3upy3Tzl5X0i70Y6qyvVW44Gq8u/6EPBnOIk5fd4PgM42swesjOVAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b56940562929fd77ed409b4fafc32e22ed94ad2876063dd103ebeb70c3b2aa4c","last_reissued_at":"2026-07-05T03:30:50.737397Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:30:50.737397Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Guided Image Captioning Performance across Domains","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bo Pang, Edwin G. Ng, Piyush Sharma, Radu Soricut","submitted_at":"2020-12-04T00:05:02Z","abstract_excerpt":"Image captioning models generally lack the capability to take into account user interest, and usually default to global descriptions that try to balance readability, informativeness, and information overload. On the other hand, VQA models generally lack the ability to provide long descriptive answers, while expecting the textual question to be quite precise. We present a method to control the concepts that an image caption should focus on, using an additional input called the guiding text that refers to either groundable or ungroundable concepts in the image. Our model consists of a Transforme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.02339","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.02339/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.02339","created_at":"2026-07-05T03:30:50.737458+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.02339v3","created_at":"2026-07-05T03:30:50.737458+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.02339","created_at":"2026-07-05T03:30:50.737458+00:00"},{"alias_kind":"pith_short_12","alias_value":"WVUUAVRJFH6X","created_at":"2026-07-05T03:30:50.737458+00:00"},{"alias_kind":"pith_short_16","alias_value":"WVUUAVRJFH6XP3KA","created_at":"2026-07-05T03:30:50.737458+00:00"},{"alias_kind":"pith_short_8","alias_value":"WVUUAVRJ","created_at":"2026-07-05T03:30:50.737458+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.04579","citing_title":"Firebolt-VL: Efficient Vision-Language Understanding with Cross-Modality Modulation","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL","json":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL.json","graph_json":"https://pith.science/api/pith-number/WVUUAVRJFH6XP3KATNH27QZOEL/graph.json","events_json":"https://pith.science/api/pith-number/WVUUAVRJFH6XP3KATNH27QZOEL/events.json","paper":"https://pith.science/paper/WVUUAVRJ"},"agent_actions":{"view_html":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL","download_json":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL.json","view_paper":"https://pith.science/paper/WVUUAVRJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.02339&json=true","fetch_graph":"https://pith.science/api/pith-number/WVUUAVRJFH6XP3KATNH27QZOEL/graph.json","fetch_events":"https://pith.science/api/pith-number/WVUUAVRJFH6XP3KATNH27QZOEL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL/action/storage_attestation","attest_author":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL/action/author_attestation","sign_citation":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL/action/citation_signature","submit_replication":"https://pith.science/pith/WVUUAVRJFH6XP3KATNH27QZOEL/action/replication_record"}},"created_at":"2026-07-05T03:30:50.737458+00:00","updated_at":"2026-07-05T03:30:50.737458+00:00"}