{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:KZOM6IISSFJRF3B4LKENJHMHSV","short_pith_number":"pith:KZOM6IIS","schema_version":"1.0","canonical_sha256":"565ccf2112915312ec3c5a88d49d879545029d4ad94c66bf61c53112e7a2153c","source":{"kind":"arxiv","id":"2211.09371","version":3},"attestation_state":"computed","paper":{"title":"CapEnrich: Enriching Caption Semantics for Web Images via Cross-modal Pre-trained Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Linli Yao, Qin Jin, Weijing Chen","submitted_at":"2022-11-17T06:55:49Z","abstract_excerpt":"Automatically generating textual descriptions for massive unlabeled images on the web can greatly benefit realistic web applications, e.g. multimodal retrieval and recommendation. However, existing models suffer from the problem of generating ``over-generic'' descriptions, such as their tendency to generate repetitive sentences with common concepts for different images. These generic descriptions fail to provide sufficient textual semantics for ever-changing web images. Inspired by the recent success of Vision-Language Pre-training (VLP) models that learn diverse image-text concept alignment d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.09371","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-11-17T06:55:49Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"a12ce2ecb2e813c620c24d361c54efe9cc509528b19769b70d9f918b0e9d176f","abstract_canon_sha256":"00816b214f800f1a543370301a8ba4d5a36b465e50576e5ee0702062588cf2c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:52:23.930948Z","signature_b64":"4y/aRYWF14hpusfiWloLncm81XW1RxQTb4glorTjY76hbsSnVrSgtrUXTBsWZZF2dN5d6mIVHCJTXoYOEKlADA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"565ccf2112915312ec3c5a88d49d879545029d4ad94c66bf61c53112e7a2153c","last_reissued_at":"2026-07-05T05:52:23.930310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:52:23.930310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CapEnrich: Enriching Caption Semantics for Web Images via Cross-modal Pre-trained Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Linli Yao, Qin Jin, Weijing Chen","submitted_at":"2022-11-17T06:55:49Z","abstract_excerpt":"Automatically generating textual descriptions for massive unlabeled images on the web can greatly benefit realistic web applications, e.g. multimodal retrieval and recommendation. However, existing models suffer from the problem of generating ``over-generic'' descriptions, such as their tendency to generate repetitive sentences with common concepts for different images. These generic descriptions fail to provide sufficient textual semantics for ever-changing web images. Inspired by the recent success of Vision-Language Pre-training (VLP) models that learn diverse image-text concept alignment d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.09371","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.09371/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.09371","created_at":"2026-07-05T05:52:23.930398+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.09371v3","created_at":"2026-07-05T05:52:23.930398+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.09371","created_at":"2026-07-05T05:52:23.930398+00:00"},{"alias_kind":"pith_short_12","alias_value":"KZOM6IISSFJR","created_at":"2026-07-05T05:52:23.930398+00:00"},{"alias_kind":"pith_short_16","alias_value":"KZOM6IISSFJRF3B4","created_at":"2026-07-05T05:52:23.930398+00:00"},{"alias_kind":"pith_short_8","alias_value":"KZOM6IIS","created_at":"2026-07-05T05:52:23.930398+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.22613","citing_title":"RICO: Improving Accuracy and Completeness in Image Recaptioning via Visual Reconstruction","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV","json":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV.json","graph_json":"https://pith.science/api/pith-number/KZOM6IISSFJRF3B4LKENJHMHSV/graph.json","events_json":"https://pith.science/api/pith-number/KZOM6IISSFJRF3B4LKENJHMHSV/events.json","paper":"https://pith.science/paper/KZOM6IIS"},"agent_actions":{"view_html":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV","download_json":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV.json","view_paper":"https://pith.science/paper/KZOM6IIS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.09371&json=true","fetch_graph":"https://pith.science/api/pith-number/KZOM6IISSFJRF3B4LKENJHMHSV/graph.json","fetch_events":"https://pith.science/api/pith-number/KZOM6IISSFJRF3B4LKENJHMHSV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV/action/storage_attestation","attest_author":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV/action/author_attestation","sign_citation":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV/action/citation_signature","submit_replication":"https://pith.science/pith/KZOM6IISSFJRF3B4LKENJHMHSV/action/replication_record"}},"created_at":"2026-07-05T05:52:23.930398+00:00","updated_at":"2026-07-05T05:52:23.930398+00:00"}