{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7","short_pith_number":"pith:PGC3JFW4","canonical_record":{"source":{"id":"2110.06615","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-10-13T10:17:06Z","cross_cats_sorted":[],"title_canon_sha256":"42129f3d067aff27a0778280fa7daf510541e4985125dd094f712c5e0608dcab","abstract_canon_sha256":"3dab5b423d2c709068c1dfa05af8a560fa8494f3f9981d44a5748dcd9b1b446e"},"schema_version":"1.0"},"canonical_sha256":"7985b496dc8a7560e7793a5f6ced0ec7dffdcdd1901906c5021a1c2160f51759","source":{"kind":"arxiv","id":"2110.06615","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2110.06615","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"arxiv_version","alias_value":"2110.06615v1","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.06615","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"pith_short_12","alias_value":"PGC3JFW4RJ2W","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"pith_short_16","alias_value":"PGC3JFW4RJ2WBZ3Z","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"pith_short_8","alias_value":"PGC3JFW4","created_at":"2026-07-05T03:22:27Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7","target":"record","payload":{"canonical_record":{"source":{"id":"2110.06615","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-10-13T10:17:06Z","cross_cats_sorted":[],"title_canon_sha256":"42129f3d067aff27a0778280fa7daf510541e4985125dd094f712c5e0608dcab","abstract_canon_sha256":"3dab5b423d2c709068c1dfa05af8a560fa8494f3f9981d44a5748dcd9b1b446e"},"schema_version":"1.0"},"canonical_sha256":"7985b496dc8a7560e7793a5f6ced0ec7dffdcdd1901906c5021a1c2160f51759","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:22:27.842958Z","signature_b64":"oqIXvdCcQfyr0bYeTlIn3Rmf8RAcStQmmmm7Ebbzi+fvgX6C8rFYGvqPQTsA9RTjS+qG3C+80RFbfa/eqvsIBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7985b496dc8a7560e7793a5f6ced0ec7dffdcdd1901906c5021a1c2160f51759","last_reissued_at":"2026-07-05T03:22:27.842585Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:22:27.842585Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2110.06615","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:22:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KZEV79niRy1nqe4QqvAEBsKqQI7gJlDFf/7Rzq0khS85zcyLLv258sSaAhTLXVxtnaAfbfNrVhV2XqxCMHCTAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:21:09.650026Z"},"content_sha256":"ed476a050522ea5f265005b86f6d45f518d778dd785e0df0b8bc921c386caf2f","schema_version":"1.0","event_id":"sha256:ed476a050522ea5f265005b86f6d45f518d778dd785e0df0b8bc921c386caf2f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"CLIP4Caption: CLIP for Video Caption","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dian Li, Fengyun Rao, Mingkang Tang, Xiu Li, Zhanyu Wang, Zhenhua Liu","submitted_at":"2021-10-13T10:17:06Z","abstract_excerpt":"Video captioning is a challenging task since it requires generating sentences describing various diverse and complex videos. Existing video captioning models lack adequate visual representation due to the neglect of the existence of gaps between videos and texts. To bridge this gap, in this paper, we propose a CLIP4Caption framework that improves video captioning based on a CLIP-enhanced video-text matching network (VTM). This framework is taking full advantage of the information from both vision and language and enforcing the model to learn strongly text-correlated video features for text gen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.06615","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.06615/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:22:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vSnReW92g0Qnxy8Y5GJkpRjOlkc6CKYvoY9g99NZeel1Z5PHai9UXF00nOIkUmc7OJOPU3gBY8o0Zk40dAqmAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:21:09.650623Z"},"content_sha256":"a5d346fafca77822998248918118627b093c670345c6167c5710359580f0d274","schema_version":"1.0","event_id":"sha256:a5d346fafca77822998248918118627b093c670345c6167c5710359580f0d274"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7/bundle.json","state_url":"https://pith.science/pith/PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T19:21:09Z","links":{"resolver":"https://pith.science/pith/PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7","bundle":"https://pith.science/pith/PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7/bundle.json","state":"https://pith.science/pith/PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:PGC3JFW4RJ2WBZ3ZHJPWZ3IOY7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3dab5b423d2c709068c1dfa05af8a560fa8494f3f9981d44a5748dcd9b1b446e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-10-13T10:17:06Z","title_canon_sha256":"42129f3d067aff27a0778280fa7daf510541e4985125dd094f712c5e0608dcab"},"schema_version":"1.0","source":{"id":"2110.06615","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2110.06615","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"arxiv_version","alias_value":"2110.06615v1","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.06615","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"pith_short_12","alias_value":"PGC3JFW4RJ2W","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"pith_short_16","alias_value":"PGC3JFW4RJ2WBZ3Z","created_at":"2026-07-05T03:22:27Z"},{"alias_kind":"pith_short_8","alias_value":"PGC3JFW4","created_at":"2026-07-05T03:22:27Z"}],"graph_snapshots":[{"event_id":"sha256:a5d346fafca77822998248918118627b093c670345c6167c5710359580f0d274","target":"graph","created_at":"2026-07-05T03:22:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2110.06615/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Video captioning is a challenging task since it requires generating sentences describing various diverse and complex videos. Existing video captioning models lack adequate visual representation due to the neglect of the existence of gaps between videos and texts. To bridge this gap, in this paper, we propose a CLIP4Caption framework that improves video captioning based on a CLIP-enhanced video-text matching network (VTM). This framework is taking full advantage of the information from both vision and language and enforcing the model to learn strongly text-correlated video features for text gen","authors_text":"Dian Li, Fengyun Rao, Mingkang Tang, Xiu Li, Zhanyu Wang, Zhenhua Liu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-10-13T10:17:06Z","title":"CLIP4Caption: CLIP for Video Caption"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.06615","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ed476a050522ea5f265005b86f6d45f518d778dd785e0df0b8bc921c386caf2f","target":"record","created_at":"2026-07-05T03:22:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3dab5b423d2c709068c1dfa05af8a560fa8494f3f9981d44a5748dcd9b1b446e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-10-13T10:17:06Z","title_canon_sha256":"42129f3d067aff27a0778280fa7daf510541e4985125dd094f712c5e0608dcab"},"schema_version":"1.0","source":{"id":"2110.06615","kind":"arxiv","version":1}},"canonical_sha256":"7985b496dc8a7560e7793a5f6ced0ec7dffdcdd1901906c5021a1c2160f51759","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7985b496dc8a7560e7793a5f6ced0ec7dffdcdd1901906c5021a1c2160f51759","first_computed_at":"2026-07-05T03:22:27.842585Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:22:27.842585Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oqIXvdCcQfyr0bYeTlIn3Rmf8RAcStQmmmm7Ebbzi+fvgX6C8rFYGvqPQTsA9RTjS+qG3C+80RFbfa/eqvsIBA==","signature_status":"signed_v1","signed_at":"2026-07-05T03:22:27.842958Z","signed_message":"canonical_sha256_bytes"},"source_id":"2110.06615","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ed476a050522ea5f265005b86f6d45f518d778dd785e0df0b8bc921c386caf2f","sha256:a5d346fafca77822998248918118627b093c670345c6167c5710359580f0d274"],"state_sha256":"e61c998962aaa2f6533ae8c722f00ed1db5f4c6102ed845be6e7e38012af24d3"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2Lsp94psZXDcylWdlKvzLVizZA6cKRLFmuZxmozdUIpUIrj/EJndKeNEXR0prHVOhHbYx1mEEMaCrWvlxZcbCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T19:21:09.654400Z","bundle_sha256":"24e74a23d148ea7b6153083650477849f0ba361768f8876aa13df570c0993f56"}}