{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:B4EX4JW6AKNOJOCILNW3AIET2U","short_pith_number":"pith:B4EX4JW6","canonical_record":{"source":{"id":"2502.13363","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-19T01:53:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"892734b7dd7200458b395bbbd95e3f92695b873c80af785b8545491392f643fb","abstract_canon_sha256":"b8ddddff0736c2b6bdd59a6cf4ecf98433c159e989892744fd0bb5599746e557"},"schema_version":"1.0"},"canonical_sha256":"0f097e26de029ae4b8485b6db02093d52c6ccca0dd3efa949e64307dcf5b42ed","source":{"kind":"arxiv","id":"2502.13363","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.13363","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"arxiv_version","alias_value":"2502.13363v1","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13363","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"pith_short_12","alias_value":"B4EX4JW6AKNO","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"pith_short_16","alias_value":"B4EX4JW6AKNOJOCI","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"pith_short_8","alias_value":"B4EX4JW6","created_at":"2026-07-05T10:16:45Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:B4EX4JW6AKNOJOCILNW3AIET2U","target":"record","payload":{"canonical_record":{"source":{"id":"2502.13363","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-19T01:53:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"892734b7dd7200458b395bbbd95e3f92695b873c80af785b8545491392f643fb","abstract_canon_sha256":"b8ddddff0736c2b6bdd59a6cf4ecf98433c159e989892744fd0bb5599746e557"},"schema_version":"1.0"},"canonical_sha256":"0f097e26de029ae4b8485b6db02093d52c6ccca0dd3efa949e64307dcf5b42ed","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:45.943493Z","signature_b64":"oxW6GCIsmJwElBkMZTlLV5sN13TbQYeKqbYj7k1cMDREjafGeDDtourdOzI9sBL1bzRU1jZnu8Hmpl40ZDpVBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f097e26de029ae4b8485b6db02093d52c6ccca0dd3efa949e64307dcf5b42ed","last_reissued_at":"2026-07-05T10:16:45.943004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:45.943004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.13363","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:16:45Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2uFmkn+K6yZZcGl1S+GMgjKsIziT/uQoX7sjUTMBUhqKn2LDmg8nHyCitM8sUfxVe0iP/IsJGnvVH/suXB81Ag==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T13:14:49.403310Z"},"content_sha256":"6027c851ea134b0b02714534ab78927b5a12e1d8da642890ef2ef162d970bf44","schema_version":"1.0","event_id":"sha256:6027c851ea134b0b02714534ab78927b5a12e1d8da642890ef2ef162d970bf44"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:B4EX4JW6AKNOJOCILNW3AIET2U","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Pretrained Image-Text Models are Secretly Video Captioners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chunhui Zhang, Soroush Vosoughi, Yiren Jian, Zhongyu Ouyang","submitted_at":"2025-02-19T01:53:03Z","abstract_excerpt":"Developing video captioning models is computationally expensive. The dynamic nature of video also complicates the design of multimodal models that can effectively caption these sequences. However, we find that by using minimal computational resources and without complex modifications to address video dynamics, an image-based model can be repurposed to outperform several specialised video captioning systems. Our adapted model demonstrates top tier performance on major benchmarks, ranking 2nd on MSRVTT and MSVD, and 3rd on VATEX. We transform it into a competitive video captioner by post trainin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13363","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.13363/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:16:45Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kzTUv3gEbA7Cab7WcyQGdG1SEG17nQIurEzjYnYuhFYLvIdiDb9fGvQemaB4lT439gLZYFL6K4nd5pclBidBCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T13:14:49.403893Z"},"content_sha256":"3e18fcd2ca7ad9fac74b7fe86f7a7f5ce14c2c0f20e8183836089972e0fcc91f","schema_version":"1.0","event_id":"sha256:3e18fcd2ca7ad9fac74b7fe86f7a7f5ce14c2c0f20e8183836089972e0fcc91f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/B4EX4JW6AKNOJOCILNW3AIET2U/bundle.json","state_url":"https://pith.science/pith/B4EX4JW6AKNOJOCILNW3AIET2U/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/B4EX4JW6AKNOJOCILNW3AIET2U/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T13:14:49Z","links":{"resolver":"https://pith.science/pith/B4EX4JW6AKNOJOCILNW3AIET2U","bundle":"https://pith.science/pith/B4EX4JW6AKNOJOCILNW3AIET2U/bundle.json","state":"https://pith.science/pith/B4EX4JW6AKNOJOCILNW3AIET2U/state.json","well_known_bundle":"https://pith.science/.well-known/pith/B4EX4JW6AKNOJOCILNW3AIET2U/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:B4EX4JW6AKNOJOCILNW3AIET2U","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b8ddddff0736c2b6bdd59a6cf4ecf98433c159e989892744fd0bb5599746e557","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-19T01:53:03Z","title_canon_sha256":"892734b7dd7200458b395bbbd95e3f92695b873c80af785b8545491392f643fb"},"schema_version":"1.0","source":{"id":"2502.13363","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.13363","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"arxiv_version","alias_value":"2502.13363v1","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13363","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"pith_short_12","alias_value":"B4EX4JW6AKNO","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"pith_short_16","alias_value":"B4EX4JW6AKNOJOCI","created_at":"2026-07-05T10:16:45Z"},{"alias_kind":"pith_short_8","alias_value":"B4EX4JW6","created_at":"2026-07-05T10:16:45Z"}],"graph_snapshots":[{"event_id":"sha256:3e18fcd2ca7ad9fac74b7fe86f7a7f5ce14c2c0f20e8183836089972e0fcc91f","target":"graph","created_at":"2026-07-05T10:16:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.13363/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Developing video captioning models is computationally expensive. The dynamic nature of video also complicates the design of multimodal models that can effectively caption these sequences. However, we find that by using minimal computational resources and without complex modifications to address video dynamics, an image-based model can be repurposed to outperform several specialised video captioning systems. Our adapted model demonstrates top tier performance on major benchmarks, ranking 2nd on MSRVTT and MSVD, and 3rd on VATEX. We transform it into a competitive video captioner by post trainin","authors_text":"Chunhui Zhang, Soroush Vosoughi, Yiren Jian, Zhongyu Ouyang","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-19T01:53:03Z","title":"Pretrained Image-Text Models are Secretly Video Captioners"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13363","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6027c851ea134b0b02714534ab78927b5a12e1d8da642890ef2ef162d970bf44","target":"record","created_at":"2026-07-05T10:16:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b8ddddff0736c2b6bdd59a6cf4ecf98433c159e989892744fd0bb5599746e557","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-19T01:53:03Z","title_canon_sha256":"892734b7dd7200458b395bbbd95e3f92695b873c80af785b8545491392f643fb"},"schema_version":"1.0","source":{"id":"2502.13363","kind":"arxiv","version":1}},"canonical_sha256":"0f097e26de029ae4b8485b6db02093d52c6ccca0dd3efa949e64307dcf5b42ed","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0f097e26de029ae4b8485b6db02093d52c6ccca0dd3efa949e64307dcf5b42ed","first_computed_at":"2026-07-05T10:16:45.943004Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:16:45.943004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oxW6GCIsmJwElBkMZTlLV5sN13TbQYeKqbYj7k1cMDREjafGeDDtourdOzI9sBL1bzRU1jZnu8Hmpl40ZDpVBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T10:16:45.943493Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.13363","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6027c851ea134b0b02714534ab78927b5a12e1d8da642890ef2ef162d970bf44","sha256:3e18fcd2ca7ad9fac74b7fe86f7a7f5ce14c2c0f20e8183836089972e0fcc91f"],"state_sha256":"194362c8fbcfcf85069bd631f4f039f37e64ef67cfdf6f67be76d67300b945e8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4SdK6pM8rPOA4i8GlToGAeWuBAITpliWViS5D/trl8GQnyYFTvC6aljMTNN5lRCprKQvgJj/r8qy99L19bHeBQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T13:14:49.407976Z","bundle_sha256":"088ca4658adac56e2ad9e58608836d9b57b2a38f26f8e550a2e5e50b35d8f94b"}}