{"as_of":"2026-08-12T23:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f91179f7faaf3b92caccca299ad556e8173e566eacd74e803640c8a103cb7f81","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T12:10:17.560221Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.11013/citation-record","integrity":"/paper/2608.11013/integrity","json":"/paper/2608.11013/citation-record.json","paper":"/paper/2608.11013"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.166177Z","title":"Learning to compose topic-aware mixture of experts for zero-shot video captioning,","venue":null,"work_id":"d992e1f1-ed03-42aa-b396-96708904b728","year":2019},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.374700Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:0eb342ac571937a8fc2e3671f94149b8f18fe0a390ce20104431280cad5f5b20","observation_id":"4d7ac9b3-8ceb-4ae3-87d9-7bf9299bb4cc","resolution":{"observed_at":"2026-08-12T12:10:18.171021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03032","last_updated":"2023-03-06T11:02:47Z","snapshot_observed_at":"2026-08-05T10:22:16.844750Z","submitted_at":"2023-03-06T11:02:47Z","title":"DeCap: Decoding CLIP Latents for Zero-Shot Captioning via Text-Only Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03032","snapshot_observed_at":"2026-08-12T12:10:17.379818Z","title":"Decap: Decoding clip la- tents for zero-shot captioning via text-only training,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.379818Z"},"links":{"cited_paper":"/paper/2303.03032","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:fcfa5c887168889e82d9b00d4491b4e2624e2af6655898fa1464add0d73815c3","observation_id":"f533bef9-e475-42cf-936d-a43eb374ae32","resolution":{"observed_at":"2026-08-12T12:10:17.379818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.08567","last_updated":"2024-01-16T18:52:27Z","snapshot_observed_at":"2026-08-09T01:05:56.184796Z","submitted_at":"2024-01-16T18:52:27Z","title":"Connect, Collapse, Corrupt: Learning Cross-Modal Tasks with Uni-Modal Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.08567","snapshot_observed_at":"2026-08-12T12:10:17.384647Z","title":"Connect, collapse, corrupt: Learning cross-modal tasks with uni-modal data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.384647Z"},"links":{"cited_paper":"/paper/2401.08567","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:20693d9d8be390128ece21b3f181dcb36482be797c224764d23cb35865a5b792","observation_id":"19c991cc-536b-47da-9036-bc94ea65deb0","resolution":{"observed_at":"2026-08-12T12:10:17.384647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18046","last_updated":"2024-09-26T16:47:32Z","snapshot_observed_at":"2026-08-12T22:36:43.985308Z","submitted_at":"2024-09-26T16:47:32Z","title":"IFCap: Image-like Retrieval and Frequency-based Entity Filtering for Zero-shot Captioning","version":1},"cited_work":{"arxiv_id":"2409.18046","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.18046","snapshot_observed_at":"2026-08-12T12:10:17.749134Z","title":"IFCap: Image-like Retrieval and Frequency-based Entity Filtering for Zero-shot Captioning","venue":"cs.CV","work_id":"7eaf6507-7223-4775-b3b5-34afa8a7e319","year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.389553Z"},"links":{"cited_paper":"/paper/2409.18046","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:f2a83c12c3ed10fecb666ed676d53a0f4bcd563ed6038df49a9d30fd9004816d","observation_id":"da62bf44-5334-46a7-a68d-666868ee2414","resolution":{"observed_at":"2026-08-12T12:10:17.754368Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.152428Z","title":"Improving cross-modal alignment with synthetic pairs for text-only image captioning,","venue":null,"work_id":"0204be7b-c349-4ad6-a312-1384bab47e5f","year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.394603Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:80e49b6f669c704b84380da4ca5e7fa0ce9c1e94029809443d96c0784bfc6c94","observation_id":"64294703-8e7a-4721-ae98-2d64fe6fc190","resolution":{"observed_at":"2026-08-12T12:10:18.156961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.138858Z","title":"Retta: Retrieval-enhanced test-time adaptation for zero-shot video cap- tioning,","venue":null,"work_id":"34957f81-817c-44a1-b4d8-15facf6e970e","year":2025},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.399148Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:207e13c5e7be82f44302d9e3b0395d9751a9cd47d898ff81c1c83ea3c1e5ba48","observation_id":"507ee31f-a271-46ee-9576-c776fcacef4d","resolution":{"observed_at":"2026-08-12T12:10:18.143378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.124780Z","title":"Text-only training for image captioning using noise-injected CLIP,","venue":null,"work_id":"52fdfe79-a2ab-4478-94ce-4fa08156cbe2","year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.404015Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:b5482af63410a23769998994061e8fcc804908e8c17b8d18cd2221bef6e0a913","observation_id":"624c36ab-f79f-4326-8ab3-91b525f424ff","resolution":{"observed_at":"2026-08-12T12:10:18.129207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.109872Z","title":"Sequence to sequence-video to text,","venue":null,"work_id":"1c386464-9241-4107-ae99-29ac464c5d02","year":2015},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.408247Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:be216b87b10b8504bdafee944869bf6ffe3b978baad13b92cf0188262e7b0dcb","observation_id":"ebb39956-b509-4919-8638-3a7cb53d208b","resolution":{"observed_at":"2026-08-12T12:10:18.115003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.095172Z","title":"Bidirectional long- short term memory for video description,","venue":null,"work_id":"9c6ce22e-1fe4-4562-84c2-0d52a2c6bc34","year":2016},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.412383Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:6e2a46d5525abbfa1e2d54da5d712d5a506f779e12638323ef7d5349c80945e0","observation_id":"87b366b9-07bc-4024-99ce-0f2f9d010979","resolution":{"observed_at":"2026-08-12T12:10:18.100003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.080160Z","title":"Graph convolutional network meta- learning with multi-granularity pos guidance for video captioning,","venue":null,"work_id":"40a1639f-9a93-4cfb-9ef1-cc346e901b45","year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.416548Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:1e7d64e0292aebdc29375ae5b6c5c6f2f44b10d7328dc73c2c22bbc87fe72fa3","observation_id":"351d23f9-35a5-44a6-bddd-c03b3be6856e","resolution":{"observed_at":"2026-08-12T12:10:18.084924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.065887Z","title":"Describing videos by exploiting temporal structure,","venue":null,"work_id":"ad54e7c2-4892-4397-90dd-23c4fee97137","year":2015},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.420779Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:db3ff0f48a15b0672eb79da4d6e77dffbe9ae7ad051842cb287114029c189702","observation_id":"825e202e-8fe2-4ee3-b98c-b07524bac78e","resolution":{"observed_at":"2026-08-12T12:10:18.070508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.051257Z","title":"Icocap: Improving video captioning by compounding images,","venue":null,"work_id":"d8421efb-1797-45c1-8057-e641f8ec5f1a","year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.425189Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:6339bd0ad00dcc74f2167e6d2af605122ae45b9bbd290438eed147c2d737cb2f","observation_id":"fde556e8-833d-4bdd-b1f7-9c59505c39da","resolution":{"observed_at":"2026-08-12T12:10:18.056297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.036077Z","title":"Memory- based augmentation network for video captioning,","venue":null,"work_id":"fa012426-1171-4cfe-ae85-371ab4539b69","year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.429108Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:c9f503d025508227e555a429acaf646987c03f493790ada35291fab330f9ea06","observation_id":"2b072796-1967-427d-9af9-b0957ac999b3","resolution":{"observed_at":"2026-08-12T12:10:18.041638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.021144Z","title":"Attention is all you need,","venue":null,"work_id":"92ac39a2-fd95-4902-ba95-bfec8e828a5f","year":2017},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.433335Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:7030a60cd0dc327165f4217f2a754cf1d0ae2de87ef776359f7bdb50560bfc90","observation_id":"51c5fec0-3cc5-4a59-8530-6b7ade97ba1b","resolution":{"observed_at":"2026-08-12T12:10:18.026051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:18.005696Z","title":"Hierarchical modular network for video captioning,","venue":null,"work_id":"b4f233fe-7716-434f-863e-e541f1da61e1","year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.437339Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:0191288b17dd01be80a4a349583a6427720111dbbb023c1a332fa7d5bb587716","observation_id":"73b61cdb-cd4a-451d-9b67-0ce755e24ddf","resolution":{"observed_at":"2026-08-12T12:10:18.010478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.991372Z","title":"Swinbert: End-to-end transformers with sparse attention for video captioning,","venue":null,"work_id":"4b5bad91-69af-4237-ad31-e5bf3669e9bf","year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.441498Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:ac89fe6c8d40bd7a1ab906de2e63a45c1ec74994e90a4d81e89334d4b12a9bbe","observation_id":"841c5f3c-b9ce-4f86-9946-389a85bccb3b","resolution":{"observed_at":"2026-08-12T12:10:17.995895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-12T12:10:17.446549Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.446549Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:e17172b203e5471c4b69126b06a7be3df0c841d4db28234571c6b94064c0f217","observation_id":"bc11fe1d-6959-434e-bc19-90eb76ead7d6","resolution":{"observed_at":"2026-08-12T12:10:17.446549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.976310Z","title":"Ma-lmm: Memory-augmented large multimodal model for long-term video understanding,","venue":null,"work_id":"3d830e38-05ad-4941-9630-fa379b43e52b","year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.452167Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:da4f0458658ab798dfba250d373dc3f245a4f3f0340ca7f079f0cca67030348e","observation_id":"8cabbb01-3849-49c3-a761-3b511f14332c","resolution":{"observed_at":"2026-08-12T12:10:17.981085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.00575","last_updated":"2022-11-01T16:36:01Z","snapshot_observed_at":"2026-07-06T14:13:07.158001Z","submitted_at":"2022-11-01T16:36:01Z","title":"Text-Only Training for Image Captioning using Noise-Injected CLIP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.00575","snapshot_observed_at":"2026-08-12T12:10:17.456353Z","title":"Text-only training for image captioning using noise-injected clip,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.456353Z"},"links":{"cited_paper":"/paper/2211.00575","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:c90ea4e3339615a54af62cc78b8b307682fadd1484fe3a35aaa7a398f5585cd7","observation_id":"a9dfdb99-f0a2-475d-b9ba-b17f6bc85d72","resolution":{"observed_at":"2026-08-12T12:10:17.456353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.02655","last_updated":"2022-05-30T19:45:05Z","snapshot_observed_at":"2026-08-11T01:58:52.550156Z","submitted_at":"2022-05-05T13:56:18Z","title":"Language Models Can See: Plugging Visual Controls in Text Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.02655","snapshot_observed_at":"2026-08-12T12:10:17.461499Z","title":"Language models can see: Plugging visual controls in text generation,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.461499Z"},"links":{"cited_paper":"/paper/2205.02655","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:3ea0fcf49092cebc655a0713a3e823c209f3a1a34ea89451957c860db552645d","observation_id":"86c4d617-89d4-4634-bf51-4ffe29d951be","resolution":{"observed_at":"2026-08-12T12:10:17.461499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.962739Z","title":"Zerocap: Zero-shot image-to-text generation for visual-semantic arithmetic,","venue":null,"work_id":"2eeaebe3-ebce-48f6-a37c-fb764acb6bd9","year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.466408Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:1ddb6dd75f1c3cebeaff9b53aafc74d42de028a9c20210130e69d292a9377022","observation_id":"e0f70ae5-8cda-4d16-8160-57cd5ed326ae","resolution":{"observed_at":"2026-08-12T12:10:17.967231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.949127Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"31b9ab93-d4f3-4989-9976-9cd94f73ea85","year":2021},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.471210Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:76b5480006823bb1db334b0609986afb869db3576881669bf46777227cd81657","observation_id":"4241089a-f193-4dc9-8fb1-ad9b694a9576","resolution":{"observed_at":"2026-08-12T12:10:17.953545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.13273","last_updated":"2023-05-08T02:29:28Z","snapshot_observed_at":"2026-07-06T15:20:05.729156Z","submitted_at":"2023-04-26T04:06:20Z","title":"From Association to Generation: Text-only Captioning by Unsupervised Cross-modal Mapping","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.13273","snapshot_observed_at":"2026-08-12T12:10:17.475874Z","title":"From association to genera- tion: Text-only captioning by unsupervised cross-modal mapping,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.475874Z"},"links":{"cited_paper":"/paper/2304.13273","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:55bd4c2044d28d788939936c88094fca6a1acc7cca717794055a82a22278ae20","observation_id":"45b26696-9dfd-472a-bc6e-f1d27156b202","resolution":{"observed_at":"2026-08-12T12:10:17.475874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.480381Z","title":"Clip4clip: An empirical study of clip for end to end video clip retrieval and captioning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.480381Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:ffb98b8fdee15741ec4db09f7d972bdbca9d490c112cc5f4a991a304b1f8220b","observation_id":"8131c68b-a5d3-46e6-b0ef-e539345c6f38","resolution":{"observed_at":"2026-08-12T12:10:17.480381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.06072","last_updated":"2025-03-26T08:33:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-12T11:47:11Z","title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.06072","snapshot_observed_at":"2026-08-12T12:10:17.485268Z","title":"Cogvideox: Text-to-video diffusion models with an expert transformer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.485268Z"},"links":{"cited_paper":"/paper/2408.06072","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:60adafc113ea04193018757f72c3536904e727c6885d9c4aef3ba939de9f0672","observation_id":"986d18f8-79b1-4ef9-9ab4-04171e9586e4","resolution":{"observed_at":"2026-08-12T12:10:17.485268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.490054Z","title":"Language models are unsupervised multitask learners,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.490054Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:4115550633b2c452ebd58793e8df1db5017c785b566e4d1fb65f8a0b000d00d0","observation_id":"2c2ce2d1-fb0e-4d80-b6c6-f90609bcdebc","resolution":{"observed_at":"2026-08-12T12:10:17.490054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.05614","last_updated":"2020-02-15T01:31:21Z","snapshot_observed_at":"2026-08-11T02:32:49.379748Z","submitted_at":"2020-01-16T02:18:27Z","title":"Delving Deeper into the Decoder for Video Captioning","version":3},"cited_work":{"arxiv_id":"2001.05614","doi":null,"metadata_source":"pith","pith_arxiv_id":"2001.05614","snapshot_observed_at":"2026-08-12T12:10:17.651668Z","title":"Delving Deeper into the Decoder for Video Captioning","venue":"cs.CV","work_id":"77561c51-3435-4dee-8734-513e83e4bcdc","year":2020},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.494306Z"},"links":{"cited_paper":"/paper/2001.05614","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:f22770fe6aa18d375c44d71dc5add8a1f92f5916ed952583208deefce059ba95","observation_id":"013a8cef-d9a6-40d7-a267-57d1516bd0d8","resolution":{"observed_at":"2026-08-12T12:10:17.658495Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.917113Z","title":"Improving video captioning with temporal composition of a visual-syntactic embedding,","venue":null,"work_id":"8dccefd4-2d40-498b-8b2a-b905cc28a9cf","year":2021},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.499050Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:fb40c8ecfdcf2fdd8d0b3778b0d0f9db28f12670d1c5edbaf7d283efde0e42f5","observation_id":"3301e15f-319a-4d25-8899-1ec6326b9f3c","resolution":{"observed_at":"2026-08-12T12:10:17.921709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.11100","last_updated":"2022-07-27T21:52:21Z","snapshot_observed_at":"2026-07-06T13:34:13.533549Z","submitted_at":"2022-07-22T14:19:31Z","title":"Zero-Shot Video Captioning with Evolving Pseudo-Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.11100","snapshot_observed_at":"2026-08-12T12:10:17.503202Z","title":"Zero- shot video captioning with evolving pseudo-tokens,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.503202Z"},"links":{"cited_paper":"/paper/2207.11100","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:d40d75633bf73f95c3cb38b8376adef054a279dca1fb0b76a811ebb8574b7a69","observation_id":"d3429b60-cc3d-4cf0-a7cf-308a47018340","resolution":{"observed_at":"2026-08-12T12:10:17.503202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.902660Z","title":"MultiCapCLIP: Auto-encoding prompts for zero-shot multilingual visual captioning,","venue":null,"work_id":"ff74b2d5-a9b3-49dc-bed8-467122077e4e","year":2023},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.507689Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:9fce2bed20dc020e997e35459bcacf4d6846138451be404ee6a87bdccc0a9c1e","observation_id":"ce135537-b377-42d9-81ff-aa36e85e9a18","resolution":{"observed_at":"2026-08-12T12:10:17.907312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.887937Z","title":"Visual instruction tuning,","venue":null,"work_id":"9dd2fc88-f55c-487b-9322-550defa0a69f","year":2023},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.511806Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:62c5d25dfd2c5c4f86734c540c4d461554e76567f329b2716b26b7fc672de4be","observation_id":"785c6cf8-21b3-4bff-8e1a-04dec522b27b","resolution":{"observed_at":"2026-08-12T12:10:17.892387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03051","last_updated":"2025-04-09T06:24:14Z","snapshot_observed_at":"2026-08-12T22:31:09.310411Z","submitted_at":"2024-10-04T00:13:54Z","title":"AuroraCap: Efficient, Performant Video Detailed Captioning and a New Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03051","snapshot_observed_at":"2026-08-12T12:10:17.515974Z","title":"Auroracap: Efficient, perfor- mant video detailed captioning and a new benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.515974Z"},"links":{"cited_paper":"/paper/2410.03051","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:916e20b3a4fa13967649be6aafd625d701431dca6f4bf4633d174daee521ba51","observation_id":"ec9bfde8-4dd7-4aca-8880-33a90c8adda8","resolution":{"observed_at":"2026-08-12T12:10:17.515974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.520437Z","title":"Msr-vtt: A large video description dataset for bridging video and language,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.520437Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:9f4b0d5f5b3837e155292f94e2941cbb64f55aef343f8ed01b38d20ee67fd828","observation_id":"e2fbcb5a-fe08-4b75-8113-53e73c86bfd5","resolution":{"observed_at":"2026-08-12T12:10:17.520437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.525100Z","title":"Collecting highly parallel data for paraphrase evaluation,","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.525100Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:c2627680f6b48f04b8141fc386f7bcf4f358ad5bdc6ca7dcd8d332dad563e24d","observation_id":"a5c7b881-8e63-4578-b5d8-2f4ea125bd26","resolution":{"observed_at":"2026-08-12T12:10:17.525100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.529439Z","title":"Vatex: A large-scale, high-quality multilingual dataset for video-and-language research,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.529439Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:5bb256c95f9419a8ba11fd14ef8a11518a73e56dfa621dc2215085d88128f455","observation_id":"6b88c278-12d7-4e40-8221-9daaf3a9ee04","resolution":{"observed_at":"2026-08-12T12:10:17.529439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.846681Z","title":"Bleu: a method for automatic evaluation of machine translation,","venue":null,"work_id":"4d21a743-2889-498c-8995-37b855b490b1","year":2002},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.534038Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:0b151bc52f219ceb105c0c29a7b7fb4259e5fc1efabcb2b4b1d4c344c12516c1","observation_id":"1673f31e-d513-4bdb-a491-338327be8df2","resolution":{"observed_at":"2026-08-12T12:10:17.851757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.831892Z","title":"Meteor universal: Language specific translation evaluation for any target language,","venue":null,"work_id":"b3896ec5-e42e-4261-931f-0c759971fc6c","year":2014},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.538541Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:7679c367049d23d64cff20a03c1a17a6c4587bc4d541036a9335b167e9a48377","observation_id":"34951392-678c-4a7e-b98b-3cd0a53ea77f","resolution":{"observed_at":"2026-08-12T12:10:17.836508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.542918Z","title":"Rouge: A package for automatic evaluation of summaries,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.542918Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:4dc844b41908c178b91bf199fb099d462b80707726849a49eedb51449b1d9430","observation_id":"3db6d68b-92f4-4f6b-99bb-bc988827387c","resolution":{"observed_at":"2026-08-12T12:10:17.542918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.808867Z","title":"Cider: Consensus- based image description evaluation,","venue":null,"work_id":"8881e8c5-4ae6-41f0-b07f-fd3f120a22dd","year":2015},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.547305Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:8cc37a468f2adba43e69ca18b0b9cb273addd0ec25c070c02c3b07328e3a9d16","observation_id":"242319e9-d6db-45e9-86f0-0eb8885f8a5f","resolution":{"observed_at":"2026-08-12T12:10:17.813288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-12T12:10:17.551480Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.551480Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:8b607949d88fb607efb91b0639186777f7211e524603ff05cda3a060ce226e15","observation_id":"ee924bf3-d658-4431-97e2-b98db66504c4","resolution":{"observed_at":"2026-08-12T12:10:17.551480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:10:17.794732Z","title":"Expanding language-image pretrained models for general video recognition,","venue":null,"work_id":"f995cff4-7c3c-4c93-a009-b1710f9cc6ab","year":2022},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.555961Z"},"links":{"citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:b6cba3a76d68bce5ff4708d3f24917797df14cba58a93084f274a58960683c77","observation_id":"11e8d915-f8fd-4515-ba2d-437559911d98","resolution":{"observed_at":"2026-08-12T12:10:17.799454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20314","last_updated":"2025-04-19T02:22:42Z","snapshot_observed_at":"2026-08-12T22:34:51.361078Z","submitted_at":"2025-03-26T08:25:43Z","title":"Wan: Open and Advanced Large-Scale Video Generative Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20314","snapshot_observed_at":"2026-08-12T12:10:17.560221Z","title":"Wan: Open and advanced large-scale video generative models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T12:10:17.560221Z"},"links":{"cited_paper":"/paper/2503.20314","citing_paper":"/paper/2608.11013"},"observation_digest":"sha256:98b901fb937d690bbffb8626663868add8db283089ff1cec0220503006673a7b","observation_id":"25108cc2-59b5-4d44-8fa3-526165b86e40","resolution":{"observed_at":"2026-08-12T12:10:17.560221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.11013","last_updated":"2026-08-11T14:57:44Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T22:26:37.856148Z","submitted_at":"2026-08-11T14:57:44Z","title":"Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":2,"verified_fuzzy":23},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2608.11013."}