{"as_of":"2026-08-06T22:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:db6afe775de87ee3588bf041840049c0754115446941a4942d75cf14f96550c2","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":32,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:08:56.989600Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.380967Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2203.12601","last_updated":"2022-11-18T05:57:09Z","snapshot_observed_at":"2026-08-02T11:48:59.131827Z","submitted_at":"2022-03-23T17:55:09Z","title":"R3M: A Universal Visual Representation for Robot Manipulation","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-15T13:26:53.843613Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2203.12601"},"observation_digest":"sha256:3936323e96dbf71e97aeb6b46bc875c8a572d52d27f3c1ebadc70536e5033d64","observation_id":"fe34e4e4-4f16-43b4-b379-342a81ff67a2","resolution":{"observed_at":"2026-05-15T13:26:53.979248Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2204.00598","last_updated":"2022-05-27T17:52:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-01T17:43:13Z","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-16T09:50:00.546571Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2204.00598"},"observation_digest":"sha256:721d7f8e3a46428ce5df03f26bd9a531b7c0bbc67e56182d694b40d318ba4d2c","observation_id":"72761ef6-7d17-4890-8c76-6cabcc30f73d","resolution":{"observed_at":"2026-05-16T09:50:00.608579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2212.03191","last_updated":"2022-12-07T12:20:55Z","snapshot_observed_at":"2026-07-06T14:27:34.639236Z","submitted_at":"2022-12-06T18:09:49Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T00:36:53.235740Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2212.03191"},"observation_digest":"sha256:43171c38731dde0087122811969c69d80157d51af4cad7478ee33b4583221ec2","observation_id":"1eaf6c1f-7f3f-4dbd-8fe1-793a40eb840d","resolution":{"observed_at":"2026-05-17T00:36:53.372393Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:648fc95388711b1f2631add66f94da7eb9f0f237b1099d46eef7b6c176fc2377","observation_id":"b9b1d4cd-aa91-48eb-b583-b569e1e5952e","resolution":{"observed_at":"2026-05-13T23:30:00.675840Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:0ab5c2057f2ab68c0c567c3d2567bc289a8403a6e49c2918e215faf6107dfb79","observation_id":"eed293b8-bc7f-4447-a1a7-672174934fd9","resolution":{"observed_at":"2026-05-15T06:30:22.554543Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2310.01852","last_updated":"2024-01-22T03:11:15Z","snapshot_observed_at":"2026-08-02T22:12:20.187464Z","submitted_at":"2023-10-03T07:33:27Z","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","version":7},"reference_index":177,"source":"arxiv_source","source_observed_at":"2026-05-17T03:27:58.952076Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2310.01852"},"observation_digest":"sha256:162fe9e860e925835eba4ab9a209625a6aa7614fcb798c76bd77c4935edef797","observation_id":"f4791b68-3222-428b-9824-0255dc411a5c","resolution":{"observed_at":"2026-05-17T03:27:59.117146Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2404.08471","last_updated":"2024-02-15T18:59:11Z","snapshot_observed_at":"2026-08-02T05:40:31.086014Z","submitted_at":"2024-02-15T18:59:11Z","title":"Revisiting Feature Prediction for Learning Visual Representations from Video","version":1},"reference_index":292,"source":"arxiv_source","source_observed_at":"2026-05-12T12:40:23.709098Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2404.08471"},"observation_digest":"sha256:19fe201fbb5367926cbd93ca7ce956413c3d729de7dbef756e0879c2c6416821","observation_id":"9d4cba46-2e95-48a7-8e42-e8a936162640","resolution":{"observed_at":"2026-05-12T12:40:24.082290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2503.13821","last_updated":"2026-04-03T20:02:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-18T01:57:48Z","title":"Stitch-a-Demo: Video Demonstrations from Multistep Descriptions","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-23T00:30:55.729900Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2503.13821"},"observation_digest":"sha256:377ba64426ac4e5a93b2a04b375e38f80a0fd8bb0a8175acb26854254ffdfaa9","observation_id":"56f2d5c3-e0e1-47d8-a3ce-3dc8cb32a4d8","resolution":{"observed_at":"2026-05-23T00:32:18.250901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2507.04590","last_updated":"2025-07-07T00:51:57Z","snapshot_observed_at":"2026-08-02T08:05:44.477432Z","submitted_at":"2025-07-07T00:51:57Z","title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T14:10:14.929207Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.04590"},"observation_digest":"sha256:cc4ec23663ed43cac7ff6edcada2dc610b3ba5abdc42bd575b7cf4576db910db","observation_id":"61529d6c-d42b-45a7-b6ba-b13560387e97","resolution":{"observed_at":"2026-05-18T14:10:15.136859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T17:08:56.989600Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11892","last_updated":"2025-07-16T04:15:06Z","snapshot_observed_at":"2026-08-06T16:56:43.650358Z","submitted_at":"2025-07-16T04:15:06Z","title":"From Coarse to Nuanced: Cross-Modal Alignment of Fine-Grained Linguistic Cues and Visual Salient Regions for Dynamic Emotion Recognition","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:08:56.989600Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.11892"},"observation_digest":"sha256:f23e7f9c542b9f655382cace220694c76a51e3911f2a7977237dea9cc743cf7e","observation_id":"0ff0bbb6-eb8a-41d8-ba58-e1eb879b040d","resolution":{"observed_at":"2026-08-06T17:08:56.989600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T15:07:31.643197Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16716","last_updated":"2025-07-22T15:54:53Z","snapshot_observed_at":"2026-08-06T15:01:04.578746Z","submitted_at":"2025-07-22T15:54:53Z","title":"Enhancing Remote Sensing Vision-Language Models Through MLLM and LLM-Based High-Quality Image-Text Dataset Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T15:07:31.643197Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.16716"},"observation_digest":"sha256:7ab96f322357660d287da3cc03358bc8e51eb7461551cfe7212d7a82879ac547","observation_id":"4b2d807e-fe8f-405d-9432-e9dd4bc960fb","resolution":{"observed_at":"2026-08-06T15:07:31.643197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T13:24:30.193705Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.20740","last_updated":"2025-07-28T11:46:35Z","snapshot_observed_at":"2026-08-06T13:24:19.529284Z","submitted_at":"2025-07-28T11:46:35Z","title":"Implicit Counterfactual Learning for Audio-Visual Segmentation","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T13:24:30.193705Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.20740"},"observation_digest":"sha256:70da2cf0f87433fce341599466e0438e4ef29f7bb3a2013d8f701ecfbfc314b1","observation_id":"2f93e96f-c19c-41b4-8489-c070a09d9865","resolution":{"observed_at":"2026-08-06T13:24:30.193705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T12:55:40.226378Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21353","last_updated":"2025-07-28T21:46:05Z","snapshot_observed_at":"2026-08-06T12:55:36.123689Z","submitted_at":"2025-07-28T21:46:05Z","title":"Group Relative Augmentation for Data Efficient Action Detection","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T12:55:40.226378Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.21353"},"observation_digest":"sha256:fe79893dd719b4eaa290cac748b8e8b2edf616e2c118d0e4d6f6127892d312ee","observation_id":"90e63cc7-bf95-4ea5-aa75-fa1289e5b836","resolution":{"observed_at":"2026-08-06T12:55:40.226378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-06T11:46:45.517124Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.22431","last_updated":"2025-07-30T07:21:36Z","snapshot_observed_at":"2026-08-06T11:46:43.177001Z","submitted_at":"2025-07-30T07:21:36Z","title":"HQ-CLIP: Leveraging Large Vision-Language Models to Create High-Quality Image-Text Datasets and CLIP Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T11:46:45.517124Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2507.22431"},"observation_digest":"sha256:d7fe6719f23a633269013a68f1bb3b8789b45282187c2676b5391fcc5b389cb2","observation_id":"2335abde-285a-4cac-8447-b1a2c9cbfa25","resolution":{"observed_at":"2026-08-06T11:46:45.517124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2508.06964","last_updated":"2026-04-12T08:38:14Z","snapshot_observed_at":"2026-08-06T08:43:14.125473Z","submitted_at":"2025-08-09T12:20:13Z","title":"Adversarial Video Promotion Against Text-to-Video Retrieval","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-19T00:05:07.182361Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2508.06964"},"observation_digest":"sha256:23491e5b4b4fdd7a953034c83bb027e859b5239fae29701c81b5bc303f145c33","observation_id":"330c954d-ded8-446f-a47b-22ddfd931cff","resolution":{"observed_at":"2026-05-19T00:06:55.166881Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-05T18:50:58.248635Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.14039","last_updated":"2025-08-19T17:59:39Z","snapshot_observed_at":"2026-08-05T18:50:55.444502Z","submitted_at":"2025-08-19T17:59:39Z","title":"Beyond Simple Edits: Composed Video Retrieval with Dense Modifications","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T18:50:58.248635Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2508.14039"},"observation_digest":"sha256:ac0d9b984201467ba110adef85357ae6a6821c2897943c102d620e24f53834b0","observation_id":"d6b50866-ab14-445b-a0f7-1b0e75e07e7a","resolution":{"observed_at":"2026-08-05T18:50:58.248635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-08-04T19:37:42.187507Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.09151","last_updated":"2026-06-08T07:16:30Z","snapshot_observed_at":"2026-08-06T12:45:51.887065Z","submitted_at":"2025-09-11T05:06:30Z","title":"Video Understanding by Design: How Datasets Shape Video Models","version":2},"reference_index":224,"source":"pdf_text","source_observed_at":"2026-08-04T19:37:42.187507Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2509.09151"},"observation_digest":"sha256:d94ec914457e09268714fb4bad0b238f9279ce3bbe9204e0021855acc025ab1d","observation_id":"85192c1e-362f-4e51-b9a8-049c42e00291","resolution":{"observed_at":"2026-08-04T19:37:42.187507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2511.12034","last_updated":"2026-05-12T07:28:48Z","snapshot_observed_at":"2026-08-02T17:34:14.031307Z","submitted_at":"2025-11-15T05:01:43Z","title":"Calibrated Multimodal Representation Learning with Missing Modalities","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T22:08:07.217659Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2511.12034"},"observation_digest":"sha256:79cdc841f0d08e44922b2990930452a258ebec9da69a3dfd7253312cb146d29a","observation_id":"9761cf3d-85dc-4d2b-94c8-8d97a3f83eff","resolution":{"observed_at":"2026-05-17T22:10:22.692713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2512.03963","last_updated":"2026-04-14T11:28:58Z","snapshot_observed_at":"2026-07-31T23:59:38.983651Z","submitted_at":"2025-12-03T16:57:00Z","title":"TempR1: Improving Temporal Understanding of MLLMs via Temporal-Aware Multi-Task Reinforcement Learning","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-17T02:18:21.718091Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2512.03963"},"observation_digest":"sha256:17180dd1f74966cc5b694b30567255c30c0367a78670275257bde61036ddf69c","observation_id":"a30d6193-0b77-4a33-8a28-fdcf02fa13f9","resolution":{"observed_at":"2026-05-17T02:18:52.335434Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2512.13511","last_updated":"2026-05-01T11:23:38Z","snapshot_observed_at":"2026-07-06T22:39:04.164003Z","submitted_at":"2025-12-15T16:38:59Z","title":"Adapting MLLMs for Nuanced Video Retrieval","version":3},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-16T22:20:09.051957Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2512.13511"},"observation_digest":"sha256:cdfef71811b8c67c4ba96bebb634b71352eb1d29cfbc56dd011973c0a1e15293","observation_id":"51e029eb-ce1e-4431-9dbe-a30d6a95526c","resolution":{"observed_at":"2026-05-16T22:21:18.921271Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2602.02513","last_updated":"2026-05-19T07:10:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-23T06:39:01Z","title":"Learning ORDER-Aware Multimodal Representations for Composite Materials Design","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T14:36:44.656655Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2602.02513"},"observation_digest":"sha256:84193d1cc91a99603e23b9a11d7cd69e07713e83bcc62bae0daf439a60cf8c9a","observation_id":"9d82ce30-3dc5-4cb9-8455-d148e1611781","resolution":{"observed_at":"2026-05-21T14:40:14.550097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2603.09921","last_updated":"2026-07-02T16:25:34Z","snapshot_observed_at":"2026-07-14T23:55:23.688065Z","submitted_at":"2026-03-10T17:18:53Z","title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-15T13:11:54.384284Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2603.09921"},"observation_digest":"sha256:44c9e0ad844385639281e7311261896a136e95dbfb9f63b2742c289c562c1af6","observation_id":"98a3f7aa-d2a1-49c0-a7f6-9169612aa447","resolution":{"observed_at":"2026-05-15T13:15:50.570008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-14T23:55:24.006436Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding.arXiv preprint arXiv:2109.14084, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.09921","last_updated":"2026-07-02T16:25:34Z","snapshot_observed_at":"2026-07-14T23:55:23.688065Z","submitted_at":"2026-03-10T17:18:53Z","title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-14T23:55:24.006436Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2603.09921"},"observation_digest":"sha256:04b856800fdf9bb4be0613dc2ee273b85da01662795959a3190d6013c2230da4","observation_id":"e5989aae-07c8-4169-a9bf-75f9fcec2bff","resolution":{"observed_at":"2026-07-14T23:55:24.006436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-13T21:37:55.887477Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding.arXiv preprint arXiv:2109.14084, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.20190","last_updated":"2026-06-09T22:16:34Z","snapshot_observed_at":"2026-08-04T20:13:17.293244Z","submitted_at":"2026-03-20T17:59:25Z","title":"CoVR-R:Reason-Aware Composed Video Retrieval","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-13T21:37:55.887477Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2603.20190"},"observation_digest":"sha256:b881d942c65cf59ce114dcecf317b09e9d3538dc352f1cdd4545102ae19cfd48","observation_id":"79bc04fd-414f-4150-a5c8-710534ff86cb","resolution":{"observed_at":"2026-07-13T21:37:55.887477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2604.04875","last_updated":"2026-04-06T17:26:04Z","snapshot_observed_at":"2026-07-06T22:53:46.999911Z","submitted_at":"2026-04-06T17:26:04Z","title":"DIRECT: Video Mashup Creation via Hierarchical Multi-Agent Planning and Intent-Guided Editing","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T19:44:52.768021Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.04875"},"observation_digest":"sha256:8734e615004ff29add4e825f9fcb9597f7596f8df8f5cbec17cd3b6e5d19570c","observation_id":"c3f96828-e72e-4f38-953b-6c85b0fefb02","resolution":{"observed_at":"2026-05-10T22:30:53.943204Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2604.08337","last_updated":"2026-04-09T15:10:25Z","snapshot_observed_at":"2026-08-03T10:29:21.842229Z","submitted_at":"2026-04-09T15:10:25Z","title":"InstAP: Instance-Aware Vision-Language Pre-Train for Spatial-Temporal Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T18:06:33.139310Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.08337"},"observation_digest":"sha256:2849d6c111916edcc3837fd3a940a407868b09f6d12daacbf19d4f3f6fa2cdcf","observation_id":"a286e855-f8a7-4198-b58a-1c9456c2bb14","resolution":{"observed_at":"2026-05-11T05:30:58.326399Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2604.14684","last_updated":"2026-05-26T13:10:47Z","snapshot_observed_at":"2026-07-12T20:05:14.844082Z","submitted_at":"2026-04-16T06:40:44Z","title":"DETR-ViP: Detection Transformer with Robust Discriminative Visual Prompts","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T12:07:21.203513Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.14684"},"observation_digest":"sha256:b51e1f5a3c73c13eb6fa9d5a88f79175c26e89a0b43bf9ada5ca35f1eafcd550","observation_id":"bf843ac4-63bf-4441-91ca-748db06036a3","resolution":{"observed_at":"2026-05-10T12:10:21.931008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-12T20:05:20.702092Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding.arXiv preprint arXiv:2109.14084,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.14684","last_updated":"2026-05-26T13:10:47Z","snapshot_observed_at":"2026-07-12T20:05:14.844082Z","submitted_at":"2026-04-16T06:40:44Z","title":"DETR-ViP: Detection Transformer with Robust Discriminative Visual Prompts","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-12T20:05:20.702092Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2604.14684"},"observation_digest":"sha256:b465b76f03d2a07eb952b530fa2b8f287cc7db25e8eb8728ad3daf56650f3bcb","observation_id":"ee162317-38b2-4cf0-8ca7-b03ee279b6d7","resolution":{"observed_at":"2026-07-12T20:05:20.702092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2605.00051","last_updated":"2026-04-29T15:29:19Z","snapshot_observed_at":"2026-08-02T05:45:46.015951Z","submitted_at":"2026-04-29T15:29:19Z","title":"Learning from the Unseen: Generative Data Augmentation for Geometric-Semantic Accident Anticipation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-09T20:04:39.623342Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2605.00051"},"observation_digest":"sha256:f2f461036719f4128eee1edf81b026d2fbd923e3f3f2687fbb201a9fe7947fc1","observation_id":"e05b8581-ffea-4cf1-8fd9-ae8478f83ced","resolution":{"observed_at":"2026-05-11T15:26:08.280521Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2605.21059","last_updated":"2026-05-20T11:44:01Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T11:44:01Z","title":"Multimodal LLMs under Pairwise Modalities","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-21T05:37:56.792564Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2605.21059"},"observation_digest":"sha256:2a655a3d90de563c96a174cc98047f5ecbe29c86d69a8594d2acb73994b8d9e5","observation_id":"6d4afa95-e025-4962-a74a-fc809fa40d2b","resolution":{"observed_at":"2026-05-21T05:39:40.686881Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2606.02273","last_updated":"2026-06-01T13:59:46Z","snapshot_observed_at":"2026-07-06T23:42:44.661608Z","submitted_at":"2026-06-01T13:59:46Z","title":"Vision-language Models for Driver Monitoring Systems: A Driver Activity Description Dataset","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T15:29:51.424572Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2606.02273"},"observation_digest":"sha256:29bec56238f9a488f8532018d2295a9c135a0fe187f0e906c95adf98e73166e1","observation_id":"b15f4643-7da8-4008-bbbf-485042c8b020","resolution":{"observed_at":"2026-07-01T22:16:16.925691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:d2d4a371a8150329efe0c85011e97d847f5b59b4ab1c424aef15b41df48fcddf","observation_id":"8acc79c7-1e61-4a3c-85fb-d94956ca9fc6","resolution":{"observed_at":"2026-07-04T06:39:37.382403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2109.14084/citation-record","integrity":"/paper/2109.14084/integrity","json":"/paper/2109.14084/citation-record.json","paper":"/paper/2109.14084"},"outbound":[],"paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 32 inbound Pith citation observations for arXiv:2109.14084."}