{"as_of":"2026-08-04T20:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e7fddc92ff4cb298101d6a7701907c2c5d0f5cee3d627622ad6f77f36c3b51ad","coverage":[{"denominator":21,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-24T00:22:35.635679Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T19:37:41.721882Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-23T07:25:28.511257Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"cited_work":{"arxiv_id":"2406.05615","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05615","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","venue":"cs.CL","work_id":"804f2592-11f5-4591-9ff4-98ed5baf9acb","year":2024},"citing_paper":{"arxiv_id":"2412.07160","last_updated":"2026-04-26T01:15:25Z","snapshot_observed_at":"2026-08-02T08:51:24.009214Z","submitted_at":"2024-12-10T03:41:07Z","title":"Motion-aware Contrastive Learning for Temporal Panoptic Scene Graph Generation","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-23T07:23:51.435139Z"},"links":{"cited_paper":"/paper/2406.05615","citing_paper":"/paper/2412.07160"},"observation_digest":"sha256:b5dc0dd9e60830b511731a007f08a3cb82854bb89cfc100c96b66c610479f8ee","observation_id":"76956ff7-933b-42b2-a40a-ee85b8d72a96","resolution":{"observed_at":"2026-05-23T07:25:28.513235Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05615","snapshot_observed_at":"2026-08-04T19:37:41.721882Z","title":"Video-language understanding: A survey from model architecture, model training, and data perspectives,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09151","last_updated":"2026-06-08T07:16:30Z","snapshot_observed_at":"2026-08-04T19:37:40.169661Z","submitted_at":"2025-09-11T05:06:30Z","title":"Video Understanding by Design: How Datasets Shape Video Models","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-04T19:37:41.721882Z"},"links":{"cited_paper":"/paper/2406.05615","citing_paper":"/paper/2509.09151"},"observation_digest":"sha256:daa8d2dae737da5a21cfc4b14c3747acab43a5c9c7ad32c2f853ced45a310db7","observation_id":"acd93843-1245-4273-9ddf-fc8d3bb2bc92","resolution":{"observed_at":"2026-08-04T19:37:41.721882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05615","snapshot_observed_at":"2026-07-14T09:11:59.440912Z","title":"Available: https://arxiv.org/abs/2406.05615","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10797","last_updated":"2026-07-12T15:09:33Z","snapshot_observed_at":"2026-08-02T11:07:19.413264Z","submitted_at":"2026-07-12T15:09:33Z","title":"Compositional Context Fine-Tuning Vision-Language Model for Complex Assembly Action Understanding from Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-14T09:11:59.440912Z"},"links":{"cited_paper":"/paper/2406.05615","citing_paper":"/paper/2607.10797"},"observation_digest":"sha256:4e45cee388c39d79d15af17cb657ac1e9e1f5855e0a9a5444dfe399b5b1efc79","observation_id":"0e4e6dec-ef78-4322-abe8-37b149c03e8d","resolution":{"observed_at":"2026-07-14T09:11:59.440912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.05615/citation-record","integrity":"/paper/2406.05615/integrity","json":"/paper/2406.05615/citation-record.json","paper":"/paper/2406.05615"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2205.08508","last_updated":"2022-05-17T17:26:23Z","snapshot_observed_at":"2026-07-06T13:10:52.823147Z","submitted_at":"2022-05-17T17:26:23Z","title":"A CLIP-Hitchhiker's Guide to Long Video Retrieval","version":1},"cited_work":{"arxiv_id":"2205.08508","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2205.08508","snapshot_observed_at":"2026-07-04T06:19:37.851711Z","title":"A CLIP-Hitchhiker’s Guide to Long Video Retrieval","venue":null,"work_id":"fc7b88db-0f37-4471-931b-e39676b12213","year":2022},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2205.08508","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:c39b7474f6ed3834a6a39e7a39d2f4a5e7cac218436b862a664b21f927e80ab3","observation_id":"976b9646-8b96-49e8-925e-a3f348526e55","resolution":{"observed_at":"2026-05-24T00:23:39.603943Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.01340","last_updated":"2018-08-03T20:17:05Z","snapshot_observed_at":"2026-07-06T06:54:00.483796Z","submitted_at":"2018-08-03T20:17:05Z","title":"A Short Note about Kinetics-600","version":1},"cited_work":{"arxiv_id":"1808.01340","doi":null,"metadata_source":"pith","pith_arxiv_id":"1808.01340","snapshot_observed_at":"2026-07-09T11:16:11.465544Z","title":"A Short Note about Kinetics-600","venue":"cs.CV","work_id":"851b1623-6feb-441e-8849-b07f1753f22e","year":2018},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/1808.01340","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:9afcd2296ec64323f3b863bd1c9bb5b0bdd06c61000b5494d30c0f53759b63bb","observation_id":"123c8b5e-ffc0-47de-aec6-9e0e9d898a3a","resolution":{"observed_at":"2026-05-24T00:23:39.597715Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":"1810.04805","doi":"10.1111/jofi.12885","metadata_source":"pith","pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-07-12T02:22:28.232339Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"cs.CL","work_id":"ed240a10-5b19-406c-baa5-30803f465785","year":2018},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:8eb95edd8a36798a8b442eedbcba734ab14aed20cbdb2ea0ef51bef0d816f4a2","observation_id":"0302d9cb-47c8-42b9-b052-6c0b97b4e89c","resolution":{"observed_at":"2026-05-24T00:23:39.659765Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2011.11760","last_updated":"2020-11-10T21:49:14Z","snapshot_observed_at":"2026-07-06T10:17:10.235209Z","submitted_at":"2020-11-10T21:49:14Z","title":"Multimodal Pretraining for Dense Video Captioning","version":1},"cited_work":{"arxiv_id":"2011.11760","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2011.11760","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Proceedings of the IEEE/CVF Conference on Computer Vision and Pat- tern Recognition, pages 16783–16792","venue":null,"work_id":"03359a82-c215-4fa0-a397-fef8dede8820","year":2020},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2011.11760","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:ad04816521b780843d63bb3a9857352faa1b7ed561a1d4a7e603847a05978501","observation_id":"1c8d84ce-ab50-46cd-a837-aef0d9ea6995","resolution":{"observed_at":"2026-05-24T00:23:39.584385Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.06950","last_updated":"2017-04-14T19:20:10Z","snapshot_observed_at":"2026-07-06T05:23:35.178736Z","submitted_at":"2016-12-21T02:29:53Z","title":"Temporal Tessellation: A Unified Approach for Video Analysis","version":2},"cited_work":{"arxiv_id":"1612.06950","doi":null,"metadata_source":"pith","pith_arxiv_id":"1612.06950","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Temporal Tessellation: A Unified Approach for Video Analysis","venue":"cs.CV","work_id":"c0d95105-b33c-4cd9-872a-07db9c6818e2","year":2016},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/1612.06950","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:7911803b044b8cc04cfbaf7a2a49b78a89826e06bd9209eca6a7a7b944ffc003","observation_id":"9058bebf-c827-460b-af80-512bfe559eb3","resolution":{"observed_at":"2026-05-24T00:23:39.629257Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Proceedings of the 2018 Con- ference on Empirical Methods in Natural Language Processing, pages 1369–1379","venue":null,"work_id":"707d39ce-0a76-475c-bbe6-684948ec36be","year":2018},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:dd04135cf082076a3cbd6d10e422a0d9b1ac585a7fedf63887408bd07e9c2577","observation_id":"1fde12c8-86c0-49aa-87e6-857c917fffef","resolution":{"observed_at":"2026-05-24T00:23:40.854609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:b779805e88573da9f2b0eb5f2e4f633f7345e8814335ce2867c1beb6518618c7","observation_id":"bd69b2c4-4d4e-44ed-91b8-551ffb3f7747","resolution":{"observed_at":"2026-05-24T00:23:39.617174Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.13230","last_updated":"2021-06-24T17:59:46Z","snapshot_observed_at":"2026-07-06T11:22:38.453675Z","submitted_at":"2021-06-24T17:59:46Z","title":"Video Swin Transformer","version":1},"cited_work":{"arxiv_id":"2106.13230","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2106.13230","snapshot_observed_at":"2026-07-03T19:08:49.804199Z","title":"In European Conference on Computer Vision, pages 413–430","venue":null,"work_id":"4eab6a90-c0e0-4827-9c27-5744f88b5eac","year":2021},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2106.13230","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:35af72e6f8ca61253a980246151cb40706e1a974bbe90e3728d7bbaa7cfae12d","observation_id":"f61f62ae-f6e4-414c-9216-2a8e3757797d","resolution":{"observed_at":"2026-05-24T00:23:39.590980Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17486","last_updated":"2024-03-26T08:32:39Z","snapshot_observed_at":"2026-07-06T17:50:50.473384Z","submitted_at":"2024-03-26T08:32:39Z","title":"KDMCSE: Knowledge Distillation Multimodal Sentence Embeddings with Adaptive Angular margin Contrastive Learning","version":1},"cited_work":{"arxiv_id":"2403.17486","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17486","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Proceedings of the IEEE/CVF international conference on computer vision, pages 2630–2640","venue":null,"work_id":"7538371f-fb95-42ad-b3e9-5bb212408116","year":2019},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2403.17486","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:e81647645992ff7d9db91d58744b7ce5974d1fd0c1724a7d14286141e7aa1c5e","observation_id":"babc48cd-1c10-4681-aff7-5fdd4ed290fd","resolution":{"observed_at":"2026-05-24T00:23:39.635473Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Pro- ceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pages 18983–18992","venue":null,"work_id":"09fe8ede-19d4-4626-ad22-007ac7908ca6","year":2019},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:b8e1a1dbc60e5a82b632b98c8a159f68bb43bcddce742dcade6b6472b785ab99","observation_id":"8ed671ab-8510-49f7-84d3-584a2422b938","resolution":{"observed_at":"2026-05-24T00:23:40.859286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Proceedings of the 29th ACM International Conference on Multimedia, pages 2871– 2879","venue":null,"work_id":"c6ef863e-d209-41e6-87db-6a107589a3d5","year":null},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:cc651fe0c8e51cc9b8a5b3eb6d8e2214814740747a056084da429c0d17dc77d7","observation_id":"a6a1c026-ba9f-49ab-a19a-b13543fa9066","resolution":{"observed_at":"2026-05-24T00:23:40.880575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1811.00347","last_updated":"2018-12-07T07:03:52Z","snapshot_observed_at":"2026-08-04T13:33:48.224677Z","submitted_at":"2018-11-01T12:47:11Z","title":"How2: A Large-scale Dataset for Multimodal Language Understanding","version":2},"cited_work":{"arxiv_id":"1811.00347","doi":null,"metadata_source":"pith","pith_arxiv_id":"1811.00347","snapshot_observed_at":"2026-07-04T16:49:57.270995Z","title":"How2: A Large-scale Dataset for Multimodal Language Understanding","venue":"cs.CL","work_id":"62673cc1-e85d-4c0f-93e5-b4b329842f9f","year":2018},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/1811.00347","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:b7fb30a5305b0102dd048ea800582dcdfaf5af01c64f36e9ba434ccea667f6af","observation_id":"bc340d84-edd0-475c-a159-daf7183c25e8","resolution":{"observed_at":"2026-05-24T00:23:39.623096Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2312.17432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T19:16:01.292273Z","title":"In Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pages 1207–1216","venue":null,"work_id":"ed4332e5-301f-4911-a5d9-cb08ea33992a","year":2023},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:797b6606104b9a91430a40c5fd5d94e97869b96841aaf475d9872989415c3b15","observation_id":"8141bcfc-d6f8-429f-a758-d68c80b39b17","resolution":{"observed_at":"2026-05-24T00:23:39.648338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.08252","last_updated":"2021-05-18T03:21:37Z","snapshot_observed_at":"2026-07-06T11:10:20.716791Z","submitted_at":"2021-05-18T03:21:37Z","title":"Weakly Supervised Dense Video Captioning via Jointly Usage of Knowledge Distillation and Cross-modal Matching","version":1},"cited_work":{"arxiv_id":"2105.08252","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2105.08252","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In 2017 IEEE In- ternational Conference on Image Processing (ICIP), pages 4197–4201","venue":null,"work_id":"5131a18c-6785-402a-bcad-335f7344f4a7","year":2017},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2105.08252","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:93b98a7deab05113a36132c03053f10c1599befc05f00280fa1bc5f463bc2c7c","observation_id":"e7f0a07b-598f-4cc0-a715-39259373732c","resolution":{"observed_at":"2026-05-24T00:23:39.641818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In International Conference on Machine Learning, pages 3891–3900","venue":null,"work_id":"9f1a344c-ebd4-4f6d-985e-e61f78b3bf42","year":2015},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:2188a8783283fba56ea1b19ebc4394f6a083d2a9cd374c8eef17db49ae7a0055","observation_id":"3a21679c-6baf-4c27-8972-71413f860338","resolution":{"observed_at":"2026-05-24T00:23:40.884387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03166","last_updated":"2024-10-24T22:35:27Z","snapshot_observed_at":"2026-07-06T15:51:12.179086Z","submitted_at":"2023-07-06T17:47:52Z","title":"VideoGLUE: Video General Understanding Evaluation of Foundation Models","version":3},"cited_work":{"arxiv_id":"2307.03166","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2307.03166","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2307.03166","venue":null,"work_id":"a82fe8c7-3bc3-45d6-a94d-7f353ce36320","year":2019},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2307.03166","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:9536fa08f515216ece3526b76e8c317db79aa4b6c9863f6fe926ce7e949c2d76","observation_id":"d0697f62-7cba-45cb-8c4a-b15b5003c3ca","resolution":{"observed_at":"2026-05-24T00:23:39.609764Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Proceedings of the IEEE/CVF international conference on com- puter vision, pages 6023–6032","venue":null,"work_id":"060e1ea7-11f5-40a7-9794-e84ab61f96d3","year":null},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:22e8d89c95f5a13cd46594e37453bb6719775ad01c22502cdf9f3fadca306b34","observation_id":"09739cbe-1798-4c2e-95e3-13223114382e","resolution":{"observed_at":"2026-05-24T00:23:40.868250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems, 34:23634–23651","venue":null,"work_id":"5b85370f-be95-4753-b0ab-5c4ad9b8fd44","year":null},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:e238596db7fb6f6e5db756a9135c003aa63a743ed8fa2e664d6943ff1ca8764a","observation_id":"a23216ec-1ea4-4366-aa10-9b5c3ca7fbb5","resolution":{"observed_at":"2026-05-24T00:23:40.872743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.07868","last_updated":"2024-04-11T06:21:29Z","snapshot_observed_at":"2026-08-02T01:58:00.490115Z","submitted_at":"2023-01-19T03:42:56Z","title":"MV-Adapter: Multimodal Video Transfer Learning for Video Text Retrieval","version":2},"cited_work":{"arxiv_id":"2301.07868","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2301.07868","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Thirty-First AAAI Conference on Artificial Intelligence","venue":null,"work_id":"d8837d3e-917a-49a9-8ba2-80481e523e50","year":2023},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2301.07868","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:48be0aa2bbcfcc97e30c997d19e0006734370aca2af4e7a786a9d50c9f0cbc0b","observation_id":"a348eece-2361-4d7f-a2e4-20785c3b50c4","resolution":{"observed_at":"2026-05-24T00:23:39.654123Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"set up the stand ✔ 2","venue":null,"work_id":"7a961d3f-d746-4780-9370-9dedef373354","year":null},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:6e9c6f39d42f9dc951198039dadfb00b4cc87b37083189a13f7428685ae0f03b","observation_id":"16710594-e48d-4b97-b214-637010526ebf","resolution":{"observed_at":"2026-05-24T00:23:40.863410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video moment retrieval 38s 48s 60s 64s Q: People in scuba gear are swimming around","venue":null,"work_id":"ec040a86-0e49-4ceb-ad78-06c466f05a87","year":2021},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:89cc91f41d964cb5ac6f4c9b734b8d91fc2104c2fd5814fce964330b4ff444d7","observation_id":"ddf9681e-4bdf-4e58-86b2-ad5f53b17c5b","resolution":{"observed_at":"2026-05-24T00:23:40.876556Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives"},"reference_resolution":{"displayed":21,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":0,"verified_exact":10,"verified_fuzzy":7},"total_outbound_references":21},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 21 of 21 outbound references and 3 inbound Pith citation observations for arXiv:2406.05615."}