{"as_of":"2026-08-11T12:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6eb39b656dcd9ea8740b08d2d1882576d11e386713f0d5dd7c2a06fc1e26e314","coverage":[{"denominator":190,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T23:20:32.330351Z","state":"measured"},{"denominator":200,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":200,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":286,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T11:24:00.590366Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T21:36:34.348434Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2412.04468","last_updated":"2026-04-25T07:16:42Z","snapshot_observed_at":"2026-08-03T08:48:57.969106Z","submitted_at":"2024-12-05T18:59:55Z","title":"NVILA: Efficient Frontier Visual Language Models","version":3},"reference_index":147,"source":"pdf_text","source_observed_at":"2026-05-23T07:42:22.478647Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2412.04468"},"observation_digest":"sha256:2368e3e8769e93285d30a6cb11530e1f74624af260166085cc30d35e8297106a","observation_id":"5a3218de-1b04-4090-986d-e0b103dd39fa","resolution":{"observed_at":"2026-05-23T07:42:43.192285Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-10T08:02:13.965614Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-05-22T09:27:43.919941Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2412.14171"},"observation_digest":"sha256:c1326628450705ac5e48e184073d16315ecfc69a7db54b561975e10ce67d96cf","observation_id":"1e36f211-c965-4223-a32c-3496cfddc48c","resolution":{"observed_at":"2026-05-22T09:27:44.083800Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-11T11:24:00.590366Z","title":"Video instruction tuning with synthetic data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.15509","last_updated":"2025-01-25T02:06:55Z","snapshot_observed_at":"2026-08-11T11:20:02.450317Z","submitted_at":"2024-12-20T02:45:37Z","title":"PolySmart @ TRECVid 2024 Video Captioning (VTT)","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:00.590366Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2412.15509"},"observation_digest":"sha256:fd5577995406a3c5f5a1cd2fc0c692ea53c2bfbadc98b6484a12a928e3d9b917","observation_id":"ebb78dfb-1ae9-4a2a-a40d-bfa9bd502f71","resolution":{"observed_at":"2026-08-11T11:24:00.590366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-11T10:33:19.406036Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16524","last_updated":"2024-12-21T08:01:08Z","snapshot_observed_at":"2026-08-11T10:27:40.470375Z","submitted_at":"2024-12-21T08:01:08Z","title":"LLaVA-SLT: Visual Language Tuning for Sign Language Translation","version":1},"reference_index":113,"source":"pdf_text","source_observed_at":"2026-08-11T10:33:19.406036Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2412.16524"},"observation_digest":"sha256:b51bc28243f468c627bed515d82e3e5a018de234fa5b87098b29bbd774e453fa","observation_id":"19db2b63-5f14-419c-bca5-07c4340fefef","resolution":{"observed_at":"2026-08-11T10:33:19.406036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-11T05:29:51.267504Z","title":"Video instruction tuning with synthetic data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.17415","last_updated":"2025-04-07T11:20:37Z","snapshot_observed_at":"2026-08-11T05:24:51.743510Z","submitted_at":"2024-12-23T09:26:38Z","title":"VidCtx: Context-aware Video Question Answering with Image Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T05:29:51.267504Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2412.17415"},"observation_digest":"sha256:3cf392d2c1b72641f30e2b51167dfd23df144047a0da0529c3f4fd7322f1be22","observation_id":"976db303-747f-4c80-bb8d-17d02eaadee7","resolution":{"observed_at":"2026-08-11T05:29:51.267504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-18T04:02:43.261543Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.00574"},"observation_digest":"sha256:60b8b0e440557bfa53a7b05a9b9953e2953e5eda6010170b44c345b81bdd884b","observation_id":"84022ade-c828-472f-a4f0-b9028aa05dfa","resolution":{"observed_at":"2026-05-18T04:02:43.632932Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-10T22:27:56.504052Z","title":"Video instruction tuning with synthetic data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01645","last_updated":"2025-05-13T06:38:44Z","snapshot_observed_at":"2026-08-10T23:44:47.423609Z","submitted_at":"2025-01-03T05:32:37Z","title":"HLV-1K: A Large-scale Hour-Long Video Benchmark for Time-Specific Long Video Understanding","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:56.504052Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.01645"},"observation_digest":"sha256:d67857a665fd24266273926bea82f5cd671f6731a40f5e8fc022772fb5d65afc","observation_id":"10a626aa-374f-43e8-b0d1-20146cd3af0c","resolution":{"observed_at":"2026-08-10T22:27:56.504052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-10T23:09:25.171264Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01986","last_updated":"2025-07-24T18:44:26Z","snapshot_observed_at":"2026-08-10T23:44:53.833723Z","submitted_at":"2024-12-30T17:31:37Z","title":"FrameFusion: Combining Similarity and Importance for Video Token Reduction on Large Vision Language Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:25.171264Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.01986"},"observation_digest":"sha256:8fb7de3a5e4c0d996e588a4be9513b98ec85afb06c232e92a3ef0f97eca64165","observation_id":"3c991fea-0792-4d41-a4f4-08a52fb229d2","resolution":{"observed_at":"2026-08-10T23:09:25.171264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:82f853364ec707da7623ab1fea8116f8ac49fba56c9e2d21bf7427e5ed6137d7","observation_id":"2daef667-331a-46dd-8bfb-59fbce9c1109","resolution":{"observed_at":"2026-05-23T05:45:28.399033Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-10T21:23:58.063067Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05037","last_updated":"2025-03-27T09:39:11Z","snapshot_observed_at":"2026-08-10T21:17:52.245219Z","submitted_at":"2025-01-09T07:51:14Z","title":"LongViTU: Instruction Tuning for Long-Form Video Understanding","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T21:23:58.063067Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.05037"},"observation_digest":"sha256:e1941156e63d7cba5dcc9f19c4894713ecd13a45e0a7bd00145e8bc5c6f95fad","observation_id":"b5267eb6-df57-4449-8bc5-3b6215c980dc","resolution":{"observed_at":"2026-08-10T21:23:58.063067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-10T17:18:41.330419Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.12368","last_updated":"2025-05-20T11:36:34Z","snapshot_observed_at":"2026-08-10T21:22:00.940261Z","submitted_at":"2025-01-21T18:47:32Z","title":"InternLM-XComposer2.5-Reward: A Simple Yet Effective Multi-Modal Reward Model","version":2},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-08-10T17:18:41.330419Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.12368"},"observation_digest":"sha256:59be667f1b2ef16697fb1d2e7c9b059bc1577b61bc5def7923e65e46bf4c91fe","observation_id":"e86efaa9-45dc-49a9-bece-0435d3fd93e1","resolution":{"observed_at":"2026-08-10T17:18:41.330419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:2d2df5fa4d40c5cc1d9a27dd2a48646373824f51a08a92271fc27e1402f516cb","observation_id":"ccf57c0c-21c5-4ba8-9e87-41ac7bf780c8","resolution":{"observed_at":"2026-05-11T01:19:59.773347Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2501.13826","last_updated":"2025-01-23T16:51:47Z","snapshot_observed_at":"2026-07-06T20:25:03.950783Z","submitted_at":"2025-01-23T16:51:47Z","title":"Video-MMMU: Evaluating Knowledge Acquisition from Multi-Discipline Professional Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-14T00:32:41.059558Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.13826"},"observation_digest":"sha256:a718626e6544fc04dde0cf13f562a8e05065174a80455a32e4f532813a3f104e","observation_id":"0e5a4491-57c8-48bc-8d32-d175db17ac2d","resolution":{"observed_at":"2026-05-14T00:32:41.197198Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-10T15:35:30.317413Z","title":"Video instruction tuning with synthetic data, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13919","last_updated":"2025-09-01T07:50:58Z","snapshot_observed_at":"2026-08-11T01:32:58.320917Z","submitted_at":"2025-01-23T18:58:03Z","title":"Temporal Preference Optimization for Long-Form Video Understanding","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T15:35:30.317413Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.13919"},"observation_digest":"sha256:bb6b0b956a3c21b4ff4a33fc67dd9eb1e4fb987f72c7d1ff3c67b6755c8bb34b","observation_id":"10ba334a-bb4c-4f00-b439-b1f38cf09b18","resolution":{"observed_at":"2026-08-10T15:35:30.317413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-10T14:18:18.086229Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.15513","last_updated":"2025-06-10T14:30:19Z","snapshot_observed_at":"2026-08-10T14:10:29.461568Z","submitted_at":"2025-01-26T13:10:12Z","title":"TinyLLaVA-Video: Towards Smaller LMMs for Video Understanding with Group Resampler","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T14:18:18.086229Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2501.15513"},"observation_digest":"sha256:e9395724776b4d5c639f800b7e77038c0e1bc9ede1937e6960e5efebc3df379e","observation_id":"7ce12156-d8cc-4ea6-ba88-ee60aab2a8c3","resolution":{"observed_at":"2026-08-10T14:18:18.086229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-09T15:04:36.856590Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01549","last_updated":"2025-02-03T17:30:19Z","snapshot_observed_at":"2026-08-09T16:01:36.816495Z","submitted_at":"2025-02-03T17:30:19Z","title":"VideoRAG: Retrieval-Augmented Generation with Extreme Long-Context Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-09T15:04:36.856590Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2502.01549"},"observation_digest":"sha256:4d48bc6270fdbe93ae394dd415388e14d50457b2ccba7034820625d22f7feadd","observation_id":"c0aa3591-a22b-459d-80b9-f05cf52d2255","resolution":{"observed_at":"2026-08-09T15:04:36.856590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-08T23:26:20.596294Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04144","last_updated":"2025-03-25T04:54:54Z","snapshot_observed_at":"2026-08-11T09:39:14.956670Z","submitted_at":"2025-02-06T15:25:05Z","title":"HD-EPIC: A Highly-Detailed Egocentric Video Dataset","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-08T23:26:20.596294Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2502.04144"},"observation_digest":"sha256:708c04b2453aa416bc275d24c6b8c58edee946008c20101f825b9bbffb31e28b","observation_id":"a1d6fdb8-f008-4e9b-bf41-016f8e7b30e9","resolution":{"observed_at":"2026-08-08T23:26:20.596294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2502.04326","last_updated":"2026-03-01T04:35:41Z","snapshot_observed_at":"2026-07-06T20:32:23.502480Z","submitted_at":"2025-02-06T18:59:40Z","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","version":3},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-17T05:53:26.066674Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2502.04326"},"observation_digest":"sha256:fc977e53cd8c67e234560096ee4a926e518c2e7b44a3b6101c4125e2137ca847","observation_id":"4a483b44-5e09-4120-b675-e950839ff857","resolution":{"observed_at":"2026-05-17T05:53:26.245939Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T18:23:50.551232Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.10391","last_updated":"2025-02-14T18:59:51Z","snapshot_observed_at":"2026-08-08T01:24:46.892879Z","submitted_at":"2025-02-14T18:59:51Z","title":"MM-RLHF: The Next Step Forward in Multimodal LLM Alignment","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T18:23:50.551232Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2502.10391"},"observation_digest":"sha256:670648a9dfa605e25622ba25618ff1dacc2c8e64aeaefc75187ca23453478752","observation_id":"73da6a09-7f01-4c25-b393-89a6b3c6cfe2","resolution":{"observed_at":"2026-08-07T18:23:50.551232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2503.05236","last_updated":"2026-02-25T13:17:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-07T08:36:05Z","title":"Unified Reward Model for Multimodal Understanding and Generation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-14T00:44:30.558048Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2503.05236"},"observation_digest":"sha256:aad5a8fa0aff7e7b1e0374cec7bbba7c061ecc64acd0cf9f0daf7f3c035cd8ea","observation_id":"07ed3bcc-cb45-4f90-b7e0-6f2b455b14b6","resolution":{"observed_at":"2026-05-14T00:44:30.925258Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2503.21755","last_updated":"2025-08-20T15:49:30Z","snapshot_observed_at":"2026-08-10T02:47:20.223653Z","submitted_at":"2025-03-27T17:57:01Z","title":"VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-14T18:42:02.940250Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2503.21755"},"observation_digest":"sha256:3b56dd03664e47014bf7e00d1937d5314ec06b58259ee95fb0508d51e03e90ef","observation_id":"28953e7a-4738-41e5-8848-730399223951","resolution":{"observed_at":"2026-05-14T18:42:03.151308Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2504.05299","last_updated":"2025-04-07T17:58:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T17:58:57Z","title":"SmolVLM: Redefining small and efficient multimodal models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T20:23:50.552549Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2504.05299"},"observation_digest":"sha256:673ecd049048723a0074cc8dce7dd82f5b57ec24fa37e52e2801a01d62d605e2","observation_id":"b57506b3-c9fb-473e-94f1-db33721ee035","resolution":{"observed_at":"2026-05-13T20:23:51.750015Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-15T20:56:07.247122Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2504.06958"},"observation_digest":"sha256:19ebebd2e2c96ad2665d51dd67287122aceba284c0616c95f314724e87f4c3f4","observation_id":"3caa9af6-8bcb-4af9-a24a-5ae85ccc00ba","resolution":{"observed_at":"2026-05-15T20:56:07.764632Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T15:34:41.495843Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-10T08:26:12.719387Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.495843Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:fdd686f6886052d51e018e51cce0372d7403270691cbc9294c80fb9736c44ca4","observation_id":"aece33d9-7349-46da-b737-66bdd2dbf49b","resolution":{"observed_at":"2026-08-07T15:34:41.495843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T15:34:55.910000Z","title":"Video instruction tuning with synthetic data.arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14682","last_updated":"2025-05-20T17:59:26Z","snapshot_observed_at":"2026-08-07T20:35:28.075030Z","submitted_at":"2025-05-20T17:59:26Z","title":"UniGen: Enhanced Training & Test-Time Strategies for Unified Multimodal Understanding and Generation","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:55.910000Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.14682"},"observation_digest":"sha256:07ad14e872adb2a558efda9724eee8b165bfa6f4d6301f68d1c0ce5cf8a7a699","observation_id":"37f6e468-d31e-4e85-a9f6-ac9bc1835124","resolution":{"observed_at":"2026-08-07T15:34:55.910000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T15:20:59.633738Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15447","last_updated":"2025-05-21T12:29:40Z","snapshot_observed_at":"2026-08-11T01:33:01.559626Z","submitted_at":"2025-05-21T12:29:40Z","title":"ViaRL: Adaptive Temporal Grounding via Visual Iterated Amplification Reinforcement Learning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:20:59.633738Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.15447"},"observation_digest":"sha256:0b15eeb9a1fdc4b13752c2f1820393739f6785d003c2e5b81578cda22c509b42","observation_id":"75ba5317-b5ea-4f21-b12a-517e7916add4","resolution":{"observed_at":"2026-08-07T15:20:59.633738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T15:20:48.883282Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15529","last_updated":"2025-05-21T13:52:17Z","snapshot_observed_at":"2026-08-09T21:52:59.287019Z","submitted_at":"2025-05-21T13:52:17Z","title":"Clapper: Compact Learning and Video Representation in VLMs","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T15:20:48.883282Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.15529"},"observation_digest":"sha256:3bc746d001aeff83079380209d3d34abeaec2266fc8895f462438bd69c9515a8","observation_id":"8eacf7cc-13ab-4b30-9181-67af9725b7e1","resolution":{"observed_at":"2026-08-07T15:20:48.883282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2505.16933","last_updated":"2025-06-04T05:52:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-22T17:23:26Z","title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T03:46:06.074416Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.16933"},"observation_digest":"sha256:353af7ee63853d862d980abd53a10593d28e3d5ace3c3c15cdfc138529b74361","observation_id":"331ed5d7-2084-415c-bfeb-bc831b564724","resolution":{"observed_at":"2026-05-17T03:46:06.403500Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T14:27:45.034743Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18855","last_updated":"2025-05-24T20:09:04Z","snapshot_observed_at":"2026-08-09T21:52:56.657385Z","submitted_at":"2025-05-24T20:09:04Z","title":"Inference Compute-Optimal Video Vision Language Models","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T14:27:45.034743Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.18855"},"observation_digest":"sha256:b0b8360f5ec2169a1ab4981e847ffb62d1c7f59a5d13b6e954831e872c74c2d7","observation_id":"08d2a24f-71fa-421f-afa5-f205f726ef01","resolution":{"observed_at":"2026-08-07T14:27:45.034743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T14:14:47.117724Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19650","last_updated":"2025-05-27T11:57:17Z","snapshot_observed_at":"2026-08-09T09:30:25.697403Z","submitted_at":"2025-05-26T08:09:44Z","title":"Modality Curation: Building Universal Embeddings for Advanced Multimodal Information Retrieval","version":2},"reference_index":112,"source":"pdf_text","source_observed_at":"2026-08-07T14:14:47.117724Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.19650"},"observation_digest":"sha256:a844266eab3632c250a2496b2f602fad854bdc414040a8054a1ffb2d0653322e","observation_id":"1acf10f4-d309-4cd2-9127-00201219f46b","resolution":{"observed_at":"2026-08-07T14:14:47.117724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T14:09:16.030499Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19877","last_updated":"2025-05-26T12:05:16Z","snapshot_observed_at":"2026-08-10T09:10:24.017088Z","submitted_at":"2025-05-26T12:05:16Z","title":"Vad-R1: Towards Video Anomaly Reasoning via Perception-to-Cognition Chain-of-Thought","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T14:09:16.030499Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.19877"},"observation_digest":"sha256:eae7402a4d46fecb39878842630f7eb03e05e930130a4ab8c473d5a31fb2bd22","observation_id":"98a25c67-145d-45dc-be12-5c0ffaeba027","resolution":{"observed_at":"2026-08-07T14:09:16.030499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T14:08:10.697724Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20100","last_updated":"2025-05-26T15:08:37Z","snapshot_observed_at":"2026-08-10T18:54:11.396770Z","submitted_at":"2025-05-26T15:08:37Z","title":"AdaTP: Attention-Debiased Token Pruning for Video Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T14:08:10.697724Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.20100"},"observation_digest":"sha256:b8ec8e7500818768755870a41dd625ff836286748c5799785aaa58793db6ce4b","observation_id":"f3a496cd-ad53-4c95-84c7-41cd139ded15","resolution":{"observed_at":"2026-08-07T14:08:10.697724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T14:03:05.687068Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20124","last_updated":"2025-05-27T12:10:27Z","snapshot_observed_at":"2026-08-08T09:10:32.146078Z","submitted_at":"2025-05-26T15:24:06Z","title":"TUNA: Comprehensive Fine-grained Temporal Understanding Evaluation on Dense Dynamic Videos","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-07T14:03:05.687068Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.20124"},"observation_digest":"sha256:695537dc5f0f33044565f953e1865736cd5c27d249dda5bcc474fa6f82cd5d55","observation_id":"b1a20b4b-0c71-4321-a18f-87146e3c31f0","resolution":{"observed_at":"2026-08-07T14:03:05.687068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2505.20279","last_updated":"2026-04-21T02:48:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-26T17:56:30Z","title":"VLM-3R: Vision-Language Models Augmented with Instruction-Aligned 3D Reconstruction","version":5},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-19T12:54:01.013242Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.20279"},"observation_digest":"sha256:44207b9d6c8d729239482503d9cf3e31c7a612e5c632d76965aed712d1f6acce","observation_id":"9f3976c4-30c8-426f-a89b-b2cd0d54fcb2","resolution":{"observed_at":"2026-05-19T12:57:17.940393Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T13:46:16.849356Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21079","last_updated":"2025-05-27T12:03:30Z","snapshot_observed_at":"2026-08-10T03:26:05.211596Z","submitted_at":"2025-05-27T12:03:30Z","title":"Uni3D-MoE: Scalable Multimodal 3D Scene Understanding via Mixture of Experts","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T13:46:16.849356Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.21079"},"observation_digest":"sha256:0655cd6b990498968f206981c601a6ea00032689adefeba48fd340bd59ca3022","observation_id":"123279b9-48f2-448e-98d8-9f2e0c7aed8f","resolution":{"observed_at":"2026-08-07T13:46:16.849356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2505.21374","last_updated":"2025-05-27T16:05:01Z","snapshot_observed_at":"2026-08-06T04:32:21.352745Z","submitted_at":"2025-05-27T16:05:01Z","title":"Video-Holmes: Can MLLM Think Like Holmes for Complex Video Reasoning?","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T05:40:55.944288Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.21374"},"observation_digest":"sha256:f672cf311f9bd92e96031151b14de7100804a242c50d6fdff80cee4e28d4b243","observation_id":"62ee6a8e-3681-43ed-8861-7810bfeeed6b","resolution":{"observed_at":"2026-05-17T05:40:56.015162Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T13:10:46.144299Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22457","last_updated":"2025-05-28T15:13:34Z","snapshot_observed_at":"2026-08-09T21:52:56.063992Z","submitted_at":"2025-05-28T15:13:34Z","title":"Fostering Video Reasoning via Next-Event Prediction","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T13:10:46.144299Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.22457"},"observation_digest":"sha256:cb9b7c68df78e0f031944885f23b44a345e0598b3e765f804ee7f8323a9bc210","observation_id":"5065b583-6453-4089-8ddf-b0e1f36b6f9d","resolution":{"observed_at":"2026-08-07T13:10:46.144299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T13:09:26.216112Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22566","last_updated":"2025-05-28T16:43:01Z","snapshot_observed_at":"2026-08-10T20:28:52.024760Z","submitted_at":"2025-05-28T16:43:01Z","title":"Universal Visuo-Tactile Video Understanding for Embodied Interaction","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T13:09:26.216112Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.22566"},"observation_digest":"sha256:884064b90fafce5d813d97fb09a0bd3c364b8d03a4aadd3e748624e678f6fdef","observation_id":"120b31ac-4054-4861-882b-b452c3ddd294","resolution":{"observed_at":"2026-08-07T13:09:26.216112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T12:48:03.349403Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23484","last_updated":"2025-05-29T14:34:25Z","snapshot_observed_at":"2026-08-09T05:32:01.732887Z","submitted_at":"2025-05-29T14:34:25Z","title":"VCapsBench: A Large-scale Fine-grained Benchmark for Video Caption Quality Evaluation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:03.349403Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.23484"},"observation_digest":"sha256:05cf36a183b162c641c50cc8d2186eaf3646d16c3df578af751361c05533bbe5","observation_id":"604cc1a4-c2bd-4554-9788-ffd5f7ee5bf6","resolution":{"observed_at":"2026-08-07T12:48:03.349403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2505.23617","last_updated":"2026-05-10T04:23:42Z","snapshot_observed_at":"2026-08-10T22:38:07.341464Z","submitted_at":"2025-05-29T16:25:35Z","title":"One Trajectory, One Token: Grounded Video Tokenization via Panoptic Sub-object Trajectory","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-19T12:54:31.765909Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.23617"},"observation_digest":"sha256:371cfdb06a33e58e08c0df9ea5322d2eb622c7bcab4c1e438094d092bac6dde7","observation_id":"d0c641c0-38e4-4aeb-a236-fcdb37168342","resolution":{"observed_at":"2026-05-19T12:57:17.881781Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T08:34:36.824053Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:d379cbd7280f1c70cd139874433b752a3c9139146415c91311ae4a24895da6f8","observation_id":"7b318f2c-c5ba-4ee7-b673-9e291a49eed9","resolution":{"observed_at":"2026-05-16T08:34:36.995325Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T00:59:13.826054Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:f62de40aaee95229a9ed50dfb629d9f199462d70ad63b0916079f97bb4c6d05b","observation_id":"0051c6d8-b97b-495d-8d5f-7257e36b7dad","resolution":{"observed_at":"2026-05-22T01:00:51.308224Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T12:37:24.258273Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.258273Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ab38168341b90a98a9fc072aeff418578827bcb17cf3621ffee92eec755bdeb8","observation_id":"b30ebedc-b77a-4463-b3a6-ef8c7a773f6d","resolution":{"observed_at":"2026-08-07T12:37:24.258273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T11:58:08.604138Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00928","last_updated":"2025-06-01T09:45:41Z","snapshot_observed_at":"2026-08-10T22:33:09.846402Z","submitted_at":"2025-06-01T09:45:41Z","title":"Deep Temporal Reasoning in Video Language Models: A Cross-Linguistic Evaluation of Action Duration and Completion through Perfect Times","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T11:58:08.604138Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.00928"},"observation_digest":"sha256:c392c5abdccfab8eaaaa2fedb7e6e69c31da28a5d682941094893c8df8f37afc","observation_id":"5762f209-5a99-4b11-9c1c-55cd14564bbe","resolution":{"observed_at":"2026-08-07T11:58:08.604138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T11:59:11.136490Z","title":"Video instruction tuning with synthetic data, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00993","last_updated":"2025-06-01T12:49:39Z","snapshot_observed_at":"2026-08-08T18:44:33.509277Z","submitted_at":"2025-06-01T12:49:39Z","title":"FlexSelect: Flexible Token Selection for Efficient Long Video Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.136490Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.00993"},"observation_digest":"sha256:bdd10d53518fb4c240c9aef01a47ae99828d79144f8f3dc9a41c0243362164a3","observation_id":"7f6d8b62-bb23-49b7-b3fc-4b9f9fd99e74","resolution":{"observed_at":"2026-08-07T11:59:11.136490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T11:51:29.980105Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01274","last_updated":"2026-06-11T17:06:50Z","snapshot_observed_at":"2026-08-10T04:03:56.011290Z","submitted_at":"2025-06-02T03:08:07Z","title":"ReFoCUS: Reinforcement-guided Frame Optimization for Contextual Understanding","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:29.980105Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.01274"},"observation_digest":"sha256:d8184e038d4beff4622ee05b5930c59545cef625fe69c389945e84c3a57c2f37","observation_id":"9410b52a-a52a-4521-9fcd-8ae683fdb1c7","resolution":{"observed_at":"2026-08-07T11:51:29.980105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T11:52:05.475245Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01300","last_updated":"2025-06-02T04:23:21Z","snapshot_observed_at":"2026-08-10T13:34:27.916880Z","submitted_at":"2025-06-02T04:23:21Z","title":"ReAgent-V: A Reward-Driven Multi-Agent Framework for Video Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:05.475245Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.01300"},"observation_digest":"sha256:8340d0eb6723526d03958539c5f49c841a0b399a19934020ba7fc7ae26182fb7","observation_id":"759fd701-e7b1-4f1a-86f2-762e8626f470","resolution":{"observed_at":"2026-08-07T11:52:05.475245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T11:40:57.687621Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01725","last_updated":"2025-06-02T14:30:09Z","snapshot_observed_at":"2026-08-08T07:45:08.111238Z","submitted_at":"2025-06-02T14:30:09Z","title":"VideoCap-R1: Enhancing MLLMs for Video Captioning via Structured Thinking","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:57.687621Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.01725"},"observation_digest":"sha256:7a44ba16756fae67ece561d70d5e6325ec83ae373d5a19b7e1b186d19c38748d","observation_id":"19ab9372-e3a4-457d-8ffd-1ef64bb7114f","resolution":{"observed_at":"2026-08-07T11:40:57.687621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T11:36:40.962864Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01872","last_updated":"2025-06-02T17:01:40Z","snapshot_observed_at":"2026-08-10T02:46:37.041378Z","submitted_at":"2025-06-02T17:01:40Z","title":"Is Extending Modality The Right Path Towards Omni-Modality?","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T11:36:40.962864Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.01872"},"observation_digest":"sha256:bcf8ca8a531eaa8c67ba441bbf1ce5d239a6a037bcfddd7acd1fc58b6d39362a","observation_id":"3078cda2-ac9b-406d-9bb7-d6b3f3c12bea","resolution":{"observed_at":"2026-08-07T11:36:40.962864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T10:56:05.418612Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03990","last_updated":"2025-06-04T14:17:42Z","snapshot_observed_at":"2026-08-09T05:23:11.832180Z","submitted_at":"2025-06-04T14:17:42Z","title":"DynTok: Dynamic Compression of Visual Tokens for Efficient and Effective Video Understanding","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T10:56:05.418612Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.03990"},"observation_digest":"sha256:025797d4ba00592c0d2b7288fba61d87d31c5613166755c241c7ee4b088e623d","observation_id":"500a1ece-fc59-4481-a16d-988122648280","resolution":{"observed_at":"2026-08-07T10:56:05.418612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T10:28:46.132778Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05260","last_updated":"2025-06-05T17:21:16Z","snapshot_observed_at":"2026-08-09T11:16:28.299508Z","submitted_at":"2025-06-05T17:21:16Z","title":"LeanPO: Lean Preference Optimization for Likelihood Alignment in Video-LLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:46.132778Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.05260"},"observation_digest":"sha256:11d08b845f4fd8c9ce10d24b9499b99f1462f31146dcbd74d2c441f44cb7ec8c","observation_id":"e09aaaa5-b745-416b-afaa-963f8e4c80b2","resolution":{"observed_at":"2026-08-07T10:28:46.132778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T10:29:42.916824Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05287","last_updated":"2025-06-05T17:44:12Z","snapshot_observed_at":"2026-08-09T15:14:29.691315Z","submitted_at":"2025-06-05T17:44:12Z","title":"EOC-Bench: Can MLLMs Identify, Recall, and Forecast Objects in an Egocentric World?","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T10:29:42.916824Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.05287"},"observation_digest":"sha256:7664261ebd8d282ff61165a0ffa6f5c605998780a55aa2b3648c844facacf2f8","observation_id":"02d568b3-97ea-4d33-bb01-cfcee5cc0e46","resolution":{"observed_at":"2026-08-07T10:29:42.916824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T10:25:36.347575Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05349","last_updated":"2025-06-24T07:36:56Z","snapshot_observed_at":"2026-08-07T23:48:12.277994Z","submitted_at":"2025-06-05T17:59:58Z","title":"VideoMathQA: Benchmarking Mathematical Reasoning via Multimodal Understanding in Videos","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:25:36.347575Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.05349"},"observation_digest":"sha256:79871e180e60bc472ed43faf4b94873f9e7d22be4c2bb31fc0d7dfdf5754c9d0","observation_id":"c26d1035-6bcc-4792-800a-cc0fad81dce9","resolution":{"observed_at":"2026-08-07T10:25:36.347575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2506.05425","last_updated":"2026-04-28T02:01:09Z","snapshot_observed_at":"2026-07-31T07:37:26.215945Z","submitted_at":"2025-06-05T05:51:35Z","title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-19T11:36:36.687324Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.05425"},"observation_digest":"sha256:d705d40a6c1fc4ae386eef0f165624436660c0523b447a5cdce65a8c328ccb79","observation_id":"bc43a348-2f3f-474c-a8be-147f491a82d5","resolution":{"observed_at":"2026-05-19T11:37:15.762639Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T10:20:39.039310Z","title":"Video instruction tuning with synthetic data.ArXiv, abs/2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05689","last_updated":"2025-06-06T02:35:26Z","snapshot_observed_at":"2026-08-09T04:48:26.376897Z","submitted_at":"2025-06-06T02:35:26Z","title":"Pts3D-LLM: Studying the Impact of Token Structure for 3D Scene Understanding With Large Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:39.039310Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.05689"},"observation_digest":"sha256:05adf3e43f55e258dcbec552ecc04bc99536d3a51c0d8029004a7fbf8734c1a6","observation_id":"94c33dc7-c577-4677-a974-ecd3400ac393","resolution":{"observed_at":"2026-08-07T10:20:39.039310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T06:00:57.024929Z","title":"Video Instruction Tuning With Synthetic Data.arXiv e-prints, art","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06275","last_updated":"2025-06-06T17:58:36Z","snapshot_observed_at":"2026-08-10T20:35:02.128344Z","submitted_at":"2025-06-06T17:58:36Z","title":"Movie Facts and Fibs (MF$^2$): A Benchmark for Long Movie Understanding","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:57.024929Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.06275"},"observation_digest":"sha256:f1a52876fe56b5b03b3149909553c18df2c63e340c65b73586c61dfba3500cf5","observation_id":"3b014608-1e4e-4c3d-bae1-f2fe67a62cb5","resolution":{"observed_at":"2026-08-07T06:00:57.024929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T05:51:07.923531Z","title":"Video Instruction Tuning With Synthetic Data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06928","last_updated":"2025-06-07T21:32:19Z","snapshot_observed_at":"2026-08-10T12:00:26.470686Z","submitted_at":"2025-06-07T21:32:19Z","title":"How Important are Videos for Training Video LLMs?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T05:51:07.923531Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.06928"},"observation_digest":"sha256:8d41d9ba802d9ef561257bcb744d93dfd74de201b456b77c752107de7d46f731","observation_id":"b6f15c9d-18d7-4e73-9bcc-4a2501bf2bc6","resolution":{"observed_at":"2026-08-07T05:51:07.923531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T05:26:47.315568Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07971","last_updated":"2025-06-09T17:45:18Z","snapshot_observed_at":"2026-08-09T13:06:51.941159Z","submitted_at":"2025-06-09T17:45:18Z","title":"CyberV: Cybernetics for Test-time Scaling in Video Understanding","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T05:26:47.315568Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.07971"},"observation_digest":"sha256:af11fcf1e7a8b029c479d4f0b5ddaec4c0381c86a155b3397a817655cbb36a50","observation_id":"0ac43119-00d2-4316-80ae-1ac7c5cda771","resolution":{"observed_at":"2026-08-07T05:26:47.315568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T05:24:30.527178Z","title":"Video instruction tuning with synthetic data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08003","last_updated":"2025-06-09T17:59:42Z","snapshot_observed_at":"2026-08-07T10:34:46.576486Z","submitted_at":"2025-06-09T17:59:42Z","title":"Audio-Sync Video Generation with Multi-Stream Temporal Control","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T05:24:30.527178Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.08003"},"observation_digest":"sha256:92fd442594017caa7bfe5cf1549043b9898e8e7139f6edca5ab72eb4abb20cdc","observation_id":"5a25c88f-a587-46d3-8cd0-b7f0467bacc4","resolution":{"observed_at":"2026-08-07T05:24:30.527178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2506.09965","last_updated":"2025-06-19T03:46:55Z","snapshot_observed_at":"2026-07-31T21:40:49.363128Z","submitted_at":"2025-06-11T17:41:50Z","title":"Reinforcing Spatial Reasoning in Vision-Language Models with Interwoven Thinking and Visual Drawing","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-17T04:58:10.202784Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.09965"},"observation_digest":"sha256:fc6d62152dda2c1359e7a322c69c6f55e1806a155d520dc95e61daff42b5b592","observation_id":"21d0b0a7-8220-479c-b170-bdf75b947399","resolution":{"observed_at":"2026-05-17T04:58:10.283650Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T04:29:39.770200Z","title":"Camera zoom in, Disneyland","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10639","last_updated":"2025-06-12T12:25:37Z","snapshot_observed_at":"2026-08-08T15:16:39.506526Z","submitted_at":"2025-06-12T12:25:37Z","title":"GigaVideo-1: Advancing Video Generation via Automatic Feedback with 4 GPU-Hours Fine-Tuning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T04:29:39.770200Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.10639"},"observation_digest":"sha256:fe334733cfba67d8a08fe26543d0f52f1f34ea4896f3c5ba6378ab51a1ae4576","observation_id":"28c45976-fa18-49ff-ad2e-50198112175c","resolution":{"observed_at":"2026-08-07T04:29:39.770200Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2506.15564","last_updated":"2025-09-22T01:24:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-18T15:39:15Z","title":"Show-o2: Improved Native Unified Multimodal Models","version":3},"reference_index":145,"source":"pdf_text","source_observed_at":"2026-05-12T18:51:15.428692Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.15564"},"observation_digest":"sha256:1cdda58a06ce4573dd63f38ac5f5ba20566122994fc16d164e21e9368dbec6ff","observation_id":"9abb6d21-375d-41a4-a060-60b64bdc1499","resolution":{"observed_at":"2026-05-12T18:51:15.968478Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2506.20670","last_updated":"2025-06-25T17:59:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-25T17:59:42Z","title":"MMSearch-R1: Incentivizing LMMs to Search","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-16T15:27:04.228144Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.20670"},"observation_digest":"sha256:6153496c322db5f89c9f30879b2cea6e73a03f174479e621305a42c3d9183e29","observation_id":"2d883cd6-d552-4a7f-a565-86d5b4dc3436","resolution":{"observed_at":"2026-05-16T15:27:04.354589Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T22:15:05.019965Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22139","last_updated":"2025-07-22T07:42:31Z","snapshot_observed_at":"2026-08-09T21:52:46.816951Z","submitted_at":"2025-06-27T11:30:51Z","title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T22:15:05.019965Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.22139"},"observation_digest":"sha256:ea1883077bd984bb53891728fd6933e22723741b906e113e7662420d5a4a6180","observation_id":"0c307f2e-7cdd-4c08-b992-c4fc22cea03a","resolution":{"observed_at":"2026-08-06T22:15:05.019965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T22:10:22.971564Z","title":"Video instruction tuning with synthetic data.arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22434","last_updated":"2025-06-27T17:59:27Z","snapshot_observed_at":"2026-08-09T16:37:51.559760Z","submitted_at":"2025-06-27T17:59:27Z","title":"MiCo: Multi-image Contrast for Reinforcement Visual Reasoning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T22:10:22.971564Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.22434"},"observation_digest":"sha256:cad4c5be3a4c4677b404c92abc78987bfeda8e0d24b5c55d95fe9f5192a06f5b","observation_id":"d6e43932-032a-497b-a74f-48c42b77f00f","resolution":{"observed_at":"2026-08-06T22:10:22.971564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T21:37:08.967515Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-10T08:26:14.921423Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.967515Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:a64dddaeb8033122c9d13b2002ed66307fddf0ce6dad32f84919705f1fee5728","observation_id":"113072f7-b149-48b4-98a8-8e9f4e1254e5","resolution":{"observed_at":"2026-08-06T21:37:08.967515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T21:12:03.623119Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00817","last_updated":"2025-07-01T14:48:27Z","snapshot_observed_at":"2026-08-09T05:23:11.226777Z","submitted_at":"2025-07-01T14:48:27Z","title":"CAVALRY-V: A Large-Scale Generator Framework for Adversarial Attacks on Video MLLMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T21:12:03.623119Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.00817"},"observation_digest":"sha256:1c28aa39c33c69de2d939f7aa06adeb165026d5af04af4011531deeca5b2f55d","observation_id":"b3f8ab5d-e1e0-4dc3-a967-4662dd2d825b","resolution":{"observed_at":"2026-08-06T21:12:03.623119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T20:53:21.033536Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01492","last_updated":"2025-07-02T08:51:45Z","snapshot_observed_at":"2026-08-11T05:45:50.144023Z","submitted_at":"2025-07-02T08:51:45Z","title":"AVC-DPO: Aligned Video Captioning via Direct Preference Optimization","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:53:21.033536Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.01492"},"observation_digest":"sha256:39335402fb60002af53a126bc59dfc87ed928693a434917143d8cd1786d694b0","observation_id":"9d1e3cba-0d00-4e50-a1e7-d1699e39c8d3","resolution":{"observed_at":"2026-08-06T20:53:21.033536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T19:41:38.650060Z","title":"doi: 10.18653/ v1/2023.emnlp-demo.49","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.04976","last_updated":"2025-07-07T13:19:43Z","snapshot_observed_at":"2026-08-11T01:58:59.750704Z","submitted_at":"2025-07-07T13:19:43Z","title":"Can Video LLMs Refuse to Answer? Alignment for Answerability in Video Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:41:38.650060Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.04976"},"observation_digest":"sha256:c70ae4352ab1b748d37e2387b6d0cfd266382ae2ce33adc097e6c8fd5dbfed29","observation_id":"80aecfca-840e-43ab-83a4-dc31abeba65e","resolution":{"observed_at":"2026-08-06T19:41:38.650060Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T19:37:06.449859Z","title":"Zhang, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05240","last_updated":"2026-07-09T06:53:43Z","snapshot_observed_at":"2026-08-06T19:26:50.907830Z","submitted_at":"2025-07-07T17:49:41Z","title":"StreamVLN: Streaming Vision-and-Language Navigation via SlowFast Context Modeling","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:37:06.449859Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.05240"},"observation_digest":"sha256:66cf5cf4bba001ff20039732ae60f06dea479f0e62e238d99bc8cf806c5e7550","observation_id":"f86124a1-b5ac-4052-a9e4-c56c69400990","resolution":{"observed_at":"2026-08-06T19:37:06.449859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T18:32:50.312891Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07990","last_updated":"2025-07-10T17:59:02Z","snapshot_observed_at":"2026-08-09T00:35:41.482303Z","submitted_at":"2025-07-10T17:59:02Z","title":"Multi-Granular Spatio-Temporal Token Merging for Training-Free Acceleration of Video LLMs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T18:32:50.312891Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.07990"},"observation_digest":"sha256:a01dc08eac5dddfbc1304fd9a1a8a901523762b2958ddc2434f64f272f6758a6","observation_id":"dd7c2631-a2ca-43c3-be08-e90da971380a","resolution":{"observed_at":"2026-08-06T18:32:50.312891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T18:24:54.749240Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08513","last_updated":"2025-07-23T22:34:55Z","snapshot_observed_at":"2026-08-09T16:59:07.724320Z","submitted_at":"2025-07-11T12:00:10Z","title":"Advancing Multimodal LLMs by Large-Scale 3D Visual Instruction Dataset Generation","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T18:24:54.749240Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.08513"},"observation_digest":"sha256:7db1b09ecdc40775213052886cd349311da015e680c6a297f70d4d630d69472f","observation_id":"c59fd0ac-0659-49c7-88c2-f56a3efe1277","resolution":{"observed_at":"2026-08-06T18:24:54.749240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T17:38:19.891685Z","title":"arXiv:2410.02713 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10302","last_updated":"2025-07-14T14:05:19Z","snapshot_observed_at":"2026-08-09T06:56:01.559909Z","submitted_at":"2025-07-14T14:05:19Z","title":"DisCo: Towards Distinct and Coherent Visual Encapsulation in Video MLLMs","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T17:38:19.891685Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.10302"},"observation_digest":"sha256:34c213c5f872f0283c0f166f33295f441d992c7d0d100aefe3de110fbe63b5ba","observation_id":"904869a4-664e-4648-b434-788a107d910f","resolution":{"observed_at":"2026-08-06T17:38:19.891685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T15:38:45.527724Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15428","last_updated":"2025-07-21T09:27:45Z","snapshot_observed_at":"2026-08-08T22:42:41.641573Z","submitted_at":"2025-07-21T09:27:45Z","title":"EgoPrune: Efficient Token Pruning for Egomotion Video Reasoning in Embodied Agent","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T15:38:45.527724Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.15428"},"observation_digest":"sha256:2e8d4969cc1629ba56255944b3f746dacde79c6cfdd1c5e0e9212d8361823ca4","observation_id":"da2a2cfb-ce36-4457-8938-c5ab46742040","resolution":{"observed_at":"2026-08-06T15:38:45.527724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2507.16815","last_updated":"2025-09-18T16:26:53Z","snapshot_observed_at":"2026-08-03T17:16:24.219569Z","submitted_at":"2025-07-22T17:59:46Z","title":"ThinkAct: Vision-Language-Action Reasoning via Reinforced Visual Latent Planning","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-19T03:18:14.655384Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.16815"},"observation_digest":"sha256:b0f613fafae79124305f403e246c886de3cbc43a6a9c23e27972c5c448f314ea","observation_id":"393861fd-ebd2-4293-882b-fc6cb6c5ca62","resolution":{"observed_at":"2026-05-19T03:22:00.935931Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T15:14:17.441895Z","title":"Zhang, Y .; Zhang, R.; Gu, J.; Zhou, Y .; Lipka, N.; Yang, D.; and Sun, T","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16878","last_updated":"2025-07-22T12:29:13Z","snapshot_observed_at":"2026-08-09T16:03:32.935657Z","submitted_at":"2025-07-22T12:29:13Z","title":"CausalStep: A Benchmark for Explicit Stepwise Causal Reasoning in Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:17.441895Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.16878"},"observation_digest":"sha256:0915f3f49e0c8169b3211671192c24c9836d3a05f0c41ec07cfe95f03e39a9a4","observation_id":"8f43ed20-4d5f-404e-a453-983c4777fad8","resolution":{"observed_at":"2026-08-06T15:14:17.441895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T14:19:39.606964Z","title":"Video in- struction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19599","last_updated":"2025-07-25T18:11:23Z","snapshot_observed_at":"2026-08-07T08:25:24.125935Z","submitted_at":"2025-07-25T18:11:23Z","title":"Object-centric Video Question Answering with Visual Grounding and Referring","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T14:19:39.606964Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.19599"},"observation_digest":"sha256:d158ca01fe4b213ab3b585943cd8dc544f0887f56728e79a7fed26577bbfe241","observation_id":"67cb2719-8b81-48f7-adeb-ca73373abd9c","resolution":{"observed_at":"2026-08-06T14:19:39.606964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T14:05:32.511719Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19821","last_updated":"2025-08-02T13:22:34Z","snapshot_observed_at":"2026-08-10T22:09:58.641713Z","submitted_at":"2025-07-26T06:38:07Z","title":"LAVA: Language Driven Scalable and Versatile Traffic Video Analytics","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T14:05:32.511719Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.19821"},"observation_digest":"sha256:67a1fd6cb701286ce094dc819ec2ddd7f6271b61797f97b4a2537c9aefbe7895","observation_id":"542037b8-ea75-435c-8d9c-f7c41a471fe3","resolution":{"observed_at":"2026-08-06T14:05:32.511719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T05:30:37.791480Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01699","last_updated":"2025-08-03T10:03:58Z","snapshot_observed_at":"2026-08-09T05:23:23.804073Z","submitted_at":"2025-08-03T10:03:58Z","title":"TimeExpert: An Expert-Guided Video LLM for Video Temporal Grounding","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T05:30:37.791480Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.01699"},"observation_digest":"sha256:3f116dff50665b0bb43bd67d59d584c442b18950d988848a5d98c61c3bb5a7f5","observation_id":"fc61fe64-0088-4c2f-b95f-349400158007","resolution":{"observed_at":"2026-08-06T05:30:37.791480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T05:20:59.194993Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01984","last_updated":"2025-08-04T01:44:41Z","snapshot_observed_at":"2026-08-08T14:04:11.219976Z","submitted_at":"2025-08-04T01:44:41Z","title":"IMoRe: Implicit Program-Guided Reasoning for Human Motion Q&A","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T05:20:59.194993Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.01984"},"observation_digest":"sha256:83b066ac2723aa2e3f349bb239f833419df7d4dd410b0a5f52ab85f8f0a00c22","observation_id":"8afe1ff4-893c-4b1b-a864-e129ce931309","resolution":{"observed_at":"2026-08-06T05:20:59.194993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-05T21:55:00.759463Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.07863","last_updated":"2025-08-11T11:26:10Z","snapshot_observed_at":"2026-08-06T19:31:41.610911Z","submitted_at":"2025-08-11T11:26:10Z","title":"Being-M0.5: A Real-Time Controllable Vision-Language-Motion Model","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.759463Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.07863"},"observation_digest":"sha256:1a3ca50c719ab86e4bf7c39adbe204bce59381a589450c51d9bf74373caab48e","observation_id":"3ed91865-da65-4d6e-a1db-a804263ec9bb","resolution":{"observed_at":"2026-08-05T21:55:00.759463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2508.10016","last_updated":"2026-05-22T12:46:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-06T16:17:29Z","title":"Training-Free Multimodal Large Language Model Orchestration","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-19T00:12:39.834892Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.10016"},"observation_digest":"sha256:14a7aca154543e8249a95ab7b48b922e9dd3298c0f320cd2bff8cc8e397c3a03","observation_id":"5a81c394-c805-4be3-bdd9-a4156c2916be","resolution":{"observed_at":"2026-05-19T00:12:54.179224Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2508.10016","last_updated":"2026-05-22T12:46:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-06T16:17:29Z","title":"Training-Free Multimodal Large Language Model Orchestration","version":4},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-25T08:02:15.950975Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.10016"},"observation_digest":"sha256:c167e6ff251aa7d9d66453b1e9f3c9825c365e6d90b4703a97f16069a6f2579c","observation_id":"c5df480c-2700-4203-8757-002aa20bbbf0","resolution":{"observed_at":"2026-05-25T08:05:30.772069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-05T20:31:10.273170Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10416","last_updated":"2025-08-14T07:39:26Z","snapshot_observed_at":"2026-08-08T02:28:41.150453Z","submitted_at":"2025-08-14T07:39:26Z","title":"CorrectNav: Self-Correction Flywheel Empowers Vision-Language-Action Navigation Model","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T20:31:10.273170Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.10416"},"observation_digest":"sha256:3cb2147caef49963a8fcd4acfeb38b3a9e7c7e1bdfc2b7cd5802bcba75bf255d","observation_id":"90743981-e64d-4adc-b237-3e7eea28453e","resolution":{"observed_at":"2026-08-05T20:31:10.273170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-05T19:03:03.428316Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.13692","last_updated":"2025-08-19T09:52:04Z","snapshot_observed_at":"2026-08-09T03:24:13.317916Z","submitted_at":"2025-08-19T09:52:04Z","title":"HumanPCR: Probing MLLM Capabilities in Diverse Human-Centric Scenes","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T19:03:03.428316Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.13692"},"observation_digest":"sha256:69d2d38208758a591337b1f333d2264d9fe3c30c1025bdb20356868e2052f7f6","observation_id":"22aecef3-2f8e-4ee9-b934-8a0d42a3adde","resolution":{"observed_at":"2026-08-05T19:03:03.428316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2508.18167","last_updated":"2026-05-14T18:20:00Z","snapshot_observed_at":"2026-08-04T20:09:49.064679Z","submitted_at":"2025-08-25T16:16:42Z","title":"DiscussLLM: Teaching Large Language Models When to Speak","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-21T22:09:03.740109Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.18167"},"observation_digest":"sha256:f9a8a26c17f2fcdb4e12c5e273097edadb71d138b61cb6ab39db4b8bdc94425a","observation_id":"e03a14a9-e0c2-4e9f-8a77-f03e31f886ec","resolution":{"observed_at":"2026-05-21T22:10:42.320593Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-05T15:40:40.536185Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19650","last_updated":"2025-08-29T02:25:23Z","snapshot_observed_at":"2026-08-07T05:11:55.108109Z","submitted_at":"2025-08-27T07:58:16Z","title":"Video-LevelGauge: Investigating Contextual Positional Bias in Large Video Language Models","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T15:40:40.536185Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2508.19650"},"observation_digest":"sha256:bf9217cd354e86555e768b3e56a76b65f5bc295129a052d7402b0d867b9c978a","observation_id":"495fcdb7-43d6-4704-b7bb-7734549ef176","resolution":{"observed_at":"2026-08-05T15:40:40.536185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-05T13:54:52.875020Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00210","last_updated":"2025-08-29T19:47:25Z","snapshot_observed_at":"2026-08-10T08:26:32.361121Z","submitted_at":"2025-08-29T19:47:25Z","title":"Beyond Pixels: Introducing Geometric-Semantic World Priors for Video-based Embodied Models via Spatio-temporal Alignment","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-05T13:54:52.875020Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2509.00210"},"observation_digest":"sha256:ea4d4d8571b55d7672cb243a39e459a82cfc97a98ac21a1702a569a4fc8da514","observation_id":"96eb1452-6673-46ff-b315-def636bb0db6","resolution":{"observed_at":"2026-08-05T13:54:52.875020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2509.02949","last_updated":"2026-04-06T19:10:00Z","snapshot_observed_at":"2026-07-06T22:22:40.845966Z","submitted_at":"2025-09-03T02:26:48Z","title":"ProMQA-Assembly: Multimodal Procedural QA Dataset on Assembly","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-18T20:03:26.144452Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2509.02949"},"observation_digest":"sha256:02cbd6d28f1fac62e2b2cbed260ee83e783b100da25143192c38c30f0d278687","observation_id":"2394c300-a708-4382-b347-33c74c3f59ff","resolution":{"observed_at":"2026-05-18T20:06:50.183542Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-04T21:27:36.408216Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07680","last_updated":"2025-09-09T17:59:39Z","snapshot_observed_at":"2026-08-09T17:44:26.515281Z","submitted_at":"2025-09-09T17:59:39Z","title":"CAViAR: Critic-Augmented Video Agentic Reasoning","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-04T21:27:36.408216Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2509.07680"},"observation_digest":"sha256:9cfd830a16baae4b77cc30261e26ede84da3e2f375e28491141575a73632ff22","observation_id":"61073f3f-8553-4916-b6cb-ec7067afd79e","resolution":{"observed_at":"2026-08-04T21:27:36.408216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-04T20:32:57.016518Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08538","last_updated":"2025-09-11T11:14:00Z","snapshot_observed_at":"2026-08-08T09:10:30.973101Z","submitted_at":"2025-09-10T12:34:07Z","title":"MESH -- Understanding Videos Like Human: Measuring Hallucinations in Large Video Models","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-04T20:32:57.016518Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2509.08538"},"observation_digest":"sha256:28cb8572eb410edb4abb5f94a3888f2dd382549d06a0eb7e2df906bb9582b247","observation_id":"cc89babc-b975-4d58-9f7b-b9b38196ef01","resolution":{"observed_at":"2026-08-04T20:32:57.016518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-04T20:20:36.989099Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08621","last_updated":"2025-09-10T14:17:53Z","snapshot_observed_at":"2026-08-09T06:09:37.758584Z","submitted_at":"2025-09-10T14:17:53Z","title":"AdsQA: Towards Advertisement Video Understanding","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-04T20:20:36.989099Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2509.08621"},"observation_digest":"sha256:4f7f3cb7af7b4bf35776137cc4d00fef5bacfd3a5a54618a449d9beaf9e2c9f1","observation_id":"94e48f6b-027d-4227-996f-575df6d5d810","resolution":{"observed_at":"2026-08-04T20:20:36.989099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-04T19:28:33.485739Z","title":"Video instruction tuning with synthetic data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09263","last_updated":"2025-09-11T08:49:22Z","snapshot_observed_at":"2026-08-11T00:06:42.054152Z","submitted_at":"2025-09-11T08:49:22Z","title":"DATE: Dynamic Absolute Time Enhancement for Long Video Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T19:28:33.485739Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2509.09263"},"observation_digest":"sha256:d84f9738f849e2f51ac98450ff082cfa96eba87919a3957dbe8c5046d30a6966","observation_id":"2a6b7a38-761c-4dc1-80c4-79e67f6e1854","resolution":{"observed_at":"2026-08-04T19:28:33.485739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-04T07:23:09.070998Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.26113","last_updated":"2026-06-18T20:06:47Z","snapshot_observed_at":"2026-08-06T20:27:36.141566Z","submitted_at":"2025-10-30T03:53:22Z","title":"EgoExo-Con: Exploring View-Invariant Video Temporal Understanding","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-04T07:23:09.070998Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2510.26113"},"observation_digest":"sha256:0c53d076aeb44eeb3c86d6d012b64ae325354075326aa8eb95d65412ff66dea3","observation_id":"6eaca711-a83a-4a5a-a29c-ad30b18f7463","resolution":{"observed_at":"2026-08-04T07:23:09.070998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2511.13026","last_updated":"2026-05-14T09:07:42Z","snapshot_observed_at":"2026-08-08T16:39:03.225299Z","submitted_at":"2025-11-17T06:25:12Z","title":"REVISOR: Beyond Textual Reflection, Towards Multimodal Introspective Reasoning in Long-Form Video Understanding","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-17T22:19:36.366837Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2511.13026"},"observation_digest":"sha256:b25e795491052fb27b940e80ea32177199008eb4fa161f06ccbe6ee8ecf69d2d","observation_id":"db8bfc4c-c46f-4564-aac7-aeb8fe4723b5","resolution":{"observed_at":"2026-05-17T22:20:22.888478Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-03T21:31:37.065333Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.15098","last_updated":"2026-06-21T15:38:16Z","snapshot_observed_at":"2026-08-03T21:31:33.690337Z","submitted_at":"2025-11-19T04:13:36Z","title":"A Comprehensive Study on Visual Token Redundancy for Discrete Diffusion-based Multimodal Large Language Models","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-03T21:31:37.065333Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2511.15098"},"observation_digest":"sha256:4a1a4d4a88dd2656bf805ec4f1fae72a3eec7595733ee2ab967a8a02f08b2c03","observation_id":"3f1a1a3e-e5ca-41fa-9fe4-c25a10aaeab6","resolution":{"observed_at":"2026-08-03T21:31:37.065333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-03T20:23:09.803468Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.20272","last_updated":"2026-07-03T06:27:17Z","snapshot_observed_at":"2026-08-03T20:22:57.328566Z","submitted_at":"2025-11-25T12:58:32Z","title":"VKnowU: Evaluating Visual Knowledge Understanding in Multimodal LLMs","version":2},"reference_index":117,"source":"pdf_text","source_observed_at":"2026-08-03T20:23:09.803468Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2511.20272"},"observation_digest":"sha256:2d1c52c0375f31023112d17a471d77670cf9f558eaeb8fefc4bb04015d6de12e","observation_id":"6a320d18-c23d-4e6d-8f39-76695c00133f","resolution":{"observed_at":"2026-08-03T20:23:09.803468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-03T20:15:37.221295Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.20644","last_updated":"2026-07-09T17:59:10Z","snapshot_observed_at":"2026-08-09T00:51:01.709796Z","submitted_at":"2025-11-25T18:59:02Z","title":"Vision-Language Memory for Spatial Reasoning","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-03T20:15:37.221295Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2511.20644"},"observation_digest":"sha256:f4a54af5bad16a6a03843fb8f5dea6979391bc97e7af77a5fa7fef0d567d1890","observation_id":"4d30f68c-5231-4f13-94d8-2390bf3c4d29","resolution":{"observed_at":"2026-08-03T20:15:37.221295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-03T19:11:53.418782Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.01803","last_updated":"2026-07-09T15:41:25Z","snapshot_observed_at":"2026-08-08T13:02:50.552789Z","submitted_at":"2025-12-01T15:36:33Z","title":"Generative Action Tell-Tales: Assessing Human Motion in Synthesized Videos","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-03T19:11:53.418782Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2512.01803"},"observation_digest":"sha256:228d823b18d7b1be7ebef6413f94b794374b8ba08eb8e3eeb399cc3b60d05fc6","observation_id":"7dfb4cde-dffd-4ca0-800e-78e26826345d","resolution":{"observed_at":"2026-08-03T19:11:53.418782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-03T18:22:32.845533Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.05774","last_updated":"2026-06-04T07:27:00Z","snapshot_observed_at":"2026-08-03T18:22:22.611333Z","submitted_at":"2025-12-05T15:03:48Z","title":"Active Video Perception: Iterative Evidence Seeking for Agentic Long Video Understanding","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-03T18:22:32.845533Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2512.05774"},"observation_digest":"sha256:215db8550f5fe5c36e68d0b8686b129b3ba4365efc80e292d9c2c3c09037616d","observation_id":"67934772-bf1f-4c7c-b7cf-dd600c3842b5","resolution":{"observed_at":"2026-08-03T18:22:32.845533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.02713/citation-record","integrity":"/paper/2410.02713/integrity","json":"/paper/2410.02713/citation-record.json","paper":"/paper/2410.02713"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2204.14198","last_updated":"2022-11-15T23:07:37Z","snapshot_observed_at":"2026-07-06T13:05:12.350238Z","submitted_at":"2022-04-29T16:29:01Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","version":2},"cited_work":{"arxiv_id":"2204.14198","doi":"10.48550/arxiv.2204.14198","metadata_source":"pith","pith_arxiv_id":"2204.14198","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","venue":"cs.CV","work_id":"a110f764-38dc-41b2-a802-53744ecea1fc","year":2022},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2204.14198","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:e7753162c526beb65ce368fd57eeacbf67cf687b32c6cb96f50d362bb95f1e30","observation_id":"4301052c-2903-45d7-9e0d-cba4e2288d59","resolution":{"observed_at":"2026-05-12T04:22:31.147880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Localizing moments in video with natural language","venue":null,"work_id":"54060e06-71cc-40ad-9c5b-2f3adf9d88ce","year":2017},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:c33dd1b28d88b44f65c1bb28435322d0b4aff3a6ebfaf9dd4dae17faa1481128","observation_id":"66b0dd14-3493-4daf-8f35-012140a2a899","resolution":{"observed_at":"2026-05-10T23:20:33.631164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Localizing moments in video with natural language","venue":null,"work_id":"d615ddc8-0f7c-4e92-9d99-5f64b773b840","year":2017},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:04d6ee690831a124277e906de9500e2a928cae1ea08aa57b49c5e677bbb7061a","observation_id":"655bad6a-4350-4479-b131-88e785e2e9f5","resolution":{"observed_at":"2026-05-10T23:20:33.633942Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Frozen in time: A joint video and image encoder for end-to-end retrieval","venue":null,"work_id":"d24a093e-c939-4159-86ac-e79139e01093","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:8102575e575ee0aea0567b2d1556b9b188d4de15f50606f3d9009eba0ff78224","observation_id":"23827f82-95b0-4124-9dea-f53285e95fac","resolution":{"observed_at":"2026-05-10T23:20:33.636531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Activitynet: A large-scale video benchmark for human activity understanding","venue":null,"work_id":"da056c16-524b-48ee-8932-184520fa61cc","year":2015},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:3d9f9edd0ca94e84b211151d462e0725a0e8d87c24b6e9c719fe8ac2e343d7e7","observation_id":"76d02b7a-a755-42fe-b659-ce2066e3eec5","resolution":{"observed_at":"2026-05-10T23:20:33.639075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Collecting highly parallel data for paraphrase evaluation","venue":null,"work_id":"5c374916-5b64-4186-9765-c782478d2a12","year":2011},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:980e70621b30c4e6643d151d2a8bbff11144f40af9cc55369610d1b50419a0bd","observation_id":"fbae0af6-2579-407e-a892-124f161217a3","resolution":{"observed_at":"2026-05-10T23:20:33.645673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":"2406.07476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-07-04T16:49:57.279303Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","venue":"cs.CV","work_id":"ccfc3f89-c510-45f1-8a35-ed1a56c0ae5c","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:3fac0df74e2855c8ab4f0d53f835f70abf03b5feada90f7d017d4840479ac45f","observation_id":"502ca65c-a6f0-457e-9304-9af098457771","resolution":{"observed_at":"2026-05-11T02:44:53.738794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Slowfast networks for video recognition","venue":null,"work_id":"d7aeae2d-f711-42cb-b117-727487a19595","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:0cd474b929f7367b295a21aa5ad4865c418e21bb1ac5e8778aa1bb2f9c8b9661","observation_id":"00a4b2d9-fc92-42a8-a72f-f26fb623fc9a","resolution":{"observed_at":"2026-05-10T23:20:33.652396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"something something","venue":null,"work_id":"dd91d474-f561-465a-8a28-b507ebff396c","year":2017},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:0ce62a71d98d4a05da4261afe50e6c286e6d47b2ad027920a0a5e47126a3889f","observation_id":"31902d00-b60d-4d16-83c7-29f32c0fd786","resolution":{"observed_at":"2026-05-10T23:20:33.655505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":"800b2f65-f47d-4146-91a8-b278cdfa4b57","year":2022},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:7d94e3f8ee4894bd82c86800065f71cf9f5d8f3c299b99f3e7f1b2c9f6bcd077","observation_id":"73279b0b-7784-475a-835f-2be857f8aed9","resolution":{"observed_at":"2026-05-10T23:20:33.658407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Agqa: A benchmark for compositional spatio-temporal reasoning","venue":null,"work_id":"3959a7a6-6b85-4559-adba-4de6cb26092a","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:07ab8291777d51835d02b426d21e3e76b7fa71692a4698f48b8e13efc630927f","observation_id":"02bbcd81-0380-4091-8d7b-d1525f9018fd","resolution":{"observed_at":"2026-05-10T23:20:33.661057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Acav100m: Automatic curation of large-scale datasets for audio-visual video representation learning","venue":null,"work_id":"dc9b41d9-0d6c-4af3-b7ae-2a1331a8433f","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:8da163e8f182efcc08b922604cdda4ed5db70e0d90f3c888a387349ad32c63d0","observation_id":"8b0bd1c8-7485-45f5-980a-f4846de20a99","resolution":{"observed_at":"2026-05-10T23:20:33.664571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Less is more: Clipbert for video-and-language learning via sparse sampling","venue":null,"work_id":"99fc575b-fde0-4463-89c1-45460a24a422","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:81bfeb2171ec6907993988e24772dabc2b850957728dae79a9025693f5d83df8","observation_id":"230286e9-5613-4870-a364-f26a277a7dd8","resolution":{"observed_at":"2026-05-10T23:20:33.670379Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llava-next: What else influences visual instruction tuning beyond data?, May 2024 a","venue":null,"work_id":"5b42d29d-9e3f-4188-baf0-5e65190e7021","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:9e69c87971b2381054bc7f6a4399a3a257b48072dd8a082b1a50d9b015cfabd5","observation_id":"06030416-0d62-4587-91a3-5c3df9d660e8","resolution":{"observed_at":"2026-05-10T23:20:33.673506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llava-next: Stronger llms supercharge multimodal capabilities in the wild, May 2024 b","venue":null,"work_id":"baad07a8-803b-41c6-af98-29e545a30cbc","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:cd473b31b55c057ed4230236957ab59e4482664cf154ae69104077a11e6ee4fe","observation_id":"c17b7a02-819e-4e02-a2c8-aaa5535a4ee5","resolution":{"observed_at":"2026-05-10T23:20:33.676462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Multimodal foundation models: From specialists to general-purpose assistants","venue":null,"work_id":"cb495da8-0270-4cdf-aa68-ce852f6287a3","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:bbbbe605a49d2b4e414d38f375bbbb80bb5f0b754dc08204e58dbc228d69d9ab","observation_id":"c3d63041-360f-4226-9e92-71874183a63f","resolution":{"observed_at":"2026-05-10T23:20:33.679896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":"2301.12597","doi":"10.48550/arxiv.2301.12597","metadata_source":"pith","pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","venue":"cs.CV","work_id":"63d03f4d-15f4-4583-8286-913c19f02294","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:e6f02c95d3f9cc60e0c86c37c2a2da8d87216ad2b35ddf466b4bda4ea62374c3","observation_id":"f9f5c995-439f-48ae-af2b-edd84930ef81","resolution":{"observed_at":"2026-05-12T00:10:50.405848Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f6a8de3bea839f69c24cd95b80a40f4d880848051f1db4657e86c7f49887040e","observation_id":"945090e3-98eb-48ff-a09f-1ec94a9f1394","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":"c960a098-b938-4788-81ae-5eb68d3c778a","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:8b5f11ae8aaaaab30b975ee99f096fee8b728d2b58c069f4dd8573f510eaac68","observation_id":"1a439b78-4e3e-4342-99f1-27e05dc28378","resolution":{"observed_at":"2026-05-10T23:20:33.684018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual instruction tuning","venue":null,"work_id":"6d02d035-035e-4179-b018-4616d2401153","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:26ddb7298cae4ea25cf6b4698196022aea20a988c88115deaa7259effd7995eb","observation_id":"d78ea22e-5e74-4614-be36-1dd0f7fc7107","resolution":{"observed_at":"2026-05-10T23:20:33.690864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video detail caption","venue":null,"work_id":"12883507-3a43-4f6e-8a36-fc480ec33b57","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:92a87f062be8210744094bbd5f523a4e3b360719c9886e1aaa8af0897045cad1","observation_id":"17e17dbd-6315-4b2c-9073-1492eeaf970b","resolution":{"observed_at":"2026-05-10T23:20:33.698073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"efc52e16-dd60-40f4-b41d-bb7ab0b93690","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:2982a26f4df4ce25a87f5e05582723c8793e616f2a7c4a86cc4cb1cb32816cf5","observation_id":"b3105ae8-a139-46f9-b45c-ebed021ea7ff","resolution":{"observed_at":"2026-05-10T23:20:33.702860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding","venue":null,"work_id":"95155570-064b-4564-b1c4-a52f7704925e","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:4465806e9027029013b80144a19ff6d1096d2427c035d387f4bf4ef2e5e41a30","observation_id":"759fdf4c-38fb-41b6-a66e-290f5ae01270","resolution":{"observed_at":"2026-05-10T23:20:33.709118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How T o100 M : L earning a T ext- V ideo E mbedding by W atching H undred M illion N arrated V ideo C lips","venue":null,"work_id":"e244d10d-4baf-41dc-ae70-59f3518e6f82","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:c12972682693047b046418c5de0eaef6fab1a033cbe4781bc6c748715e8b4f92","observation_id":"471d5b74-6463-4ae3-9a81-01fb4e080289","resolution":{"observed_at":"2026-05-10T23:20:33.715166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2ce8d358-1218-4fdd-b799-fdd4152ded50","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:3f5ad7093d180464accfaeb8d99dc9e2e1dc3dbc86d36dfb478ab3c42cf8ba92","observation_id":"f3bd7134-00d2-47b8-af6c-0d0d1a4c37fd","resolution":{"observed_at":"2026-05-10T23:20:33.717985Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hello gpt-4o","venue":null,"work_id":"9ea24406-6f8c-42b6-844e-565effee5040","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:7c505c368bc482f27e1f16e65d7d6469e8552ed0c5d08ffd5a1f5ab6d53f870a","observation_id":"85fd65e4-af8a-4e1c-b389-d5de12345a19","resolution":{"observed_at":"2026-05-10T23:20:33.720765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Perception test: A diagnostic benchmark for multimodal video models","venue":null,"work_id":"69ad3219-2b1f-49dd-8eaf-53d62ab8079f","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:d5a29f7cd94245e4b893b1a3e4b695916c22363859c113e720d2c45c302df5f7","observation_id":"659eb492-fffb-4a12-9c03-e000901435ef","resolution":{"observed_at":"2026-05-10T23:20:33.723455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"cfaf5fca-f10f-4362-9478-c7b8ece14929","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:1176001f782bfee9c2d7c407e52669cf27b0b8fa5ce13f40e44b5825aae5459e","observation_id":"c185b1b6-8ce5-490a-a6ad-bcf64fd3ec7d","resolution":{"observed_at":"2026-05-10T23:20:33.732308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.09813","last_updated":"2020-10-05T06:30:56Z","snapshot_observed_at":"2026-08-06T22:52:30.854160Z","submitted_at":"2020-04-21T08:20:25Z","title":"Making Monolingual Sentence Embeddings Multilingual using Knowledge Distillation","version":2},"cited_work":{"arxiv_id":"2004.09813","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2004.09813","snapshot_observed_at":"2026-07-02T05:16:39.247855Z","title":"org/abs/2004.09813","venue":null,"work_id":"2f9a0962-1c5c-4567-98cd-6206560ad5bf","year":2020},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2004.09813","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:a2d4231fc8aeb672e37163b8f658bb4c1bd932db256058db6dd46fa152e0c87e","observation_id":"0510af63-23de-47e2-a2ab-9d33ba745e87","resolution":{"observed_at":"2026-05-10T23:20:32.996852Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A dataset for movie description","venue":null,"work_id":"9b365862-5e05-4dc8-9afb-c71b45d97dbe","year":2015},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:c28f76222b43943b660e6cc746eb668d44bcc6361b42d897f95054526e8d8cd7","observation_id":"6828b804-2371-4c0a-92bc-c2a141f862b8","resolution":{"observed_at":"2026-05-10T23:20:33.738181Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Annotating objects and relations in user-generated videos","venue":null,"work_id":"5120c85d-9c9d-42cc-b1e0-1f74b7bbdceb","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:b9ce5b92d643146ab6f52af0fe3a78926d4aa1ea26818405882d0df15b9e7b1d","observation_id":"1044fd9b-52f3-431a-aa5c-7a565aed18b4","resolution":{"observed_at":"2026-05-10T23:20:33.746891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hollywood in homes: Crowdsourcing data collection for activity understanding","venue":null,"work_id":"56d13127-bfa9-4f76-bf4e-64f0b09250fd","year":2016},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:9bd184ef98e23563586d58694475def80c988b6cc5635d50e6fbc69167ec6bae","observation_id":"881e83fa-9b05-4b89-ae6f-f6e0a7292f14","resolution":{"observed_at":"2026-05-10T23:20:33.752336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00634","last_updated":"2024-09-24T04:41:08Z","snapshot_observed_at":"2026-08-11T06:04:22.773616Z","submitted_at":"2024-06-30T09:21:01Z","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","version":2},"cited_work":{"arxiv_id":"2407.00634","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.00634","snapshot_observed_at":"2026-07-04T17:40:00.906584Z","title":"Tarsier: Recipes for training and evaluating large video description models","venue":null,"work_id":"bf38f75f-fbc4-4c6d-ab2a-6e1bcbc38485","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2407.00634","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:83dff51fd6737a9d76164bb025ca8ec0f1c7149a197832c916fcfaded213dc81","observation_id":"0be139c6-460b-4b87-ae5c-c34e807f64df","resolution":{"observed_at":"2026-05-10T23:20:32.962559Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internvid: A large-scale video-text dataset for multimodal understanding and generation","venue":null,"work_id":"5cb6d29c-6c88-4450-a77c-61e7e1b907ef","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:4cf95dc6a10fd7971e44c94eac00333b4c499cfaeed70d69c2b655cae8191ac3","observation_id":"95d1cba9-dfb0-440c-9a95-63c8b7808591","resolution":{"observed_at":"2026-05-10T23:20:33.755253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:423a702d87bf9a147693a0ed7da4f0b65cddb08b40c327a319c504a1be5b63cf","observation_id":"13c7293b-8856-4dbe-af2f-33d45d80e6b2","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"e9a497c3-2843-4d63-8eb9-fa8dc748b1e1","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:b87624864daef766095d94f9b42a4cc8448413aa2f9fc3472cf45c59307d51e6","observation_id":"e48d6f90-c0d8-4c76-9fae-0e74d949dfa9","resolution":{"observed_at":"2026-05-10T23:20:33.758033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video question answering via gradually refined attention over appearance and motion","venue":null,"work_id":"549826de-6b8b-4f12-a8db-3e8157deaa08","year":2017},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:76337c2fbe8bc5b2a7e395ccb768eea1e89336a3b95f13f7a173b086ba29a6b6","observation_id":"1b10e2c7-aa1d-43e2-a878-7e361e573fd0","resolution":{"observed_at":"2026-05-10T23:20:33.763103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":"8ed7daf9-5f7a-4338-9968-d835a46f8d58","year":2016},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:ad05e3932571862cc3e27458d42003a5fe196024db23f93e5e6789098cae22a9","observation_id":"7650c80a-2a37-4b39-94bb-361ea1bdb894","resolution":{"observed_at":"2026-05-10T23:20:33.767009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advancing high-resolution video-language representation with large-scale video transcriptions","venue":null,"work_id":"fce15bfd-3d96-489e-81da-ff3e53b79331","year":2022},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:9ac4c6c9e9a432b7022e5b13a952a185ce8662ebf6371131ab7c555145b1cb51","observation_id":"247feb16-86fd-47c0-a678-b8d9db22fd23","resolution":{"observed_at":"2026-05-10T23:20:33.775120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"19e1149c-0af0-4e1f-8c15-9e15ef515868","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f3f4003311905bf007d09cd8530ab42c47b2a40cb48e9982fc2d6ff1c2d4454c","observation_id":"8076cbd7-5638-48d4-b610-64ead95d16c3","resolution":{"observed_at":"2026-05-10T23:20:33.778232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Social-iq: A question answering benchmark for artificial social intelligence","venue":null,"work_id":"fe1f6b88-8718-40fe-b795-0619f2987c59","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:a894843e0317470920e8280ee3e80688897d95cd71f8e7ba961a6db01ef1f21a","observation_id":"d876a09d-f8a6-44ce-9a46-70e84343ad8f","resolution":{"observed_at":"2026-05-10T23:20:33.781008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Merlot: Multimodal neural script knowledge models","venue":null,"work_id":"b68f28e9-ed54-448b-8da1-4213280e128d","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:89a85b2f5137ae20b37b780e6af75e51124e5b38e71892c98ec6eecf2ec1365e","observation_id":"5d2c33ba-49f1-4ca1-86c9-9915819579b5","resolution":{"observed_at":"2026-05-10T23:20:33.786126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"e2d41f6e-f2a0-4ea4-8547-e59d97a85519","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:1b25340dd61b9780110410433986ed9699cf605401162a2869d10977a9ac6137","observation_id":"978b6f3c-4301-4ef1-8d03-d9067de561a5","resolution":{"observed_at":"2026-05-10T23:20:33.790395Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Direct preference optimization of video large multimodal models from language model reward, 2024 d","venue":null,"work_id":"4a1fef21-c077-4c4e-985d-7841bacb95be","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f5ef864b00d14994d87fbce4d146d59fa36950679f960d36305bd746241f1592","observation_id":"cc265d68-56a3-4e44-be49-5edbc41c42b3","resolution":{"observed_at":"2026-05-10T23:20:33.793622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llava-next: A strong zero-shot video understanding model, April 2024 e","venue":null,"work_id":"2fbaca6f-e19b-4d8b-92d0-973c2b286b04","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:ef6de38213df58ae63691764f74b64076c006e2d3c8b2aa4b8a43d123610f929","observation_id":"989f9f34-5446-4f2a-bed0-7bcefbbe4e4f","resolution":{"observed_at":"2026-05-10T23:20:33.796162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"35a35948-ff33-489e-8537-70d4e1c22979","year":2017},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:267571d98f83e06e055dd0e7e49c6c628f8013bf9ff2f9b7520553482d40e552","observation_id":"038e60f8-da92-414b-9ef3-356f2041dc95","resolution":{"observed_at":"2026-05-10T23:20:33.798484Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Languagebind: Extending video-language pretraining to n-modality by language-based semantic alignment, 2023 a","venue":null,"work_id":"4ca56ba3-d7cb-48ba-bbc0-be91b46cd9cc","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:80aabf2d20adabc9579d023e101205a5d77f554706c9201961ce8de4e9453253","observation_id":"ac8ac4ed-4906-4c74-9462-d53e945408da","resolution":{"observed_at":"2026-05-10T23:20:33.801244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.12119","last_updated":"2022-07-20T15:47:22Z","snapshot_observed_at":"2026-08-10T12:45:24.532998Z","submitted_at":"2022-03-23T01:17:16Z","title":"Visual Prompt Tuning","version":2},"cited_work":{"arxiv_id":"2203.12119","doi":"10.48550/arxiv.2203.12119","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.12119","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Visual prompt tuning","venue":null,"work_id":"f53e528d-b5de-4af8-950e-05b1dc0c04a6","year":2022},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2203.12119","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:41ec5c0cf8fbf431f71b48831541f3a1870562d38bce9676c719e9d30bdd8fd8","observation_id":"815f9ce0-8cdd-4806-a4f9-961be6afd4e3","resolution":{"observed_at":"2026-05-10T23:20:33.113505Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning (ICML) , pages=","venue":null,"work_id":"455dcbad-33eb-469d-b076-a39497216b41","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:4942ff6cdff298b0e6777b49bc1b958010b8c972f50ceec921c51c7c21fa3b2a","observation_id":"851caa27-a1fa-421d-9265-2d9f1a246e0b","resolution":{"observed_at":"2026-05-10T23:20:33.803516Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-11T08:20:29.798517Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":"2106.09685","doi":"10.4088/pcc.v03n0609","metadata_source":"pith","pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","venue":"cs.CL","work_id":"0426219a-789e-4964-adc8-a04538510818","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:b6654c30f26d421df23b62daf2f65a68713c63eaddfd6850cecfb19dd9f10933","observation_id":"12e5c4b7-7708-480b-a5db-e7190070fa95","resolution":{"observed_at":"2026-05-10T23:20:32.505718Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.04366","last_updated":"2022-02-02T16:39:23Z","snapshot_observed_at":"2026-08-09T04:45:04.547249Z","submitted_at":"2021-10-08T20:22:26Z","title":"Towards a Unified View of Parameter-Efficient Transfer Learning","version":3},"cited_work":{"arxiv_id":"2110.04366","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.04366","snapshot_observed_at":"2026-07-04T15:59:57.198158Z","title":"Towards a unified view of parameter-efficient transfer learning","venue":null,"work_id":"e758988d-4ae2-44ff-94fc-4417c242b1e4","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2110.04366","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:a157a4dd86aaa4b304515aad6b2ed56ad4462f70d972b433f0dedee4d35e1af1","observation_id":"a7721538-11c9-406b-af25-5b10904daabe","resolution":{"observed_at":"2026-05-10T23:20:32.551150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.05240","last_updated":"2021-12-14T18:49:27Z","snapshot_observed_at":"2026-08-09T07:40:39.556941Z","submitted_at":"2021-04-12T07:11:40Z","title":"Factual Probing Is [MASK]: Learning vs. Learning to Recall","version":2},"cited_work":{"arxiv_id":"2104.05240","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2104.05240","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Factual probing is [mask]: Learning vs","venue":null,"work_id":"32068238-990b-42e0-8bd9-5dc4754fbecc","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2104.05240","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:32cc78d4fe34a8bdf9797b25436398fcaa434d6461771751d1b318044dc95061","observation_id":"bf8445e4-ab2f-4cda-869f-ba74c271fd6a","resolution":{"observed_at":"2026-05-10T23:20:32.671954Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2106.10199","doi":"10.48550/arxiv.2106.10199","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models","venue":"arXiv (Cornell University)","work_id":"8505729c-c88b-43bd-8e2c-2c94644ca438","year":2022},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:9b02a132f9f8fccd73adc1a4fc9ce72b28f30a723858e4cfc50a89bdd2d6e776","observation_id":"b8f9d25e-f5a3-4096-b93e-d1a79eb07e64","resolution":{"observed_at":"2026-05-10T23:20:32.683322Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in neural information processing systems (NeuIPS) , volume=","venue":null,"work_id":"d411c3a0-fb5e-4fd9-9879-60568356eeec","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f7cf071b1ace1450454cb2df3aafe7476a45c0f25f734f23f8beb9c016603c31","observation_id":"13ac991c-6321-4026-bd09-d1972bb5af00","resolution":{"observed_at":"2026-05-10T23:20:33.808080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":"1810.04805","doi":"10.1111/jofi.12885","metadata_source":"pith","pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"cs.CL","work_id":"ed240a10-5b19-406c-baa5-30803f465785","year":2018},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:2ad6f4bc1ba887557189fa258e04d86c92a8c054106d11c8711833bcf23215a3","observation_id":"65e560d1-d32c-4388-9613-7d8015d3bbf0","resolution":{"observed_at":"2026-05-10T23:20:32.722162Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in neural information processing systems (NeuIPS) , volume=","venue":null,"work_id":"1700c8fc-4847-42fa-b3c0-aa442009f6a5","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:28feaec45f8c9be254a256e2165a5f4ea3812aeaff0e77aba31ca0b155baefb3","observation_id":"548183ac-c428-4bac-9a42-8f1c91d45264","resolution":{"observed_at":"2026-05-10T23:20:33.815081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00190","last_updated":"2021-01-01T08:00:36Z","snapshot_observed_at":"2026-07-06T10:29:18.734092Z","submitted_at":"2021-01-01T08:00:36Z","title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","version":1},"cited_work":{"arxiv_id":"2101.00190","doi":null,"metadata_source":"pith","pith_arxiv_id":"2101.00190","snapshot_observed_at":"2026-07-04T13:29:51.032515Z","title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","venue":"cs.CL","work_id":"23ddcedc-909d-4b18-b28a-943a50a8cef7","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2101.00190","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:81888abb3bca0f491f335398cb8c9a9ac85fdfd86707ed3fa7cc00b855a60723","observation_id":"99b7857a-520c-418c-a08c-269cfde97a3f","resolution":{"observed_at":"2026-05-11T16:57:25.623351Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08691","last_updated":"2021-09-02T17:34:41Z","snapshot_observed_at":"2026-08-06T15:24:34.790850Z","submitted_at":"2021-04-18T03:19:26Z","title":"The Power of Scale for Parameter-Efficient Prompt Tuning","version":2},"cited_work":{"arxiv_id":"2104.08691","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.08691","snapshot_observed_at":"2026-07-04T14:59:55.372222Z","title":"The Power of Scale for Parameter-Efficient Prompt Tuning","venue":"cs.CL","work_id":"1056ba8e-7b3f-4811-be8e-9a3ed9269acb","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2104.08691","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:04989bbff7dd211ffaa1518d9266c238a3dbfc90165b80b736d9e219cab7f41e","observation_id":"c57b2c5c-6d40-46db-8a70-2d7dd357c908","resolution":{"observed_at":"2026-05-11T16:34:05.826717Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , year=","venue":null,"work_id":"548872bf-f169-42d3-ba20-b275e6420d66","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:0f70ec2f5be27655361e034049ace000485255f241fd86d1136b27657043ad30","observation_id":"0dabb73c-8813-4f71-a429-74a23878f851","resolution":{"observed_at":"2026-05-10T23:20:33.817845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Journal of Computer Vision (IJCV) , year=","venue":null,"work_id":"b31669bd-3bb8-41a7-8a04-e7c8773690df","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f3c8aaf822f19ff84b8f7a04463b7e148fc3bd7147556c64d7376fa09bf841ba","observation_id":"f1cce10b-33e4-4261-92e6-125f6063e2b1","resolution":{"observed_at":"2026-05-10T23:20:33.822720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , year=","venue":null,"work_id":"4273960d-40ed-4e55-9680-880741b22e98","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:8abf07c89d22c1ec0e25c74f820a88f06dd6226b0c0a8bc3859c6f36240a0fa1","observation_id":"ed9d954f-8fea-4fa6-a205-fd88e978b7b9","resolution":{"observed_at":"2026-05-10T23:20:33.825276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":"2010.11929","doi":"10.1175/jcli-d-22-0357.1","metadata_source":"pith","pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","venue":"cs.CV","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","year":2020},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:093254bb153ec315a0cf51316c4907775537f5a753003542816a8865b66735f3","observation_id":"8b1a8a24-e2bb-41d5-82a3-04139c63d618","resolution":{"observed_at":"2026-05-10T23:20:33.070234Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.07577","last_updated":"2022-09-04T23:31:05Z","snapshot_observed_at":"2026-08-01T19:07:06.818808Z","submitted_at":"2021-10-14T17:40:08Z","title":"UniPELT: A Unified Framework for Parameter-Efficient Language Model Tuning","version":3},"cited_work":{"arxiv_id":"2110.07577","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.07577","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unipelt: A unified framework for parameter-efficient language model tuning","venue":null,"work_id":"6eb2f2c7-67ba-4aec-a0e9-25ca187d1625","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2110.07577","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:abe93f46ca267c89e9f7ff5b2e1c55ba547d4dc2d684c9d22f9a4141cd3be617","observation_id":"b65256a4-561d-4e0a-8e13-4fbf232632f5","resolution":{"observed_at":"2026-05-10T23:20:33.080613Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1611.01578","last_updated":"2017-02-15T05:28:05Z","snapshot_observed_at":"2026-07-06T05:17:29.499249Z","submitted_at":"2016-11-05T00:41:37Z","title":"Neural Architecture Search with Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"1611.01578","doi":"10.1016/j.knosys.2015.01.010","metadata_source":"pith","pith_arxiv_id":"1611.01578","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Neural Architecture Search with Reinforcement Learning","venue":"cs.LG","work_id":"376934b8-2ad9-4443-87fe-ff630c1b89d9","year":2016},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/1611.01578","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:0e590bab7712b48f1fd51b5dbf9cf89d46eec0ceb7aef3b4ad3d03e69fed5efc","observation_id":"af0df508-c6f1-432f-a240-b6d7eadb743b","resolution":{"observed_at":"2026-05-10T23:20:33.090752Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , pages=","venue":null,"work_id":"008607e1-b2dc-4540-86c1-eb14192845c0","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:50159fa7b3a64cac10dcda16a6bb634ce712cdae984f7d0ab71a5df42de6a33a","observation_id":"44187c4e-6bee-4547-b0c5-c33f95d3041a","resolution":{"observed_at":"2026-05-10T23:20:33.827651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI) , year=","venue":null,"work_id":"0c90bc9f-e729-49c4-ab20-fb036a1410c2","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:fc87409d372a0dc73781797665709b6edaab062203874f6f064dc48803d6da22","observation_id":"0b8cfa88-2e5f-4dc9-9118-c979bc5ed0a1","resolution":{"observed_at":"2026-05-10T23:20:33.829922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning (ICML) , pages=","venue":null,"work_id":"4d01df8b-777a-4f2c-b558-91b08feeae44","year":2017},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:5f3829980bfef0dd1ba6eab8085f077b68534e8f757084ec8995659f5bcff6a2","observation_id":"cff8ec30-f885-4feb-8412-25c83ffc151e","resolution":{"observed_at":"2026-05-10T23:20:33.837518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International conference on machine learning (ICML) , pages=","venue":null,"work_id":"039084be-0c08-4720-bf1f-839fec6d7b30","year":2018},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:ebc3301841b11d50c2137011344dd3d86f552e95c1f48e58d088794fb02ad94e","observation_id":"da59bf03-ebb7-48cc-96a0-65743ee8bead","resolution":{"observed_at":"2026-05-10T23:20:33.848698Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1806.09055","last_updated":"2019-04-23T06:29:32Z","snapshot_observed_at":"2026-08-09T13:23:20.015254Z","submitted_at":"2018-06-24T00:06:13Z","title":"DARTS: Differentiable Architecture Search","version":2},"cited_work":{"arxiv_id":"1806.09055","doi":null,"metadata_source":"pith","pith_arxiv_id":"1806.09055","snapshot_observed_at":"2026-07-10T21:27:35.708872Z","title":"DARTS: Differentiable Architecture Search","venue":"cs.LG","work_id":"2f7f6e44-42e7-498e-a68f-a01e092a8903","year":2018},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/1806.09055","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:95da0f59907cf54d275791b1644df3ddf6eaaeababf6d5430d688a1852ce2fb4","observation_id":"9d6537fa-6fc8-4b2e-b057-07ce0c2cb512","resolution":{"observed_at":"2026-05-10T23:20:32.657992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV) , pages=","venue":null,"work_id":"24400c75-a203-433a-85fa-e7d24ffe63c9","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:232cd4721033ec1d34cd0ce43b287da5df9f127af2c8b982c6a8c204879e691a","observation_id":"f4feed9b-5d73-4ac1-a7cb-507cc782c787","resolution":{"observed_at":"2026-05-10T23:20:33.854265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T05:50:44.058714Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , pages=","venue":null,"work_id":"80a5e127-9efc-46c7-8ede-b932cb2eceec","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:7e124a91bef2ffb7dc24895abb6c72268fbf06bcf1099c73e8792c177a81de6e","observation_id":"f2488925-6126-4f6d-8399-b4319aca5bcc","resolution":{"observed_at":"2026-05-10T23:20:33.856889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.04867","last_updated":"2020-02-21T13:36:15Z","snapshot_observed_at":"2026-08-09T06:48:42.729935Z","submitted_at":"2019-10-01T17:06:29Z","title":"A Large-scale Study of Representation Learning with the Visual Task Adaptation Benchmark","version":2},"cited_work":{"arxiv_id":"1910.04867","doi":"10.48550/arxiv.1910.04867","metadata_source":"pith","pith_arxiv_id":"1910.04867","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Large-scale Study of Representation Learning with the Visual Task Adaptation Benchmark","venue":"cs.CV","work_id":"eb743d69-6704-47c1-bb0f-76520dd00c3d","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/1910.04867","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:d70a95ddb0a16b2bb22a9ed99691cd3977087a9d53782ed747a7e5c4258da494","observation_id":"7cbb515c-9b65-4e04-9fc8-a7d8d0718e9e","resolution":{"observed_at":"2026-05-17T22:12:05.852900Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.03801","last_updated":"2016-12-13T12:19:48Z","snapshot_observed_at":"2026-07-06T05:22:21.129782Z","submitted_at":"2016-12-12T17:32:49Z","title":"DeepMind Lab","version":2},"cited_work":{"arxiv_id":"1612.03801","doi":null,"metadata_source":"pith","pith_arxiv_id":"1612.03801","snapshot_observed_at":"2026-07-09T12:56:14.879463Z","title":"DeepMind Lab","venue":"cs.AI","work_id":"8a8d827f-5377-4733-bfe8-bc66c011d458","year":2016},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/1612.03801","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:545aa38588a233c77642f88c16757732096c9cde97547c981ddbe94be4426ead","observation_id":"cb77a37b-9a32-43d9-b740-b43eeaf694e8","resolution":{"observed_at":"2026-05-10T23:20:32.704663Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1ec0f122-87f8-41d5-810c-0f2f0888a3e1","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:226df325163b003a55bdf8329ae2cbf22a0f87f3fefe49f9d898337359628029","observation_id":"079d7a4b-b31b-487a-a4f2-96cb863087f5","resolution":{"observed_at":"2026-05-10T23:20:33.859166Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , pages=","venue":null,"work_id":"906256dd-6691-45e5-9c10-86e35a4bd042","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:ca9532a5e272572992e984956c127df07b19f7f018fa0cef02465949b25d18eb","observation_id":"becd0eca-edd6-48e1-a4cb-0ac702bedc2c","resolution":{"observed_at":"2026-05-10T23:20:33.861606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.06175","last_updated":"2022-11-11T10:04:29Z","snapshot_observed_at":"2026-08-08T03:18:33.595658Z","submitted_at":"2022-05-12T16:03:26Z","title":"A Generalist Agent","version":3},"cited_work":{"arxiv_id":"2205.06175","doi":"10.48550/arxiv.2205.06175","metadata_source":"pith","pith_arxiv_id":"2205.06175","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Generalist Agent","venue":"cs.AI","work_id":"4b0a87cd-8d54-4abc-9698-c4cb20995600","year":2022},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2205.06175","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:047212ca2ab0af955886cd9bd5fb202b17074629be754e6952814bcec7bf66f8","observation_id":"386789d2-454e-4bc1-864d-34a9202ebb3e","resolution":{"observed_at":"2026-05-13T06:24:50.082909Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , volume=","venue":null,"work_id":"ed18c027-7e22-4f06-b6c3-b0aa6c7957e8","year":2004},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:ef4b2fa66465bebe45ce23b71d58d297502ac865d8a3f4954536889f95456526","observation_id":"e151b766-d10c-4e28-8602-7d59752067f5","resolution":{"observed_at":"2026-05-10T23:20:33.864909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"European conference on computer vision (ECCV) , pages=","venue":null,"work_id":"71fdb59d-a0f0-4733-a070-490df0ded493","year":2014},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:4be95267f203e2b56f8f64ca0416b7194810dea54537d2dd28bfefa0b4d12980","observation_id":"6151fcf9-2484-4cc6-99a4-9ea3ac4333f4","resolution":{"observed_at":"2026-05-10T23:20:33.868968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.02503","last_updated":"2022-08-12T08:24:43Z","snapshot_observed_at":"2026-08-09T20:43:40.437248Z","submitted_at":"2021-03-03T16:12:22Z","title":"Domain Generalization: A Survey","version":7},"cited_work":{"arxiv_id":"2103.02503","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.02503","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Domain gen- eralization in vision: A survey","venue":null,"work_id":"dee411d0-8993-4f21-8471-2f3f8caf017a","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2103.02503","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:a2bcfd873f8184b28b137c98298a04106edb906c0f7f43de07983b7ee27cff34","observation_id":"15138065-4b00-4e76-bf61-966ddadb9054","resolution":{"observed_at":"2026-05-10T23:20:32.827044Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11432","last_updated":"2021-11-22T18:59:55Z","snapshot_observed_at":"2026-07-06T12:11:02.119174Z","submitted_at":"2021-11-22T18:59:55Z","title":"Florence: A New Foundation Model for Computer Vision","version":1},"cited_work":{"arxiv_id":"2111.11432","doi":null,"metadata_source":"pith","pith_arxiv_id":"2111.11432","snapshot_observed_at":"2026-07-04T20:00:08.247992Z","title":"Florence: A New Foundation Model for Computer Vision","venue":"cs.CV","work_id":"99823072-36a8-4b10-9ef5-a7f91da74650","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2111.11432","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:b38ed720dc2c0147fd518bdcab07ab542c6c3b6eebae986fcd01c13a4027f04d","observation_id":"3f589d9b-43fd-44f4-b4d8-c27517ce5713","resolution":{"observed_at":"2026-05-16T09:38:09.598269Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.07845","last_updated":"2022-08-24T01:41:45Z","snapshot_observed_at":"2026-08-11T11:45:06.118634Z","submitted_at":"2022-03-15T13:01:00Z","title":"Bamboo: Building Mega-Scale Vision Dataset Continually with Human-Machine Synergy","version":2},"cited_work":{"arxiv_id":"2203.07845","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.07845","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2203.07845 , year=","venue":null,"work_id":"e4107433-752f-465c-b60c-3447bf20dbb6","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2203.07845","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:0c0f487671a31aeb73ee3a245a882370dc453f435311bdcdd641f79795a40750","observation_id":"3fa01dc7-494e-42cb-8ae2-601bb5a9a1f1","resolution":{"observed_at":"2026-05-10T23:20:32.843466Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The International Journal of Robotics Research , volume=","venue":null,"work_id":"832712f2-adfa-41fd-ae80-041d6d956518","year":2013},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:aa3765cd6a1d9f348f5d0617986e566c8c2c9988f98957912319e0d8b80468dd","observation_id":"2031154a-0a67-4953-9281-e3ff212fcf98","resolution":{"observed_at":"2026-05-10T23:20:33.177342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"European conference on computer vision (ECCV) , pages=","venue":null,"work_id":"b9ca8bcc-13c6-446c-a788-9d4bd8da7d8b","year":2014},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:bc8f65458597d3a8f9171411082d68bb1b34ef3ba91b91f37ca3bd86bc3b7547","observation_id":"6468b77b-5232-4c11-b814-83c8a6f35115","resolution":{"observed_at":"2026-05-10T23:20:33.180846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , volume=","venue":null,"work_id":"95823e99-7e8a-4d9b-b727-f77d678b2e4e","year":2006},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:c034b2f196140d03af4c091d05e95aea31a5544bba4b50b92617211bfff37718","observation_id":"6699bcf1-ff8e-4feb-b031-aefe71b4b598","resolution":{"observed_at":"2026-05-10T23:20:33.191997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:11:40.302133Z","title":"Proceedings of the IEEE international conference on computer vision workshops , pages=","venue":null,"work_id":"9178b53a-3209-4879-8009-26a9046d9a60","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:39a0d1d8b52246d21e10f6fa4d6c77855d7c505f089eb2b5fb292368aa525689","observation_id":"6d4769f5-2af0-4189-a26e-0905173bd6c6","resolution":{"observed_at":"2026-05-10T23:20:33.204880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , pages=","venue":null,"work_id":"f7e5754b-8254-46ab-b575-7682649d8f0d","year":2012},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f7ebf16aa69ae10a53b0cba70edfd8d9bb35a4641cbb39e4e504e2900bff6824","observation_id":"177e2e96-0336-49b6-b186-406a605a0d42","resolution":{"observed_at":"2026-05-10T23:20:33.215197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1306.5151","last_updated":"2013-06-21T14:31:57Z","snapshot_observed_at":"2026-07-06T03:16:28.287173Z","submitted_at":"2013-06-21T14:31:57Z","title":"Fine-Grained Visual Classification of Aircraft","version":1},"cited_work":{"arxiv_id":"1306.5151","doi":null,"metadata_source":"pith","pith_arxiv_id":"1306.5151","snapshot_observed_at":"2026-07-11T03:07:53.092922Z","title":"Fine-Grained Visual Classification of Aircraft","venue":"cs.CV","work_id":"ed360110-3ce4-4959-8c74-1785cd9e537d","year":2013},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/1306.5151","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:d1ec656603be6fe7566ec6aec2e4d05f3590c64ad210ecbc4a5182848e42548f","observation_id":"1bc02244-54b3-42ed-bc46-4ebb6ea780d4","resolution":{"observed_at":"2026-05-11T17:41:07.065951Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning (ICML) , pages=","venue":null,"work_id":"7ef3fdd1-b09f-46e8-9d29-db0ca02ff1f2","year":2019},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:5345b21a069807b2a36029cd3e0baf62956ca5fc6c53ba33d6d734ddb15cb524","observation_id":"11037149-2c26-4f9c-ac01-070d7d6b68e5","resolution":{"observed_at":"2026-05-10T23:20:33.231663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems (NeuIPS) , volume=","venue":null,"work_id":"e9912131-beb7-48aa-b477-175639713df0","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:3381db8e80cfe4bdab2d75dce47116ad207d2baf09f47a27cf457efa43aa30d4","observation_id":"1fe8658e-da36-4b0b-b222-812b1063b19f","resolution":{"observed_at":"2026-05-10T23:20:33.243801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , pages=","venue":null,"work_id":"a337eed0-2658-41df-9210-55826f50bac1","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f32cb0015d9644b34c5dc7ffded201c4e67df452972385c6b15b069a18bafd91","observation_id":"c12945ab-129a-4116-8b49-11ef163ed4ad","resolution":{"observed_at":"2026-05-10T23:20:33.248943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV) , pages=","venue":null,"work_id":"56e7925f-15fe-4873-b162-89e5895b0c62","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:cde59cb32e63c024e63d9dd90138b65cf6131ab158d6ae332411b7ce86fa36fd","observation_id":"492ef995-fcaf-449d-847e-1f697b9d95e2","resolution":{"observed_at":"2026-05-10T23:20:33.253338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.07461","last_updated":"2019-02-22T23:53:34Z","snapshot_observed_at":"2026-07-06T06:34:26.609892Z","submitted_at":"2018-04-20T06:35:04Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","version":3},"cited_work":{"arxiv_id":"1804.07461","doi":null,"metadata_source":"pith","pith_arxiv_id":"1804.07461","snapshot_observed_at":"2026-07-04T09:19:44.134394Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","venue":"cs.CL","work_id":"1bb6fb0c-482d-43cf-94a8-ed18f72a5563","year":2018},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/1804.07461","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:5663d07bef5282b22e01701c2fb7ce617a2e17be1592bb28c79f2c61ce451842","observation_id":"c4c79ab6-9f45-4398-a24e-defaa9e5fae1","resolution":{"observed_at":"2026-05-12T21:24:15.760169Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , pages=","venue":null,"work_id":"22a8df11-64af-4e15-ad12-f1acb04a1376","year":2009},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:e79a002b16a02c3e16b3e50d488eaa8b60f3fe5c28b06e2d40c89cb3a8daac99","observation_id":"ecb94d70-e088-48f9-85ff-aca8c816f96a","resolution":{"observed_at":"2026-05-10T23:20:33.258796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T02:03:11.253739Z","title":"International Conference on Machine Learning (ICML) , pages=","venue":null,"work_id":"e73fd2fd-264f-4a8e-9478-dfad49e22128","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:36bcf5671a4f6708498b92cdd22e992bfd08d8de39a57aa407df4ab991ae5a57","observation_id":"da8c7e21-26a4-44d9-8fcd-85d1c68b5489","resolution":{"observed_at":"2026-05-10T23:20:33.265599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.06377","last_updated":"2021-12-19T19:23:25Z","snapshot_observed_at":"2026-07-31T02:35:46.853381Z","submitted_at":"2021-11-11T18:46:40Z","title":"Masked Autoencoders Are Scalable Vision Learners","version":3},"cited_work":{"arxiv_id":"2111.06377","doi":"10.48550/arxiv.2111.06377","metadata_source":"pith","pith_arxiv_id":"2111.06377","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Masked Autoencoders Are Scalable Vision Learners","venue":"cs.CV","work_id":"0747476e-5bd0-4596-908f-391611d9364e","year":2021},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2111.06377","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:c7776c2f2fc9071efbdaf32c2ded36d03cc46637aa5bc820d58f541b1b39b7e9","observation_id":"ba6f137f-115a-4c0f-9c18-fd65ee659570","resolution":{"observed_at":"2026-05-16T06:53:57.787532Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-10T15:38:38.318698+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-10T15:38:38.318698+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T20:44:09.179460Z","title":"2009 , publisher=","venue":null,"work_id":"75c9142a-a015-42fa-b181-87ac5c0ba487","year":2009},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:ccf5295e96626333b16803835af7d7cd27cb1c478c9c337ec294c7355d7a516f","observation_id":"8eab20aa-1df1-4fff-a0f6-48c1e6598ab5","resolution":{"observed_at":"2026-05-10T23:20:33.271777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"conference on computer vision and pattern recognition workshop , pages=","venue":null,"work_id":"a96b7e3c-26da-4c48-9c05-75bf0f281294","year":2004},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":123,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:eaa9d1be998e2714446f23e10b87467c75a2123e81b21585dcb75224ffdf087a","observation_id":"823fbbd0-89ce-4027-9d77-4e4df0d39b4b","resolution":{"observed_at":"2026-05-10T23:20:33.276208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cimpoi and S","venue":null,"work_id":"eafa9113-705b-428d-bfc0-558a0c03bbbc","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":124,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:1f43ea23ec146d75ae12cd4393db725558deee9c74c60eb3c7d5999b5d1ac5fd","observation_id":"868691b8-0914-4978-bde2-a6ffbf9a5ed5","resolution":{"observed_at":"2026-05-10T23:20:33.290469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T08:47:00.520592Z","title":null,"venue":null,"work_id":"994ce3e2-21af-4d78-9016-e57288d613c0","year":null},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:1a237e66d27e5ca6a00316b63a37eb87d4deb4296386f7b7b7800feb9aceeb82","observation_id":"db0b8a0e-8d0b-4442-a52f-fc3ec12764d5","resolution":{"observed_at":"2026-05-10T23:20:33.293151Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2010 IEEE computer society conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"0c85514a-268e-4c2b-8cf1-0b6147a11f36","year":2010},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":126,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:87905449b98a82efac6882ffc6053a71d91533483a2b7e26d9a883a8762d7348","observation_id":"1cc1e7cb-037b-4b85-b3c4-d45c773c5660","resolution":{"observed_at":"2026-05-10T23:20:33.295995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":19,"parse_uncertain":0,"unresolved":4,"verified_exact":9,"verified_fuzzy":68},"total_outbound_references":190},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 100 of 190 outbound references and 100 inbound Pith citation observations for arXiv:2410.02713."}