{"as_of":"2026-08-06T16:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0c76bde843feab92dbe6a8032038a8cba52f2d4f94671bd70fd40f9b97a5ef51","coverage":[{"denominator":100,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T01:12:46.295455Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.17798/citation-record","integrity":"/paper/2606.17798/integrity","json":"/paper/2606.17798/citation-record.json","paper":"/paper/2606.17798"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":"2412.05271","doi":"10.48550/arxiv.2412.05271","metadata_source":"pith","pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","venue":"cs.CV","work_id":"ee70bdc8-4656-4849-ada7-ce42a2278d70","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:8e938db3d0c0a881c271c16be17431f63da1ac59856f120de09a5c8ae53e2fe5","observation_id":"627e6066-a645-4ca2-b461-9f5cb92c2c27","resolution":{"observed_at":"2026-07-03T20:38:56.140145Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":"2409.12191","doi":"10.48550/arxiv.2409.12191","metadata_source":"pith","pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","venue":"cs.CV","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:71829d4484d813ae7ef013754a4172ed64bcf4303bc3dc8077c8a316c99e0e3c","observation_id":"3f98e227-1dd8-493d-b85a-463a544ec722","resolution":{"observed_at":"2026-07-03T20:38:56.173494Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":"2408.01800","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-07-10T11:37:03.161139Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","venue":"cs.CV","work_id":"0f06e436-0c76-4e3c-be5e-6168f6bc4336","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:87473c671309ac1488e4c568ec1f743ee6750bf9f1bd99ed11ad08a8f6bc775f","observation_id":"0aabb6b9-16c7-4b69-b664-000368da88b4","resolution":{"observed_at":"2026-07-03T20:38:56.208066Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"cited_work":{"arxiv_id":"2407.03320","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.03320","snapshot_observed_at":"2026-07-03T20:38:56.215264Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","venue":"cs.CV","work_id":"930005ca-91e5-4ee8-a634-1684c76d8cf9","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2407.03320","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:636d126f6ccc0c23ddf70bb1dc13d57cce23a95d7591af8e1ba0bf7885c8a539","observation_id":"0d5f52e1-f1a5-468e-80b3-7343718705bf","resolution":{"observed_at":"2026-07-03T20:38:56.216681Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":"2406.12793","doi":"10.48550/arxiv.2406.12793","metadata_source":"pith","pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","venue":"cs.CL","work_id":"de9ce5af-0d8d-4b94-9793-64968d9bc06d","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:b935d3b7134907fab16d36afbac2caf054b73ac8c6bc287febf2cc0da638e5c2","observation_id":"9d1b3b68-0735-4663-a951-812655a81338","resolution":{"observed_at":"2026-07-03T20:38:56.188794Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:eef967f11626a1a9e83e2e842af7253384085cbd8657cb776fa4dfed4d03995b","observation_id":"850ea5ce-203f-4cd4-ac50-d4befa26a7f0","resolution":{"observed_at":"2026-07-03T20:38:56.192029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":"2306.05424","doi":"10.48550/arxiv.2306.05424","metadata_source":"pith","pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","venue":"cs.CV","work_id":"51f627f4-8fae-4882-a3e9-abdf932ef27b","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:74c3768981f2b5439d813575ddc65f654f958d20e14e9f98302fa5e6a920a64c","observation_id":"b7c49bf2-115f-4904-84c1-e6acad251b41","resolution":{"observed_at":"2026-07-03T20:38:56.161743Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:cba1f37648720219c0617d9efe7e45f3bc683d5626e4f2e7c9634920d1327acc","observation_id":"8a68d7a6-8893-4eb9-a143-b4b0bfbc9d1f","resolution":{"observed_at":"2026-07-03T20:38:56.156986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Vid2seq: Large-scale pretraining of a visual language model for dense video captioning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:497f685849455601305336ef190ada3254f28340d674a92776e18338ebad1b11","observation_id":"8f3c6715-e53b-4656-b04f-e4ef8b672e7d","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.03191","last_updated":"2022-12-07T12:20:55Z","snapshot_observed_at":"2026-07-06T14:27:34.639236Z","submitted_at":"2022-12-06T18:09:49Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","version":2},"cited_work":{"arxiv_id":"2212.03191","doi":"10.48550/arxiv.2212.03191","metadata_source":"pith","pith_arxiv_id":"2212.03191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","venue":"cs.CV","work_id":"780aaeee-ac26-46b1-b6ff-64a7a624e694","year":2022},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2212.03191","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:538863a5430a439abd39645f46887231dfe71c3db700606c191ccd178c2ae4f5","observation_id":"896845f4-3d43-4c4e-b595-bd8a8b081d21","resolution":{"observed_at":"2026-07-03T20:38:56.183828Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":"2406.07476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-07-04T16:49:57.279303Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","venue":"cs.CV","work_id":"ccfc3f89-c510-45f1-8a35-ed1a56c0ae5c","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:484e3956dea8c0d78fcedae07ba5f1e19d4bbe7af09db2e63f56758c201d66b5","observation_id":"f5d290d5-32f7-4035-aa05-f1d6cad28be8","resolution":{"observed_at":"2026-07-03T20:38:56.159372Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12961","last_updated":"2025-02-27T06:09:46Z","snapshot_observed_at":"2026-07-06T19:18:17.523057Z","submitted_at":"2024-09-19T17:59:51Z","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","version":4},"cited_work":{"arxiv_id":"2409.12961","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.12961","snapshot_observed_at":"2026-07-04T13:09:50.847114Z","title":"Oryx mllm: On- demand spatial-temporal understanding at arbitrary resolution","venue":null,"work_id":"4a905ca2-7426-40db-a509-3452624d3ae5","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2409.12961","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:fd1e218636e88d75eceb3e21ba9ae7cfe6affbebb80baea4c92376801e2169d9","observation_id":"89a1a066-f3e4-47fc-9d6e-9f3370537c85","resolution":{"observed_at":"2026-07-03T20:38:56.170346Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:50d53726b0dce74a221cbc86ea58d34969067451502a63f699e1e0255aaec968","observation_id":"b81ab108-434d-43d4-8619-da982bb9e121","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:62723b041f40827807d4c79530592bcb54f3b0db3e35e1c74f5eda8febfd9e19","observation_id":"8b069b59-ddbf-4ffd-8919-b3520863a203","resolution":{"observed_at":"2026-07-03T20:38:56.143340Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.17176","last_updated":"2024-04-26T06:17:04Z","snapshot_observed_at":"2026-07-06T18:05:59.538624Z","submitted_at":"2024-04-26T06:17:04Z","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","version":1},"cited_work":{"arxiv_id":"2404.17176","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.17176","snapshot_observed_at":"2026-07-03T20:38:56.117133Z","title":"Moviechat+: Question-aware sparse memory for long video question answering","venue":null,"work_id":"e6a54789-a1e5-467c-8777-bbb1d564bd4f","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2404.17176","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:005a1ca568c0c0a8bd2af38ff000241907b6fe46cad7eeafd1bc3161597528db","observation_id":"b17d68cf-e15c-4649-9cce-ea65ece727a3","resolution":{"observed_at":"2026-07-03T20:38:56.118616Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Ma-lmm: Memory-augmented large multimodal model for long-term video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:98ba5538c645244002a7d73de4632f701e17a80aed6b94f7944ba3c14ebdb03f","observation_id":"3beb55c8-9040-4568-9f47-17f7c3fbba7c","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Longllava: Scaling multi-modal llms to 1000 images efficiently via hybrid architecture,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:aaa826e5fceea927e90a70d0602698be7d1407c638423e88ea757f877d8d58c3","observation_id":"132ece48-ee35-4cf8-acfe-229f819960e3","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2409.02889","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:39:46.752891Z","title":"Longllava: Scaling multi-modal llms to 1000 images efficiently via hybrid architecture","venue":null,"work_id":"c93e268d-1703-46de-8123-15c73f06bc0b","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:e5e9eb243d50c71f8b41911fe9a351255b1603b8d8708687f633a4b369bad241","observation_id":"5650dc60-12d8-4aa7-91b8-e87c6fd5a55a","resolution":{"observed_at":"2026-07-03T20:38:56.121280Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Longvila: Scaling long-context visual language models for long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:2c05dd0b455d87c2279f6d77eea2d68dff0270573385960cb60fc9a74b355a28","observation_id":"9cc197e0-e82a-4cbf-9f40-d22dc731d015","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-07-06T18:36:12.298104Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":"2406.16852","doi":"10.48550/arxiv.2406.16852","metadata_source":"pith","pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Long Context Transfer from Language to Vision","venue":"cs.CV","work_id":"52f1b946-568f-4819-9d8a-a87296f8852d","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:bf0915a7df5a62abaf37b27d56447342aaf3da38f3576d23d044eae2dac78606","observation_id":"b74f458d-aa20-4515-9c4f-45f2b761acae","resolution":{"observed_at":"2026-07-03T20:38:56.151751Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Videollm-online: Online video large language model for streaming video,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:f4e53fd28c00cb480c74355c77a1b93ae7c11aa8316d03de8ea4f0ab16e5e45b","observation_id":"e5176662-6b26-452f-b39e-1a184f94d697","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Videollm-mod: Efficient video-language streaming with mixture-of-depths vision computation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:3ba09fc4584c640814434cf0c11bb2d8eaa460e14d329b24ce9e48b20b89504f","observation_id":"6481aeee-4d5d-4df6-992b-464ba936c4b9","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.03663","last_updated":"2025-03-06T16:25:37Z","snapshot_observed_at":"2026-07-06T20:47:19.219786Z","submitted_at":"2025-03-05T16:52:34Z","title":"LION-FS: Fast & Slow Video-Language Thinker as Online Video Assistant","version":2},"cited_work":{"arxiv_id":"2503.03663","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.03663","snapshot_observed_at":"2026-07-03T20:38:56.133065Z","title":"Lion-fs: Fast & slow video-language thinker as online video assistant,","venue":null,"work_id":"879985b7-119b-48e5-98d7-042daf125c07","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2503.03663","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:34ddaaa2862445ab513b09e8c4503c22ef7ddd560f4c1ab7e9556c9b3e116e3f","observation_id":"99fbe1a0-8d9e-4941-9c87-0bd4597ef252","resolution":{"observed_at":"2026-07-03T20:38:56.134846Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06220","last_updated":"2025-09-07T10:23:25Z","snapshot_observed_at":"2026-07-06T20:49:09.368757Z","submitted_at":"2025-03-08T13:44:38Z","title":"StreamMind: Unlocking Full Frame Rate Streaming Video Dialogue through Event-Gated Cognition","version":3},"cited_work":{"arxiv_id":"2503.06220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06220","snapshot_observed_at":"2026-07-03T20:38:56.187640Z","title":"Stream- mind: Unlocking full frame rate streaming video dialogue through event-gated cognition","venue":null,"work_id":"4d08c82c-d8e5-44d3-9d8a-c371afea6a6e","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2503.06220","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:759ccb16be1b1f7953398d038aaea5a8d3ee197c555d1a359f076a1cbba9dcef","observation_id":"9787fc71-32b4-47b1-b609-6f718805aa53","resolution":{"observed_at":"2026-07-03T20:38:56.189303Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2411.17991","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T20:48:55.463368Z","title":"Videollm knows when to speak: Enhancing time-sensitive video comprehension with video-text duet interaction format","venue":null,"work_id":"0bc13a71-30dd-466a-974f-09324d2a7814","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:94e67ffb6228febe7d25b2e4066a24a935ca2fd227e4b67c3315067c440d8a9f","observation_id":"b495186d-efd6-406a-829c-74dc87b044fb","resolution":{"observed_at":"2026-07-03T20:48:55.465398Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Streaming long video understanding with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:daf1fe3197783451dbfb9a863b958eadc5c4a75347f2a1181529c5198b5187a4","observation_id":"2311b7f4-915f-4b60-8d5b-88630c2c589b","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Streaming dense video captioning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:bb89346d885e5fcc7806bab4e20c2f2e3f87af26054ec811803fd875cee9a3f4","observation_id":"0f4abcfa-3cb7-4b83-b26d-38db5cfda108","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-07-06T20:24:50.262766Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:738e9d69328ef7def1bda29117d97e61bcc70449fb01f06daffbef7e94db00b3","observation_id":"a9e49ad1-317e-4670-b9c3-23633883a83e","resolution":{"observed_at":"2026-07-03T20:38:56.211274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:b6fb6a9d65ce4e0917e9541c48a1fc9efe75154f1b8679a38bdc4ccf9b1b95c4","observation_id":"0500e553-c771-49b0-8cba-ee60a865a48a","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Soccernet: A scalable dataset for action spotting in soccer videos,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:29b0de79edbc7267846dc1bc2395aa572c6a4ad6276098be306a91ad0c86aa4d","observation_id":"30980586-d4d5-4533-b4f2-57eb7e146384","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.10810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T12:41:04.328023Z","title":"Svbench: A benchmark with temporal multi-turn dialogues for streaming video understanding","venue":null,"work_id":"a96af222-b449-4d3e-b31d-c2fae15db9fd","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:95d9b35a9403c24829befbd626d35ba08d96e7aaba1dde66ed30f5dfc01b033a","observation_id":"919e5318-5b27-451b-bf27-43c7421511b8","resolution":{"observed_at":"2026-07-03T20:38:56.214126Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05510","last_updated":"2025-03-27T17:40:09Z","snapshot_observed_at":"2026-07-06T20:18:58.432111Z","submitted_at":"2025-01-09T19:00:01Z","title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","version":2},"cited_work":{"arxiv_id":"2501.05510","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.05510","snapshot_observed_at":"2026-07-04T03:19:29.950974Z","title":"Ovo-bench: How far is your video-llms from real-world online video understanding?","venue":null,"work_id":"aa3866ff-d08d-457f-be75-99b6b66be18f","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2501.05510","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:588f1d80234c7155dd8814ba08170462ece0e96f177367223a9f012f33798368","observation_id":"1f0a41bd-83de-4108-83c6-2735fe6e7321","resolution":{"observed_at":"2026-07-03T20:38:56.213701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Livestar: Live streaming assistant for real-world online video understanding,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:863dd108f437feb53f4944d6a59d1e71a13db40c258603a7614eb24f7808135f","observation_id":"1bcb833d-577d-4561-af44-881459d45eae","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":"2307.09288","doi":"10.24963/ijcai.2025/706","metadata_source":"pith","pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","venue":"cs.CL","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:8125bbcce411dd97ef1f8c99ebb7602831873bcba7ae3bffc7794713703a2e7c","observation_id":"0c3c7d6c-8897-4c52-bbe5-8e1f27e12a82","resolution":{"observed_at":"2026-07-03T20:38:56.196825Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:6d34130d575fbcdcb58e7c8f6cb6d0708c4d60730c88bbbcf681e7cfc8d8c1ec","observation_id":"b06a6a4d-cdd9-4fad-9c35-2d1e87ea50c7","resolution":{"observed_at":"2026-07-03T20:38:56.199389Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:23b912e81f7ac329314ded076a97e750f22063d02313227b698175e044583e15","observation_id":"9cb03962-87e3-4920-96a9-b01fc7faf7bc","resolution":{"observed_at":"2026-07-03T20:38:56.202622Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:348e03dc7a393772afa519a96bd0e30abe610b4cc8fe5b4e48d601e0d683d737","observation_id":"2e7b8826-60c1-4f8b-b441-0ab7bcc59934","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Improving language understanding by generative pre-training,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:32f7ba069f53e1bf84db1d338e59fabffa0509e432c7df8e9736d1827377a1da","observation_id":"05ff6f68-ea22-4834-bbf0-92723e921596","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Video dataflywheel: Resolving the impossible data trinity in video-language understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:ad957cc71c772d93595f485b4c4f76ccb59f93a0e944b9b5af1f6c709abcc492","observation_id":"cd91ef08-29fb-4c34-9677-eb2dc798a317","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Object-centric rep- resentation learning for video scene understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:e871f33faabdcd4e5a79b7b29d851c19ccd60e2aa4aaf81f4d311282e245609f","observation_id":"e5671da8-4ac8-4e00-a6c0-3bd12e9e58d9","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Sharegpt4video: Improving video understanding and generation with better captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:d994d81006d94a8c7e9cc6689b01301896458ffebf70603c1b66bddbd224d2d8","observation_id":"4491d186-d388-466b-859f-fabecf21408a","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":"2404.16994","doi":"10.48550/arxiv.2404.16994","metadata_source":"pith","pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","venue":"cs.CV","work_id":"8949d4db-20f2-47c1-83a6-fcbe041b62ef","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:034402306f4aa9f8153086738b83b044686b2a5d8ffd87eec30377bafcf580dc","observation_id":"1cc93271-321e-4114-8d60-8d17cb5e6254","resolution":{"observed_at":"2026-07-03T20:38:56.205507Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Video recap: Recursive captioning of hour-long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:0020ba5c8662e78e3e278a2d03c7b05994e05b4752fa1e322ddf4c0b96e06372","observation_id":"9fecdd50-70a1-4ae9-ab49-82b7ad52c327","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15747","last_updated":"2023-11-06T12:20:33Z","snapshot_observed_at":"2026-08-06T03:16:21.214943Z","submitted_at":"2023-10-24T11:44:39Z","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","version":2},"cited_work":{"arxiv_id":"2310.15747","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15747","snapshot_observed_at":"2026-07-03T20:48:55.458276Z","title":"Large language models are temporal and causal reasoners for video question answering","venue":null,"work_id":"6942ee49-a63d-407a-98c0-fc5f721d08d4","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2310.15747","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:4c5531f8d4c4cdc952fba0b4b9dfe44a990db320e16aaf132eb624f7f6a6f670","observation_id":"42234f78-d51b-40d0-9c58-0292d2ed1089","resolution":{"observed_at":"2026-07-03T20:48:55.460863Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:0b450f7cc22d6b736dbdbd138aa06b37055c2a283a73a25b18bd3ab24407db13","observation_id":"83444394-84c3-4d72-b9d5-00147384e322","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09418","last_updated":"2024-06-13T17:59:59Z","snapshot_observed_at":"2026-07-06T18:30:32.982860Z","submitted_at":"2024-06-13T17:59:59Z","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","version":1},"cited_work":{"arxiv_id":"2406.09418","doi":"10.48550/arxiv.2406.09418","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.09418","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Videogpt+: Integrating image and video en- coders for enhanced video understanding","venue":"arXiv (Cornell University)","work_id":"b73f5b08-33ad-4759-b16b-0663cf9728b1","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.09418","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:51a468734b761a4b33e077aff3582262b7d310423e519ed40d1056e1cbaf3e78","observation_id":"5a769220-b4bb-4308-8d91-65af6e1c98b5","resolution":{"observed_at":"2026-07-03T20:38:56.200063Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Learning to answer visual questions from web videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:a39817fb36993f176c6c32967ba918f2d960f7cbe0002deeb90a4e6b78892de1","observation_id":"c968da58-0971-4b36-a17a-0caad4fe4b43","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Transformer-empowered invariant grounding for video question answering,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:87fdcb74cfa0a6d900b56349f7b8b1d6626c374871e2efdec89aa73fde636373","observation_id":"636d20b2-a71e-4f72-b293-50aa4b974c99","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Intentqa: Intent question answering in videos by cognitive context reasoning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:424f3b1c9b98a08dc9933a33b4277d517c2c6013842e50942fc9430b167e764a","observation_id":"23ca7f92-479d-4217-96f5-9a3837f06090","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Vtg-llm: Integrating timestamp knowledge into video llms for enhanced video temporal grounding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:180cc587b567d55c83f537a25740b5d967a05c1666b21a19ffb5fbb9a7e382cf","observation_id":"eeba40fe-dce8-4d29-96f5-0b6a091ba3f3","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Vtg-gpt: Tuning-free zero- shot video temporal grounding with gpt,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:7894dc46fb2debc14ee81d5ded8ce0051da4f609e4c7ba3c21d9223cd40a3584","observation_id":"1ee5b3c0-f9c4-4ee4-9339-814d4bd1a726","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10228","last_updated":"2024-03-15T11:58:18Z","snapshot_observed_at":"2026-07-06T17:45:11.343569Z","submitted_at":"2024-03-15T11:58:18Z","title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","version":1},"cited_work":{"arxiv_id":"2403.10228","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.10228","snapshot_observed_at":"2026-07-03T20:38:56.195732Z","title":"arXiv preprint arXiv:2403.10228 , year=","venue":null,"work_id":"fb2ec1aa-a3b0-43e2-9e93-2b73d3097074","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2403.10228","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:c1c702e1b1c0c33414bfe04c208ad69a6ceacdfaee9921f00c5d8dd20d3de6ee","observation_id":"bc60d46a-72b6-452b-a97e-5da54f35f31d","resolution":{"observed_at":"2026-07-03T20:38:56.197261Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Llava-next: A strong zero-shot video understanding model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:63e2f79221d2ab07a6a7c445bbf426ef50af944fe6edba37d3205161ef6de5e4","observation_id":"3e42a175-82f6-4606-bafb-acc2ecbbf98f","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":"2311.10122","doi":"10.48550/arxiv.2311.10122","metadata_source":"pith","pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","venue":"cs.CV","work_id":"e2121c51-a55e-476a-af81-7ba6970fe6cf","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:391a0c0c975a47bbbe1915f2a31091341aa4b4e16d40aa8ea312fac1b28cc030","observation_id":"523f0f51-cc18-4bc2-a16a-7727ce90f3bd","resolution":{"observed_at":"2026-07-03T20:38:56.194442Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Vila: On pre-training for visual language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:b659ef59cf7b9f3727689dd3ffbae05aa09f8e055fcca5697732ca33a8e7cb5c","observation_id":"f983d2b9-4280-42c8-bfe6-4f064420c107","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:37278e84627438bdad28f49eb031f99d90218b16aefee7efb1ed2eef77bab42c","observation_id":"39f2533c-8d43-4946-bb8c-3b45bfa48445","resolution":{"observed_at":"2026-07-03T20:38:56.186287Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Valor: Vision-audio-language omni-perception pretraining model and dataset,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:bddbad6ec5569a944ad0f024c2efb7d18c39cc12a2646261e89b84f6e1f8fe1e","observation_id":"5ee2bf18-7555-468a-92a8-09edf6b7d236","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Cap4video++: Enhancing video understanding with auxiliary captions,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:3e80869f54ec94c40947d76471e3d53a4591c731b980125486293bb4fd22cf64","observation_id":"d9b9a6b8-d576-4173-a8bb-62cfc6fa0914","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Hierarchical banzhaf interaction for general video-language representation learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:b8d1f213301b96ac6fc5fac22883e05cde595c2fb1413da6042f955cd6a80a0a","observation_id":"2a7516f1-14e1-4738-a662-9b7cb46fda77","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Video dataflywheel: Resolving the impossible data trinity in video-language understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:b3938d9d665ce63fae42d9b7e8b48f6f83a832354056765658e3e6478c415294","observation_id":"7710a38f-64b5-4172-9161-6f35811ed8d3","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.08401","last_updated":"2023-06-14T09:50:06Z","snapshot_observed_at":"2026-08-02T12:17:31.696131Z","submitted_at":"2023-06-14T09:50:06Z","title":"LiveChat: A Large-Scale Personalized Dialogue Dataset Automatically Constructed from Live Streaming","version":1},"cited_work":{"arxiv_id":"2306.08401","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.08401","snapshot_observed_at":"2026-07-03T20:38:56.192968Z","title":"Livechat: A large- scale personalized dialogue dataset automatically constructed from live streaming,","venue":null,"work_id":"e8a8b9e1-3dfd-4fb2-95c1-770e5202e0b1","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2306.08401","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:8ebdfdd8fd49a6188fa9cd0982dd03910ecd80e5503e2d86534debaf6d097f34","observation_id":"adb47ff5-f74e-4de9-87bb-1972c61e1e11","resolution":{"observed_at":"2026-07-03T20:38:56.194395Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"cited_work":{"arxiv_id":"2606.06991","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.06991","snapshot_observed_at":"2026-07-03T20:38:56.159073Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","venue":"cs.CV","work_id":"b0e2c2e3-5b6e-4821-9097-d95537244496","year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2606.06991","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:d281b451ce4468bcc5517d0d61bd8b0c9b63ee5afbf4da5f83ebed7e16bc2937","observation_id":"dd0c6591-e0ee-474c-b682-973138cd64dd","resolution":{"observed_at":"2026-07-03T20:38:56.160538Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Querystream: Advancing streaming video understanding with query-aware pruning PREPRINT, 2026 18 and proactive response,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:450b23c5d16e317904842636d213f3e939c66caa180eb90b36917697c8b366ba","observation_id":"23438b1f-c851-45af-b843-b8d276f4c6de","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01852","last_updated":"2024-01-22T03:11:15Z","snapshot_observed_at":"2026-08-02T22:12:20.187464Z","submitted_at":"2023-10-03T07:33:27Z","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","version":7},"cited_work":{"arxiv_id":"2310.01852","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01852","snapshot_observed_at":"2026-07-09T11:16:11.487266Z","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","venue":"cs.CV","work_id":"6ac9f10d-021b-4fb0-baeb-c11f5770c53d","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2310.01852","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:bfbfd1d80fd3ac882ecb750d8083aa688eb91d640742bf644e354861828ceaf6","observation_id":"265f504f-45a7-4ff8-84fd-a238ec295f6d","resolution":{"observed_at":"2026-07-03T20:38:56.172788Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:a973f4ff6b0f7282a7651510d21af1dac1b64ee584b727f8b28d8f3f98f0cd23","observation_id":"a4363b46-8f26-4c53-9b55-1e9dbf016981","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Activitynet- qa: A dataset for understanding complex web videos via question answering,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:97bcd4eb862caed8cee99c38e7b26e9aad0cbd982a436d5a8279a1efdb9cc276","observation_id":"279110ff-7009-4def-9fc2-ef645bada4bf","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.00200","last_updated":"2020-09-29T20:37:17Z","snapshot_observed_at":"2026-08-04T07:19:10.375553Z","submitted_at":"2020-05-01T03:49:26Z","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","version":2},"cited_work":{"arxiv_id":"2005.00200","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.00200","snapshot_observed_at":"2026-07-03T20:38:56.173895Z","title":"Hero: Hierarchical encoder for video+ language omni-representation pre-training","venue":null,"work_id":"d64e127a-fcfc-49d6-82f5-d45d60d42c80","year":2005},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2005.00200","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:54ec0d7f2ab4398b753b539d626d1b62d014f7c3ec2b596868aed357bcb00fdd","observation_id":"fe6319d5-aa72-41db-8405-77fa52d251b9","resolution":{"observed_at":"2026-07-03T20:38:56.175406Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Per- ception test: A diagnostic benchmark for multimodal video models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:b4fcb0d0e4317f570171d2c351d0fcbbde2ec69bbacac773a35f40e0d2bb9de7","observation_id":"a63644e3-1edd-4632-8e45-252edb43d375","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Social- iq: A question answering benchmark for artificial social intelligence,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:2db9ed20eabd0f7327b4790688ad77cd1e60867a36e80ff9979a07461996f200","observation_id":"a5450594-c4d0-4123-a469-43385212284b","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Video question answering via gradually refined attention over appearance and motion,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:6d1c2accbee1f166f38c97fde349fdf448bc3f0ca44341e5b54d7980022638dc","observation_id":"e35b6132-bdf1-40c2-9e30-b80a8b5ec0b7","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.01696","last_updated":"2019-05-07T21:34:05Z","snapshot_observed_at":"2026-07-06T06:59:26.080263Z","submitted_at":"2018-09-05T19:14:11Z","title":"TVQA: Localized, Compositional Video Question Answering","version":2},"cited_work":{"arxiv_id":"1809.01696","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.01696","snapshot_observed_at":"2026-07-04T06:39:37.647628Z","title":"TVQA: Localized, Compositional Video Question Answering","venue":"cs.CL","work_id":"c70dc07b-3c2d-40aa-bc37-27edb4560a59","year":2018},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/1809.01696","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:c924eb40a7057037b49cdfeec637bd8ca4248fcdcc96815738e2151dd1cd6784","observation_id":"a9628186-9cfb-4c4f-89c7-83640a9e721b","resolution":{"observed_at":"2026-07-03T20:38:56.179069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:a055f427e6b13f0c21058bf76a4c53c68ac5954964a6b2e503fbd239432ea29c","observation_id":"e51cedfa-774b-463a-8ff6-7d83cff31736","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Moviechat: From dense token to sparse memory for long video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:588e9b7eef15315373206465e7317c67b42a9d20946c177e56378059a6af644c","observation_id":"b076ea72-423e-4376-9fac-1e88a2ad387b","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-05T10:34:24.268925Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"cited_work":{"arxiv_id":"2406.08035","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.08035","snapshot_observed_at":"2026-07-04T16:09:57.146670Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","venue":"cs.CV","work_id":"e9bfdf40-cd28-4d57-98b8-31b7e97f9f2f","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.08035","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:ddba0b7eeb3e1332921f25edc9c2912637a079e1b0554f8f7a4f33b0d12c2185","observation_id":"2302a3cc-dbea-4b04-9bf2-0a57450b0d54","resolution":{"observed_at":"2026-07-03T20:38:56.186478Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Tgif-qa: Toward spatio- temporal reasoning in visual question answering,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:49ee3dfa06732e75748ab12c85449545b33f22cba78bcac6a35fb9b936f450b5","observation_id":"65b7985b-bcdd-44be-98b5-5707d76db13e","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Moviechat+: Question-aware sparse memory for long video question answering,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:7d08cf2cafa1a25fbf4170730c65991ecb107e206462192e32bfbc9695a20a99","observation_id":"c505d8cd-bd90-40a0-b523-fedca3bcb250","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Momentor++: Advancing video large language models with fine-grained long video reasoning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:8776853bc389b5ab5f0569bdae661b8a3b61b6369b8adead9994765f5139c532","observation_id":"6632422c-0ed1-43c9-a40b-d79d16fbdaa0","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Selongvlm: Empowering long video language models with self-corrective clip selection,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:6ade1294fef910c606f600970242506fa800c7246e7a0da4ad5149e6176ad60d","observation_id":"061d204b-0f3d-4a81-8946-65a586cc7203","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Ego-r1: Agentic chain-of-tool-thought for ultra- long egocentric video reasoning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:f2c0ac7330e520023d693351320fc44bb8adbb1b17b186c2374047f532d03329","observation_id":"4b419074-bb84-415c-858d-293a87258cff","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Hier-egopack: Hierarchical egocentric video understanding with diverse task perspectives,","venue":null,"work_id":null,"year":1917},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:59e7e7efb8243ab62576154688cd92875589b4f7e7b4f73d7277541b68949240","observation_id":"e5494989-9c1a-43e9-a108-5219e788fd33","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"cited_work":{"arxiv_id":"2606.07032","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.07032","snapshot_observed_at":"2026-07-03T20:38:56.147115Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","venue":"cs.CV","work_id":"de29a1b9-7e72-469a-aa47-64ddd3ed1710","year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2606.07032","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:972a834571bfdf7c6bf98c7e418d4d3eacb0c820cca4792256e176b6881a6f63","observation_id":"3062e5b6-9dfd-4c90-876c-a06af031dc4d","resolution":{"observed_at":"2026-07-03T20:38:56.149055Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Lvos: A benchmark for large-scale long-term video object segmentation,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:c39426c9a103eae14bd8c98b58ebd750b2c636149a6b0cac602832afefb74b2a","observation_id":"9957af5d-bb0f-4d25-a5f3-838265ce4a72","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Wild- video: Benchmarking lmms for understanding video-language interac- tion,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:1c0034789775ef6b14473c791b234a893987becca645ae636b1d73dce9cea956","observation_id":"fd2e873d-1e55-48ee-9dca-7ed7284483c2","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"A survey on video temporal grounding with multimodal large language model,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:9e4fc74cd8fa9891a04435b7e1cde6c0dc309eb83b06090622a64086e402476f","observation_id":"abf91a47-c61f-4aac-a7c3-05e74b05baf9","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Timechat-online: 80% visual tokens are naturally redundant in streaming videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:0fb4981e227554c8e49f067ce875755647e8b7aa2333fb43b9743ca8a55195d1","observation_id":"3a690608-de23-42a1-b9d5-69b565545202","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Videollamb: Long streaming video understanding with recurrent memory bridges,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:061cc2b890ec32c58403741daaccc1e60c2aaa68fcd494ba3144cbb3bbe26791","observation_id":"fd15ff1d-8407-4534-bb3b-6eae7f33d5d1","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Evaluation by moments: Past and future,","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:cec1a9f9d2787e56494c46d500df49ce403b92ade2a0ccfe7480264ff1eb1c30","observation_id":"0abc7edd-b519-4f5f-9ac9-331232039582","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Ldre: Llm-based diver- gent reasoning and ensemble for zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:877652ed9c6763693d503dd23e411ecaee2745e4f715de730fa86adf8fdf0040","observation_id":"d2ec4503-77ed-43ad-a76a-877f36af7609","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Seman- tic editing increment benefits zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:a3cc18891622a057a72ed9c62befd684acdc68e304b89516f5ed2d394200e29c","observation_id":"705236a4-f2cf-4dce-9ca9-6805022915cc","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:c839d42d9fa18286363405b2a5b54c546e8e89c075daea563d2753f91ae6f9bc","observation_id":"8513b66c-7f58-4cce-a5b5-b0fb7094c7be","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Streamingcot: A dataset for temporal dynamics and multimodal chain- of-thought reasoning in streaming videoqa,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:413c1e10bd470efba9eef3ed35462fc7f55e6bcfee554a0b67840abebb24493b","observation_id":"f0186bb1-275d-4156-b913-c06e03da4361","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":"2312.10300","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-07-03T20:38:56.149993Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":"32c962fe-1151-47f5-b3eb-dbbe2aba6841","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:068d96993540eef0b0a469dd194b96870cff17c813ecd16eda2e6910081c9efb","observation_id":"cabb546d-af1a-46f3-8abe-4ff690e6b441","resolution":{"observed_at":"2026-07-03T20:38:56.151854Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:70e5ea961e33b1f108fb8c65fdffde5f4a4a356af841f767892d86b342999ecd","observation_id":"40caaa14-d8ec-4074-a426-ddab0dadf133","resolution":{"observed_at":"2026-07-03T20:38:56.157736Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17297","last_updated":"2024-03-26T00:53:24Z","snapshot_observed_at":"2026-08-02T11:10:24.263044Z","submitted_at":"2024-03-26T00:53:24Z","title":"InternLM2 Technical Report","version":1},"cited_work":{"arxiv_id":"2403.17297","doi":"10.48550/arxiv.2403.17297","metadata_source":"pith","pith_arxiv_id":"2403.17297","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternLM2 Technical Report","venue":"cs.CL","work_id":"dfa13e0e-1c3c-4fb6-943d-a19945bacdbe","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2403.17297","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:4d2f0f15b933786c626a356f52f1a652c3ce0305684ef2cd0fcdf78bd7edad97","observation_id":"bb3f9e80-9305-4707-9787-b669bfae4b17","resolution":{"observed_at":"2026-07-03T20:38:56.181360Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17421","last_updated":"2023-10-11T05:07:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-29T17:34:51Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","version":2},"cited_work":{"arxiv_id":"2309.17421","doi":"10.48550/arxiv.2309.17421","metadata_source":"pith","pith_arxiv_id":"2309.17421","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","venue":"cs.CV","work_id":"344e9dbe-1d9b-4992-a4f1-9bc649978f46","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2309.17421","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:4c967b96720ab80ab952166811dbfa40e221e177af845d0961b78a5cfd496073","observation_id":"1138aca4-a08e-4f88-8c33-79fdb5d1b957","resolution":{"observed_at":"2026-07-03T20:38:56.129188Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:ccf7d95ae55ac5e494d551c88a8d35e51705a6642705827d1ab0a8e045c85156","observation_id":"0f42cb28-47d0-458d-933f-f909572168a9","resolution":{"observed_at":"2026-07-03T20:38:56.137610Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:22ca68df5df06c2640a952880c2ddcef081218163b29d4e08cfabef793d022af","observation_id":"5ec331c7-0732-4199-b0e4-6caacad6680a","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"G-eval: Nlg evaluation using gpt-4 with better human alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:3f4a12b9efe2b04b25621b9819900bd3e6a8376fe374e874c625e96f8a649045","observation_id":"0cd7f72c-2c80-4c72-924f-9376be71d0e2","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:44ce3365e55ab1fec9f66e684c89f073181d43ca81316e882e8afbbe705121bc","observation_id":"8bca9ac5-1dbd-4e0f-b11c-555f3302df39","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:12:46.295455Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:5385853ed2719722aa12bd26765e7322b626f089c0851e527c551ba20d94685d","observation_id":"9d01f371-dd3b-4133-b7e5-095cbc80e3d0","resolution":{"observed_at":"2026-06-27T01:12:46.295455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":58,"verified_exact":40,"verified_fuzzy":0},"total_outbound_references":100},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 100 of 100 outbound references and 0 inbound Pith citation observations for arXiv:2606.17798."}