{"as_of":"2026-08-10T03:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:172b194f564e5f4105f0ce8da02c93fdee23280c94ff75dc4943279927f4537f","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T21:21:45.795412Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-25T07:55:33.235804Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-09T21:21:45.795412Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.19098","last_updated":"2025-05-19T10:18:07Z","snapshot_observed_at":"2026-08-09T21:15:39.483633Z","submitted_at":"2025-01-31T12:45:46Z","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-09T21:21:45.795412Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2501.19098"},"observation_digest":"sha256:40c7e687e3820ac17f5d8ca64f9ca57ae3134e0f30839a5c214530976c3293c6","observation_id":"66e6dcd5-d004-49f6-b368-7eaf9ffc69f5","resolution":{"observed_at":"2026-08-09T21:21:45.795412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":"2405.19209","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"e68024d4-8bd3-4e50-a45b-d0d58d446dba","year":2025},"citing_paper":{"arxiv_id":"2504.09583","last_updated":"2025-04-13T14:06:50Z","snapshot_observed_at":"2026-07-06T21:08:37.884789Z","submitted_at":"2025-04-13T14:06:50Z","title":"AirVista-II: An Agentic System for Embodied UAVs Toward Dynamic Scene Semantic Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-25T07:51:33.652470Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2504.09583"},"observation_digest":"sha256:4252ea839975d6c7f8da50d3eb9f4d49424efcf7eb1a551809bdc574c45d8572","observation_id":"497456f7-0595-408d-ab9a-510bbceffe90","resolution":{"observed_at":"2026-05-25T07:55:33.239119Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T11:59:13.027251Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-07T11:53:14.950396Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.027251Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:51778e7569bd0c8ffb766f96920ace1e1902d51cc4dc28d0f00ffe3cda6eb07a","observation_id":"2fc4f736-9708-4578-9ce3-60917c7ecfe6","resolution":{"observed_at":"2026-08-07T11:59:13.027251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T11:11:41.249261Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05395","last_updated":"2025-09-02T17:50:58Z","snapshot_observed_at":"2026-08-09T14:06:01.941872Z","submitted_at":"2025-06-03T19:44:49Z","title":"TriPSS: A Tri-Modal Keyframe Extraction Framework Using Perceptual, Structural, and Semantic Representations","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T11:11:41.249261Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.05395"},"observation_digest":"sha256:cec99b402b3eaf2c9c09a1f46e702e5568fc1a324884b04eb216b68d1461940c","observation_id":"5c494f6b-4435-49f1-9976-d8d62903bfc5","resolution":{"observed_at":"2026-08-07T11:11:41.249261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T06:05:31.825064Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos.arXiv preprint arXiv:2405.19209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06144","last_updated":"2025-06-06T15:02:30Z","snapshot_observed_at":"2026-08-09T09:05:23.257309Z","submitted_at":"2025-06-06T15:02:30Z","title":"CLaMR: Contextualized Late-Interaction for Multimodal Content Retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T06:05:31.825064Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.06144"},"observation_digest":"sha256:eabebbabed82de786b416e86778a45db3dc917b9f8b0976b31ba3371e8389be8","observation_id":"bae2896b-f6e0-4548-b8bc-34a2e87c3ace","resolution":{"observed_at":"2026-08-07T06:05:31.825064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T06:00:56.990818Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos.arXiv preprint arXiv:2405.19209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06275","last_updated":"2025-06-06T17:58:36Z","snapshot_observed_at":"2026-08-07T21:48:03.079145Z","submitted_at":"2025-06-06T17:58:36Z","title":"Movie Facts and Fibs (MF$^2$): A Benchmark for Long Movie Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:56.990818Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.06275"},"observation_digest":"sha256:f8b4f751a9c35b8ef94e52f040a2fbd2081036f981bbeb04766619e402f60432","observation_id":"bafa5580-7d0f-462f-a507-2c1989dff989","resolution":{"observed_at":"2026-08-07T06:00:56.990818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T05:49:53.428376Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07016","last_updated":"2025-06-13T19:05:47Z","snapshot_observed_at":"2026-08-08T12:14:39.714543Z","submitted_at":"2025-06-08T06:34:29Z","title":"MAGNET: A Multi-agent Framework for Finding Audio-Visual Needles by Reasoning over Multi-Video Haystacks","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T05:49:53.428376Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.07016"},"observation_digest":"sha256:309d2a9b3e259617817e65bdd71464c213b02bbe92370e4d154558e398ac09ea","observation_id":"28f45b09-2bf9-437a-841d-7b2e7489b3af","resolution":{"observed_at":"2026-08-07T05:49:53.428376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T04:54:24.495885Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos.arXiv preprint arXiv:2405.19209,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09445","last_updated":"2025-06-11T06:52:31Z","snapshot_observed_at":"2026-08-07T04:45:44.440916Z","submitted_at":"2025-06-11T06:52:31Z","title":"TOGA: Temporally Grounded Open-Ended Video QA with Weak Supervision","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T04:54:24.495885Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.09445"},"observation_digest":"sha256:b7342fbc391177002d92edd850d002132877e8e8b39b91b4ee1d4f433c0a3204","observation_id":"392bf377-9646-4ff1-87ec-15f8401b57b1","resolution":{"observed_at":"2026-08-07T04:54:24.495885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T00:34:36.343155Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos.arXiv preprint arXiv:2405.19209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13654","last_updated":"2025-06-16T16:17:08Z","snapshot_observed_at":"2026-08-07T12:51:07.988679Z","submitted_at":"2025-06-16T16:17:08Z","title":"Ego-R1: Chain-of-Tool-Thought for Ultra-Long Egocentric Video Reasoning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T00:34:36.343155Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.13654"},"observation_digest":"sha256:3e88ce4e146e988ea57f146109d3be0750e1c5d9dad74e354348be568f7ac974","observation_id":"fd24e478-93f4-46ad-87fa-1ca87619db4e","resolution":{"observed_at":"2026-08-07T00:34:36.343155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-06T22:20:13.236841Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21891","last_updated":"2025-06-27T04:05:12Z","snapshot_observed_at":"2026-08-06T22:14:20.472000Z","submitted_at":"2025-06-27T04:05:12Z","title":"DIVE: Deep-search Iterative Video Exploration A Technical Report for the CVRR Challenge at CVPR 2025","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:20:13.236841Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.21891"},"observation_digest":"sha256:7da6a9bd92da47a5a3fdb4afc57e325a7db7ce3e9a3b07574e440a38db000ce7","observation_id":"283646a8-8b70-457c-b779-f6eb04f50ef7","resolution":{"observed_at":"2026-08-06T22:20:13.236841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-06T21:06:26.787020Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02001","last_updated":"2025-07-01T18:39:26Z","snapshot_observed_at":"2026-08-09T02:09:40.214145Z","submitted_at":"2025-07-01T18:39:26Z","title":"Temporal Chain of Thought: Long-Video Understanding by Thinking in Frames","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T21:06:26.787020Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2507.02001"},"observation_digest":"sha256:fd3c5a2c6bdc8c3672b22e4cf53264443c97bf2d9c7938e8d48b020d3d67a39e","observation_id":"f1cc9b77-687e-4840-91f8-69b982c35dc5","resolution":{"observed_at":"2026-08-06T21:06:26.787020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-06T18:10:14.223529Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09068","last_updated":"2025-07-23T13:06:44Z","snapshot_observed_at":"2026-08-07T05:26:16.655495Z","submitted_at":"2025-07-11T23:07:04Z","title":"Infinite Video Understanding","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T18:10:14.223529Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2507.09068"},"observation_digest":"sha256:f6313190c7e67be09364397b8a5b56dd557afbac39b9e9c888a177b8aee2584a","observation_id":"5d65b882-b812-4eb2-9817-d7922d9a8c0b","resolution":{"observed_at":"2026-08-06T18:10:14.223529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-06T15:53:34.166522Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14784","last_updated":"2025-08-18T09:06:46Z","snapshot_observed_at":"2026-08-09T23:46:47.099823Z","submitted_at":"2025-07-20T01:57:00Z","title":"LeAdQA: LLM-Driven Context-Aware Temporal Grounding for Video Question Answering","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T15:53:34.166522Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2507.14784"},"observation_digest":"sha256:298f5b5c230300e7f29688a7dd4f30b9c4a1d446f31a0072fc904609f5a87bdd","observation_id":"315dcd01-b057-4bad-816e-2f20f409bed2","resolution":{"observed_at":"2026-08-06T15:53:34.166522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-05T16:05:18.756354Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19024","last_updated":"2025-08-26T13:42:48Z","snapshot_observed_at":"2026-08-08T00:07:02.060969Z","submitted_at":"2025-08-26T13:42:48Z","title":"ProPy: Building Interactive Prompt Pyramids upon CLIP for Partially Relevant Video Retrieval","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T16:05:18.756354Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2508.19024"},"observation_digest":"sha256:32dc9c73fc9da0f522e5bd4a06673c4e22b5b62a52fa2a9eaed37b105433c213","observation_id":"8aa244f2-782b-4304-88de-19c590604bd3","resolution":{"observed_at":"2026-08-05T16:05:18.756354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-04T21:27:35.899881Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07680","last_updated":"2025-09-09T17:59:39Z","snapshot_observed_at":"2026-08-09T17:44:26.515281Z","submitted_at":"2025-09-09T17:59:39Z","title":"CAViAR: Critic-Augmented Video Agentic Reasoning","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-04T21:27:35.899881Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2509.07680"},"observation_digest":"sha256:7f8d55ee18d3e420b794d78a70605a1eedad5082a780472c2737a90b2b27d17b","observation_id":"d7f41027-5e6f-4fd2-b83f-dcac4abe59d1","resolution":{"observed_at":"2026-08-04T21:27:35.899881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-03T18:22:31.365284Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos.arXiv preprint arXiv:2405.19209,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.05774","last_updated":"2026-06-04T07:27:00Z","snapshot_observed_at":"2026-08-03T18:22:22.611333Z","submitted_at":"2025-12-05T15:03:48Z","title":"Active Video Perception: Iterative Evidence Seeking for Agentic Long Video Understanding","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-03T18:22:31.365284Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2512.05774"},"observation_digest":"sha256:522f3e7053b4da740626536e9192e4a832a6b5ee8b04428fc7424e1d3ea43e63","observation_id":"ecc2491c-45b2-4cab-bf74-e1e0057f03fa","resolution":{"observed_at":"2026-08-03T18:22:31.365284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":"2405.19209","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"e68024d4-8bd3-4e50-a45b-d0d58d446dba","year":2025},"citing_paper":{"arxiv_id":"2601.14724","last_updated":"2026-05-07T12:10:26Z","snapshot_observed_at":"2026-08-05T03:02:47.238051Z","submitted_at":"2026-01-21T07:26:15Z","title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:04.564442Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2601.14724"},"observation_digest":"sha256:eb88ecc167a601638d9b256e0c08e35d39ba9083db8bd592875d50c2867d7077","observation_id":"5f1de1df-84e0-4e28-bce3-6e91b5a6de45","resolution":{"observed_at":"2026-05-16T12:57:53.878322Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-02T23:33:14.887050Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos.arXiv preprint arXiv:2405.19209,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.13602","last_updated":"2026-05-30T00:19:02Z","snapshot_observed_at":"2026-08-02T23:33:07.026705Z","submitted_at":"2026-02-14T04:52:11Z","title":"Towards Sparse Video Understanding and Reasoning","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-02T23:33:14.887050Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2602.13602"},"observation_digest":"sha256:bb4cc077341f93a67d33b94b6b3b885faa3229afd2a22c4d20a761e532b49b9b","observation_id":"adbb7cc1-15e2-4710-80ed-7eb32774ec8f","resolution":{"observed_at":"2026-08-02T23:33:14.887050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":"2405.19209","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"e68024d4-8bd3-4e50-a45b-d0d58d446dba","year":2025},"citing_paper":{"arxiv_id":"2604.02891","last_updated":"2026-04-03T09:00:38Z","snapshot_observed_at":"2026-07-06T22:52:10.923214Z","submitted_at":"2026-04-03T09:00:38Z","title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T20:40:41.380829Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2604.02891"},"observation_digest":"sha256:40fe4e1c05d47295daac531c81c53ca71ef0c656f6736bdb165d2be290ad1ebf","observation_id":"23bca80f-f2ec-488a-9935-5b6b2802f143","resolution":{"observed_at":"2026-05-13T20:43:14.883302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":"2405.19209","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"e68024d4-8bd3-4e50-a45b-d0d58d446dba","year":2025},"citing_paper":{"arxiv_id":"2604.05079","last_updated":"2026-04-06T18:30:50Z","snapshot_observed_at":"2026-08-02T23:48:40.440218Z","submitted_at":"2026-04-06T18:30:50Z","title":"SVAgent: Storyline-Guided Long Video Understanding via Cross-Modal Multi-Agent Collaboration","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T20:20:08.590407Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2604.05079"},"observation_digest":"sha256:aae8799f4d5560a632eaa9ea6ac0be83f1886d9b00c77f4b843e422bd9096e3e","observation_id":"7ffd72af-016d-45fa-8211-2631d70216db","resolution":{"observed_at":"2026-05-10T22:00:48.953290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":"2405.19209","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"e68024d4-8bd3-4e50-a45b-d0d58d446dba","year":2025},"citing_paper":{"arxiv_id":"2604.14149","last_updated":"2026-04-16T15:48:38Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:59:52Z","title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-10T13:28:58.920442Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2604.14149"},"observation_digest":"sha256:82547332abae73e9962648ef19a6e77a0ec3cda3c7a2e2c14e3d9579c50cf603","observation_id":"d0820b5f-edbd-4f18-a35e-41b19a32e8bc","resolution":{"observed_at":"2026-05-10T13:30:26.614514Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":"2405.19209","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"e68024d4-8bd3-4e50-a45b-d0d58d446dba","year":2025},"citing_paper":{"arxiv_id":"2605.01662","last_updated":"2026-05-03T01:30:29Z","snapshot_observed_at":"2026-07-06T23:14:52.417213Z","submitted_at":"2026-05-03T01:30:29Z","title":"Video Active Perception: Effective Inference-Time Long-Form Video Understanding with Vision-Language Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-08T19:17:20.745425Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2605.01662"},"observation_digest":"sha256:6af6b0ee536a752274d42f84f5aaa3eb5d26452b0db2ae115debf1b595f666be","observation_id":"01fe7a43-bdd3-4e21-a48d-92919b0efde8","resolution":{"observed_at":"2026-05-09T05:55:31.687909Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-07-14T13:38:16.395135Z","title":"VideoTree : Adaptive tree-based video representation for LLM reasoning on long videos, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.10190","last_updated":"2026-07-11T08:06:04Z","snapshot_observed_at":"2026-08-08T06:11:49.761593Z","submitted_at":"2026-07-11T08:06:04Z","title":"PhysMRV: Physical Memory Retrieval and Verification for Physics Plausibility Reasoning","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-07-14T13:38:16.395135Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2607.10190"},"observation_digest":"sha256:da8536716f3e6a6e842f9b5bac1dd37dd96a1efb42928566a29768a89fefbcc5","observation_id":"d6fa54ed-9f06-42b7-bc43-25a5d061f534","resolution":{"observed_at":"2026-07-14T13:38:16.395135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2405.19209/citation-record","integrity":"/paper/2405.19209/integrity","json":"/paper/2405.19209/citation-record.json","paper":"/paper/2405.19209"},"outbound":[],"paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-05T17:33:53.862042Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2405.19209."}