{"as_of":"2026-08-12T04:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1c40988cd21810612cc1be08bdf4efa9809d85b832e9155a8d8991b2d9766f80","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":7,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":7,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:44:50.317635Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:29:44.913681Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-08-11T13:44:50.317635Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12833","last_updated":"2025-05-28T09:22:24Z","snapshot_observed_at":"2026-08-11T17:45:15.774121Z","submitted_at":"2024-12-17T11:54:47Z","title":"FocusChat: Text-guided Long Video Understanding via Spatiotemporal Information Filtering","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T13:44:50.317635Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2412.12833"},"observation_digest":"sha256:d534f55f8bf0d17f0a297427fca558c7ec8c620d0e026b64949791e104e1b407","observation_id":"b4cf9ec3-f3d7-4196-b36c-00a50127b88e","resolution":{"observed_at":"2026-08-11T13:44:50.317635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-08-06T15:53:32.306216Z","title":"Ranasinghe, X","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14784","last_updated":"2025-08-18T09:06:46Z","snapshot_observed_at":"2026-08-09T23:46:47.099823Z","submitted_at":"2025-07-20T01:57:00Z","title":"LeAdQA: LLM-Driven Context-Aware Temporal Grounding for Video Question Answering","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T15:53:32.306216Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2507.14784"},"observation_digest":"sha256:b67f3895fda1903b1248759a3d0dc08f00c2a4af8c67b83ba2b8d9d4ad4429b4","observation_id":"d6ee249b-7f99-4517-b17d-92e41ccbddc6","resolution":{"observed_at":"2026-08-06T15:53:32.306216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2604.02891","last_updated":"2026-04-03T09:00:38Z","snapshot_observed_at":"2026-08-11T16:19:37.738972Z","submitted_at":"2026-04-03T09:00:38Z","title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T20:40:41.380829Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2604.02891"},"observation_digest":"sha256:f08e7511a4a79aea81f05cfd13b985d5a92cee0bb158782ed1a42bddc16eb7eb","observation_id":"c930cb75-2276-44ae-adab-2cf010a1b749","resolution":{"observed_at":"2026-05-13T20:43:14.871880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2604.14692","last_updated":"2026-05-15T12:00:53Z","snapshot_observed_at":"2026-08-11T16:10:49.896942Z","submitted_at":"2026-04-16T06:50:20Z","title":"Chain-of-Glimpse: Search-Guided Progressive Object-Grounded Reasoning for Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T12:03:09.408019Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2604.14692"},"observation_digest":"sha256:0ce8176a03de3b209b1c281dab5f73e00dd7e5135c4c06908d836c65ca7ecd80","observation_id":"aa8772c6-d631-472d-b460-9edf5730b0a3","resolution":{"observed_at":"2026-05-10T12:05:22.177953Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2604.14692","last_updated":"2026-05-15T12:00:53Z","snapshot_observed_at":"2026-08-11T16:10:49.896942Z","submitted_at":"2026-04-16T06:50:20Z","title":"Chain-of-Glimpse: Search-Guided Progressive Object-Grounded Reasoning for Video Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-19T17:34:10.344111Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2604.14692"},"observation_digest":"sha256:6b816aa9ea4165127dae379fa9fcab04c1cb81c1da27f74278b1d4d12b119162","observation_id":"ac4c87ab-087c-4c2d-8458-7591a1866ac2","resolution":{"observed_at":"2026-05-19T17:37:41.750865Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2605.08974","last_updated":"2026-05-09T14:32:36Z","snapshot_observed_at":"2026-08-11T16:09:05.826725Z","submitted_at":"2026-05-09T14:32:36Z","title":"Tracking the Truth: Object-Centric Spatio-Temporal Monitoring for Video Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T02:31:40.463891Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2605.08974"},"observation_digest":"sha256:cdf4e4cbd083cae9a11466b09f5e22db24b0434ecff97d5bd34922238af1dafa","observation_id":"9f09208b-d686-492b-96d3-8d3ef94bd99a","resolution":{"observed_at":"2026-05-12T07:36:31.182712Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:e65c1463f1c5171a2ca40bb06637ac85db5ad7d4287f95befdc3ac8c71d0ca9c","observation_id":"0a58456e-931f-4ce1-a837-dae08856644d","resolution":{"observed_at":"2026-07-04T10:29:44.915306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2403.16998/citation-record","integrity":"/paper/2403.16998/integrity","json":"/paper/2403.16998/citation-record.json","paper":"/paper/2403.16998"},"outbound":[],"paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 7 inbound Pith citation observations for arXiv:2403.16998."}