{"as_of":"2026-08-08T21:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:db045b2b3b02c12828ef1e9d9e27ef6a7f0cfa7ac69d27b5c57de79558d5d2cd","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":7,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":7,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:32:51.011685Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T07:56:47.335501Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05185","snapshot_observed_at":"2026-08-07T12:32:51.011685Z","title":"Linvt: Empower your image- level large language model to understand videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24329","last_updated":"2025-07-31T03:03:18Z","snapshot_observed_at":"2026-08-07T12:22:55.467747Z","submitted_at":"2025-05-30T08:10:18Z","title":"DisTime: Distribution-based Time Representation for Video Large Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:51.011685Z"},"links":{"cited_paper":"/paper/2412.05185","citing_paper":"/paper/2505.24329"},"observation_digest":"sha256:ca1913835ec149f9e431bde36a4a270508a8eb22be542cfbd58fc8bc2e01b1dc","observation_id":"e2bdee07-8166-41f5-a9a7-78f04a2aaf8d","resolution":{"observed_at":"2026-08-07T12:32:51.011685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05185","snapshot_observed_at":"2026-08-07T11:59:06.759496Z","title":"Linvt: Empower your image-level large language model to understand videos, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00993","last_updated":"2025-06-01T12:49:39Z","snapshot_observed_at":"2026-08-08T18:44:33.509277Z","submitted_at":"2025-06-01T12:49:39Z","title":"FlexSelect: Flexible Token Selection for Efficient Long Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:06.759496Z"},"links":{"cited_paper":"/paper/2412.05185","citing_paper":"/paper/2506.00993"},"observation_digest":"sha256:eeea32cc9b22810dccc987ec4802cd3f58325f263039c55ff1f5c63d153e7afd","observation_id":"7a28d731-d390-4e5c-9c8d-1465ae651bdd","resolution":{"observed_at":"2026-08-07T11:59:06.759496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05185","snapshot_observed_at":"2026-08-06T21:24:45.561143Z","title":"arXiv preprint arXiv:2412.05185 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00316","last_updated":"2025-07-02T01:08:41Z","snapshot_observed_at":"2026-08-06T21:17:00.714769Z","submitted_at":"2025-06-30T23:14:49Z","title":"${\\mu}^2$Tokenizer: Differentiable Multi-Scale Multi-Modal Tokenizer for Radiology Report Generation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:24:45.561143Z"},"links":{"cited_paper":"/paper/2412.05185","citing_paper":"/paper/2507.00316"},"observation_digest":"sha256:49c23e334823cd63c7409696274a5bddfb2a379a0be72f4d1be3a927efe3f271","observation_id":"47912e91-9c6a-4594-8839-40ba677a6da4","resolution":{"observed_at":"2026-08-06T21:24:45.561143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05185","snapshot_observed_at":"2026-08-06T17:25:02.720820Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10963","last_updated":"2025-07-15T03:58:28Z","snapshot_observed_at":"2026-08-06T17:17:47.417758Z","submitted_at":"2025-07-15T03:58:28Z","title":"AROMA: Mixed-Initiative AI Assistance for Non-Visual Cooking by Grounding Multi-modal Information Between Reality and Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:25:02.720820Z"},"links":{"cited_paper":"/paper/2412.05185","citing_paper":"/paper/2507.10963"},"observation_digest":"sha256:fd58abbfd0d639867aa47626bbd30226d943f493c468dfe01b291ca367776646","observation_id":"e555be81-c022-47df-b305-a1faed9b8205","resolution":{"observed_at":"2026-08-06T17:25:02.720820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","version":2},"cited_work":{"arxiv_id":"2412.05185","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.05185","snapshot_observed_at":"2026-07-02T07:56:47.335501Z","title":"Linvt: Empower your image- level large language model to understand videos","venue":null,"work_id":"677837f8-97cd-47ce-b789-22d88af9446f","year":2024},"citing_paper":{"arxiv_id":"2602.20913","last_updated":"2026-04-15T16:09:22Z","snapshot_observed_at":"2026-08-02T12:38:41.181077Z","submitted_at":"2026-02-24T13:49:47Z","title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T20:01:31.129959Z"},"links":{"cited_paper":"/paper/2412.05185","citing_paper":"/paper/2602.20913"},"observation_digest":"sha256:d2ab19c8acdac1813e8caf81eaeb46387e0f170293f4e4433fe973a81622344f","observation_id":"9b54c22b-409b-4082-828b-5cac0d08ff71","resolution":{"observed_at":"2026-05-15T20:01:33.511427Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","version":2},"cited_work":{"arxiv_id":"2412.05185","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.05185","snapshot_observed_at":"2026-07-02T07:56:47.335501Z","title":"Linvt: Empower your image- level large language model to understand videos","venue":null,"work_id":"677837f8-97cd-47ce-b789-22d88af9446f","year":2024},"citing_paper":{"arxiv_id":"2606.06532","last_updated":"2026-06-03T17:47:49Z","snapshot_observed_at":"2026-08-07T12:55:44.925060Z","submitted_at":"2026-06-03T17:47:49Z","title":"GOPAgen: Motion-Aware and Efficient Agentic Long-Video Understanding with Structural Memory and Hierarchical Reasoning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T06:33:32.090913Z"},"links":{"cited_paper":"/paper/2412.05185","citing_paper":"/paper/2606.06532"},"observation_digest":"sha256:6e13698503f2a4dd8acb5f454e06818369a79d9e1045681c86ec436f21d5b07d","observation_id":"944fc3e0-f602-4b05-be63-7a75b5014222","resolution":{"observed_at":"2026-07-02T07:56:47.337270Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","version":2},"cited_work":{"arxiv_id":"2412.05185","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.05185","snapshot_observed_at":"2026-07-02T07:56:47.335501Z","title":"Linvt: Empower your image- level large language model to understand videos","venue":null,"work_id":"677837f8-97cd-47ce-b789-22d88af9446f","year":2024},"citing_paper":{"arxiv_id":"2606.29445","last_updated":"2026-06-28T15:11:19Z","snapshot_observed_at":"2026-08-02T18:05:44.581086Z","submitted_at":"2026-06-28T15:11:19Z","title":"Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-30T07:48:01.719339Z"},"links":{"cited_paper":"/paper/2412.05185","citing_paper":"/paper/2606.29445"},"observation_digest":"sha256:84d052147fa3fa053e5b1ef94e19164fbb25e8c3c814a6076482216ad12bbb4c","observation_id":"c3ac952d-456c-4709-bf4b-b1a34cc5cead","resolution":{"observed_at":"2026-06-30T07:54:22.394593Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.05185/citation-record","integrity":"/paper/2412.05185/integrity","json":"/paper/2412.05185/citation-record.json","paper":"/paper/2412.05185"},"outbound":[],"paper":{"arxiv_id":"2412.05185","last_updated":"2024-12-11T14:43:02Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T20:02:53.153763Z","submitted_at":"2024-12-06T17:04:42Z","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 7 inbound Pith citation observations for arXiv:2412.05185."}