{"as_of":"2026-08-10T16:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ed667b6f3c14f6af5dc29df86904312acab038be9cc06a1a76012a065bd40aac","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:32:50.149539Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T00:49:17.542022Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2505.20715","last_updated":"2026-04-18T02:55:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-27T04:50:07Z","title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-19T13:13:40.485342Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2505.20715"},"observation_digest":"sha256:fbd7b0d8480248d8c6c5fdc220123125624d3b9eb6527128abdbc2de0e24bc3e","observation_id":"6fedff7f-aaec-4ec1-8767-dea3b12c325d","resolution":{"observed_at":"2026-05-19T13:17:18.536691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-07T12:32:50.149539Z","title":"Cg- bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.24329","last_updated":"2025-07-31T03:03:18Z","snapshot_observed_at":"2026-08-09T06:10:22.489105Z","submitted_at":"2025-05-30T08:10:18Z","title":"DisTime: Distribution-based Time Representation for Video Large Language Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:50.149539Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2505.24329"},"observation_digest":"sha256:a8cfc0b0d2289df4d999384175462c34ab154498e14c9c766e48b13bfc5bc7b7","observation_id":"e150a606-02fb-4a4e-8b6f-c23c42111155","resolution":{"observed_at":"2026-08-07T12:32:50.149539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-07T10:27:05.529740Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding.arXiv preprint arXiv:2412.12075, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05328","last_updated":"2025-07-22T07:00:35Z","snapshot_observed_at":"2026-08-08T19:39:48.959316Z","submitted_at":"2025-06-05T17:58:33Z","title":"AV-Reasoner: Improving and Benchmarking Clue-Grounded Audio-Visual Counting for MLLMs","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:05.529740Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2506.05328"},"observation_digest":"sha256:9265cb50fd5f4808069cba587b242e0d2904e620c5ea4c113017bd2b8b60b794","observation_id":"02223deb-a7f6-4379-8ff1-a37f4262b12d","resolution":{"observed_at":"2026-08-07T10:27:05.529740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-07T06:00:58.009101Z","title":"Cg-bench: Clue-grounded question answering bench- mark for long video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06253","last_updated":"2025-06-06T17:25:48Z","snapshot_observed_at":"2026-08-09T07:19:17.693738Z","submitted_at":"2025-06-06T17:25:48Z","title":"Bridging Perspectives: A Survey on Cross-view Collaborative Intelligence with Egocentric-Exocentric Vision","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:58.009101Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2506.06253"},"observation_digest":"sha256:6ffd8d29f8ac9a7f67017323d0997c0e22e403b3a034664a054f144cff2c5ac1","observation_id":"e150c5ae-358f-4b63-a745-006588ae60ac","resolution":{"observed_at":"2026-08-07T06:00:58.009101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-07T06:00:56.855381Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding.arXiv preprint arXiv:2412.12075, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06275","last_updated":"2025-06-06T17:58:36Z","snapshot_observed_at":"2026-08-07T21:48:03.079145Z","submitted_at":"2025-06-06T17:58:36Z","title":"Movie Facts and Fibs (MF$^2$): A Benchmark for Long Movie Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:56.855381Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2506.06275"},"observation_digest":"sha256:ddf2316cffbfe16315005f144f6f41030f4e3db86d5c0e782e2bc1e609d1eef7","observation_id":"d426b64f-a31c-4848-bfb2-48061f7ec977","resolution":{"observed_at":"2026-08-07T06:00:56.855381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-07T04:22:55.805315Z","title":"Cg- bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.10857","last_updated":"2025-08-04T09:11:48Z","snapshot_observed_at":"2026-08-08T13:30:57.844783Z","submitted_at":"2025-06-12T16:17:17Z","title":"VRBench: A Benchmark for Multi-Step Reasoning in Long Narrative Videos","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T04:22:55.805315Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2506.10857"},"observation_digest":"sha256:caf42b48744529f260d4daff7258260621542f1801aa298c43b8aa548f8af18c","observation_id":"aa3c841a-826e-4063-b441-59d7aee0678b","resolution":{"observed_at":"2026-08-07T04:22:55.805315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-06T22:41:07.070388Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20960","last_updated":"2025-06-29T15:16:22Z","snapshot_observed_at":"2026-08-07T12:35:43.320358Z","submitted_at":"2025-06-26T02:54:24Z","title":"OmniEval: A Benchmark for Evaluating Omni-modal Models with Visual, Auditory, and Textual Inputs","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:41:07.070388Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2506.20960"},"observation_digest":"sha256:a7f18590cd69e1ad0019313e1c87c483bd45dee6361073b272b338eb8d16d9c5","observation_id":"f0c1fecd-0d5e-46e2-96fd-1bf79aa69b5e","resolution":{"observed_at":"2026-08-06T22:41:07.070388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-05T23:32:17.702112Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10922","last_updated":"2025-08-07T08:52:11Z","snapshot_observed_at":"2026-08-09T09:33:06.733360Z","submitted_at":"2025-08-07T08:52:11Z","title":"A Survey on Video Temporal Grounding with Multimodal Large Language Model","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-05T23:32:17.702112Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2508.10922"},"observation_digest":"sha256:e18ba06c2eee5efe86c10e605381b1c9e16d78cf29629d5f3d67e98d5a16d30e","observation_id":"19e47eb9-e53e-4ef6-b09f-014ff4bff5ee","resolution":{"observed_at":"2026-08-05T23:32:17.702112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2508.21094","last_updated":"2026-04-24T22:53:42Z","snapshot_observed_at":"2026-08-09T19:03:57.055397Z","submitted_at":"2025-08-27T14:33:32Z","title":"EMCompress: Video-LLMs with Endomorphic Multimodal Compression","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T20:40:42.995833Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2508.21094"},"observation_digest":"sha256:8eb6d59515ed02d9c01d5070edea5df53a1d2e7191b9fa0e9578b58f79c961bc","observation_id":"60dad2f5-2f7a-48c4-8107-13874d6f42b4","resolution":{"observed_at":"2026-05-18T20:41:50.596950Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2511.13026","last_updated":"2026-05-14T09:07:42Z","snapshot_observed_at":"2026-08-08T16:39:03.225299Z","submitted_at":"2025-11-17T06:25:12Z","title":"REVISOR: Beyond Textual Reflection, Towards Multimodal Introspective Reasoning in Long-Form Video Understanding","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T22:19:36.366837Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2511.13026"},"observation_digest":"sha256:21990cb7816142ad7860cfe2f02dcff2d42ac08e910acccef11f41755285956a","observation_id":"5b5217f4-7191-472c-8ce7-66cc9f303bd9","resolution":{"observed_at":"2026-05-17T22:20:22.941804Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2601.15724","last_updated":"2026-04-19T09:17:00Z","snapshot_observed_at":"2026-08-07T08:54:37.463054Z","submitted_at":"2026-01-22T07:47:29Z","title":"VideoThinker: Building Agentic VideoLLMs with LLM-Guided Tool Reasoning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T12:17:42.135851Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2601.15724"},"observation_digest":"sha256:4283288524a0e9cdcb78c467edab750c83ba3c4c63ed459665b1a9dd0b77c6c4","observation_id":"df201b4e-fc59-42cd-b341-1f002533f6fa","resolution":{"observed_at":"2026-05-16T12:17:51.923578Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2602.20913","last_updated":"2026-04-15T16:09:22Z","snapshot_observed_at":"2026-08-02T12:38:41.181077Z","submitted_at":"2026-02-24T13:49:47Z","title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T20:01:31.129959Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2602.20913"},"observation_digest":"sha256:6c89ef42df254f9897ea7b5ee17eddaf6cf2effbd86cd90fdf4488592bdd0f80","observation_id":"8cb99da1-f860-46d7-964b-44a6e0bcaba9","resolution":{"observed_at":"2026-05-15T20:01:33.453527Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2604.05546","last_updated":"2026-04-14T01:04:46Z","snapshot_observed_at":"2026-08-05T11:51:07.603519Z","submitted_at":"2026-04-07T07:44:11Z","title":"Efficient Inference for Large Vision-Language Models: Bottlenecks, Techniques, and Prospects","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T18:54:04.104227Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2604.05546"},"observation_digest":"sha256:316007671f3d34feead18c7dfeda6665c205ca4123b39f7b9e13940b5b4ae5e0","observation_id":"1e641f19-8de6-4e96-990a-0d80cdbc04a6","resolution":{"observed_at":"2026-05-10T23:45:50.955671Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2604.11627","last_updated":"2026-04-13T15:38:22Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-13T15:38:22Z","title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T15:23:08.671342Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2604.11627"},"observation_digest":"sha256:87807e47578c09036cdfd79a76addfb79799b238faaf256ebd11f4505cce8d48","observation_id":"267929d3-e852-4c10-a1e2-0fc519398c0b","resolution":{"observed_at":"2026-05-11T10:41:03.728687Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2604.20937","last_updated":"2026-06-19T12:00:35Z","snapshot_observed_at":"2026-08-02T23:37:34.955988Z","submitted_at":"2026-04-22T13:28:53Z","title":"Sink-Token-Aware Pruning for Fine-Grained Video Understanding in Efficient Video LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T00:43:44.921189Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2604.20937"},"observation_digest":"sha256:44dbb1a5d5bd1878f5b9b16192fc5fc63225282ec7b1758dfd76a386e511e232","observation_id":"e610b167-f3da-415c-8fe0-d640e963b60a","resolution":{"observed_at":"2026-05-10T00:44:48.414508Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-14T19:27:29.843866Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.22226","last_updated":"2026-07-13T04:05:34Z","snapshot_observed_at":"2026-07-16T23:18:42.650626Z","submitted_at":"2026-04-24T05:02:03Z","title":"Towards Temporal Compositional Reasoning in Long-Form Sports Videos","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-14T19:27:29.843866Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2604.22226"},"observation_digest":"sha256:50818865f6e5bfc7420c86b6ad54e638c9bb42c4a4a7409e107bdfd8bbfccac9","observation_id":"420a45c4-ccb8-4b28-ac77-fb6637033bd0","resolution":{"observed_at":"2026-07-14T19:27:29.843866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2605.09874","last_updated":"2026-05-11T01:59:59Z","snapshot_observed_at":"2026-07-30T14:13:44.494421Z","submitted_at":"2026-05-11T01:59:59Z","title":"EgoMemReason: A Memory-Driven Reasoning Benchmark for Long-Horizon Egocentric Video Understanding","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-12T04:37:46.090777Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2605.09874"},"observation_digest":"sha256:773f690d87a1519f2cd64747bb76698b13e9a984c723c4fc5afa74f97c01dce1","observation_id":"bb6e4c7b-c7d6-46af-9ab0-a9de1463aa6a","resolution":{"observed_at":"2026-05-12T06:06:24.690741Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2605.22907","last_updated":"2026-05-21T18:00:22Z","snapshot_observed_at":"2026-08-02T16:30:28.109405Z","submitted_at":"2026-05-21T18:00:22Z","title":"VideoOdyssey: A Benchmark for Ultra-Long-Context and Omni-Modal Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-25T05:51:49.390597Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2605.22907"},"observation_digest":"sha256:ef37a18c4e21143e4f7989a6d2bc6b6468cdf7c73c0ffcfbd2b11af2b64ea273","observation_id":"c8e94c6d-64a2-4df4-bab7-3e695f1db87e","resolution":{"observed_at":"2026-05-25T05:55:25.014007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2605.23216","last_updated":"2026-07-02T06:54:56Z","snapshot_observed_at":"2026-08-03T00:30:19.785022Z","submitted_at":"2026-05-22T04:19:29Z","title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-25T04:54:23.077914Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2605.23216"},"observation_digest":"sha256:4fa97423392a855b87adaa7ce21616284ce8b64c3b4a7088a6b2631e1a6de98a","observation_id":"6aa79895-1600-4daf-9aec-21d29475123a","resolution":{"observed_at":"2026-05-25T04:55:23.166760Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2605.23216","last_updated":"2026-07-02T06:54:56Z","snapshot_observed_at":"2026-08-03T00:30:19.785022Z","submitted_at":"2026-05-22T04:19:29Z","title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-04T00:41:02.284215Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2605.23216"},"observation_digest":"sha256:1925e318f65228fb62a01fcac2ab756cc4673256708740a8da9b431d817b1d52","observation_id":"9c8f22a7-42e6-4db0-b5e9-7ccde862993c","resolution":{"observed_at":"2026-07-04T00:49:17.545143Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2606.08231","last_updated":"2026-06-06T15:39:29Z","snapshot_observed_at":"2026-08-07T16:58:05.550639Z","submitted_at":"2026-06-06T15:39:29Z","title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-27T19:36:57.231932Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2606.08231"},"observation_digest":"sha256:7414ea8d3d5ddbd55607e7d253d2ded265a732e46fdf2cf65847d643b7dacf00","observation_id":"f043ac01-df25-4a2e-86ed-07be12eda13a","resolution":{"observed_at":"2026-07-02T21:37:25.333752Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2412.12075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-04T00:49:17.542022Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":"011136c4-f8a3-4d2a-a10f-3e6548f6353f","year":2024},"citing_paper":{"arxiv_id":"2606.13141","last_updated":"2026-06-11T10:05:49Z","snapshot_observed_at":"2026-07-06T23:51:55.044440Z","submitted_at":"2026-06-11T10:05:49Z","title":"Rethinking RAG in Long Videos: What to Retrieve and How to Use It?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T06:30:33.428489Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2606.13141"},"observation_digest":"sha256:19cc6f11e6154684e55b2d6d9adf9926e37b06a8efcec53a4d324474d1caa6ae","observation_id":"6d90e396-1fe9-4aa3-8d49-02c641377eef","resolution":{"observed_at":"2026-07-03T15:28:33.999172Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-12T05:50:16.895740Z","title":"arXiv preprint arXiv:2412.12075 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02959","last_updated":"2026-07-03T05:05:59Z","snapshot_observed_at":"2026-08-06T10:42:51.828740Z","submitted_at":"2026-07-03T05:05:59Z","title":"Incentivizing Vision Language Models to Search for Long Video Question Answering","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-12T05:50:16.895740Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2607.02959"},"observation_digest":"sha256:3413e4a99e955b432576d55e55725d069b3991ce6e4921a92978604620a66446","observation_id":"432ebe94-bbaf-4073-84f0-3a6c78567a71","resolution":{"observed_at":"2026-07-12T05:50:16.895740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-07-11T08:59:46.244502Z","title":"arXiv preprint arXiv:2412.12075 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05089","last_updated":"2026-07-06T13:50:15Z","snapshot_observed_at":"2026-07-11T08:59:45.751461Z","submitted_at":"2026-07-06T13:50:15Z","title":"TimeThink: Reasoning with Time for Video LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-11T08:59:46.244502Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2607.05089"},"observation_digest":"sha256:13e5dde79b35b2bd18d16301b9e68f5d9d0db745b56e3c6c6e2c785160b306ab","observation_id":"a86fe1e0-42ac-4ff6-a65f-c6d62ca74575","resolution":{"observed_at":"2026-07-11T08:59:46.244502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.12075/citation-record","integrity":"/paper/2412.12075/integrity","json":"/paper/2412.12075/citation-record.json","paper":"/paper/2412.12075"},"outbound":[],"paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T20:07:58.873037Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 24 inbound Pith citation observations for arXiv:2412.12075."}