{"as_of":"2026-08-08T18:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:38cece532dbe67b28f864729a8a32a65bc16cb8235f588235379eebd12e7f354","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:40:11.807797Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:38:56.149993Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-08-07T11:40:11.807797Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01725","last_updated":"2025-06-02T14:30:09Z","snapshot_observed_at":"2026-08-08T07:45:08.111238Z","submitted_at":"2025-06-02T14:30:09Z","title":"VideoCap-R1: Enhancing MLLMs for Video Captioning via Structured Thinking","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.807797Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2506.01725"},"observation_digest":"sha256:cc417bcf5b5e1212f63047457c769fb9cfdc42f7a729e16120ab83f3aea81e32","observation_id":"5eead944-f033-4430-97d5-61140c8c3ef9","resolution":{"observed_at":"2026-08-07T11:40:11.807797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-08-07T04:44:59.302001Z","title":"Shot2story: A new benchmark for comprehensive understanding of multi-shot videos, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.302001Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:7652dd22f3a7da5c7f44d37a25a74097f7d003fdf4e054d69415f9082abf7b21","observation_id":"3959a487-d5e7-4256-979c-3260fefde00a","resolution":{"observed_at":"2026-08-07T04:44:59.302001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-08-06T22:23:41.969901Z","title":"Shot2Story20K: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21785","last_updated":"2025-06-26T21:46:48Z","snapshot_observed_at":"2026-08-08T08:49:33.541875Z","submitted_at":"2025-06-26T21:46:48Z","title":"Comparing Learning Paradigms for Egocentric Video Summarization","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:23:41.969901Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2506.21785"},"observation_digest":"sha256:174ecbc70991811d478d9b471131bde1611fcdc587964a1df8cc8c60d088bd4b","observation_id":"e5aed2b2-69a5-42fe-87bd-b4dce7efe8ad","resolution":{"observed_at":"2026-08-06T22:23:41.969901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-08-06T20:29:50.482506Z","title":"Shot2story20k: A new benchmark for comprehen- sive understanding of multi-shot videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02591","last_updated":"2025-07-23T07:25:27Z","snapshot_observed_at":"2026-08-07T06:17:01.457380Z","submitted_at":"2025-07-03T12:55:16Z","title":"AuroraLong: Bringing RNNs Back to Efficient Open-Ended Video Understanding","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:29:50.482506Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2507.02591"},"observation_digest":"sha256:f8b632a0592bad946a2553b306a9783cbbff00e4f06564eb0ced5eb7ae2f9804","observation_id":"6f53dc11-111a-4c9e-9ca0-ef9b3f4c1988","resolution":{"observed_at":"2026-08-06T20:29:50.482506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-08-06T15:17:57.824025Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.16873","last_updated":"2025-07-22T08:24:33Z","snapshot_observed_at":"2026-08-07T23:46:03.737945Z","submitted_at":"2025-07-22T08:24:33Z","title":"HIPPO-Video: Simulating Watch Histories with Large Language Models for Personalized Video Highlighting","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T15:17:57.824025Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2507.16873"},"observation_digest":"sha256:36f3b8d3e746cf88ff496da74a94c507be8eae09dc1bea232f304b4d02483f94","observation_id":"6b864fc9-9403-4a05-a35b-e0262781c528","resolution":{"observed_at":"2026-08-06T15:17:57.824025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-08-05T18:39:29.914405Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.14395","last_updated":"2025-08-20T03:45:18Z","snapshot_observed_at":"2026-08-05T18:39:25.569218Z","submitted_at":"2025-08-20T03:45:18Z","title":"NoteIt: A System Converting Instructional Videos to Interactable Notes Through Multimodal Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T18:39:29.914405Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2508.14395"},"observation_digest":"sha256:88c63272844b6f0ef73e7ce9ffbd6e850e720a86df0e5bddbbbee22161e2b058","observation_id":"71c663e5-043f-4578-b719-2fe8944040cd","resolution":{"observed_at":"2026-08-05T18:39:29.914405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":"2312.10300","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-07-03T20:38:56.149993Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":"32c962fe-1151-47f5-b3eb-dbbe2aba6841","year":2023},"citing_paper":{"arxiv_id":"2512.21334","last_updated":"2026-04-10T15:00:46Z","snapshot_observed_at":"2026-08-03T08:10:30.527963Z","submitted_at":"2025-12-24T18:59:36Z","title":"Streaming Video Instruction Tuning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T19:44:11.032898Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2512.21334"},"observation_digest":"sha256:11350c9b48c36cd8fd2039eaaafc4fa6c0ced00cfc493020a2b010e43718af3a","observation_id":"072fa698-ddc5-4a3a-878a-c96338c1d383","resolution":{"observed_at":"2026-05-16T19:48:21.847902Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":"2312.10300","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-07-03T20:38:56.149993Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":"32c962fe-1151-47f5-b3eb-dbbe2aba6841","year":2023},"citing_paper":{"arxiv_id":"2604.15127","last_updated":"2026-04-17T03:24:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-16T15:13:41Z","title":"MCSC-Bench: Multimodal Context-to-Script Creation for Realistic Video Production","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T09:10:31.245928Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2604.15127"},"observation_digest":"sha256:a7f02d6e75517537dc1d372b6c120a5ddd8fa659a63f742d79521a1decaf4df9","observation_id":"de5a588d-53e0-4229-a971-0412b48b708f","resolution":{"observed_at":"2026-05-10T09:13:29.933725Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":"2312.10300","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-07-03T20:38:56.149993Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":"32c962fe-1151-47f5-b3eb-dbbe2aba6841","year":2023},"citing_paper":{"arxiv_id":"2604.23789","last_updated":"2026-05-09T04:03:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-26T16:28:46Z","title":"MuSS: A Large-Scale Dataset and Cinematic Narrative Benchmark for Multi-Shot Subject-to-Video Generation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-08T06:28:42.129881Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2604.23789"},"observation_digest":"sha256:ecb6545aedc63b79ebeff9ffea37758d5a4f00c2ee37860bfac2e808d39274e2","observation_id":"1b527f97-cdc2-4c25-a66d-72bd07c71120","resolution":{"observed_at":"2026-05-11T21:11:19.082732Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":"2312.10300","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-07-03T20:38:56.149993Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":"32c962fe-1151-47f5-b3eb-dbbe2aba6841","year":2023},"citing_paper":{"arxiv_id":"2604.23789","last_updated":"2026-05-09T04:03:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-26T16:28:46Z","title":"MuSS: A Large-Scale Dataset and Cinematic Narrative Benchmark for Multi-Shot Subject-to-Video Generation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T00:50:10.509727Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2604.23789"},"observation_digest":"sha256:327a4983ff7fae13ad2c324e42cfeb48875ce0228ae8e2dd45a39e76ec0b0708","observation_id":"78b7ed1d-47e0-4726-993f-b16a86d3fa47","resolution":{"observed_at":"2026-05-12T00:51:15.120188Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":"2312.10300","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-07-03T20:38:56.149993Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":"32c962fe-1151-47f5-b3eb-dbbe2aba6841","year":2023},"citing_paper":{"arxiv_id":"2606.08615","last_updated":"2026-06-07T13:00:19Z","snapshot_observed_at":"2026-08-05T22:03:12.300616Z","submitted_at":"2026-06-07T13:00:19Z","title":"Harnessing Streaming Video in the Wild","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T18:47:55.910417Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2606.08615"},"observation_digest":"sha256:de75a0b93ba40f72c0ba5e551aeeea821cdaf32a7e4042ac29d05852cc130d75","observation_id":"afdedb8a-49e6-476f-9ab3-44a88148bfaf","resolution":{"observed_at":"2026-07-02T22:37:25.681985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":"2312.10300","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-07-03T20:38:56.149993Z","title":"Shot2story20k: A new benchmark for comprehensive understanding of multi-shot videos","venue":null,"work_id":"32c962fe-1151-47f5-b3eb-dbbe2aba6841","year":2023},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-08-08T11:16:26.006355Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:221e2fe940c8dbc9ecd4d6bc6008bfbdbeeffb9f7ccfb36fd1aaa672c9bd6bf4","observation_id":"cabb546d-af1a-46f3-8abe-4ff690e6b441","resolution":{"observed_at":"2026-07-03T20:38:56.151854Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2312.10300/citation-record","integrity":"/paper/2312.10300/integrity","json":"/paper/2312.10300/citation-record.json","paper":"/paper/2312.10300"},"outbound":[],"paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2312.10300."}