{"as_of":"2026-08-07T09:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3ba402431d4b3864810468908337f5498a2c661d515da93cd9ba1f3b1898550a","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:49:34.963022Z","state":"measured"},{"denominator":63,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":63,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T01:52:44.785582Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:46:56.833017Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2511.04570","last_updated":"2026-04-07T09:55:11Z","snapshot_observed_at":"2026-07-30T01:17:08.732833Z","submitted_at":"2025-11-06T17:25:23Z","title":"Thinking with Video: Video Generation as a Promising Multimodal Reasoning Paradigm","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-18T00:54:24.641649Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2511.04570"},"observation_digest":"sha256:d0cb562420f4ed9b53849b96cb45176c266d663f04d5b67c93eba0576efb3ced","observation_id":"558a108b-cffe-40fe-a71d-db26eebee2c9","resolution":{"observed_at":"2026-05-18T00:55:35.070077Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2605.01657","last_updated":"2026-05-03T00:52:51Z","snapshot_observed_at":"2026-08-02T06:05:08.178689Z","submitted_at":"2026-05-03T00:52:51Z","title":"Act2See: Emergent Active Visual Perception for Video Reasoning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-08T19:34:53.683729Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2605.01657"},"observation_digest":"sha256:32daf6cea086b79cd2f940be61e6457bb211ff9c7b9996e29aaa7491960dc0f3","observation_id":"51a2c363-4137-4637-9b3a-ed92ff0003fd","resolution":{"observed_at":"2026-05-09T05:45:22.948194Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2605.17283","last_updated":"2026-05-17T06:39:05Z","snapshot_observed_at":"2026-08-02T16:55:37.869194Z","submitted_at":"2026-05-17T06:39:05Z","title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-05-20T14:43:46.517807Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2605.17283"},"observation_digest":"sha256:77943186ea82a9dfcbd5691001e96583fbcf4e6b859bb6145e87aa1a05343441","observation_id":"05811938-693a-4663-9912-8d5666776b84","resolution":{"observed_at":"2026-05-20T14:48:23.408448Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2606.05736","last_updated":"2026-06-04T05:55:15Z","snapshot_observed_at":"2026-07-06T23:45:42.379051Z","submitted_at":"2026-06-04T05:55:15Z","title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-28T01:52:44.785582Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2606.05736"},"observation_digest":"sha256:7c66abf422ad1eeb218877fc4440a5fcba55f6fe847fafc0a6f077e24dc1ed9f","observation_id":"aed15bde-fead-4786-81d4-3742e7779b81","resolution":{"observed_at":"2026-07-02T12:46:56.834326Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.09876/citation-record","integrity":"/paper/2507.09876/integrity","json":"/paper/2507.09876/citation-record.json","paper":"/paper/2507.09876"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:49:34.724807Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.724807Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:a6dfd42853f79aa68d8ccbabe10e287a0a8f128e97d82750b46b7d8d8b3b9348","observation_id":"7cc29dba-3559-498c-aca7-61e5c70afd2c","resolution":{"observed_at":"2026-08-06T17:49:34.724807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:39.360226Z","title":"Hadzic, Taran Kota, Jimming He, Cristobal Eyzaguirre, Zane Durante, Manling Li, Jiajun Wu, and Li Fei-Fei","venue":null,"work_id":"b3b73293-0410-4cd9-bde7-71c615144615","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.729419Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:32ee057c1e9feb4c225f067860dc176e74547bc2685d7da71152d52ad8f0bcc5","observation_id":"27dccd33-300a-42ae-bdef-f6fc5854b9cf","resolution":{"observed_at":"2026-08-06T17:49:39.444946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:39.212057Z","title":null,"venue":null,"work_id":"30974041-ee16-4b7c-85fd-8165cbb26cf7","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.734176Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:46f775d6510054e9a7e5a8ecea4dafedc8a18a7ff8ef0558cc2954191d2a9245","observation_id":"015d76b4-6404-43c0-8a9a-186d849caf1f","resolution":{"observed_at":"2026-08-06T17:49:39.276953Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09567","snapshot_observed_at":"2026-08-06T17:49:34.738049Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.738049Z"},"links":{"cited_paper":"/paper/2503.09567","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:e497d6bf2953df930d4fd4833f08a23d1c35112a8786688b7db5a50098f4ec46","observation_id":"9f229cfb-9c15-459d-ba5f-197b93d0830e","resolution":{"observed_at":"2026-08-06T17:49:34.738049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:39.029215Z","title":null,"venue":null,"work_id":"55bf8c91-fc28-418a-af82-2a71afcac42e","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.742594Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:1509b343c7b5fee184c75c967fd8cc35ac2500496d43e899e88ee2fef617280f","observation_id":"27225dd8-3d9e-4ee7-ba66-f364c93899cb","resolution":{"observed_at":"2026-08-06T17:49:39.124749Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01903","last_updated":"2025-08-05T16:19:40Z","snapshot_observed_at":"2026-08-06T20:38:04.040927Z","submitted_at":"2025-07-02T17:19:20Z","title":"AI4Research: A Survey of Artificial Intelligence for Scientific Research","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.01903","snapshot_observed_at":"2026-08-06T17:49:34.746783Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.746783Z"},"links":{"cited_paper":"/paper/2507.01903","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:aa3c19bd500585ca308388593ceb4746bd2db320d91ddd12ef5004f90654a8ac","observation_id":"6b51a282-7bef-46aa-bf45-3d01fdc5038b","resolution":{"observed_at":"2026-08-06T17:49:34.746783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.751721Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.751721Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:f2c4385397b032e2d091fecab7b1a319224e2fef72cf677ead8b374c60c971d5","observation_id":"381c4fb1-79fd-43e4-874f-f9ace96b357c","resolution":{"observed_at":"2026-08-06T17:49:34.751721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11623","last_updated":"2024-10-15T14:08:53Z","snapshot_observed_at":"2026-07-06T19:33:52.737083Z","submitted_at":"2024-10-15T14:08:53Z","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11623","snapshot_observed_at":"2026-08-06T17:49:34.756138Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.756138Z"},"links":{"cited_paper":"/paper/2410.11623","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:984dd0a4ce488c2d385d84d5ba2ddada1f4fcff5facd31ad132ad628c8a9c213","observation_id":"7cbc93df-fee0-4c19-be81-59e670680f03","resolution":{"observed_at":"2026-08-06T17:49:34.756138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.760649Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.760649Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:91b60d3a62ab1e421692615c5572200726c7118b5a0ddb06d2f8b4ee2444875f","observation_id":"2d461dc0-9426-464a-83f7-fcb862011ed3","resolution":{"observed_at":"2026-08-06T17:49:34.760649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.904588Z","title":null,"venue":null,"work_id":"593d96d0-894b-4a58-a81b-a3e14233e415","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.764444Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:984159199bb1b0dfa8f77a3a656f22eebc761232281e6db066237c9e9f5686da","observation_id":"6a1f1955-fb2e-4578-932a-2a72e29473a0","resolution":{"observed_at":"2026-08-06T17:49:38.973447Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.768376Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.768376Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:5348d6f4ed38c13c7bd3da5ba3d1ba87e616bf7fb564223e95bb31c27f3dbea0","observation_id":"cdb2164e-c249-4acf-8c87-f125a9f45085","resolution":{"observed_at":"2026-08-06T17:49:34.768376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.750542Z","title":null,"venue":null,"work_id":"994cea36-9b09-40e4-9c58-f8072bdde726","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.771662Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:4ebcffdfb7f39b17240b7be5f888e7ba7794e14ec149338dd87ca70821f3118b","observation_id":"1c9ba49d-6aa0-4872-a967-faf832a01226","resolution":{"observed_at":"2026-08-06T17:49:38.817718Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.563468Z","title":null,"venue":null,"work_id":"6f94576b-3e2e-4606-ac54-bbfbad1c2fb9","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.775143Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:f7c9774920b2102a00745cc74cb800b13c8d178255f2fd1e9aa5d4c440cd571a","observation_id":"e2612be1-665c-4f42-aef0-71123dcd7b7c","resolution":{"observed_at":"2026-08-06T17:49:38.675109Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.409474Z","title":null,"venue":null,"work_id":"1a670a72-f8a9-4260-951c-e0fc4480028c","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.778450Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:29ae9cc4bb8a09159b722c9bcee86898355557b9d1bd59329d258654e2d2b782","observation_id":"fc9b25a2-4771-49c2-9ba0-ada62172607b","resolution":{"observed_at":"2026-08-06T17:49:38.456307Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.222112Z","title":null,"venue":null,"work_id":"1fd3f72e-f343-4cbd-8854-a51a9187a0ce","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.781657Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:1a572b819e81043aefbcf13e93d9c84ecb32f9c53bcf475fdf5247a2e96e73e4","observation_id":"95d610e5-de64-4d0d-9cfd-0afd5792bf15","resolution":{"observed_at":"2026-08-06T17:49:38.318985Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-06T17:49:34.784841Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.784841Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:340c4d692390ff6f9109d7b11cd415c09c3feac392e9f4601bd8cfcbfa505685","observation_id":"47f20178-e7b6-4f49-b8e5-6375af45dfaf","resolution":{"observed_at":"2026-08-06T17:49:34.784841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.055256Z","title":null,"venue":null,"work_id":"54ac5287-742b-457e-a8d9-f575278ec290","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.788646Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:29a6ed5083be3f6671a00ced2118efd8f886600588d900ee75a30e0c64082516","observation_id":"40a9f754-3d02-443c-a189-4b4386002a73","resolution":{"observed_at":"2026-08-06T17:49:38.144409Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13903","last_updated":"2023-11-09T06:50:26Z","snapshot_observed_at":"2026-07-06T15:31:18.144952Z","submitted_at":"2023-05-23T10:26:42Z","title":"Let's Think Frame by Frame with VIP: A Video Infilling and Prediction Dataset for Evaluating Video Chain-of-Thought","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13903","snapshot_observed_at":"2026-08-06T17:49:34.791982Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.791982Z"},"links":{"cited_paper":"/paper/2305.13903","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:e8f3aa63608ca172e1f6ebf663bb66a9277797e9f64e6e657f0b0034cf02fcf7","observation_id":"34ee85ce-7b5b-4e31-ad35-2200692b9b20","resolution":{"observed_at":"2026-08-06T17:49:34.791982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06428","last_updated":"2025-02-11T14:59:25Z","snapshot_observed_at":"2026-08-03T22:34:50.903495Z","submitted_at":"2025-02-10T13:03:05Z","title":"CoS: Chain-of-Shot Prompting for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06428","snapshot_observed_at":"2026-08-06T17:49:34.796119Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.796119Z"},"links":{"cited_paper":"/paper/2502.06428","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:f013e94327ec58f7c9c61e09ea2841af1c5ed7da1d4bac4fe13546a8468094f4","observation_id":"2608cca2-ea4e-443e-af2f-9c56d48093d5","resolution":{"observed_at":"2026-08-06T17:49:34.796119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.875522Z","title":null,"venue":null,"work_id":"9516e826-7298-4780-a9a8-622c87748f15","year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.800952Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:ae5473d12d5eb3dab9707860d1cced5bf0d90d831a33c1fbcce1bacfcfdabc8a","observation_id":"c83bbcbc-7685-4a7b-a003-7427cf56afec","resolution":{"observed_at":"2026-08-06T17:49:37.973782Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.688204Z","title":null,"venue":null,"work_id":"454b30bb-bf69-44eb-a089-c53530aaebea","year":2009},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.804448Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:22b11ca9c53fa17238eefbee83bd9bf5c3470e95dec7f2d2a5e5ef7f66051820","observation_id":"8a6a185a-9cf7-4d98-8d98-077db591bbc2","resolution":{"observed_at":"2026-08-06T17:49:37.797446Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.537544Z","title":null,"venue":null,"work_id":"000bba05-3748-4efa-9fa0-289ed7a5ba54","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.808242Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:ea6655450c72ba61c8b25ef7a8c76ad09c5b1067a7962578d5b4b8575a557d2a","observation_id":"a2931641-5298-47e4-9ec9-3fe25df594a7","resolution":{"observed_at":"2026-08-06T17:49:37.614690Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.812074Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.812074Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:58c233020137a982be0289f503d33d7dc5e6279164a4c4a309e39f235ea5d4ee","observation_id":"cf46c7f1-1d0f-44e2-aeb7-1e2d646f3f9a","resolution":{"observed_at":"2026-08-06T17:49:34.812074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.03426","last_updated":"2020-09-18T01:56:41Z","snapshot_observed_at":"2026-08-02T15:32:07.466568Z","submitted_at":"2018-02-09T19:39:33Z","title":"UMAP: Uniform Manifold Approximation and Projection for Dimension Reduction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.03426","snapshot_observed_at":"2026-08-06T17:49:34.816549Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.816549Z"},"links":{"cited_paper":"/paper/1802.03426","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:56dff88fa0dd93d715615593e38200bb0f1ebd1fe9e547d410e2b5638bbc3c6a","observation_id":"705c3741-17e4-41b5-bba2-d05810d930e0","resolution":{"observed_at":"2026-08-06T17:49:34.816549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.361000Z","title":null,"venue":null,"work_id":"cd9a4de3-c38c-4950-835b-cd7b5509b207","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.821127Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:c8834332dec70b4c9bcd1bfacfca99dbe9aefb624239b0ff6439dd6a7cb93e84","observation_id":"99ae4b9e-a7b2-4628-8444-243c9f642b85","resolution":{"observed_at":"2026-08-06T17:49:37.432290Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12819","last_updated":"2025-08-25T02:35:43Z","snapshot_observed_at":"2026-07-06T18:17:21.217971Z","submitted_at":"2024-05-21T14:24:01Z","title":"Large Language Models Meet NLP: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12819","snapshot_observed_at":"2026-08-06T17:49:34.825536Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.825536Z"},"links":{"cited_paper":"/paper/2405.12819","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:13894f5570fd64f7a30f7ced6d03331b0fd4e1a75506fd51beb229f41efe9c99","observation_id":"8f5a3dfb-fe81-4984-a699-8009d92c7ac0","resolution":{"observed_at":"2026-08-06T17:49:34.825536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.178248Z","title":null,"venue":null,"work_id":"da0b000a-2c92-4c67-9a58-31a15aa1ece5","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.829913Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:236a910dca6fd26d1ecb848564cb675a75e855453913531d1825a9e0c907d1d7","observation_id":"543ab5dc-eed6-4180-9e03-722476cbd432","resolution":{"observed_at":"2026-08-06T17:49:37.243312Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.833545Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.833545Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:5da687e969614b222b708d7dd5b7363abed783f8dc339b9e4a5352ff78c8dd01","observation_id":"417c138a-f82d-4e25-8533-f5fecc693e0f","resolution":{"observed_at":"2026-08-06T17:49:34.833545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.836951Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.836951Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:ac584a1820629cdd1a805b0ea9069aeced04e4e793fa3494d3258ed1e607f4e7","observation_id":"aaf1e309-0123-40fd-9642-f6257d239dee","resolution":{"observed_at":"2026-08-06T17:49:34.836951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-06T17:49:34.840557Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.840557Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:12ea68b3c1b234f1239c50de6b7bcd5346da4233484d898be1f00143cfa97d35","observation_id":"cdb59fb9-855e-4629-98b2-c2e8e7362bd0","resolution":{"observed_at":"2026-08-06T17:49:34.840557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.971843Z","title":null,"venue":null,"work_id":"a763b27b-245d-481c-9267-3d6309ddb2a2","year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.844247Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:64a1e23be115addd13b048c10dc801813ad65c03dbae4c431135081b94763db5","observation_id":"0b118f7e-1b0c-483d-bf9e-cf69301697dd","resolution":{"observed_at":"2026-08-06T17:49:37.066785Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.04091","last_updated":"2023-05-26T07:06:48Z","snapshot_observed_at":"2026-07-06T15:24:07.662207Z","submitted_at":"2023-05-06T16:34:37Z","title":"Plan-and-Solve Prompting: Improving Zero-Shot Chain-of-Thought Reasoning by Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.04091","snapshot_observed_at":"2026-08-06T17:49:34.848049Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.848049Z"},"links":{"cited_paper":"/paper/2305.04091","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:1d214e9c3e02b58004956156b02561acc807bf3ba331a5c85c55d9ddef5bc9c7","observation_id":"c8b75389-fff8-4870-8dad-a03a8435da70","resolution":{"observed_at":"2026-08-06T17:49:34.848049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.820936Z","title":null,"venue":null,"work_id":"21b6cdfc-a337-49bb-a379-71aa7fdc5ad6","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.852024Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:7ce66d1202bc000280cf611b167885eb19fc2c47415842470715dd89b8721c2b","observation_id":"d4367e60-f478-420a-ac14-fdb8ded6754f","resolution":{"observed_at":"2026-08-06T17:49:36.875213Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.06020","last_updated":"2025-09-05T16:04:23Z","snapshot_observed_at":"2026-08-07T08:43:01.208163Z","submitted_at":"2025-05-09T13:08:27Z","title":"ArtRAG: Retrieval-Augmented Generation with Structured Context for Visual Art Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.06020","snapshot_observed_at":"2026-08-06T17:49:34.856382Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.856382Z"},"links":{"cited_paper":"/paper/2505.06020","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:52193fb675b47a887b61e4ade64ff2fdf6e450e6829bc2288ee249414fb307b7","observation_id":"e94e0bfa-8ec6-4cfa-aa30-3dd8f9994c8e","resolution":{"observed_at":"2026-08-06T17:49:34.856382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-06T17:49:34.861248Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.861248Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:cb921c02015fa1af745ff1d3d96cbb2d9f017322363752a2ce318696672e38c8","observation_id":"96400931-ec18-4727-a50d-ab57e09dac6f","resolution":{"observed_at":"2026-08-06T17:49:34.861248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12605","last_updated":"2025-03-23T13:47:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-16T18:39:13Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.12605","snapshot_observed_at":"2026-08-06T17:49:34.865290Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.865290Z"},"links":{"cited_paper":"/paper/2503.12605","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:35c19bac7f016f688f772577c00bdcdefe233e3f4f10b4225e488f95a29f79f0","observation_id":"4039a2da-2b2a-458d-952c-ce10cad2ac97","resolution":{"observed_at":"2026-08-06T17:49:34.865290Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.692949Z","title":null,"venue":null,"work_id":"3a8d330d-761e-4335-8f28-5cf6c5eeca57","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.870094Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:c8e980a154f66df5586befa370fccd99375c5868696208d49249771c23a6cf19","observation_id":"a7b84ee6-057a-4995-bdb6-9daf7fb89eb0","resolution":{"observed_at":"2026-08-06T17:49:36.762340Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.554713Z","title":null,"venue":null,"work_id":"ce7939ee-2133-4eee-98f9-d2fc25076126","year":2022},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.874308Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:7203dfae77e6ee59bc8cd2734bcc1915e9ffedf02971bff6ef892fce64d38ffa","observation_id":"07208b62-5b89-4fe8-a7d5-d9468087a54e","resolution":{"observed_at":"2026-08-06T17:49:36.621591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.877874Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.877874Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:34bf8fd8c57a87a9dc7accfd9e2419690a9ae5c7dc2d7519eb7f892b70260844","observation_id":"725d5f29-1d0b-4b20-ad2e-647f8f1ac83f","resolution":{"observed_at":"2026-08-06T17:49:34.877874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.271919Z","title":null,"venue":null,"work_id":"402f7410-6513-4bc5-8b2a-62048f70db1c","year":2021},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.885407Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:cbc1fdb571c6a900022f3944d19956786ac213dd960901e79da83e063f06e6a2","observation_id":"aa414a16-3ddf-4612-b5aa-f74256674037","resolution":{"observed_at":"2026-08-06T17:49:36.340889Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.154008Z","title":null,"venue":null,"work_id":"c1142786-5a6f-400d-b5aa-a7a5fdbc5c3a","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.889875Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:b0d702725ebfedd121ad94a026ae7da715285ddc00b12ef5aa25530a56eac400","observation_id":"4021bd4b-b315-4079-99d1-370aadc0c1c5","resolution":{"observed_at":"2026-08-06T17:49:36.190327Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09193","last_updated":"2023-11-15T18:39:21Z","snapshot_observed_at":"2026-08-02T16:45:54.807090Z","submitted_at":"2023-11-15T18:39:21Z","title":"The Role of Chain-of-Thought in Complex Vision-Language Reasoning Task","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09193","snapshot_observed_at":"2026-08-06T17:49:34.898886Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.898886Z"},"links":{"cited_paper":"/paper/2311.09193","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:4ce58f6602de32200afa9080b667f84ce74c900e5c70a202e33918c53f653c99","observation_id":"75444443-fc2f-4161-a82e-a565c3d2ca3c","resolution":{"observed_at":"2026-08-06T17:49:34.898886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.032658Z","title":"In Proceedings of the 32nd ACM International Conference on Multimedia","venue":null,"work_id":"3f081fa8-4d61-46b5-9130-9f4d74acf83f","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.894306Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:e086a2534277635324f33cd97c135dcb1df233c14ba80c984302eb58aa7e9c8b","observation_id":"81bbbdab-83a1-4ccd-b08e-068c332d6e55","resolution":{"observed_at":"2026-08-06T17:49:36.091285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:35.945026Z","title":null,"venue":null,"work_id":"4f4db40f-bad7-4244-ba3f-7590a24b0cce","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.907424Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:c9f9859abe9cfbbbfa5ff36d0a6f7ba3fbc29545d9fb242f7c2f94948db5e4f0","observation_id":"f5c01e62-6511-4ff8-a8b5-38d91ddb0294","resolution":{"observed_at":"2026-08-06T17:49:35.989539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15676","last_updated":"2025-02-05T22:01:59Z","snapshot_observed_at":"2026-08-07T05:02:10.975472Z","submitted_at":"2024-04-24T06:12:00Z","title":"Beyond Chain-of-Thought: A Survey of Chain-of-X Paradigms for LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15676","snapshot_observed_at":"2026-08-06T17:49:34.903388Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.903388Z"},"links":{"cited_paper":"/paper/2404.15676","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:02aaebb71a6b52afaba3ce39c1ec6454ab2bb6db69c78cd2ce7163554ae848d9","observation_id":"738d0cee-20c3-4b20-9962-7b8c1bdad03c","resolution":{"observed_at":"2026-08-06T17:49:34.903388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.914835Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.914835Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:d5a877ea07d2e42ca6d395335f025ea348fbce0295bc9c70925dfedebe16d7d5","observation_id":"de0f54f3-7733-48c3-a7ca-14a2344bf4da","resolution":{"observed_at":"2026-08-06T17:49:34.914835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.911048Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.911048Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:1f27233a01894c73b9281d3c6152dea240408db7c34514f9e2f43061eb6818f6","observation_id":"2dce07eb-d539-47f4-b7e2-3ff2e92a9a68","resolution":{"observed_at":"2026-08-06T17:49:34.911048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.922692Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.922692Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:641216b81e0c87ac40e48f61926b8febca367c6eededcc079effc7ac71397266","observation_id":"da72bc3b-1d72-4d21-b00e-4fb3eae1daec","resolution":{"observed_at":"2026-08-06T17:49:34.922692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T17:49:34.919181Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.919181Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:894f1ba02b67cac2b7573fbd285779d5cb9a60228af94928085950dbedd5184d","observation_id":"fb1ad447-fb3d-417a-bbda-ba92963928ef","resolution":{"observed_at":"2026-08-06T17:49:34.919181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:35.790666Z","title":null,"venue":null,"work_id":"ef90d9de-ebd3-4d2a-8e7a-2621f889c9f9","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.931465Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:c223af29317817023e1a30f5f3908ba16061815d878eb39b5fb7080d550074e4","observation_id":"c1269154-5f01-4e48-9464-a743702026bd","resolution":{"observed_at":"2026-08-06T17:49:35.848311Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-07-06T18:05:03.284784Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-06T17:49:34.927360Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.927360Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:b81d0383474a966e5e26b4be0be73b203c0f6e3e6ec32a5a5657c904b8563b01","observation_id":"5dfb38a4-4956-47d4-9875-0bea4f982ff9","resolution":{"observed_at":"2026-08-06T17:49:34.927360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:35.655271Z","title":null,"venue":null,"work_id":"bc6f80c7-2768-4e0e-911d-d84d45495aaf","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.939634Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:26eab301727c7a253689f06454ac4f0699fd225926f6b3fa6206a49304fefb4d","observation_id":"5e3c5ab6-bafd-4e4b-bf04-bdd6ff960e1a","resolution":{"observed_at":"2026-08-06T17:49:35.695185Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-06T17:49:34.935033Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.935033Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:dca7d02a9cbfe6a758007a5398cb5dd9d70b489a0723d8cd8d895bb114c0e6a3","observation_id":"880af460-c720-488d-9016-1775980c506b","resolution":{"observed_at":"2026-08-06T17:49:34.935033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-06T17:49:34.948907Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.948907Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:fc773e39e0a4cd6dfa3fd729cb175559e434dc2528a76cc60f64bd9cb1abdf6d","observation_id":"77f6682e-d6f3-4804-97bc-14acf2efeb61","resolution":{"observed_at":"2026-08-06T17:49:34.948907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19108","last_updated":"2025-05-25T11:54:32Z","snapshot_observed_at":"2026-08-06T23:04:47.135343Z","submitted_at":"2025-05-25T11:54:32Z","title":"CCHall: A Novel Benchmark for Joint Cross-Lingual and Cross-Modal Hallucinations Detection in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.19108","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19108","snapshot_observed_at":"2026-08-06T17:49:35.067988Z","title":"CCHall: A Novel Benchmark for Joint Cross-Lingual and Cross-Modal Hallucinations Detection in Large Language Models","venue":"cs.CL","work_id":"30e0f882-481f-431b-9347-64fdecf1b230","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.944193Z"},"links":{"cited_paper":"/paper/2505.19108","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:9daf6d917428a5a45cbacf9734228eb0b2970f0173f96929900f9c2241ce9120","observation_id":"e905d582-7990-4c7a-a245-8611b3d2a1ed","resolution":{"observed_at":"2026-08-06T17:49:35.074557Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-06T17:49:34.958286Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.958286Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:03c1f984e600eb88dad301888d070ebf8ddaefd3cab8b2608dc807fe160aa97a","observation_id":"dd4dff17-aeac-4d24-8f70-b2c93b50ac69","resolution":{"observed_at":"2026-08-06T17:49:34.958286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.953969Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.953969Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:e444d3155e4c6f86e597b343f8204c8f57bd883bce1c7c225154df2078cac474","observation_id":"5048f173-f12d-46c6-b396-352c5b0f0fb0","resolution":{"observed_at":"2026-08-06T17:49:34.953969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16038","last_updated":"2024-01-30T14:37:10Z","snapshot_observed_at":"2026-08-06T12:45:18.285554Z","submitted_at":"2024-01-30T14:37:10Z","title":"A Survey on Generative AI and LLM for Video Generation, Understanding, and Streaming","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16038","snapshot_observed_at":"2026-08-06T17:49:34.963022Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.963022Z"},"links":{"cited_paper":"/paper/2404.16038","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:957d0dbe697dec5d9e4e88e95888d853a54576af1dcc026342be846a7861bab5","observation_id":"71857a80-a104-4735-aadb-a2a0ed40550d","resolution":{"observed_at":"2026-08-06T17:49:34.963022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.426483Z","title":"In European Conference on Computer Vision","venue":null,"work_id":"f92c5565-9054-4640-b2a8-153b8bc451a8","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.881390Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:7daf36f3c6f5169a4f2b8b6ec93c80b8fc4337f479c600f9d2c403d08d18aa8c","observation_id":"88264853-d586-461a-bbb6-a98f2bc7408b","resolution":{"observed_at":"2026-08-06T17:49:36.505134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T17:42:42.583231Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":55,"verified_exact":1,"verified_fuzzy":3},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 4 inbound Pith citation observations for arXiv:2507.09876."}