{"as_of":"2026-08-10T07:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:92455949a999075c1800d30fbd41b80f1cc29ea74ac7632173fa34541a1147ad","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T17:50:00.740770Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.26680/citation-record","integrity":"/paper/2605.26680/integrity","json":"/paper/2605.26680/citation-record.json","paper":"/paper/2605.26680"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:b539c2bcacc71da522133dc583abb05944fd2434464ebb8d7a75e3e37137c521","observation_id":"cd60909a-2a0e-43f3-8e2c-d3ed7f1c64cf","resolution":{"observed_at":"2026-06-29T17:53:47.110244Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:f7b5aeed71b03c4f2538c5ea4c635158f7a319a8d3cfef6c57fc345be8689e04","observation_id":"a2dca62f-22ee-4a28-bac8-b4589b824905","resolution":{"observed_at":"2026-06-29T17:53:47.126670Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:16.864468+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:16.864468+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19392","last_updated":"2024-07-02T15:30:03Z","snapshot_observed_at":"2026-08-02T12:22:52.096368Z","submitted_at":"2024-06-27T17:59:45Z","title":"ReXTime: A Benchmark Suite for Reasoning-Across-Time in Videos","version":2},"cited_work":{"arxiv_id":"2406.19392","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.19392","snapshot_observed_at":"2026-06-29T17:53:47.141859Z","title":"Rextime: A benchmark suite for reasoning-across-time in videos.arXiv preprint arXiv:2406.19392, 2024","venue":null,"work_id":"98f6b524-b595-4873-a17c-76e8f83a3925","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2406.19392","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:c6ee5739bcd55ca830a62f244ebaec8cb3812708ec7db0d47385a596400352ea","observation_id":"e2916a80-bca9-4063-b06e-8d2eedb67c33","resolution":{"observed_at":"2026-06-29T17:53:47.143553Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:7494fd1f24e785d6265bedb61191293206cee1cc91811999db976b3b1cb795c6","observation_id":"9a1382c3-4616-4011-9027-4ced28821d41","resolution":{"observed_at":"2026-06-29T17:53:47.054431Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.22315","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.830378Z","title":"Videozoomer: Reinforcement-learned temporal focusing for long video reasoning","venue":null,"work_id":"ed057419-0436-4abb-bdc4-60bf5755a879","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:ebaa6138addccd4303ed9901813073e599747b598819344b1ddaa58e484d986f","observation_id":"e6b53214-d894-4308-ad19-8923e8722b6b","resolution":{"observed_at":"2026-06-29T17:53:47.149003Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":"2503.21776","doi":"10.48550/arxiv.2503.21776","metadata_source":"pith","pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","venue":"cs.CV","work_id":"0ce88332-564c-4361-8e2a-3850eb1ace9c","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:2085b86e187f572e5129f945b45aac207b43e2b2a2edbbd4e59820395f7a2885","observation_id":"b83c7ede-0965-4cb3-bd97-00d5d0cf9956","resolution":{"observed_at":"2026-06-29T17:53:47.146062Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:abc2478aac9ac75b113fbf7b74a8fc6c5fd96f57dbc3732ff2d24baefa279fd6","observation_id":"7f4a8675-abfe-4692-a958-19d83c573a17","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.24786","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T12:09:48.884636Z","title":"Love- r1: Advancing long video understanding with an adaptive zoom-in mechanism via multi-step reasoning","venue":null,"work_id":"c40ac341-47ce-43cd-83ba-adebdd00b99b","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:50b05fdc5184c481ec2226cfa3f7f04a26ab70bf0fb5c339fbe6ea22e2a92856","observation_id":"c78ab457-45da-4bcc-a101-5a727d13b26f","resolution":{"observed_at":"2026-06-29T17:53:47.061147Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Tall: Temporal activity localization via language query","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:fb4480db2bef19d8dc88a34b52a38a9a5bcdaceca8ce3f3ea45244e238ab6e40","observation_id":"fb944604-a3fd-49d6-a8d4-57514e97112e","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Gemini 3 Pro Model Card","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:f4e439f6870b802a2ad26972af66731ad6d0d64dabc686369109fdfe3347132f","observation_id":"394b3f26-cdff-4683-9944-e0aa643365c6","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:c038dd3251ae0dfd8a4a0066767b9bf3606847c53cb1b638f07490a2ac4fce59","observation_id":"72b1b12a-568c-471e-b8e7-cbc03326b264","resolution":{"observed_at":"2026-06-29T17:53:47.063996Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.24304","doi":"10.48550/arxiv.2509.24304","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Framethinker: Learning to think with long videos via multi-turn frame spotlighting","venue":"arXiv (Cornell University)","work_id":"4ed1536d-cea3-451a-bb12-5f4188bc89ec","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:533c4ee9afc213c89af61ec4ed37f6874e3ab28ad5294c80d9afd08e30b19ec4","observation_id":"0d0bc326-c0a6-4b3d-8765-bd9697045bbf","resolution":{"observed_at":"2026-06-29T17:53:47.055731Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06749","last_updated":"2026-02-28T21:10:52Z","snapshot_observed_at":"2026-08-07T18:44:26.813869Z","submitted_at":"2025-03-09T20:06:45Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":"2503.06749","doi":"10.48550/arxiv.2503.06749","metadata_source":"pith","pith_arxiv_id":"2503.06749","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","venue":"cs.CV","work_id":"38998646-34ee-4605-b661-ab356f16d6e5","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2503.06749","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:bcc4ac6066efa2a85ecb9f7640bfbd1985a410184108a8d70caa1c352c43c0b7","observation_id":"c17be9e3-4f0d-4877-8f56-8a2df063a1b4","resolution":{"observed_at":"2026-06-29T17:53:47.066375Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:dabcb45396b9c5f0448aaff5722ba1dc8aa330f7d17ce83f1f1c1a132d1c2d63","observation_id":"ef8e1294-98b4-4ddf-bb3f-eeee7d438180","resolution":{"observed_at":"2026-06-29T17:53:47.115262Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Chat-univi: Unified visual representation empowers large language models with image and video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:91c972cbb47466bba258f2a0d54e3dcf579479a8832cb6045a4d2d1e69e38ec3","observation_id":"8054942e-8613-4f4b-bc66-225174eb4613","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07491","last_updated":"2025-06-23T13:45:50Z","snapshot_observed_at":"2026-08-09T09:48:45.884814Z","submitted_at":"2025-04-10T06:48:26Z","title":"Kimi-VL Technical Report","version":3},"cited_work":{"arxiv_id":"2504.07491","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.07491","snapshot_observed_at":"2026-07-08T06:34:41.884360Z","title":"Kimi-VL Technical Report","venue":"cs.CV","work_id":"c876520f-8a20-44f3-b92a-bf7d35bd430f","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2504.07491","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:d5f0152d895a1a1cbd5197103a1c9cd4751c302440632be27d824ca4ebef023b","observation_id":"e0eea99c-922c-43ce-a610-f15bafc3d2aa","resolution":{"observed_at":"2026-06-29T17:53:47.117940Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Dense-captioning events in videos","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:ce51470d64a701136f74f2e75baf334dcf57e61fcc7f48a6364b1cf8412b73ae","observation_id":"15a68bf9-b97b-4b4a-8ea3-abfaaf930ed2","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:0a2d578ce44b922ab3161203a3ffcd0aa89e4499c6fbfb461f5e109283f8b4b2","observation_id":"1951d29b-8c83-4961-a868-06c5ff0cfc84","resolution":{"observed_at":"2026-06-29T17:53:47.132160Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01908","last_updated":"2025-06-02T17:28:26Z","snapshot_observed_at":"2026-08-07T11:29:22.385642Z","submitted_at":"2025-06-02T17:28:26Z","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","version":1},"cited_work":{"arxiv_id":"2506.01908","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01908","snapshot_observed_at":"2026-07-04T16:49:57.219030Z","title":"Reinforcement learning tuning for videollms: Reward design and data efficiency","venue":null,"work_id":"058b4756-2b75-4b6d-aa11-ba523aab08e0","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2506.01908","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:e71ab5cc1094938e0f56687224e7a77b29ca3d73b75cf601691b931f4d4b7054","observation_id":"5f329e02-4e84-42af-94a8-3b9914d312ba","resolution":{"observed_at":"2026-06-29T17:53:47.135590Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"cited_work":{"arxiv_id":"2504.06958","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.06958","snapshot_observed_at":"2026-07-04T16:49:57.434938Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","venue":"cs.CV","work_id":"7be17d59-6cde-455a-99c3-06e28659839f","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2504.06958","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:d2d99ff8ab9851aaac5c12f1f6f2b73fd1f41c2d959cafb61d49888c9cf221c7","observation_id":"ec21962f-3b64-4db1-883d-9015dceab48c","resolution":{"observed_at":"2026-06-29T17:53:47.109477Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.07801","last_updated":"2026-05-22T06:26:26Z","snapshot_observed_at":"2026-07-06T22:45:00.816348Z","submitted_at":"2026-02-08T03:45:50Z","title":"VideoTemp-o3: Harmonizing Temporal Grounding and Video Understanding in Agentic Thinking-with-Videos","version":4},"cited_work":{"arxiv_id":"2602.07801","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.07801","snapshot_observed_at":"2026-07-02T17:27:15.692225Z","title":"VideoTemp-o3: Harmonizing Temporal Grounding and Video Understanding in Agentic Thinking-with-Videos","venue":"cs.CV","work_id":"163fafac-9454-4660-b9f6-fea1651a94fc","year":2026},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2602.07801","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:ce0c7c73ecf36e62fac58f07f74a6eec6db9fd4d41d6c461e627aca00cc1ae11","observation_id":"a3c115e8-2e73-4b43-85f0-1d82a03280be","resolution":{"observed_at":"2026-06-29T17:53:47.138226Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.06485","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T21:37:25.381494Z","title":"Video-rts: Rethinking reinforcement learning and test-time scaling for efficient and enhanced video reasoning.arXiv preprint arXiv:2507.06485, 2025","venue":null,"work_id":"792cc6cf-83a1-4293-a847-f347b4b1f5f2","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:1ef24964264f1b00bd8e8b6820a154bfd0168b96f116e8230f64fbc0fb5448a4","observation_id":"107391d0-6271-4530-93eb-512804475c8d","resolution":{"observed_at":"2026-06-29T17:53:47.124073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":"2306.05424","doi":"10.48550/arxiv.2306.05424","metadata_source":"pith","pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","venue":"cs.CV","work_id":"51f627f4-8fae-4882-a3e9-abdf932ef27b","year":2023},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:a46dd579a2064cb4522d0e7caafc22a852b84b129527fa475d8c1f335b991943","observation_id":"0c2ea2c2-2550-4613-8d19-f7645ee5f965","resolution":{"observed_at":"2026-06-29T17:53:47.094370Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.20579","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:49:57.203105Z","title":"Open-o3 video: Grounded video reasoning with explicit spatio-temporal evidence","venue":null,"work_id":"fb25eed0-9409-44e6-af47-191ebb595de1","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:7e18f6284428fb773bb34ab5db09b44f92e75609ac00d6d5a638d36d90b808b4","observation_id":"ff8b3412-52c2-45bb-a696-002569f432df","resolution":{"observed_at":"2026-06-29T17:53:47.087447Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Visual cot: Advancing multi-modal language models with a comprehensive dataset and benchmark for chain-of-thought reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:b55536f3e7ab0abe5be1a80f35befb4d7b51a76952fe2041c2fa6423cce259c0","observation_id":"1644c19d-11a6-47f4-8706-7495d3d7a138","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:9b3d3ddd43fa7c6fc6e528f82b5afabefafd075f6081f2c2ac76eaf0874a02ce","observation_id":"a7303139-5c10-4923-a01b-e1d20d2ad53c","resolution":{"observed_at":"2026-06-29T17:53:47.103700Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06157","last_updated":"2024-05-30T09:11:02Z","snapshot_observed_at":"2026-08-10T06:26:38.514349Z","submitted_at":"2024-05-30T09:11:02Z","title":"Temporal Grounding of Activities using Multimodal Large Language Models","version":1},"cited_work":{"arxiv_id":"2407.06157","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06157","snapshot_observed_at":"2026-06-29T17:53:47.084226Z","title":"Temporal grounding of activities using multimodal large language models","venue":null,"work_id":"4f1e7494-621d-4884-9bb3-74cd3c3867ba","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2407.06157","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:fae8a53b3cf4abcc63b3e21b355dfc62b5c04ca4f8571866eaf75ebcbb075bce","observation_id":"0e945c59-2149-4293-8372-a9166d2c4bb0","resolution":{"observed_at":"2026-06-29T17:53:47.085857Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Adaptive keyframe sampling for long video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:2002ef0cf0465c84a27e9050903ebbd2fdf5a52120da1627a48a88353a1b152f","observation_id":"4a0e107d-5a71-460e-9250-32ae08b5c9d0","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.16141","last_updated":"2025-06-19T08:49:13Z","snapshot_observed_at":"2026-08-08T06:18:41.273900Z","submitted_at":"2025-06-19T08:49:13Z","title":"GRPO-CARE: Consistency-Aware Reinforcement Learning for Multimodal Reasoning","version":1},"cited_work":{"arxiv_id":"2506.16141","doi":"10.48550/arxiv.2506.16141","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.16141","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Grpo- care: Consistency-aware reinforcement learning for multimodal reasoning","venue":"ArXiv.org","work_id":"41a7dfb7-5564-4132-89bf-b7301ad2fd75","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2506.16141","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:cb4335df337412886777da2e49d8b40cc45f3e54a84912a5e42171853ab6326f","observation_id":"9b56a2e3-91fe-42c4-a10f-3c2b1b66a312","resolution":{"observed_at":"2026-06-29T17:53:47.092273Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.12434","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T17:40:00.905697Z","title":"Videorft: Incentivizing video reasoning capability in mllms via reinforced fine-tuning","venue":null,"work_id":"89c249fe-60da-4724-bb0c-770442676a88","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:5ea53309ffbabae785633baf7239ae3fc40db5497fa740e9a5dcc98685dbadff","observation_id":"5fbac1b6-d5cb-4ca0-b42d-7dab98014b6b","resolution":{"observed_at":"2026-06-29T17:53:47.101742Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"cited_work":{"arxiv_id":"2406.08035","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.08035","snapshot_observed_at":"2026-07-04T16:09:57.146670Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","venue":"cs.CV","work_id":"e9bfdf40-cd28-4d57-98b8-31b7e97f9f2f","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2406.08035","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:9529a59ab4e3b33a032b6ab1134f20d42ca5c7b24d7d5132bf0b33fa822725b7","observation_id":"1d80d592-0544-49b3-8c17-2ea82b75c705","resolution":{"observed_at":"2026-06-29T17:53:47.131542Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12605","last_updated":"2025-03-23T13:47:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-16T18:39:13Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":"2503.12605","doi":"10.48550/arxiv.2503.12605","metadata_source":"pith","pith_arxiv_id":"2503.12605","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","venue":"cs.CV","work_id":"a8fbb385-8ed1-4952-9ee8-8d03be2b6821","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2503.12605","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:050d5ae94e0f65cc466433da87f7ba82c76057cfdb5d5cb46bfe08a724f05b15","observation_id":"92f3acad-7a0e-4a6e-85a3-d4edb8fd8e77","resolution":{"observed_at":"2026-06-29T17:53:47.115143Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Chain-of-thought prompting elicits reasoning in large language models.Advances in Neural Information Processing Systems, 35:24824–24837, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:8ddb575480b6c4ee221deae7ccacc77f6d9a8742e97e0bd2118fcf2c96429b82","observation_id":"0391f5a4-5d2d-453f-829f-e644916d4f07","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Can I trust your answer? Visually grounded video question answering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:86f67c1e2367dc54acec28483498872c9d1bc62dc56c32c59e07957a3b8a33bc","observation_id":"36096b60-118f-48c3-8771-3ba85442b49e","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Videochat-r1.5: Visual test-time scaling to reinforce multimodal reasoning by iterative perception","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:fc208b4757072326aab16d6badfee99bceb7fbb037e8384f8e2cc20fc23b6242","observation_id":"3d8fc918-8948-46c9-893b-e1e52b821ff8","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.20785","last_updated":"2026-05-21T10:39:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-25T19:22:48Z","title":"LongVT: Incentivizing \"Thinking with Long Videos\" via Native Tool Calling","version":3},"cited_work":{"arxiv_id":"2511.20785","doi":null,"metadata_source":"pith","pith_arxiv_id":"2511.20785","snapshot_observed_at":"2026-07-02T23:57:29.009867Z","title":"LongVT: Incentivizing \"Thinking with Long Videos\" via Native Tool Calling","venue":"cs.CV","work_id":"18448f92-5a5c-48d8-bc8e-62464eb1c501","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2511.20785","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:75601f7fb6a43b7bab27e652a486b57a7c63ccbe11e56e978d34b1ac602b8597","observation_id":"f51a4f7d-3e4e-4c37-8c49-fe4634c4e81d","resolution":{"observed_at":"2026-06-29T17:53:47.134508Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.27280","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T23:39:04.715431Z","title":"Focus: Efficient keyframe selection for long video understanding","venue":null,"work_id":"2397701a-d29b-4144-9a17-058f8b43049d","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:8f5fe81aedf4da73d30b8110062ba57f861606e8fedafb3a75d60cad137ff3de","observation_id":"0029ac56-f577-430e-b1a3-1fbcf02a33d1","resolution":{"observed_at":"2026-06-29T17:53:47.137255Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Frame-voyager: Learning to query frames for video large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:30eecee10328c3492eb61eb6bc95845b75e7f439061f53259962649b6c4388f8","observation_id":"838367aa-8689-4197-9e11-14311c95be71","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.23224","last_updated":"2026-05-21T09:05:36Z","snapshot_observed_at":"2026-08-02T12:29:31.448544Z","submitted_at":"2026-01-30T17:47:30Z","title":"Video-o3: Native Interleaved Clue Seeking for Long Video Multi-Hop Reasoning","version":2},"cited_work":{"arxiv_id":"2601.23224","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.23224","snapshot_observed_at":"2026-07-02T17:27:15.492915Z","title":"Video-o3: Native Interleaved Clue Seeking for Long Video Multi-Hop Reasoning","venue":"cs.CV","work_id":"d9291b6d-e10d-4ad4-b474-2cba910d9f4e","year":2026},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2601.23224","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:1a6d31e71f5d52bdaa00b03b3db066b49ceb1d05be3b2a689282c9241fb801f2","observation_id":"f1224666-04e5-4bcb-be0a-0ba2d4c7743c","resolution":{"observed_at":"2026-06-29T17:53:47.121262Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:008f2c2e8ec87ddcb2aad68add20fe91e860f222c5787f82fb103b4dbb4dc646","observation_id":"585e08e9-4107-4b7b-b8a9-e61d522f013e","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.04416","last_updated":"2025-09-03T07:11:03Z","snapshot_observed_at":"2026-08-09T07:51:37.523530Z","submitted_at":"2025-08-06T13:03:21Z","title":"Thinking With Videos: Multimodal Tool-Augmented Reinforcement Learning for Long Video Reasoning","version":2},"cited_work":{"arxiv_id":"2508.04416","doi":"10.48550/arxiv.2508.04416","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.04416","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Thinking with videos: Multimodal tool-augmented reinforcement learning for long video reasoning","venue":"arXiv (Cornell University)","work_id":"824594ad-21b0-49b7-88f8-fac8553528e3","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2508.04416","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:2c0e39a2b5ac47f3751e4774c9e056ac9254e346a52fbe0916cf68d9310ffed9","observation_id":"c00334ef-161f-491f-be05-6bc80315fd53","resolution":{"observed_at":"2026-06-29T17:53:47.123574Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12937","last_updated":"2025-08-04T04:22:09Z","snapshot_observed_at":"2026-08-06T21:34:54.534751Z","submitted_at":"2025-03-17T08:51:44Z","title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","version":2},"cited_work":{"arxiv_id":"2503.12937","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.12937","snapshot_observed_at":"2026-07-10T05:16:48.080500Z","title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","venue":"cs.AI","work_id":"e1614961-16d2-43bc-908a-8c57da5b151c","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2503.12937","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:8de78eb0027406c2d8f320cedc949c9e342d6d74134d5ed991c7299dbc21c37a","observation_id":"dbed8afa-7fa0-4bc1-af93-846a9da60741","resolution":{"observed_at":"2026-06-29T17:53:47.126233Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:184be5a5b3cd1d68d83cf3e94d6deea42cb89e3b1ff2ba806f480b745de3b419","observation_id":"aff93130-9265-4d0a-b03c-6ca38dc78df7","resolution":{"observed_at":"2026-06-29T17:53:47.139711Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14362","last_updated":"2026-03-01T04:59:56Z","snapshot_observed_at":"2026-08-02T12:23:34.946873Z","submitted_at":"2025-05-20T13:48:11Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.14362","doi":"10.48550/arxiv.2505.14362","metadata_source":"pith","pith_arxiv_id":"2505.14362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","venue":"cs.CV","work_id":"5f6cf57b-2407-4127-b39c-d8a61494e474","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2505.14362","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:8d364ae14c56337a134adaac65fb58e57fca7c7a60d45a63e1dcc910da68b45d","observation_id":"8f0e0d7e-22b1-4299-a4e3-4b78ef8ff049","resolution":{"observed_at":"2026-06-29T17:53:47.140730Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.04264","doi":"10.48550/arxiv.2406.04264","metadata_source":"pith","pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","venue":"cs.CV","work_id":"346256da-dd21-4cc3-9a98-519467614854","year":2024},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:a9d314cbe2be6876ceb759a1de0614162815d3225346b0470397ac924a671993","observation_id":"4c05d73f-2aa9-4308-9574-8d453e49aa30","resolution":{"observed_at":"2026-06-29T17:53:47.080480Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-10T07:07:56.707005Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":"2504.10479","doi":"10.48550/arxiv.2504.10479","metadata_source":"pith","pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","venue":"cs.CV","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","year":2025},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:8cd674bcb66ae676cd49974de6fb3655b52b590aa01a88e89741eaab16c5470e","observation_id":"6afaf91e-c776-440d-8e98-62407c20a90d","resolution":{"observed_at":"2026-06-29T17:53:47.129478Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:8125f1d8daf19520eb52d554b4e3cc7b79669e4679dcc4893c60d29a846b2304","observation_id":"4931a756-9708-46db-88ca-18a2999ec714","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Explain why this portion is critical","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:4659dfa1c1b180c34b6d21e72b88630bd975acbfe0bfdabe379158a019304f78","observation_id":"1ec86a77-a6e5-43c9-86e3-c15b10774c4c","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:feaaeb54161494dabb31b71c669344806380f7fbd878107b904aba0ad9874ffa","observation_id":"4262fb4f-cbfe-4cb2-957e-1f27ade60ac6","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"zoom_in_cot","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:bd9c585a2f905ee37869e688e95a02692eb1d7c2ce073d9c4db0322b2a86023b","observation_id":"b5c64ec9-779e-47a3-9855-b09c6bfb9178","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Using the visual information in the segment, reason step-by-step to reach the answer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:e513acf3caa181007ee85a9443b439a82c57e91af127f9538ec98d3f6ba625bb","observation_id":"0e959838-c1dd-4c85-9232-371c320b6787","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"answer_cot","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:33797b3d19aa78dd4c6787bde97e5a621ac4a88959fe4fbe2939ca6f55b0e159","observation_id":"26696af3-b55b-4b4c-9e8a-e38be90fb1cf","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:10a3a810c2dce11f3c5e3c16f55bcca68400d25cf9ce76ae56b97d05b6485e36","observation_id":"abf719dc-f887-4c9f-abdb-568734f7f3b6","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:f86db7f14e80786930b583b300a30a359481d16605d219cbaa89e10bb256789c","observation_id":"e5c16279-b35f-4827-8bc8-6787148f2b7a","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Keep the reasoning concise and non-repetitive","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:5bb4f8f22a8ce1954e4d6772afda8e9d1e69f731333eb2d4d024fbc5ebcddc2f","observation_id":"e0db0caf-164f-44a7-9846-382b920b2d3f","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:d1df4b5b2e190d294003cb7c1dc99566acc399f9674e3b59160fe070c9b39919","observation_id":"0219894b-ceeb-4d60-8b6c-e4be36b5bfef","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T17:50:00.740770Z","title":"Max retrieval / injection","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-29T17:50:00.740770Z"},"links":{"citing_paper":"/paper/2605.26680"},"observation_digest":"sha256:7e4885d799d8160deb05157024f72d1dcc1f753e3189fa745ab83878019ad50c","observation_id":"84ab49aa-8ad4-4259-8c7b-1a2b2a60c609","resolution":{"observed_at":"2026-06-29T17:50:00.740770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2605.26680","last_updated":"2026-05-26T08:16:16Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:16:16Z","title":"DynFrame: Adaptive Reasoning-Driven Multimodal Framework with Dynamic Frame Augmentation for Complex Video Understanding"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":23,"verified_exact":33,"verified_fuzzy":0},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 0 inbound Pith citation observations for arXiv:2605.26680."}