{"as_of":"2026-08-07T19:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ef9e0a702c63e086377ea76f8ccb0143ed67adfd98fb458fa338a6d8228183b8","coverage":[{"denominator":74,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":74,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-15T20:48:44.933542Z","state":"measured"},{"denominator":76,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":76,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T12:10:52.100410Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-06-29T22:13:59.633393Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"cited_work":{"arxiv_id":"2602.17555","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.17555","snapshot_observed_at":"2026-06-29T22:13:59.633393Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","venue":"cs.CV","work_id":"34223ca6-ceac-45d4-98ad-4c6d94723a11","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2602.17555","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:4653928e286ee5fa432aa6347320061310ce4295af86fc5696f184c34f0c891e","observation_id":"ca9de3d6-bab6-4863-899b-95acdc4c3104","resolution":{"observed_at":"2026-06-29T22:13:59.634880Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.17555","snapshot_observed_at":"2026-07-11T12:10:52.100410Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.04872","last_updated":"2026-07-06T09:47:23Z","snapshot_observed_at":"2026-08-05T15:57:57.173805Z","submitted_at":"2026-07-06T09:47:23Z","title":"EventCoT: Event-centric Video Chain-of-thought for Reasoning Temporal Localization","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-07-11T12:10:52.100410Z"},"links":{"cited_paper":"/paper/2602.17555","citing_paper":"/paper/2607.04872"},"observation_digest":"sha256:a5b4b8dbfb6527f10a1139cb6b8a4a064cf059cbfac156a98898c46d8fc2bd01","observation_id":"85dba891-455d-446f-8815-b7e93f4c9785","resolution":{"observed_at":"2026-07-11T12:10:52.100410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2602.17555/citation-record","integrity":"/paper/2602.17555/integrity","json":"/paper/2602.17555/citation-record.json","paper":"/paper/2602.17555"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The claude 3 model family: Opus, sonnet, haiku.Claude-3 Model Card, 1(1):4","venue":null,"work_id":"b43ef26d-361f-46b3-a5f6-c74536435f1b","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:d58d33a276e472999fd078f94fb60201bed458c1055bf07048cf65cb5f60c856","observation_id":"9f5fb39e-c0c8-4292-af5c-a5aa4623a85d","resolution":{"observed_at":"2026-05-15T20:50:17.912633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:16730a892b70d37e42c3175047ff4f6023d2947377dcb7fe8d4bc4aafb477a7b","observation_id":"415449e0-40c5-4efb-b274-21a9fc0545b0","resolution":{"observed_at":"2026-05-15T20:50:17.355161Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06486","last_updated":"2025-03-09T07:07:03Z","snapshot_observed_at":"2026-08-07T17:19:21.993775Z","submitted_at":"2025-03-09T07:07:03Z","title":"PerturboLLaVA: Reducing Multimodal Hallucinations with Perturbative Visual Training","version":1},"cited_work":{"arxiv_id":"2503.06486","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06486","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Perturbollava: Reducing multimodal hallucinations with per- turbative visual training.arXiv preprint arXiv:2503.06486","venue":null,"work_id":"7a75722a-70bd-4c38-81cc-ddbe39ab29cd","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2503.06486","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:4cece155a925f4e488060aafb864efa5f59fb934c54e4276181ab5ecf72767fc","observation_id":"f0f7b3eb-afa1-4583-88dd-bcfc239420b3","resolution":{"observed_at":"2026-05-15T20:50:17.363150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rextime: A benchmark suite for reasoning-across-time in videos.Advances in Neural In- formation Processing Systems, 37:28662–28673","venue":null,"work_id":"ff67c7e1-a377-46ec-aca7-3b2da5457373","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:e0ec7ed7bf41be1cd9f5db0f0eab0d9d1fa1822ca1fe539e514c2a4c121182bf","observation_id":"c8c6babe-6981-4bb2-a476-a7642d796942","resolution":{"observed_at":"2026-05-15T20:50:17.903131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sharegpt4video: Improving video understand- ing and generation with better captions.Advances in Neural Information Processing Systems, 37:19472–19495","venue":null,"work_id":"6dd48033-ac11-4d41-91dd-b567f0fcc51f","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:327ce1f927f03618ab82ddef3f22018c29e2f84be87df625abf6cd613338ee7e","observation_id":"10bc9c55-0139-423a-be98-cd0f5bc24946","resolution":{"observed_at":"2026-05-15T20:50:17.867739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.07966","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.111453Z","title":"Scaling rl to long videos","venue":null,"work_id":"51d8c593-7e1f-4b0e-a674-da3e22e07f8f","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:20a0d80ad3f3af0e21b095d54f476643721b63e38a4848f52d5c42aca929948c","observation_id":"d019d1a7-b3ae-45af-aaa4-56e92d853f1c","resolution":{"observed_at":"2026-05-15T20:50:17.360259Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":"2406.07476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-07-04T16:49:57.279303Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","venue":"cs.CV","work_id":"ccfc3f89-c510-45f1-8a35-ed1a56c0ae5c","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:f7a3e3bca2a43b910f5804c1b30d69c258e4883fb5a92b0cae953f2993b3a9f0","observation_id":"3a46ea84-1d00-424c-9e5c-d0430f0c79a3","resolution":{"observed_at":"2026-05-15T20:50:17.352986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11495","last_updated":"2025-03-14T15:21:44Z","snapshot_observed_at":"2026-08-07T17:03:39.921455Z","submitted_at":"2025-03-14T15:21:44Z","title":"V-STaR: Benchmarking Video-LLMs on Video Spatio-Temporal Reasoning","version":1},"cited_work":{"arxiv_id":"2503.11495","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.11495","snapshot_observed_at":"2026-07-04T03:29:31.466663Z","title":"V-star: Bench- marking video-llms on video spatio-temporal reasoning","venue":null,"work_id":"1a8331e9-feb0-4852-b486-606942c93723","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2503.11495","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:819d2d30c51ec5f397a328b3a61a5744d29d1d42683fb1604271a008656f138a","observation_id":"4434c415-058f-4bc4-b1d9-d3381d95eb76","resolution":{"observed_at":"2026-05-15T20:50:17.357781Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Spatial-temporal trans- former for dynamic scene graph generation","venue":null,"work_id":"55bf98de-ebfc-4929-8703-8f997ac6ba59","year":2021},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:2805f721fc87aef23bae7a3478108cf532aada4616408b18ac615f63d7c5f59a","observation_id":"7b3d732e-4471-4481-9e57-33ed2db58000","resolution":{"observed_at":"2026-05-15T20:50:17.901120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.17018","doi":"10.48550/arxiv.2505.17018","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sophiavl-r1: Reinforcing mllms reasoning with thinking reward","venue":"ArXiv.org","work_id":"de003be6-a854-4f8d-84ec-4a852d8422c6","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:c850a263a38b87217efa8e6fe325ac71c5d743e0b29bc51271b3f1e534db6202","observation_id":"d710b2ce-a93c-40f1-ab5d-6408887bcae1","resolution":{"observed_at":"2026-05-15T20:50:17.326138Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing.Advances in Neural Information Processing Sys- tems, 37:89098–89124","venue":null,"work_id":"58848ce5-a7b1-4215-a46b-b60d0968c15d","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:b7ce3209d9e8569b0bc56d11bf16ad06321cf3a46dea853d85b079454cfdc7f7","observation_id":"1e388425-27dd-41c1-8682-95d614269836","resolution":{"observed_at":"2026-05-15T20:50:17.917780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-of-thought: Step-by-step video reasoning from perception to cognition","venue":null,"work_id":"3272d41b-d0aa-47b1-a913-89fef0f0ee23","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:461e410cf5e270153f38e5b4b042219f9222462ed45937dd069489033420b8aa","observation_id":"0ae663e0-08ca-419d-8925-c9b064e5d3cd","resolution":{"observed_at":"2026-05-15T20:50:17.912993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":"2503.21776","doi":"10.48550/arxiv.2503.21776","metadata_source":"pith","pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","venue":"cs.CV","work_id":"0ce88332-564c-4361-8e2a-3850eb1ace9c","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:6d7f753a74fcfb44fd4d62c4ef0c15b5e400f8e73031e01f5cb48a8dd1c2a2c3","observation_id":"a4ffcf43-2769-47e2-88ed-92569688896c","resolution":{"observed_at":"2026-05-15T20:50:17.350762Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":"7279f254-03c9-4be5-b8e8-ac4642583959","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:6604bd0a0e3df3806d09d3fb67a34e7202e54bbadb5c5bb4f3466acf995cdf7a","observation_id":"fa7a02a4-645d-4a61-a3d5-1c8085a543b2","resolution":{"observed_at":"2026-05-15T20:50:17.907308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.22355","last_updated":"2025-07-07T15:42:11Z","snapshot_observed_at":"2026-08-07T04:25:28.195802Z","submitted_at":"2025-06-27T16:05:34Z","title":"Embodied AI Agents: Modeling the World","version":3},"cited_work":{"arxiv_id":"2506.22355","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.22355","snapshot_observed_at":"2026-07-04T11:29:51.053006Z","title":"Embodied AI Agents: Modeling the World","venue":null,"work_id":"54a68230-d29b-4113-96dc-ae0da1ebc38b","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2506.22355","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:90dc352489c5cbec79ff769b05a712c475cd15ae0e754837a6c5956168ad276f","observation_id":"2029d3f8-9b70-4a18-99c2-0fb321b3d092","resolution":{"observed_at":"2026-05-15T20:50:17.337127Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.00318","last_updated":"2026-04-04T04:38:43Z","snapshot_observed_at":"2026-07-06T21:34:05.672636Z","submitted_at":"2025-05-31T00:08:21Z","title":"Chain-of-Frames: Advancing Video Understanding in Multimodal LLMs via Frame-Aware Reasoning","version":2},"cited_work":{"arxiv_id":"2506.00318","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.00318","snapshot_observed_at":"2026-07-02T17:27:15.716678Z","title":"Chain-of-Frames: Advancing Video Understanding in Multimodal LLMs via Frame-Aware Reasoning","venue":"cs.CV","work_id":"9208ed98-514d-40a0-901a-15cd610da796","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2506.00318","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:31e774465c5939de9ad7adda7ddd28939751b01069a80314c9229f891ac9c128","observation_id":"c45a468b-95a2-4c3b-977b-f9ea127e6646","resolution":{"observed_at":"2026-05-15T20:50:17.331346Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Real-time scene understanding for blind users: Enhancing vision-language models for accessibility","venue":null,"work_id":"d1e1d585-6381-43e2-a7c9-aa2984ca7f1b","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:378a24e770bfad14ee46807d156b1cbb3ca6093e00e4cc826ddb279b1033ebff","observation_id":"2c65fb75-4f22-4c20-8e04-1431f1850f7b","resolution":{"observed_at":"2026-05-15T20:50:17.908979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deepseek-r1 incentivizes reasoning in llms through reinforcement learning.Nature, 645(8081):633– 638","venue":null,"work_id":"06aeab51-9f75-4e8d-94d6-422b4e0cd375","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:04af7ae61f040554065ad988115a61bde498267091f7e90e4ff82cc658db726f","observation_id":"ebb49bc6-80ca-470d-b840-06d10b4dd78f","resolution":{"observed_at":"2026-05-15T20:50:17.909349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05643","last_updated":"2025-03-03T10:28:30Z","snapshot_observed_at":"2026-08-07T05:23:25.250251Z","submitted_at":"2024-10-08T02:46:30Z","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","version":3},"cited_work":{"arxiv_id":"2410.05643","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05643","snapshot_observed_at":"2026-07-02T12:16:57.678370Z","title":"Trace: Temporal grounding video llm via causal event modeling","venue":null,"work_id":"b2af5095-2950-48b5-af5e-270218b8d041","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2410.05643","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:88c75c29f005498c8b8f819643b4e82cc1a657c6522b7a3573402ef9ce2fa596","observation_id":"86ef4eb3-55aa-463f-95eb-5e51f1234ed0","resolution":{"observed_at":"2026-05-15T20:50:17.305860Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09445","last_updated":"2025-06-11T06:52:31Z","snapshot_observed_at":"2026-08-07T04:45:44.440916Z","submitted_at":"2025-06-11T06:52:31Z","title":"TOGA: Temporally Grounded Open-Ended Video QA with Weak Supervision","version":1},"cited_work":{"arxiv_id":"2506.09445","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09445","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Toga: Tempo- rally grounded open-ended video qa with weak supervision","venue":null,"work_id":"85ab639a-924f-4916-a6a7-0a2e42dc9efb","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2506.09445","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:9c5c4771ebab462bcb1cddbcedc551e1cdff826b798fc291139cbe2d935b51a3","observation_id":"6d894c32-3ecb-45dd-a86a-c6762ad69e19","resolution":{"observed_at":"2026-05-15T20:50:17.342680Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videoespresso: A large-scale chain-of-thought dataset for fine-grained video reasoning via core frame selection","venue":null,"work_id":"42b31453-7418-4ffd-afb1-9ca6a57da995","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:70dd312db8ea2cb8a3ade63350825fbf3f27391d0d95606382c5762cc4e1be89","observation_id":"50e746e7-4b5e-4bc7-993e-949296716072","resolution":{"observed_at":"2026-05-15T20:50:17.897299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"To- wards open-vocabulary scene graph generation with prompt- based finetuning","venue":null,"work_id":"3d225cf9-53a5-4998-be84-cb41dac713e2","year":2022},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:71c1ef5b4948a8a41da39b33a4707961c4771aa42fdb31bda9ee26221a2e278a","observation_id":"31cdb3d0-9a54-4998-a147-75949029aa1b","resolution":{"observed_at":"2026-05-15T20:50:17.899558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Glm-4.1 v-thinking: Towards versatile multi- modal reasoning with scalable reinforcement learning.arXiv e-prints, pages arXiv–2507","venue":null,"work_id":"db0b6f23-a399-40a3-9aad-d02e15bb99f2","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:2a82cecac5cd9f71d8d4c875513549a15ca1b13e445c4370be64b7bd1820aa3f","observation_id":"915d2a63-00a5-4147-a73e-01012cd22d7c","resolution":{"observed_at":"2026-05-15T20:50:17.919486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.06317","last_updated":"2025-08-08T13:47:00Z","snapshot_observed_at":"2026-08-07T09:55:36.226227Z","submitted_at":"2025-08-08T13:47:00Z","title":"Uncertainty-quantified Rollout Policy Adaptation for Unlabelled Cross-domain Temporal Grounding","version":1},"cited_work":{"arxiv_id":"2508.06317","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.06317","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Uncertainty-quantified roll- out policy adaptation for unlabelled cross-domain temporal grounding","venue":null,"work_id":"3b36b78f-3ca9-4663-b65a-f491584c1840","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2508.06317","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:c688785a5515066b893cac064adad32c2493a9adff1a3a1d0c1974ae9d57c829","observation_id":"21156c8a-d0fd-423f-94f0-37dd52a0d419","resolution":{"observed_at":"2026-05-15T20:50:17.331446Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vtimellm: Empower llm to grasp video moments","venue":null,"work_id":"6691db29-d031-4427-84ea-feaea308b737","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:02d36749f243b707797166abd45fb1fc252638009ea945688caae686756f0846","observation_id":"5d5cdca8-a824-44a3-af0a-987e97696a04","resolution":{"observed_at":"2026-05-15T20:50:17.897202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lita: Language instructed temporal-localization assistant","venue":null,"work_id":"26fac5ae-4d3b-4a5b-a270-a4a4d76dde6f","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:4eff9ac6d49287c1a092a7eece7260ec03533604d5d4982337d3ec2be7ba816c","observation_id":"7e8bd92e-e70b-4914-88c3-f42ae7fee634","resolution":{"observed_at":"2026-05-15T20:50:17.904883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Building a mind palace: Structuring environment-grounded semantic graphs for ef- fective long video analysis with llms","venue":null,"work_id":"dfa03589-7e50-4c7d-b563-c4cbab0acbe5","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:827c742be457f28ace612ec0ff926edbc6b15e4e173341114cbbe55964a5bcb4","observation_id":"7c28ffae-31e7-4afc-a5ce-581409848079","resolution":{"observed_at":"2026-05-15T20:50:17.906816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Action genome: Actions as compositions of spatio- temporal scene graphs","venue":null,"work_id":"96cd1286-0ccd-4c5f-b981-cf5e85564489","year":2020},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:447e612da979b08bb52067c34080fb08b789be900ff1999d26e2c6742eb9eb42","observation_id":"f99db999-b1f3-466b-a32d-73d1bafa01d6","resolution":{"observed_at":"2026-05-15T20:50:17.916053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.12132","doi":"10.48550/arxiv.2509.12132","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Look again, think slowly: Enhancing visual reflection in vision-language models","venue":"ArXiv.org","work_id":"88b0b898-273d-458c-bdcf-b56805958874","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:d57cbfcd6f0942fe74cff468458dc52546d6d17be9f28a9940c67bbdd0f07014","observation_id":"39b8bd07-c8a7-407e-adcd-ebcbe0fd6a97","resolution":{"observed_at":"2026-05-15T20:50:17.355679Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chat-univi: Unified visual representation em- powers large language models with image and video un- derstanding","venue":null,"work_id":"43262ad0-df74-43fb-af23-e93d86e17e02","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:3d2f5123d20bbfbb916ed4679dbe69d4c3e43627769728d2c6d8712b15e8a0ce","observation_id":"f4875134-737b-4d6d-82d2-0bcf503a9e06","resolution":{"observed_at":"2026-05-15T20:50:17.914708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Do you remember? dense video captioning with cross-modal memory retrieval","venue":null,"work_id":"5100207a-4ede-4178-ad94-2ca04957d35c","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:8ed2c3dc2d8cce2c3cfccc3f90a2577dfdbd7982edbeba6e7ce05418253d3ab7","observation_id":"f6e7061f-110a-4648-82be-f4b883ee572d","resolution":{"observed_at":"2026-05-15T20:50:17.924034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vidhalluc: Evaluating temporal hallucinations in multimodal large lan- guage models for video understanding","venue":null,"work_id":"08bf3b39-2ca3-424e-9372-64597cd9721c","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:8455e785254a63f71ac25c64ced5d0b8d7a1f993d9cb3f2175f3fee62e1bf265","observation_id":"6cf79116-d81c-4194-bfaf-59bcefa6f1de","resolution":{"observed_at":"2026-05-15T20:50:17.929153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01908","last_updated":"2025-06-02T17:28:26Z","snapshot_observed_at":"2026-08-07T11:29:22.385642Z","submitted_at":"2025-06-02T17:28:26Z","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","version":1},"cited_work":{"arxiv_id":"2506.01908","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01908","snapshot_observed_at":"2026-07-04T16:49:57.219030Z","title":"Reinforcement learning tuning for videollms: Reward design and data efficiency","venue":null,"work_id":"058b4756-2b75-4b6d-aa11-ba523aab08e0","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2506.01908","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:8f7d8e098d70ddd051a59261e3bd11b7bf7a53e47c7e693d39dbc4dd89c517e2","observation_id":"d62ebeaa-5f0d-4f4c-bb7e-d320650c0026","resolution":{"observed_at":"2026-05-15T20:50:17.320193Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Inten- tqa: Context-aware video intent reasoning","venue":null,"work_id":"d15bd3b0-34bd-432e-82ea-9f1b0d8bea5d","year":2023},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:7b3662b25886ad1aa93b77b62b464d71458d8870e3bfa9d2597031361eb21530","observation_id":"3e96dc93-a236-49ae-93a4-c01ecf63556c","resolution":{"observed_at":"2026-05-15T20:50:17.891407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06166","last_updated":"2024-10-08T16:10:29Z","snapshot_observed_at":"2026-08-07T05:06:22.564018Z","submitted_at":"2024-10-08T16:10:29Z","title":"Temporal Reasoning Transfer from Text to Video","version":1},"cited_work":{"arxiv_id":"2410.06166","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.06166","snapshot_observed_at":"2026-07-04T10:29:44.916128Z","title":"Tem- poral reasoning transfer from text to video","venue":null,"work_id":"6ed3ce19-4b99-4851-8f1c-ed95e050473b","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2410.06166","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:b89ef1829915266d54b7ce28f810ca36b799cd026050baeadfaea7fafbf9594e","observation_id":"8230c58b-072f-4f6f-a24b-35f25ae5a698","resolution":{"observed_at":"2026-05-15T20:50:17.317205Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Embodied agent inter- face: Benchmarking llms for embodied decision making","venue":null,"work_id":"fd529fa8-5e02-44d1-967d-0bda4ade142f","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:2182e3d837c90b777d64523eb8c2693498dda3897b854765d8b7f5e8d7af3026","observation_id":"8c64cd1d-55e7-4260-9a2d-361a8ed12191","resolution":{"observed_at":"2026-05-15T20:50:17.880624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"From pixels to graphs: Open-vocabulary scene graph generation with vision-language models","venue":null,"work_id":"3e735e4d-3958-4080-a7c8-4f765b060ac4","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:7759d08ff470177298ffbf38df47c956f01a3bc000de079239e7881b30ec4308","observation_id":"200d6909-d5c2-49af-b057-b5f5a31ee70f","resolution":{"observed_at":"2026-05-15T20:50:17.882447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"cited_work":{"arxiv_id":"2504.06958","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.06958","snapshot_observed_at":"2026-07-04T16:49:57.434938Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","venue":"cs.CV","work_id":"7be17d59-6cde-455a-99c3-06e28659839f","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2504.06958","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:9811bad08fd2ea832de9ca001476bb17b40efe221d071524075d07175e8670dc","observation_id":"7b18476e-6052-409e-920d-660b4714a53b","resolution":{"observed_at":"2026-05-15T20:56:07.931779Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Factorizable net: an efficient subgraph-based framework for scene graph generation","venue":null,"work_id":"865d6f15-171a-4045-ba64-5999676141db","year":2018},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:ce111759d7d3bb9a74e29232ed2f43860fab7d781bf4ca9955753025dd8a6c62","observation_id":"f8633f1c-9ec2-4432-b97c-13e15abbe6fc","resolution":{"observed_at":"2026-05-15T20:50:17.893095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-llava: Learning united visual repre- sentation by alignment before projection","venue":null,"work_id":"8466d8db-6fed-4259-9953-816984c01aba","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:95d4be67f7be1aad2e4c84c6d8415727ebf4c3f627353ef6767afde29ea60377","observation_id":"e58adc24-469a-4f4d-ae59-0221806007ba","resolution":{"observed_at":"2026-05-15T20:50:17.871435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vila: On pre-training for vi- sual language models","venue":null,"work_id":"4cbc08de-9b00-4ce0-b70a-ee13e4d5a4fd","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:c371c7f625001f0bd96a364e5238dd5f7067f09abdb6f930fce876437600665f","observation_id":"33aef3a7-863e-4a5a-8fe4-0a1613a0a488","resolution":{"observed_at":"2026-05-15T20:50:17.899383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Univtg: Towards unified video- language temporal grounding","venue":null,"work_id":"6fcb4d85-d021-4523-a8f5-ef0273e6f4d1","year":2023},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:616de20fbcf143f4800c9561a042ee61d456ed75706a1117c3472c4d517bc28e","observation_id":"6d7f2c74-79b1-47a0-b6d3-5cfe3a3be4b9","resolution":{"observed_at":"2026-05-15T20:50:17.882299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21523","last_updated":"2025-06-20T08:41:41Z","snapshot_observed_at":"2026-08-07T14:44:18.138977Z","submitted_at":"2025-05-23T05:08:40Z","title":"More Thinking, Less Seeing? Assessing Amplified Hallucination in Multimodal Reasoning Models","version":3},"cited_work":{"arxiv_id":"2505.21523","doi":"10.48550/arxiv.2505.21523","metadata_source":"pith","pith_arxiv_id":"2505.21523","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Advances in Neural Information Processing Systems , year =","venue":"cs.CL","work_id":"bb9acd95-b27a-4709-9dfa-75f62a65c716","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2505.21523","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:c75c6b44ee3a74b0700d6ddef47e5aa52bb9d48074943159c07c3bae3bf3d0f7","observation_id":"42004759-81a5-471a-9f69-a12430e11910","resolution":{"observed_at":"2026-05-15T20:50:17.334014Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20715","last_updated":"2026-04-18T02:55:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-27T04:50:07Z","title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","version":2},"cited_work":{"arxiv_id":"2505.20715","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20715","snapshot_observed_at":"2026-07-02T17:27:15.832913Z","title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","venue":"cs.CV","work_id":"35af37e5-9956-4ce9-be10-932e2342d463","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2505.20715","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:b6dfb1bf09fee1a136f1744449966e5cc5227eaefb3a1668be39671d4f7da457","observation_id":"6b13fddb-5b7b-4527-826c-4a725636c01f","resolution":{"observed_at":"2026-05-15T20:50:17.314324Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.06077","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"When thinking drifts: Evidential grounding for robust video reasoning","venue":null,"work_id":"c3305151-f358-45ca-882a-b94421317644","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:224cee3043f730e98e4c793cc5472b6947637ac0794b0e7db76779d369cdec6e","observation_id":"9dfaf9cd-7642-4161-8223-70b988d07863","resolution":{"observed_at":"2026-05-15T20:50:17.344715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-chatgpt: Towards detailed video un- derstanding via large vision and language models","venue":null,"work_id":"f95780eb-75fb-4bd3-bfe6-1a11a0a54646","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:a9a23ddcf3f12697c0101a6b3cc1a47658a0ce0eb0b2f26e3e495e6766305772","observation_id":"cc03584b-6285-49d5-9323-ce1fb009052c","resolution":{"observed_at":"2026-05-15T20:50:17.884546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.08835","last_updated":"2024-07-03T18:05:02Z","snapshot_observed_at":"2026-08-04T18:08:42.450729Z","submitted_at":"2023-11-15T10:22:35Z","title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","version":4},"cited_work":{"arxiv_id":"2311.08835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.08835","snapshot_observed_at":"2026-07-04T03:29:31.161365Z","title":"Correlation-guided query-dependency calibration in video representation learning for temporal grounding","venue":null,"work_id":"930bf918-21bb-485f-a9a4-7951c1ac3b74","year":2023},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2311.08835","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:b42097008719bdc60519c893eaafd043532a25788e758218293ff701d1ec4f29","observation_id":"257d8de6-f7ab-479f-a1c0-9d9ceca21ecb","resolution":{"observed_at":"2026-05-15T20:50:17.328936Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unbiased scene graph generation in videos","venue":null,"work_id":"3e2d58f8-02f4-4bd8-a75d-9a97580af93a","year":2023},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:4a3edb1ef0fd033da5aaa02f9650b84a51f26b8453fbc766da6d7c1974e3e87d","observation_id":"ef59aff4-b3a7-416c-a779-3371d284ebe1","resolution":{"observed_at":"2026-05-15T20:50:17.865731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hig: Hierarchical interlacement graph approach to scene graph generation in video understanding","venue":null,"work_id":"bf12583e-8636-4c4c-bbf2-f99cfbdc75a8","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:3b3b60560b27eb95c35115dca462878b01d75b31bd8dd8bfa356993f4fd83218","observation_id":"4fcf03e9-060d-4cd6-9dd5-99f900caba39","resolution":{"observed_at":"2026-05-15T20:50:17.860904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hyperglm: Hypergraph for video scene graph generation and anticipation","venue":null,"work_id":"a53683fd-f045-4b7d-bf0a-778518d6fe70","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:84d49e85f0529515d70de3ce939c459507edd96169b74dfbc058a43fbb0555c3","observation_id":"96258dd8-77d7-4a12-bad6-7a836e461633","resolution":{"observed_at":"2026-05-15T20:50:17.921448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4 technical report","venue":null,"work_id":"5823f0ff-286b-468c-bb85-2d60310c21a5","year":null},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:951341805152d81aedf85330435ff891488422d97d367eda1ead349fddd78894","observation_id":"4c634398-e264-4c3e-a689-5b5c796bb710","resolution":{"observed_at":"2026-05-15T20:50:17.914392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T00:25:49.260079Z","title":"Gpt-4o system card","venue":null,"work_id":"ebd64501-fd2c-4ca6-981d-bf988eb57341","year":null},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:208ca4d470b864baaa840a1d22bc8c8f60b610bab77e8a7b38d825de30c484b4","observation_id":"20cc96f2-af3b-453c-9e46-92bf9ad55239","resolution":{"observed_at":"2026-05-15T20:50:17.910864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01407","last_updated":"2026-08-05T08:51:38Z","snapshot_observed_at":"2026-08-07T18:15:18.676717Z","submitted_at":"2025-04-02T06:47:19Z","title":"ZoomV: Temporal Zoom-in for Efficient Long Video Understanding","version":3},"cited_work":{"arxiv_id":"2504.01407","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.01407","snapshot_observed_at":"2026-08-06T02:01:23.268869Z","title":"Timesearch: Hierarchical video search with spotlight and reflection for human-like long video understanding","venue":null,"work_id":"e4620934-0280-4240-b1ef-1b160dfc987a","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2504.01407","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:4b103fecffa67e0ca589152ca8e031ebda2fdf66fdaf9452f34c8b85c0528d2c","observation_id":"ac77e235-4cbe-41f3-a396-d25bf3c38437","resolution":{"observed_at":"2026-08-06T02:01:23.268869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Question- answering dense video events","venue":null,"work_id":"fd76f5e4-1cb9-4122-b96f-dc70b9c836bc","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:970dd35f52c7fcc17a38d5b4a19277f33a36971aa47ca137087d38799f68783a","observation_id":"e4b213bb-51c7-4153-a0e0-5a20dfa05fb9","resolution":{"observed_at":"2026-05-15T20:50:17.859108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Step: Enhancing video-llms’ com- positional reasoning by spatio-temporal graph-guided self- training","venue":null,"work_id":"254a526f-8e13-46e5-86a3-85662f6fe712","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:9137b362a3834a662ac4b85183f73a93bed78280d1238a783a76ff782ada3be8","observation_id":"b22e985c-0c3e-4a94-a3b4-34efded62655","resolution":{"observed_at":"2026-05-15T20:50:17.925612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"fcda677c-bd85-4e86-9a74-1c45320fbe4e","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:0335c38972787afe7bcef9a2065ea42278effc45d7be4e41cca0e80e4efa7db5","observation_id":"52eaa934-63d4-4398-9834-78c42842010b","resolution":{"observed_at":"2026-05-15T20:50:17.864409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07615","last_updated":"2025-04-14T15:15:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-10T10:05:15Z","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","version":2},"cited_work":{"arxiv_id":"2504.07615","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.07615","snapshot_observed_at":"2026-07-11T03:17:51.904109Z","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","venue":"cs.CV","work_id":"d36889cb-edb6-448f-9a50-36df8b1623e5","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2504.07615","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:fb9935ceec3ee3c03d286daec53480aab03ad3a68aa8b492d8b40fe5f28f8fae","observation_id":"9ede4606-d07b-4c39-bc6f-5ba72652507a","resolution":{"observed_at":"2026-05-15T20:50:17.290213Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:2b9885bc232a7773b1900e6558fb25ae7c3a71eb3a5f4d230fd6a70178b8ea5f","observation_id":"fb90480b-a89e-4a6f-a611-2bf5662a0432","resolution":{"observed_at":"2026-05-15T20:50:17.348363Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12387","last_updated":"2024-04-18T17:59:48Z","snapshot_observed_at":"2026-07-06T18:02:19.428801Z","submitted_at":"2024-04-18T17:59:48Z","title":"Reka Core, Flash, and Edge: A Series of Powerful Multimodal Language Models","version":1},"cited_work":{"arxiv_id":"2404.12387","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.12387","snapshot_observed_at":"2026-07-04T10:09:45.386126Z","title":"arXiv preprint arXiv:2404.12387 , year=","venue":null,"work_id":"e8a6f1f4-737f-4845-a995-3910ae4edc61","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2404.12387","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:a5280617b511786fd2670edadeca7c38d8d9f3e3e94a5af84fd69d219e0b319a","observation_id":"e28eb4f0-7cfa-45e4-880c-b2695d6e4ce8","resolution":{"observed_at":"2026-05-15T20:50:17.299732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Causal ai scientist: Facilitating causal data science with large language models","venue":null,"work_id":"3b40c803-16a8-456c-94f9-8f275d9f977d","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:b081123d08f71ee57a4ddb130ce82489fac06476def2b1db799ca0111fe6849f","observation_id":"511600ad-fa45-463a-b372-d1339b3197bd","resolution":{"observed_at":"2026-05-15T20:50:17.931383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"cited_work":{"arxiv_id":"2508.18265","doi":"10.48550/arxiv.2508.18265","metadata_source":"pith","pith_arxiv_id":"2508.18265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","venue":"cs.CV","work_id":"b8f5e260-fff5-444e-bcf5-2c42cfefd83d","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2508.18265","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:d90ccd805102071a27475dd411794ef23fbae670cc0100ac0931f2cc37dcd5a8","observation_id":"e5003bc1-b6e4-4d96-84f5-f96518f6cdbd","resolution":{"observed_at":"2026-05-15T20:50:17.295417Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13377","last_updated":"2025-06-29T08:11:35Z","snapshot_observed_at":"2026-08-02T21:01:10.455745Z","submitted_at":"2025-03-17T17:04:20Z","title":"Time-R1: Post-Training Large Vision Language Model for Temporal Video Grounding","version":3},"cited_work":{"arxiv_id":"2503.13377","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.13377","snapshot_observed_at":"2026-07-04T09:49:44.696023Z","title":"Time-R1: Post-Training Large Vision Language Model for Temporal Video Grounding","venue":"cs.CV","work_id":"ef2c21b6-ae25-436a-bac3-f8d625541320","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2503.13377","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:a20463b448cb5d4cbf474426f342ad06b5cf97574fe30417b106cd0ba007c0f7","observation_id":"4495bbd2-de89-44cc-a83e-b0b39d756171","resolution":{"observed_at":"2026-05-17T02:40:06.763282Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sportshhi: A dataset for human-human interaction detection in sports videos","venue":null,"work_id":"a66cac17-e3c3-46b8-86c1-930250dcb1a9","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:f108f205fecd0ee842f2aeb067f1072530fd69d6654a27298c97e154eca14c71","observation_id":"d43ac4bc-f95d-4211-8c3b-b48c6a578c23","resolution":{"observed_at":"2026-05-15T20:50:17.927223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.14677","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T03:14:31.670525Z","title":"Visionary-r1: Mitigating shortcuts in visual reasoning with reinforcement learning","venue":null,"work_id":"c227c7de-f573-4665-b921-30284b6bdb9a","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:60e5af019f426d97fdf43d26da1d5f073270b70af73913e3f9d4b91d46996aca","observation_id":"d476a100-c10d-4a3b-b461-6bea06375118","resolution":{"observed_at":"2026-05-15T20:50:17.340125Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Can i trust your answer? visually grounded video question answering","venue":null,"work_id":"663cdb73-3954-47e3-8ac4-1b4cd606b3f5","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:0bc3527d4a7aa1958997e04617f98ad1b4cc3e22903eccce006398025436595b","observation_id":"08557017-0441-455c-9ffa-9aaec944151f","resolution":{"observed_at":"2026-05-15T20:50:17.918002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":"2404.16994","doi":"10.48550/arxiv.2404.16994","metadata_source":"pith","pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","venue":"cs.CV","work_id":"8949d4db-20f2-47c1-83a6-fcbe041b62ef","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:ae104b94bb1df1fd5c6ea2557d5f5cc0772452e8680ca2b6387055bb5c1498b8","observation_id":"a077a12b-9366-4d12-bdf0-aa1653f2517d","resolution":{"observed_at":"2026-05-15T20:50:17.342034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Panoptic video scene graph generation","venue":null,"work_id":"d3e0dcba-c6a4-407d-a4c9-77b92eb7aa2d","year":2023},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:b623f84a2c220b8577b96da3fe2bcadd8bfca6d085fa9ab6b0283a2daf71ffe0","observation_id":"57132801-4c3f-436a-8365-f5cafe8907da","resolution":{"observed_at":"2026-05-15T20:50:17.916364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Thinking in space: How mul- timodal large language models see, remember, and recall spaces","venue":null,"work_id":"b785fb3d-164a-470c-b8a2-0de92ecffbe4","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:657f90156ee43a7b3b616b784d15c3ebec176e8c1fc63b9e2e72ce81ccfb6e61","observation_id":"44b0c5ee-6c92-4316-93b8-d39258c929d7","resolution":{"observed_at":"2026-05-15T20:50:17.919647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":"2501.13106","doi":"10.48550/arxiv.2501.13106","metadata_source":"pith","pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","venue":"cs.CV","work_id":"38f52461-37fd-4266-bc46-9dea31be2824","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:99953effa775689b89448b612ed9692c0b7e38fdb0bcd8d9b3720859824680cb","observation_id":"b05026ca-166f-4fb1-ba7a-1d870bce46dd","resolution":{"observed_at":"2026-05-15T20:50:17.353150Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.04416","last_updated":"2025-09-03T07:11:03Z","snapshot_observed_at":"2026-08-06T10:35:42.926353Z","submitted_at":"2025-08-06T13:03:21Z","title":"Thinking With Videos: Multimodal Tool-Augmented Reinforcement Learning for Long Video Reasoning","version":2},"cited_work":{"arxiv_id":"2508.04416","doi":"10.48550/arxiv.2508.04416","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.04416","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Thinking with videos: Multimodal tool-augmented reinforcement learning for long video reasoning","venue":"arXiv (Cornell University)","work_id":"824594ad-21b0-49b7-88f8-fac8553528e3","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2508.04416","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:39169c4ae3ecc8b361d8778ec1334a3e44fff65990cf2818faffe669307a366a","observation_id":"8ef12da1-0c3f-4e51-aeab-d8f38fddd796","resolution":{"observed_at":"2026-05-15T20:50:17.308558Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vtime- 11 cot: Thinking by drawing for video temporal grounding and reasoning","venue":null,"work_id":"42831356-bade-4fb3-9d6a-af87cc7cef98","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:4c95e90b283de2d6e22e3fc1a77700f1ace417dcbf4b5f5e8a7d25292ee0ba98","observation_id":"d712bd22-8129-47e5-bdc7-8484d71ea0e1","resolution":{"observed_at":"2026-05-15T20:50:17.905072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.08817","last_updated":"2025-06-12T15:51:33Z","snapshot_observed_at":"2026-08-07T04:59:07.065080Z","submitted_at":"2025-06-10T14:08:56Z","title":"Video-CoT: A Comprehensive Dataset for Spatiotemporal Understanding of Videos Based on Chain-of-Thought","version":3},"cited_work":{"arxiv_id":"2506.08817","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.08817","snapshot_observed_at":"2026-07-01T21:26:14.022696Z","title":"Video-cot: A comprehensive dataset for spatiotemporal understanding of videos based on chain-of- thought","venue":null,"work_id":"8d9acd65-4bc7-4f3b-805e-dfb46b9f4a1f","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2506.08817","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:be4b230acd3314c528b90d15f39d54386b9a51e2b358f5e56709afbabc7408c3","observation_id":"3c1f2cfd-c49a-4438-a077-9db5e8271014","resolution":{"observed_at":"2026-05-15T20:50:17.350435Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":"2410.02713","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-07-09T21:36:34.348434Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","venue":"cs.CV","work_id":"e598f516-d992-449a-ab6d-6c788b3a1d7b","year":2024},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:0831cc5467fd03858953519a5247f79a2e81e69ce8d120a0a8f490f7e6243530","observation_id":"05b6845a-ca89-4feb-9deb-3c6c34cc87a3","resolution":{"observed_at":"2026-05-15T20:50:17.302649Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Towards video thinking test: A holistic benchmark for advanced video reasoning and understanding","venue":null,"work_id":"3fd9e35d-7829-4626-ab2b-8bc9c6b211cf","year":2025},"citing_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-15T20:48:44.933542Z"},"links":{"citing_paper":"/paper/2602.17555"},"observation_digest":"sha256:f8cfa5b0009e25b45958ae6986a4c6d3422ccdbf819a7f27982c8b68fc60c24f","observation_id":"94d69096-4dc9-45a8-af32-517308399139","resolution":{"observed_at":"2026-05-15T20:50:17.901551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking"},"reference_resolution":{"displayed":74,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":31,"verified_fuzzy":42},"total_outbound_references":74},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 74 of 74 outbound references and 2 inbound Pith citation observations for arXiv:2602.17555."}