{"as_of":"2026-08-06T08:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b136b2b982f04a2b297d6c3267bda29a78bf3cd5fb89d28c0a0dc27a6bd9e6ad","coverage":[{"denominator":78,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":78,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T18:38:16.204012Z","state":"measured"},{"denominator":80,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":80,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T22:38:43.102769Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-01T13:55:45.219247Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"cited_work":{"arxiv_id":"2604.08014","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.08014","snapshot_observed_at":"2026-07-01T13:55:45.219247Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","venue":"cs.CV","work_id":"83fb5500-3f58-46d4-a4b5-6bb0b753fdbc","year":2026},"citing_paper":{"arxiv_id":"2605.11723","last_updated":"2026-05-28T12:50:43Z","snapshot_observed_at":"2026-08-01T23:43:09.636688Z","submitted_at":"2026-05-12T08:08:33Z","title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T06:00:31.582714Z"},"links":{"cited_paper":"/paper/2604.08014","citing_paper":"/paper/2605.11723"},"observation_digest":"sha256:3106f052e589667b11576ec0794c08266abfce05c924d14ceb212d1e80c38a06","observation_id":"dcaffa3d-fcad-4423-a16d-c6a73eac24b3","resolution":{"observed_at":"2026-05-13T06:02:22.173753Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"cited_work":{"arxiv_id":"2604.08014","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.08014","snapshot_observed_at":"2026-07-01T13:55:45.219247Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","venue":"cs.CV","work_id":"83fb5500-3f58-46d4-a4b5-6bb0b753fdbc","year":2026},"citing_paper":{"arxiv_id":"2605.11723","last_updated":"2026-05-28T12:50:43Z","snapshot_observed_at":"2026-08-01T23:43:09.636688Z","submitted_at":"2026-05-12T08:08:33Z","title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-30T22:38:43.102769Z"},"links":{"cited_paper":"/paper/2604.08014","citing_paper":"/paper/2605.11723"},"observation_digest":"sha256:52530f6bf533c5ab4d06f71f424d50b37c8a5e297a5266f8834cc6b12b594f69","observation_id":"fef6ae39-0ad0-4864-950d-ea767d2c4109","resolution":{"observed_at":"2026-07-01T13:55:45.220662Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2604.08014/citation-record","integrity":"/paper/2604.08014/integrity","json":"/paper/2604.08014/citation-record.json","paper":"/paper/2604.08014"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:a8e47085b4cee4ca6e33ffbb31a0f52026140456ad987efe11b90c730536adfe","observation_id":"03e527c6-e867-48f1-a7fc-585e082f69af","resolution":{"observed_at":"2026-05-11T00:15:52.986438Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05336","last_updated":"2025-07-05T11:38:26Z","snapshot_observed_at":"2026-07-06T21:37:28.044836Z","submitted_at":"2025-06-05T17:59:29Z","title":"VideoMolmo: Spatio-Temporal Grounding Meets Pointing","version":2},"cited_work":{"arxiv_id":"2506.05336","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05336","snapshot_observed_at":"2026-07-03T15:48:34.853226Z","title":"Videomolmo: Spatio-temporal grounding meets pointing","venue":null,"work_id":"11238605-33ed-4866-a7b0-1df17e186797","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2506.05336","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:036b3177dbd03b21e11b2ab99ae98cdb59393f2413fa5cfd1dfb456d963b3258","observation_id":"4e0c4fc2-6f94-494a-abad-4663b2914a11","resolution":{"observed_at":"2026-05-11T00:15:52.997911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:07ace125e212667c0ee700aefdd224760dce70fef8ee4279d8bcf96bf6e9ee10","observation_id":"e44d4599-a473-4eda-a4a5-7e6051b95d3c","resolution":{"observed_at":"2026-05-11T00:15:53.011675Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:c9257657b64e8d6de25f43e54c47cd4925aa0b21b8c217d982c8c4e0b84b8c8e","observation_id":"eb9e1fee-8d64-4930-915f-f3ced21b8f90","resolution":{"observed_at":"2026-05-11T00:15:52.973268Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"24dda928-8f44-436f-bcb0-6a88eefcb7b7","year":2023},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:9b06cbda681d63e41b1c8746f52f987e755fb61fc443fc7818c6cba7495617fb","observation_id":"9dd28c23-b887-4066-89e8-be85fb2793dc","resolution":{"observed_at":"2026-05-16T17:01:08.927714Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09478","last_updated":"2023-11-07T18:25:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-14T03:22:07Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","version":3},"cited_work":{"arxiv_id":"2310.09478","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.09478","snapshot_observed_at":"2026-07-08T22:35:40.586981Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","venue":"cs.CV","work_id":"fb62cd1b-3991-40be-a987-3cfa5772b5b5","year":2023},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2310.09478","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:7bdb35e5b0ae246eefe2b4b2d7a7a8228fa30f7cda462d69d4e671ef265287ab","observation_id":"6b54e20b-ce52-43da-989e-31865a44cde9","resolution":{"observed_at":"2026-05-16T07:13:09.112362Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d2432363-0e9c-47d7-aa58-806f8e12cfca","year":2021},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:d8a3cf48a64ec400d680aee801175580552bc76ea993d4d39a24fe357db63512","observation_id":"89464ff6-6ed3-4848-a31f-d40fd1d97299","resolution":{"observed_at":"2026-05-16T17:01:08.974618Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":"2406.07476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-07-04T16:49:57.279303Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","venue":"cs.CV","work_id":"ccfc3f89-c510-45f1-8a35-ed1a56c0ae5c","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:96cc4d63d831f734e3f3f87fda5b1ba21d147f0aef1bcfdde2796ce51d269f28","observation_id":"689bedc6-fe71-40ae-b1c6-136dcc660e08","resolution":{"observed_at":"2026-05-11T02:44:53.738794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:8ccbdcebe71ee41eda56523d438b93fc830319d53aec8650f53fbb7ab947c959","observation_id":"66d69e12-b12a-40a9-9e7f-1c9525f784ac","resolution":{"observed_at":"2026-05-11T00:15:52.673332Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c604e90f-2d70-4aed-9361-8f3fdc175283","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:ebd9dbefd47404eb0643d5a3c85ab30aab7b25e53c09c0069dca4d7570f53983","observation_id":"cc7da782-fcb4-45f1-a46d-abdaf7734a93","resolution":{"observed_at":"2026-05-16T17:01:08.968113Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b9e6e01d-674d-40f7-a284-0f8336d9a51f","year":2017},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:8f07f7b54f4e9c9532d03cd4a6d0570e46bb0db4545f7bbd7a2a95c6aef1bb0c","observation_id":"d85e0dda-2334-47eb-a5f5-75d50cd5c5bb","resolution":{"observed_at":"2026-05-16T17:01:08.963560Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:18b62012a6ca4d734461586950426f9eb4717510397268f238e1681b3dc8fe33","observation_id":"e30c2298-a832-45f9-979a-1a0aacad71d4","resolution":{"observed_at":"2026-05-11T00:15:52.878364Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T17:26:25.954614Z","title":null,"venue":null,"work_id":"d60cc48f-93fd-4505-9ccb-c47cfb514d0f","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:02c8f7a96df594e29097070edef566174cc7d4ef2c93026fb72a340a8a3dd324","observation_id":"f733cdac-c99a-473b-995a-eb4a492b05fb","resolution":{"observed_at":"2026-05-16T17:01:08.965885Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11168","last_updated":"2025-02-16T15:38:33Z","snapshot_observed_at":"2026-07-06T20:37:23.683063Z","submitted_at":"2025-02-16T15:38:33Z","title":"Knowing Your Target: Target-Aware Transformer Makes Better Spatio-Temporal Video Grounding","version":1},"cited_work":{"arxiv_id":"2502.11168","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.11168","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Knowing your target: Target-aware transformer makes better spatio-temporal video grounding","venue":null,"work_id":"3050449c-0e12-4279-a07e-f119659c2fdf","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2502.11168","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:197aa34aef93f6b2a72a61321850c5d3b9bc3e4c5d3eef812883beafbcca62f7","observation_id":"3f730f31-bc37-4b13-b640-8a50767e92c7","resolution":{"observed_at":"2026-05-11T00:15:52.697378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.21375","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:49:57.224424Z","title":"arXiv preprint arXiv:2511.21375 (2025)","venue":null,"work_id":"1a1f6a40-fdc8-4cc4-afcf-af374eb098ec","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:2b34dd277b21911791285b573ba39afdc39711bac5c581407ef21f7aa2c76cae","observation_id":"30edabf3-1b54-4af0-94a5-3a2e4cab767e","resolution":{"observed_at":"2026-05-11T00:15:52.740655Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-02T16:13:31.498470Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2505.07062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.07062","snapshot_observed_at":"2026-07-08T16:15:06.198778Z","title":"Seed1.5-VL Technical Report","venue":"cs.CV","work_id":"0e8e025f-ca1e-49cc-aee2-33f3a0201f3c","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2505.07062","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:95680c93e4f2b658c5fc0437e27c365621fd41dddf27fbe6499d66e0b15f8c7a","observation_id":"1313a7be-ad69-4774-a5ac-00d1a6eb1146","resolution":{"observed_at":"2026-05-11T05:26:06.820639Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:5c1fcd4278a0963284ae075a1ce08276ec48c014f268aa87e6f021c006bdbc91","observation_id":"0e0e84d8-b2b2-4b31-a071-1826ec27ae9d","resolution":{"observed_at":"2026-05-11T00:15:52.634455Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"aafb814f-09c5-4fce-89dc-4c09475d1f5a","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:be4cf37e3631145ce10f61766c6b434485d53caa44fd641c9fc4ea0a415fa52d","observation_id":"6e0009aa-6dc0-427f-a45e-a00966516850","resolution":{"observed_at":"2026-05-16T17:01:08.979074Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4c39c275-edc1-4417-93f8-d764d615c6e5","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:8ed9353322e7f8a2ac644f65840b676e478eafa08d134e95dc174a76073d4c46","observation_id":"a00676f5-7bef-49e4-a4d0-346253b959cb","resolution":{"observed_at":"2026-05-16T17:01:08.923150Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05643","last_updated":"2025-03-03T10:28:30Z","snapshot_observed_at":"2026-08-01T19:33:10.793870Z","submitted_at":"2024-10-08T02:46:30Z","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","version":3},"cited_work":{"arxiv_id":"2410.05643","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05643","snapshot_observed_at":"2026-07-02T12:16:57.678370Z","title":"Trace: Temporal grounding video llm via causal event modeling","venue":null,"work_id":"b2af5095-2950-48b5-af5e-270218b8d041","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2410.05643","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:441e141cd231af66a34f4f6bf109c1c48ec115d1330fa71a3a24ed7db7e69623","observation_id":"b2406288-18d5-453f-b276-06fdbcfcaaba","resolution":{"observed_at":"2026-05-11T00:15:52.645222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"14e4b163-5b53-4f24-80ab-17c6054faad3","year":2022},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:32c932214a266bcea41f79ee374873d0c82a09b1e1f0c4d43138efe6c74e5c8e","observation_id":"3c331f08-101c-49c7-bffe-12c5a26e584d","resolution":{"observed_at":"2026-05-16T17:01:08.950231Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7d0e69ed-2a12-4537-9aa0-7037f4a9119c","year":2019},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:73a42a08be99451bbe0a7d4938ff67b232b4708a01fcd40a30a23635daffd8e9","observation_id":"8b0d8697-112b-40c7-8c23-c8fc5122137a","resolution":{"observed_at":"2026-05-16T17:03:08.454373Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2474b64e-b504-491c-b748-099f1e4f2c31","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:f8f5e0034f2b25fada87c8301371d6b8bee63efbeb85fcd16202918db33efa16","observation_id":"21c60a39-af6f-43cc-b45e-4064a1324dfa","resolution":{"observed_at":"2026-05-16T17:01:08.939950Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In Proceedings of the IEEE/CVF International Conference on Computer Vision","venue":null,"work_id":"5b6ab4f4-8759-441d-8387-96e85909019d","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:3b47469b7efd59a0b206c85f7296c5328ccb21279888400878ffb01a8f96a9ef","observation_id":"c09a265f-a2b8-4643-94b6-abd8825f7278","resolution":{"observed_at":"2026-05-16T17:03:08.456635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7bc46e08-5a46-4573-8fb1-2206561216f4","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:c05fdd0562f60aae7666442141b19040562fd6228c594aa614022815a9e3547f","observation_id":"d392f44d-712e-471d-8a2b-589a814fc9dc","resolution":{"observed_at":"2026-05-16T17:03:08.452529Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4871d6cb-4053-4e91-bced-09ce0cb20c0f","year":2022},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:a47d1ae2a85f16349ed6d5fb9c175d25030a8a36366d0da716ea2527359c8339","observation_id":"f802ab8f-a916-4e0a-8f82-34210f84d052","resolution":{"observed_at":"2026-05-16T17:01:08.970173Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1baa4f68-fa8e-4640-82bb-1c849d41f83c","year":2014},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:ccbbfe15e748026801b512988326a859b0c76152dc000909f3d7202e90f8924f","observation_id":"f9a2a2cb-4aa2-4016-b8f5-bd8a1f773586","resolution":{"observed_at":"2026-05-16T17:01:08.981123Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ef043387-0360-4132-9be6-20a5dbd58fcb","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:a6b02127a6e4cc12afab9fedd3eaaa71dd4637ed752f33b0a9e67c1df986a7c8","observation_id":"895659a5-cf6e-4867-85ba-5a464570fe1c","resolution":{"observed_at":"2026-05-16T17:01:08.936702Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:a7b2644be48814eea864f37c3493501e0c3952ec2482174a3d58aa5d417fba98","observation_id":"c675a1a4-a757-4695-b289-669b3f43708c","resolution":{"observed_at":"2026-05-11T00:15:52.956688Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"dfdb8f34-db01-4346-8e5d-4b65907ee8f1","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:75032a2ef4e7e455aea5afb56b9862cca044d3a9bf4fd4b266a1648517a6755a","observation_id":"3046611d-a5fc-4547-9af1-29f8deee0f28","resolution":{"observed_at":"2026-05-16T17:01:08.971503Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"65b2ffa8-96d4-4f17-903b-e21baae19b6d","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:fbb11208281437f31ef26a8078ab8d85f7b7d3eaf8f0b3cd5fb0cb2b19c1e54b","observation_id":"007b591c-b86c-4842-a96a-fe533fbf348a","resolution":{"observed_at":"2026-05-16T17:01:08.972325Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the Computer Vision and Pattern Recognition Conference","venue":null,"work_id":"50c383f6-ca47-4eed-a96d-ae8a209d98b1","year":2026},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:db54443bc627667965c0d57df224fdf6d692fbdcae0bb6105e7bbd2cc4aaa83f","observation_id":"07f8147d-d76b-4727-9128-41422ad5f1ae","resolution":{"observed_at":"2026-05-16T17:03:08.449600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"52f5584f-a47f-4743-8d24-bd3030c0197c","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:50a90e544608900db334fc065cebb2243401f6a20304b4bc58a8d00783f34176","observation_id":"4b5c7932-22ef-4fd7-a099-105df56dceaa","resolution":{"observed_at":"2026-05-16T17:01:08.912937Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T09:26:09.660516Z","title":null,"venue":null,"work_id":"022fa398-e6c8-4a91-b874-06fdb2e19138","year":2014},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:cb8dc8a481c48f2379d088def830cddb1833dd60f0843288e8fdd23918ea69a1","observation_id":"6151143f-74f3-476e-8ddf-c96013f441bd","resolution":{"observed_at":"2026-05-16T17:01:08.962910Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cc4aff04-894d-475b-a007-91dd9a81d3e2","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:3d2aa61ebf539798ee92b840b565618af02678d2d69185fb3bb56ff05d16a919","observation_id":"dfe6b685-b6cf-41f1-aaba-815399150bc5","resolution":{"observed_at":"2026-05-16T17:01:08.903843Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":"21a89d3f-41aa-4aca-8def-590a7c283e00","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:5f75c120b6541cda859ec7da6a0f4b3f988e97858f557c2f5c166225d45f65a1","observation_id":"0054ae11-fc59-4a6f-a21c-f9f07917f1ad","resolution":{"observed_at":"2026-05-16T17:01:08.942455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e96875d8-b541-42f5-a789-9a187ddd5c8d","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:797cd937b74710575362ce693f56c89c540ddc5e40d07050ab352dfd3b6231b6","observation_id":"1b854232-bea8-44c2-8290-3ad7759b73dc","resolution":{"observed_at":"2026-05-16T17:01:08.960533Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":"1711.05101","doi":"10.1137/1.9781611972825.47","metadata_source":"pith","pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Decoupled Weight Decay Regularization","venue":"cs.LG","work_id":"07ef7360-d385-4033-83f7-8384a6325204","year":2017},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:7070855b332f98a9ffc6ac25a065fdf9ab3ca7307c0ad9332e41dac820310779","observation_id":"96c16b5a-5069-47c1-8780-f795ab43e17e","resolution":{"observed_at":"2026-05-11T00:15:52.828736Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fe24be7a-5165-4178-9d2d-3acf172c32ad","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:73720a3815621bbbe6f9288785de74adfebf9c896acd92405acf4dbe26d71ae5","observation_id":"3d9fc9ee-4422-4440-a8e1-56c3b2a3292c","resolution":{"observed_at":"2026-05-16T17:01:08.960845Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"671c2dc8-438a-4de6-8e07-e19166af2ec2","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:33474f1c0f09e2e58c7a7e85af70c93cdd43a747cf96b920e8e126f434bb655c","observation_id":"0511f31c-12f9-4161-a871-c0b1bedc9a71","resolution":{"observed_at":"2026-05-16T17:01:08.977009Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e016937b-a8d4-4bb2-9a6e-af38a2ef54ce","year":2016},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:475e5829fcfb1708807cf7b0550574745397fbf792ede7f6389e8a18a3e3cd56","observation_id":"bd7f4352-d542-438b-8b76-1920be3684c3","resolution":{"observed_at":"2026-05-16T17:01:08.950371Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"30b2a9ba-b3ac-4fc9-9fcf-a2b1cad29b40","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:d1aa8a8ac611aaca0e9d7c80c77e440bbcc0e57c04c2fd95ff5064674bea12c1","observation_id":"f8cb001b-9d68-4ebf-91fd-613de1a2339f","resolution":{"observed_at":"2026-05-16T17:01:08.934005Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"ec584eec-219c-4c6d-808a-13793c196ab9","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:b09a7116a6825fac25f62db67f7af29dc7dadce8641781a3f3dafe5feeea37db","observation_id":"2819e5e8-f657-4708-a268-9ead1824e835","resolution":{"observed_at":"2026-05-16T17:03:08.458749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"37409251-9c49-491a-9a1b-db81724626c4","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:1137cba758bee9ba30e1da6323640f671bb74f72d637034a94b9d598a6fe10d1","observation_id":"67bd6e26-429b-41ba-ad21-e380f2fbcd36","resolution":{"observed_at":"2026-05-16T17:01:08.969665Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a842862e-41b1-48ae-8c8d-49a6d59b461e","year":2021},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:76c7b89628898fad5f25ab8667266ae4462786aa321b3bf31c25a8b0173c7d08","observation_id":"47e4b230-eef9-4319-90e9-5e00901a0bfe","resolution":{"observed_at":"2026-05-16T17:01:08.953283Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9770c8f3-d6b0-42ac-9977-a1bdcc58bf00","year":2021},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:2ce40e13a02eba46a2c6431fa494ca8af5f2bb14cda3fb756827e364b44dbd4b","observation_id":"03d04dcf-574b-4688-a28b-03070d28b5f6","resolution":{"observed_at":"2026-05-16T17:01:08.926172Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"228a2fda-fdc4-4a01-bf00-825035e7f04c","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:d36a34ee677ceb3427d4b603b6379456a13c838d07f2ea24ad03d50d52b8d3a9","observation_id":"dfb94d0a-dbba-4e43-9c7d-bfa9f8be20d7","resolution":{"observed_at":"2026-05-16T17:01:08.911844Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01132","last_updated":"2024-12-02T05:15:32Z","snapshot_observed_at":"2026-07-06T19:59:56.438747Z","submitted_at":"2024-12-02T05:15:32Z","title":"Eyes on the Road: State-of-the-Art Video Question Answering Models Assessment for Traffic Monitoring Tasks","version":1},"cited_work":{"arxiv_id":"2412.01132","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.01132","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"979a861d-1636-435b-8d60-22182c5d78f5","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2412.01132","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:ae66ce5f68ee65a923540e597bf57284aea97da8db5e35d2ccc9d3f7f17d9963","observation_id":"f316b1d1-7c9a-494e-96fa-35b68ef33342","resolution":{"observed_at":"2026-05-11T00:15:52.858304Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21980","last_updated":"2025-07-22T15:39:40Z","snapshot_observed_at":"2026-08-04T17:30:49.978270Z","submitted_at":"2025-06-27T07:41:15Z","title":"R1-Track: Direct Application of MLLMs to Visual Object Tracking via Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2506.21980","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.21980","snapshot_observed_at":"2026-06-30T07:24:21.251853Z","title":"R1-track: Direct application of mllms to visual object tracking via reinforcement learning","venue":null,"work_id":"603a5ed6-b789-43ec-a241-77b44c0ff954","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2506.21980","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:3c02d5bec744a8fc1a0202a45467b92908d4881cf2b75ffebc23850691b86eac","observation_id":"089fe88c-1ecb-499f-81a2-338d51b865be","resolution":{"observed_at":"2026-05-11T00:15:52.937357Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07844","last_updated":"2024-07-22T03:26:21Z","snapshot_observed_at":"2026-08-05T18:50:42.664160Z","submitted_at":"2024-07-10T17:05:49Z","title":"OV-DINO: Unified Open-Vocabulary Detection with Language-Aware Selective Fusion","version":2},"cited_work":{"arxiv_id":"2407.07844","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.07844","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ov-dino: Unified open-vocabulary detection with language-aware selective fusion","venue":null,"work_id":"40d2ff23-09ad-45c6-9df7-2702e18d4558","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2407.07844","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:552c16505038faa271e571cc8ad7d65b6f67da28b9ad24311dc079d5490b8c68","observation_id":"9ffa1cd3-1c7a-4f2a-8b55-aad5ad78509a","resolution":{"observed_at":"2026-05-11T00:15:52.714098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"bd1c0ab1-5a67-4997-a22f-31c57c772021","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:6fdc63cc911a3f40e1ea6e071d5fb6e59671087fb4eeb76e888864e448220e16","observation_id":"966695bf-1c53-4c2b-9c39-747705c3d438","resolution":{"observed_at":"2026-05-16T17:01:08.921026Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13983","last_updated":"2025-04-11T05:22:55Z","snapshot_observed_at":"2026-07-06T20:54:32.016138Z","submitted_at":"2025-03-18T07:40:36Z","title":"SpaceVLLM: Endowing Multimodal Large Language Model with Spatio-Temporal Video Grounding Capability","version":3},"cited_work":{"arxiv_id":"2503.13983","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.13983","snapshot_observed_at":"2026-07-01T13:55:45.245984Z","title":"Spacevllm: Endowing multimodal large language model with spatio-temporal video grounding capability","venue":null,"work_id":"16cd5604-619a-4668-9b30-8669d0bd34d9","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2503.13983","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:128c408c4171767f3677ff7c53b1c4981752c23b36bcc42464909a2d747c1e41","observation_id":"52f84a49-4bfb-48a1-8a83-47d56b384bab","resolution":{"observed_at":"2026-05-11T00:15:52.654477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"54662adf-583d-423b-ac7f-a5bc100b32af","year":2023},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:a7f083ebf09e7254bc42433b71ddfff15ee8e0e2ed4b7b79c212a74b951f39d3","observation_id":"51298b5c-090c-48cc-b61f-6ae9ea5f2fb5","resolution":{"observed_at":"2026-05-16T17:01:08.965370Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9de417b5-604e-4e47-b300-59f3b61a81b7","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:da891e28db4a35e1fd12be76cec4bef517698c39d514383aca9d38eb13406492","observation_id":"7f55d807-058f-4abc-9a08-a981ea1ba8ef","resolution":{"observed_at":"2026-05-16T17:01:08.958236Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.08224","last_updated":"2025-08-13T05:32:22Z","snapshot_observed_at":"2026-08-05T21:38:43.899246Z","submitted_at":"2025-08-11T17:43:45Z","title":"Capabilities of GPT-5 on Multimodal Medical Reasoning","version":2},"cited_work":{"arxiv_id":"2508.08224","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.08224","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2508.08224 , year=","venue":null,"work_id":"34a11023-802a-400d-8b18-ae8fd1635f3b","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2508.08224","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:ffc3d343d7f5d0a94cceed0d6b1608a7216a0211b3e6f7543faf93edf53153b2","observation_id":"47a02ffc-5ad4-4e0a-9370-de99b3808546","resolution":{"observed_at":"2026-05-11T00:15:52.778358Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.05221","last_updated":"2025-08-07T10:02:07Z","snapshot_observed_at":"2026-08-05T23:33:19.402646Z","submitted_at":"2025-08-07T10:02:07Z","title":"ReasoningTrack: Chain-of-Thought Reasoning for Long-term Vision-Language Tracking","version":1},"cited_work":{"arxiv_id":"2508.05221","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.05221","snapshot_observed_at":"2026-06-30T07:24:21.262066Z","title":"Reasoningtrack: Chain-of-thought reasoning for long-term vision-language tracking","venue":null,"work_id":"6e9195bc-4a62-4f9d-ade8-ac76f8468598","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2508.05221","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:b470e669001bd84c7a83f2212c07e46f4c60f3863457b89d64fee9734e48347c","observation_id":"9712d64c-8111-4ab7-a570-3004793e8fe0","resolution":{"observed_at":"2026-05-11T00:15:52.842242Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10228","last_updated":"2024-03-15T11:58:18Z","snapshot_observed_at":"2026-07-06T17:45:11.343569Z","submitted_at":"2024-03-15T11:58:18Z","title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","version":1},"cited_work":{"arxiv_id":"2403.10228","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.10228","snapshot_observed_at":"2026-07-03T20:38:56.195732Z","title":"arXiv preprint arXiv:2403.10228 , year=","venue":null,"work_id":"fb2ec1aa-a3b0-43e2-9e93-2b73d3097074","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2403.10228","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:3f1e7832e89461d6ddc308588f71db9b6837b21b9adfc1061610c1d77a070620","observation_id":"f06b4652-9297-48bf-a0b1-87467b1b6445","resolution":{"observed_at":"2026-05-11T00:15:52.667017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"de1cdfc8-f07c-4bb4-91dd-bfd2ae030b11","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:31590cb83da535bda2c7b4be3bcbbc33426dbb6ac0bc5a1c6c0cebe3c3c80d6c","observation_id":"7f3332ce-1bfc-4977-90c5-2f2b42be7db9","resolution":{"observed_at":"2026-05-16T17:01:08.945824Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"094b4e6c-4d5b-4d3c-8443-55442b585faf","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:5e87c5b43e8d0ee6c3f60a35a94295380a4873d51a6208426483d30eb431676d","observation_id":"ede8304a-8de8-4b7b-886f-e7078fb0d046","resolution":{"observed_at":"2026-05-16T17:01:08.923891Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"38e7f8eb-6e2e-449c-a855-a06161290795","year":2021},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:2bb90717458a5b90e38b2b4a948e7e3d556129ebea498e985dbef4839e0e4a55","observation_id":"97bd188b-0fea-47aa-869a-162614f932f1","resolution":{"observed_at":"2026-05-16T17:01:08.915312Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ddcb8d21-0ed8-4707-89b7-99d31a478b23","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:807fd7c0fc95aa50130d2e45c5dba69e426e69db8f85ae44a76f644305170f48","observation_id":"ad62aba3-729e-48da-918b-a3813783ba9d","resolution":{"observed_at":"2026-05-16T17:01:08.947018Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ee01e727-9002-448e-a22d-c5dbc35694a9","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:0087de7bb17c62c24adf68a748f585ebb126b1808b70a3bc302d0d4c8a504a2a","observation_id":"ff4adf8e-d13f-4636-b8c2-da9cafd2db8f","resolution":{"observed_at":"2026-05-16T17:01:08.939153Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:6cb063e1c73ffa83ba145b45b5e3638099de1efe9339a40261444edb3ca5ea99","observation_id":"ac64b546-be49-486f-8215-00dbdc5c3862","resolution":{"observed_at":"2026-05-11T00:15:52.816730Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"01979873-822c-4812-b0ba-a9c49d1d2968","year":2022},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:f52915254b26c618a79cc2f0e9c8945cc8a6cbaa1ba367f734fbaf167400d5ad","observation_id":"d4338866-c3e6-4647-ace4-15f2127f0164","resolution":{"observed_at":"2026-05-16T17:01:08.933081Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b87a3fbf-fda4-478e-979a-6972c0225ec8","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:70647c719f092109223e0335c69c95adb654427a8d2ad1ced43bf8c7a4bc524e","observation_id":"42891a3c-abd8-4789-a31e-6a943be979b4","resolution":{"observed_at":"2026-05-16T17:01:08.953108Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.03332","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"783d395e-9e81-406c-9c06-1610a6061995","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:29dbd86722010c1c702b29e51319d1e4d667107025976dc7c93cf6c0b8bf2221","observation_id":"0db98ccb-098a-4f3f-b0fc-f84a03105cd3","resolution":{"observed_at":"2026-05-11T00:15:52.683254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.15178","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"10a84b72-b3c8-4b8c-8bb7-5221b08820d8","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:ab393a292bcd574e0861ba17e8b3285dc7810679c680d1a6f47f82a586994e8d","observation_id":"fb385005-0799-4810-b9d5-eb0c52d56d29","resolution":{"observed_at":"2026-05-11T00:15:52.661029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10500","last_updated":"2025-03-13T16:02:30Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T16:02:30Z","title":"OmniSTVG: Toward Spatio-Temporal Omni-Object Video Grounding","version":1},"cited_work":{"arxiv_id":"2503.10500","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10500","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omnistvg: Toward spatio-temporal omni-object video grounding","venue":null,"work_id":"8c9620f7-eab5-4f97-ba2f-1707c4671a34","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2503.10500","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:4bd48048a439fbccc33cfe92cba350467640db98236f8b6565ff8846a521ba3e","observation_id":"c58e50b0-e33d-4bd9-9cf1-23a78c262372","resolution":{"observed_at":"2026-05-11T00:15:52.900670Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01442","last_updated":"2020-03-08T00:09:07Z","snapshot_observed_at":"2026-07-06T08:26:38.349660Z","submitted_at":"2019-10-03T13:16:36Z","title":"CLEVRER: CoLlision Events for Video REpresentation and Reasoning","version":2},"cited_work":{"arxiv_id":"1910.01442","doi":"10.48550/arxiv.1910.01442","metadata_source":"pith","pith_arxiv_id":"1910.01442","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CLEVRER: CoLlision Events for Video REpresentation and Reasoning","venue":"cs.CV","work_id":"595fdbf7-89d5-4593-8f57-24e8b469a43c","year":2019},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/1910.01442","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:c1e831240a9f4a80d3446881007a2b151cb52ce9284a44afa4238061e48adefb","observation_id":"1ffa13e2-a46b-48ef-a74a-111ef11e5c8b","resolution":{"observed_at":"2026-05-16T18:01:50.618167Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-23T20:53:53.698369+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T20:53:53.698369+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.06471","last_updated":"2025-08-08T17:21:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-08T17:21:06Z","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","version":1},"cited_work":{"arxiv_id":"2508.06471","doi":"10.48550/arxiv.2508.06471","metadata_source":"pith","pith_arxiv_id":"2508.06471","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","venue":"cs.CL","work_id":"5bb4e5d7-985e-431b-bb0d-75576cdc2950","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2508.06471","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:b1c531c4d6316f32f950bea1d1a04a8f4f416c9abb932be6fe9f349e862746cd","observation_id":"be9ad18d-9541-4556-ba95-b0b211f37482","resolution":{"observed_at":"2026-05-11T17:50:08.654972Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-23T05:23:01.474969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T05:23:01.474969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-06T08:21:30.398010Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:c1024e0879defe0a512e0d3dd88ab5eade0ad645043be81820c96a4125aea582","observation_id":"616db61f-0708-4d9d-b4f9-5467b24966ae","resolution":{"observed_at":"2026-05-11T00:15:52.911593Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"483f7fdd-9c5f-4aab-8d17-ca9fb817553f","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:d523fcfd69949f49406600da35d4ba837fe97120308de0a49a501ff672ce2e67","observation_id":"379c3f38-9768-4c07-8fdf-d426f3a223e8","resolution":{"observed_at":"2026-05-16T17:01:08.967678Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2008.06941","last_updated":"2020-08-22T11:11:32Z","snapshot_observed_at":"2026-08-01T19:16:06.824442Z","submitted_at":"2020-08-16T15:39:56Z","title":"Object-Aware Multi-Branch Relation Networks for Spatio-Temporal Video Grounding","version":2},"cited_work":{"arxiv_id":"2008.06941","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2008.06941","snapshot_observed_at":"2026-07-02T02:06:26.944849Z","title":"Object-aware multi-branch relation networks for spatio-temporal video grounding","venue":null,"work_id":"bf57be4a-2a0a-4fd1-8bb9-9ed315d3c1e3","year":2008},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2008.06941","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:518d7c509dc257b8dcb02472ed5b2ddbb2ee11fd54f68e6044ab9935d6623b16","observation_id":"cb6194a0-2c77-490e-aa60-3cc793f4f6f7","resolution":{"observed_at":"2026-05-11T00:15:52.947338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"94aed302-3693-46ec-85e5-783c3ea09299","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:c1703eb0cfa4d7e5343ef12c4d82a42cd29acb0aa002444c41a96c787b5c5fdf","observation_id":"bc92fbcb-83ef-473b-ab63-8c4b64108fb8","resolution":{"observed_at":"2026-05-16T17:01:08.928512Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":"82b0a49d-5c1a-406f-9a64-c5788282931f","year":null},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:8ad29f6814afd18ebc2dcba2da28e5c87234870dc2808c6d086411b529ae9f52","observation_id":"18750a5a-ae32-48f5-a051-0d052b4be2f7","resolution":{"observed_at":"2026-05-16T17:01:08.908758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.13313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5bbefd9d-2ec2-411b-94a4-b4cd0c82f8eb","year":2026},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:e0b72485ea82b7f281c35c891ac2cbab8049169c2a1ec3f701e6c10bf6d4c193","observation_id":"0ddd6f15-8067-4fa1-abbd-ba9db88528a0","resolution":{"observed_at":"2026-05-11T00:15:52.627661Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":"2504.10479","doi":"10.48550/arxiv.2504.10479","metadata_source":"pith","pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","venue":"cs.CV","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","year":2025},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:e93f7a6ebdcd125449692d77495d4d6cb13b6b430e863fa97ca11fe1e1be1384","observation_id":"76af5c33-de7a-42d7-9de3-a0a5171e2c6d","resolution":{"observed_at":"2026-05-11T00:15:52.615342Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.04159","last_updated":"2021-03-18T03:14:26Z","snapshot_observed_at":"2026-07-06T10:02:45.105181Z","submitted_at":"2020-10-08T17:59:21Z","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","version":4},"cited_work":{"arxiv_id":"2010.04159","doi":"10.48550/arxiv.2010.04159","metadata_source":"pith","pith_arxiv_id":"2010.04159","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","venue":"cs.CV","work_id":"876f9fe8-c712-4550-9c26-cc18ab69abf2","year":2020},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2010.04159","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:fb17a1c093d9566894a2718d6a662e0a83bb5d84f5c81e68bcf3f9f892327564","observation_id":"fb089b24-05d2-4fd7-ade0-098a9f8457eb","resolution":{"observed_at":"2026-05-11T09:47:17.069034Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T00:17:26.235002Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding"},"reference_resolution":{"displayed":78,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":40,"verified_exact":33,"verified_fuzzy":5},"total_outbound_references":78},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 78 of 78 outbound references and 2 inbound Pith citation observations for arXiv:2604.08014."}