{"as_of":"2026-08-08T09:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b1f14bf5d6d5524599e0479a19715215ea38093f94dc6ffb76ccc4b4747035f6","coverage":[{"denominator":85,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":85,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T10:58:20.690107Z","state":"measured"},{"denominator":86,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":86,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-22T09:20:32.920925Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-22T09:21:20.658722Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"cited_work":{"arxiv_id":"2509.03501","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.03501","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ryoo, Silvio Savarese, Caiming Xiong, and Juan Carlos Niebles","venue":null,"work_id":"74747d35-9880-4f4d-9f35-53c9f2416dbe","year":2025},"citing_paper":{"arxiv_id":"2605.21625","last_updated":"2026-05-20T18:36:57Z","snapshot_observed_at":"2026-07-06T23:32:01.202542Z","submitted_at":"2026-05-20T18:36:57Z","title":"Flat-Pack Bench: Evaluating Spatio-Temporal Understanding in Large Vision-Language Models through Furniture Assembly","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-22T09:20:32.920925Z"},"links":{"cited_paper":"/paper/2509.03501","citing_paper":"/paper/2605.21625"},"observation_digest":"sha256:18d7996692b840e39af78fbbfa5228339c506e2ca0e91d30c491423722b3d414","observation_id":"7d5ed8f9-de26-4309-b4df-0653128618b2","resolution":{"observed_at":"2026-05-22T09:21:20.660900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2509.03501/citation-record","integrity":"/paper/2509.03501/integrity","json":"/paper/2509.03501/citation-record.json","paper":"/paper/2509.03501"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-05T10:58:15.439297Z","title":"Phi-3 technical report: A highly capable language model locally on your phone","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.439297Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:8b5c12cac07812cc365b3bff483655b187d1cef172f172117712fbc73e6343a8","observation_id":"f32d129a-eba2-490c-9936-8d5edfae3144","resolution":{"observed_at":"2026-08-05T10:58:15.439297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:15.478984Z","title":"Vicas: A dataset for combining holistic and pixel-level video un- derstanding using captions with grounded segmentation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.478984Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:cc52e9fbdacc1463ff5b6b97334cb6e32bd76e62ba329dc64dcd0ea64025f4ad","observation_id":"b3ccf81e-1b56-427c-9b08-7e20054f5d45","resolution":{"observed_at":"2026-08-05T10:58:15.478984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T10:58:15.511934Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.511934Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:eb7b071998cc25513fd2e7bc397e6017c421b03a3157e613851fcd3adbae5db8","observation_id":"5393e087-47aa-431c-a067-46362548aeee","resolution":{"observed_at":"2026-08-05T10:58:15.511934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:15.546854Z","title":"PySceneDetect: Video Scene Cut Detection","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.546854Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:655c9e465d5f86f3922e4337542a18287c052fa659d0c8712e618ebeaeff450f","observation_id":"0acf6b67-da61-46a4-8d88-8b9b91104798","resolution":{"observed_at":"2026-08-05T10:58:15.546854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:15.594910Z","title":"Sharegpt4video: Improving video understanding and generation with better captions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.594910Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:2798f3d946387b7afa0fbac015c3c0a3eaf7162dfcc1ac09fda768d4776d46c6","observation_id":"0942c527-958a-49d8-a7a2-3e6248af82cb","resolution":{"observed_at":"2026-08-05T10:58:15.594910Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:15.625667Z","title":"Panda-70m: Captioning 70m videos with multiple cross-modality teachers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.625667Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:81bc23d16c3b694e3b29826bf365d472ee2c57403564a4b063645403ed002e0c","observation_id":"8a097da6-4d02-45a7-9027-91034c18167b","resolution":{"observed_at":"2026-08-05T10:58:15.625667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-05T10:58:15.702318Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test- time scaling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.702318Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:d1e60e6475a50f46988162e744869c464452473d3e1025a49dc1c2c1557c1c1d","observation_id":"5b0f2dbe-11db-4cda-bb90-e1f1021855e8","resolution":{"observed_at":"2026-08-05T10:58:15.702318Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-05T10:58:15.748440Z","title":"Videollama 2: Advancing spatial- temporal modeling and audio understanding in video-llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.748440Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:5de87a277f587f99c2fb42b569d4921f62f924766af07d08068d21ce5db38450","observation_id":"2d03d88a-5f72-4a3d-bf4a-d778e64fa9fc","resolution":{"observed_at":"2026-08-05T10:58:15.748440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13180","last_updated":"2025-07-23T19:22:35Z","snapshot_observed_at":"2026-08-07T23:47:53.537198Z","submitted_at":"2025-04-17T17:59:56Z","title":"PerceptionLM: Open-Access Data and Models for Detailed Visual Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13180","snapshot_observed_at":"2026-08-05T10:58:15.783270Z","title":"Per- ceptionlm: Open-access data and models for detailed visual understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.783270Z"},"links":{"cited_paper":"/paper/2504.13180","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:5a17b546d03675aa1d890cb8a37e1b5de6829efe687dab54fcbf8a33ef48bad2","observation_id":"048c6c3c-bdd1-4fc5-86b3-6ebce32f8450","resolution":{"observed_at":"2026-08-05T10:58:15.783270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01426","last_updated":"2025-06-15T21:53:18Z","snapshot_observed_at":"2026-08-07T18:41:52.024477Z","submitted_at":"2025-01-02T18:59:45Z","title":"Unifying Specialized Visual Encoders for Video Language Models","version":2},"cited_work":{"arxiv_id":"2501.01426","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01426","snapshot_observed_at":"2026-08-05T10:58:21.356963Z","title":"Unifying Specialized Visual Encoders for Video Language Models","venue":"cs.CV","work_id":"10d9662b-52ee-4edb-bdd9-1c64b635908e","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.827629Z"},"links":{"cited_paper":"/paper/2501.01426","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:8eac6f22a58695ae1c3a02dbf785f69e3b4757b823d316fcfd23f452c485a733","observation_id":"bd4597e5-0732-488b-8a8c-dc1b7b90ad62","resolution":{"observed_at":"2026-08-05T10:58:21.479672Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:15.862688Z","title":"Videorefer benchmark evaluation for general mllms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.862688Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:1e8d292846714e2835f366b6a1e9e35cbec22d97d7872242c0f893c4ffb53b27","observation_id":"39097890-3252-438c-801d-864c4e3436f5","resolution":{"observed_at":"2026-08-05T10:58:15.862688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:36.284245Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions","venue":null,"work_id":"115c4be7-bb93-4cdd-baed-cad9f06abd10","year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.886354Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:a1acbb6bd119a081557aeba6f4e46e81fd414c0407ac493eda0e2f27572775db","observation_id":"ea89e910-4024-4328-91d2-40ecbdec82b8","resolution":{"observed_at":"2026-08-05T10:58:36.378923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:36.003304Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":"df39bd0c-7ad7-416b-ad1b-1128615ad806","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.959077Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:df1aeff8e4c708f32a1a20e5b2378c5b0e53bc2c699dc6288ff0c9c36fe45d5b","observation_id":"20a1c4c3-7e96-4250-857e-91bb626e0f3c","resolution":{"observed_at":"2026-08-05T10:58:36.140120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:35.780463Z","title":null,"venue":null,"work_id":"11c258a8-5493-47b1-98d3-d5687bd8a1bb","year":2018},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.964475Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:3ff8168a2be663e64c3f4e3af245fd0490192fe78b820bf6567c558b9e4ead1c","observation_id":"c4716635-c7ba-48e6-befe-dd37db008796","resolution":{"observed_at":"2026-08-05T10:58:35.911323Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:35.532219Z","title":"Vtg-llm: Integrating timestamp knowledge into video llms for enhanced video temporal grounding","venue":null,"work_id":"8cbc11f6-5a05-4aa0-b63d-c39a90200dc9","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:15.997832Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:d34c4b67f8dd9874cede587efeabb32089291f1c2a0476c188b587a55fa16f99","observation_id":"dc4f1d6e-e92b-4609-96e1-011c9c7cfa23","resolution":{"observed_at":"2026-08-05T10:58:35.677128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08326","last_updated":"2025-03-22T09:03:54Z","snapshot_observed_at":"2026-08-04T01:44:36.508625Z","submitted_at":"2025-01-14T18:58:04Z","title":"Omni-RGPT: Unifying Image and Video Region-level Understanding via Token Marks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08326","snapshot_observed_at":"2026-08-05T10:58:16.021381Z","title":"Omni-rgpt: Unifying image and video region-level understanding via token marks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.021381Z"},"links":{"cited_paper":"/paper/2501.08326","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:df6f63f0cf34de7080e434126e45b6b0e09a0ad78497f6714f63ae087436b3d4","observation_id":"239fdc4d-74dd-4aac-8b13-50ca7b2927b2","resolution":{"observed_at":"2026-08-05T10:58:16.021381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16500","last_updated":"2024-08-29T12:59:12Z","snapshot_observed_at":"2026-08-05T11:54:14.447608Z","submitted_at":"2024-08-29T12:59:12Z","title":"CogVLM2: Visual Language Models for Image and Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16500","snapshot_observed_at":"2026-08-05T10:58:16.100830Z","title":"Cogvlm2: Visual language mod- els for image and video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.100830Z"},"links":{"cited_paper":"/paper/2408.16500","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:3fe87586b007f31545b0646956ccdc678d5f39106a653e3db8101e32c6a8c0ef","observation_id":"6b30dc52-6e55-4609-94cd-8a3165b86594","resolution":{"observed_at":"2026-08-05T10:58:16.100830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:35.351238Z","title":"Vtimellm: Empower llm to grasp video moments","venue":null,"work_id":"7783cb4f-59e9-4db7-9fa9-1b48f124c7a8","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.153565Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:66bd9fbb5c4ed77468d260d8018fff1191701f7a5653f93bc66def64a73812c1","observation_id":"8d6e30f3-90d9-4439-b413-0fe2d25dcc04","resolution":{"observed_at":"2026-08-05T10:58:35.448603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:34.966780Z","title":"Tgif-qa: Toward spatio-temporal reasoning in visual question answering","venue":null,"work_id":"978211c8-0154-4068-8904-c0997b890796","year":2017},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.212317Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:eff7c918c457018f8f6e5c0c23d34f187192dae5527561cce172c3d46f410fed","observation_id":"ca09e8d2-9d88-44e3-bebf-71617c12196e","resolution":{"observed_at":"2026-08-05T10:58:35.167898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:34.746617Z","title":"Referring to any person, 2025","venue":null,"work_id":"c2baa46f-c543-4537-9a98-93016cf3ee69","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.255854Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:a2868ee1231726356e0286cce8020027193cdf9334b7ad7cfa576aa929b845c6","observation_id":"c7f1e1c6-471c-4efa-b6ad-ef27fa137087","resolution":{"observed_at":"2026-08-05T10:58:34.871035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:34.386749Z","title":"Miradata: A large-scale video dataset with long durations and structured captions","venue":null,"work_id":"859acaa2-52c5-4f17-949f-f6da19f369e9","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.276413Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:e11da9c7df7e6e4e94b2039a1d79a3e351a2298ee97c2c94772dc57dba0797c0","observation_id":"8d707d43-44e9-44ae-bad0-275f8ebeb5f5","resolution":{"observed_at":"2026-08-05T10:58:34.590640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10781","last_updated":"2025-09-09T12:36:30Z","snapshot_observed_at":"2026-08-07T17:06:00.315976Z","submitted_at":"2025-03-13T18:21:07Z","title":"Large-scale Pre-training for Grounded Video Caption Generation","version":3},"cited_work":{"arxiv_id":"2503.10781","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.10781","snapshot_observed_at":"2026-08-05T10:58:21.128916Z","title":"Large-scale Pre-training for Grounded Video Caption Generation","venue":"cs.CV","work_id":"46cf00a2-f06d-46f9-b85a-93229ad834c6","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.322852Z"},"links":{"cited_paper":"/paper/2503.10781","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:1d8847314462883cfe280aba1173c6bc5f69ff5c1769e71d85c646aac2b1a84c","observation_id":"64164d42-fc27-40eb-8b7d-6a7dd2404fe1","resolution":{"observed_at":"2026-08-05T10:58:21.238931Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:34.121515Z","title":"Detecting mo- ments and highlights in videos via natural language queries","venue":null,"work_id":"6f19d591-c65f-4cea-9cc3-5d3b9d2a08a6","year":2021},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.360030Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:8c171d4e6226c3325f29f3d3190f33d42422957519e5ca709bd54e16c3a001fd","observation_id":"490269bf-c118-43c1-9d77-91433c9dc0bd","resolution":{"observed_at":"2026-08-05T10:58:34.262660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T10:58:16.394371Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.394371Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:f2f640475f73546db175fd56d10a90b6f9a2bb784cd8b5309f5a9f917c0df109","observation_id":"bdbad2d6-08e4-4336-8bb6-36cb5609d366","resolution":{"observed_at":"2026-08-05T10:58:16.394371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T10:58:16.444568Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.444568Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:ea3b82ad99d4c03b15a991a0d056cab3e0d4a91da446a7812f833f7a6a38fd35","observation_id":"565fbcb7-361f-4c0f-a99e-47a1c40b960c","resolution":{"observed_at":"2026-08-05T10:58:16.444568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:33.882316Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":"76f7d80d-7ef7-4d32-a4d8-b85f013b4a86","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.510722Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:f24300843b024d85575c214c966506faaf420767838fcafc2a96461c457f35dc","observation_id":"4e2dcb1b-3744-4395-a5b1-e1c1d46511d2","resolution":{"observed_at":"2026-08-05T10:58:33.986752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:33.554576Z","title":"Temporal reasoning transfer from text to video","venue":null,"work_id":"4079f003-6963-4133-bdb3-49010e564f89","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.553789Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:ffbc11c2f1013e217913db4fc815f5354ebe4abe0753898aa2626b8b4170a35e","observation_id":"5f901906-4b2c-4118-aa92-62de57bd8c5d","resolution":{"observed_at":"2026-08-05T10:58:33.713478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:33.180914Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"bc01d536-c6bf-402a-8303-1e654fbd3e3b","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.591382Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:f1b54020c5c2c54584830c7b318cad7bee94b1a7034d65201c5bdf673ad79f65","observation_id":"b7e4c9cd-d50d-4af7-8df4-24e422889d97","resolution":{"observed_at":"2026-08-05T10:58:33.367721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16072","last_updated":"2025-04-22T17:51:41Z","snapshot_observed_at":"2026-08-07T16:00:09.944765Z","submitted_at":"2025-04-22T17:51:41Z","title":"Describe Anything: Detailed Localized Image and Video Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16072","snapshot_observed_at":"2026-08-05T10:58:16.670856Z","title":"Describe anything: Detailed localized image and video captioning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.670856Z"},"links":{"cited_paper":"/paper/2504.16072","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:5548f987ca1ad527ef83979ff961dc7c62e4d1259b1ef50687599d69bdd465a4","observation_id":"17a543bd-35a1-449b-abf9-58aa9b4c13ea","resolution":{"observed_at":"2026-08-05T10:58:16.670856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:16.708257Z","title":"Unleashing hour-scale video train- ing for long video-language understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.708257Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:cc8c984fedf042c6020045f77a4d1f05e98e1c740c4f4cdfccff5cc7c99f5a48","observation_id":"51407507-def0-4822-9ef0-dcab963e35dc","resolution":{"observed_at":"2026-08-05T10:58:16.708257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-05T10:58:16.722899Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.722899Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:6dfa806711003fb33298ffc6ab8207bf0af0b4b5e19db30adbed1aa01bba6360","observation_id":"2b0f3896-8ba2-43c2-ac5a-aa020adf157d","resolution":{"observed_at":"2026-08-05T10:58:16.722899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-05T10:58:16.779731Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.779731Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:42fe1830bc290342509c0379b6a82104ebd1dd6554fc6e1e995218d635ca81d3","observation_id":"7d10a3ff-980e-41d7-a53f-13c215e27703","resolution":{"observed_at":"2026-08-05T10:58:16.779731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00476","snapshot_observed_at":"2026-08-05T10:58:16.815963Z","title":"Tempcom- pass: Do video llms really understand videos?arXiv preprint arXiv:2403.00476, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.815963Z"},"links":{"cited_paper":"/paper/2403.00476","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:606922c30a45b45d376ab70de228055414eafd42d0a03cbf3ed0249e2a7e5ee7","observation_id":"10c054de-5eed-4652-8f0e-67c4dddb1291","resolution":{"observed_at":"2026-08-05T10:58:16.815963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:32.872413Z","title":"Groma: Localized visual tokenization for grounding multimodal large language models","venue":null,"work_id":"50d8885a-cbc8-46fe-b1ef-0e9e1d829ddf","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.858999Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:545324ae90c73f3d6635f8d5a59fcc8079d49baf4e2218c86bd3b01f39d81648","observation_id":"8baa8687-9803-408e-b636-62050abd02b7","resolution":{"observed_at":"2026-08-05T10:58:33.010944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-05T10:58:16.903200Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.903200Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:dcdbc344ce13bc7cb347c5ae9f7d793c8f432e7f20d66208ebcdc71d424387b1","observation_id":"a6ad5f79-7adc-411a-9357-50dd0a011a34","resolution":{"observed_at":"2026-08-05T10:58:16.903200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.13681","last_updated":"2022-02-18T05:50:50Z","snapshot_observed_at":"2026-08-05T16:29:39.348508Z","submitted_at":"2020-11-27T11:43:45Z","title":"Point and Ask: Incorporating Pointing into Visual Question Answering","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2011.13681","snapshot_observed_at":"2026-08-05T10:58:16.959800Z","title":"Point and ask: Incorporating pointing into visual question answering","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.959800Z"},"links":{"cited_paper":"/paper/2011.13681","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:62ca330302331bc44484f80b2747a6f1593ec4c38eae16ec5db9e4bd1eb90a66","observation_id":"bd380dbb-488a-4b83-8b6c-16e9af6165b4","resolution":{"observed_at":"2026-08-05T10:58:16.959800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.13435","last_updated":"2023-12-13T17:24:10Z","snapshot_observed_at":"2026-07-06T16:51:06.781541Z","submitted_at":"2023-11-22T14:48:30Z","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.13435","snapshot_observed_at":"2026-08-05T10:58:17.017701Z","title":"Pg-video-llava: Pixel grounding large video- language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.017701Z"},"links":{"cited_paper":"/paper/2311.13435","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:618587a62a62681a2f137c15dfcb4f0675183464d023a4de37532e5a7c574da7","observation_id":"94876515-aa3e-476b-a6bb-d9e3c9eeeb79","resolution":{"observed_at":"2026-08-05T10:58:17.017701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:32.514834Z","title":"Videoglamm: A large multimodal model for pixel-level vi- sual grounding in videos","venue":null,"work_id":"fb99ba21-9c91-43c6-80f2-e1fa3fd68b8b","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.050270Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:b035d813ab786788a9a871da451b4d09171375a09dcbab8fc87adbde0fee5950","observation_id":"716af1ec-e4e1-4635-9627-c571d526b777","resolution":{"observed_at":"2026-08-05T10:58:32.712204Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11435","last_updated":"2024-06-02T05:40:18Z","snapshot_observed_at":"2026-08-05T15:20:33.981747Z","submitted_at":"2024-02-18T03:04:38Z","title":"Momentor: Advancing Video Large Language Model with Fine-Grained Temporal Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11435","snapshot_observed_at":"2026-08-05T10:58:17.121792Z","title":"Momen- tor: Advancing video large language model with fine-grained temporal reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.121792Z"},"links":{"cited_paper":"/paper/2402.11435","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:4bd958ed1f002e582fb69378281bb83cc9690eeb20b782327eea4b7ce1374442","observation_id":"5ff4291b-5505-4f60-b17f-e45a306f10fb","resolution":{"observed_at":"2026-08-05T10:58:17.121792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:32.109460Z","title":"Artemis: Towards referential understanding in com- plex videos","venue":null,"work_id":"dfa3c7df-0bee-4875-b7fa-281f9bd0c562","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.184642Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:06352a82f798f05d5364290c724b9de8581a8b821117db4ed99bd91d65f389f5","observation_id":"0b2eee42-add2-4e91-92c4-61e8bce2a2ad","resolution":{"observed_at":"2026-08-05T10:58:32.345406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:31.781628Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":"f8591bae-7830-4a39-a284-1ae353bf3817","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.240520Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:2b87c12a1b315d629ea9390ab965d99979f6835e8bfc001714b873fa2d6b93d2","observation_id":"f1ab99bb-d21e-4920-99d6-f9af787c780e","resolution":{"observed_at":"2026-08-05T10:58:31.969203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-05T10:58:17.383157Z","title":"Sam 2: Segment anything in images and videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.383157Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:f44c6b5967a9e92bb6156d36f943dcde418dc480c8baec4c4a849d8ee1f183c3","observation_id":"cd859a89-9070-41aa-871c-aff310a6b94a","resolution":{"observed_at":"2026-08-05T10:58:17.383157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:31.446174Z","title":"Grounded sam 2: Ground and track anything in videos with grounding dino, florence-2 and sam 2","venue":null,"work_id":"b293c760-0940-43ae-b153-694287541776","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.432596Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:de409f8152c8a4e05eab5dc2abfdf7a3d3132db6d9e0b954d26f8fe6b30db285","observation_id":"a4b7911d-4e68-4380-aa44-c84057af99cf","resolution":{"observed_at":"2026-08-05T10:58:31.616302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16267","last_updated":"2025-06-09T19:33:32Z","snapshot_observed_at":"2026-07-06T19:37:16.203681Z","submitted_at":"2024-10-21T17:59:11Z","title":"xGen-MM-Vid (BLIP-3-Video): You Only Need 32 Tokens to Represent a Video Even in VLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16267","snapshot_observed_at":"2026-08-05T10:58:17.487286Z","title":"xgen-mm-vid (blip-3-video): You only need 32 tokens to represent a video even in vlms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.487286Z"},"links":{"cited_paper":"/paper/2410.16267","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:32a4a6a89e8194ec61336d364195c67ea4240fd6fe9853d612b3a2aaa45a3e97","observation_id":"2f9abcff-b849-49cd-b4bc-18e75325fbd6","resolution":{"observed_at":"2026-08-05T10:58:17.487286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00459","last_updated":"2024-09-26T09:54:57Z","snapshot_observed_at":"2026-08-07T22:39:17.691418Z","submitted_at":"2024-03-30T19:46:59Z","title":"NumeroLogic: Number Encoding for Enhanced LLMs' Numerical Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00459","snapshot_observed_at":"2026-08-05T10:58:17.549752Z","title":"Numerologic: Num- ber encoding for enhanced llms’ numerical reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.549752Z"},"links":{"cited_paper":"/paper/2404.00459","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:905eba17209d6aff0edfe2fa841fb2f433e3677e1fced4a31b1353455b9d05b7","observation_id":"99bcb17b-27a2-48d3-93ce-16d3844b925f","resolution":{"observed_at":"2026-08-05T10:58:17.549752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:17.583118Z","title":"Sama: Towards multi-turn referen- tial grounded video chat with large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.583118Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:9c08fc373c9a7419170aa139a1543868fc030a5f7c68110d95eb47bbda4dad59","observation_id":"1bbee3d4-1b02-4f24-acb0-d7539bf57205","resolution":{"observed_at":"2026-08-05T10:58:17.583118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:31.213549Z","title":"Qwen2.5: A party of foundation models, 2024","venue":null,"work_id":"497efae3-c670-4305-a3c9-317aaa75492d","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.612753Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:2a42966c5f47bcdbddf55bf32094d321bd52376eebc31bfd636a0deb455be27a","observation_id":"0c54bd9b-9a9c-4f0e-9a94-cc1157014dd7","resolution":{"observed_at":"2026-08-05T10:58:31.332165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:30.812056Z","title":"Natural language processing with Python and spaCy: A practical introduction","venue":null,"work_id":"4bfb245b-e4d1-485f-a86b-92c642f971af","year":2020},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.667045Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:9b4d0a7949f77ceb511c5bcf19b82b486105e1ca23850fcec6c14828c98596ef","observation_id":"7214d2b9-9d9e-41f2-8fbf-0bf0ee11b373","resolution":{"observed_at":"2026-08-05T10:58:31.040447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03290","last_updated":"2025-08-21T05:15:19Z","snapshot_observed_at":"2026-08-02T03:09:09.488305Z","submitted_at":"2024-10-04T10:04:37Z","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03290","snapshot_observed_at":"2026-08-05T10:58:17.747639Z","title":"Grounded-videollm: Sharpening fine-grained tem- poral grounding in video large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.747639Z"},"links":{"cited_paper":"/paper/2410.03290","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:375429927acad786c41def9d9bf3c37be10d45fc7efea877645a65d0e31329ab","observation_id":"41ca1c2b-a3bf-4f21-bef5-9b4434fbc12a","resolution":{"observed_at":"2026-08-05T10:58:17.747639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:30.485425Z","title":"Elysium: Exploring object-level perception in videos via mllm","venue":null,"work_id":"2acc4488-60fc-4259-93cf-01d0aad07ce1","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.831748Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:d640855c770ccfc9693a5a2e4442d78d260085eb6be42eeaeef53ee1c590ea60","observation_id":"765d890c-9a71-4e79-97ce-21c300b094b2","resolution":{"observed_at":"2026-08-05T10:58:30.665992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:30.224140Z","title":"Tarsier: Recipes for training and evaluating large video description models, 2024","venue":null,"work_id":"5df30db5-ade1-40c0-a15b-00961f616371","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.857705Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:38194e53e3b98cfe6d8dc2222ef44f193a2bbdb1a7926e94130fba273ba426cf","observation_id":"ff965332-63fc-4e5b-87ca-42b14e74a0f2","resolution":{"observed_at":"2026-08-05T10:58:30.357181Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:29.906580Z","title":"Language as queries for referring video object segmen- tation","venue":null,"work_id":"d7fd916f-a78e-4975-8eb3-76473e424fad","year":2022},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:17.889827Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:6f83881cadedf7611bac8f903eb53fa4354c26c8c4bdbb94afddc1c7de5a3373","observation_id":"b15deeb9-3579-48ff-a139-a73346e85ea6","resolution":{"observed_at":"2026-08-05T10:58:30.042271Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05037","last_updated":"2025-03-27T09:39:11Z","snapshot_observed_at":"2026-07-06T20:18:35.429115Z","submitted_at":"2025-01-09T07:51:14Z","title":"LongViTU: Instruction Tuning for Long-Form Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05037","snapshot_observed_at":"2026-08-05T10:58:18.003545Z","title":"Longvitu: In- struction tuning for long-form video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.003545Z"},"links":{"cited_paper":"/paper/2501.05037","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:6ceba393e52c614bf9e8180d940e2064fe296fb856afa277340f03b930d38582","observation_id":"88627f00-7b73-4104-9607-25201e2aaa50","resolution":{"observed_at":"2026-08-05T10:58:18.003545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:29.570489Z","title":"Number it: Temporal grounding videos like flipping manga","venue":null,"work_id":"56e70fcc-aaa2-4927-8862-4c9a8ebf7328","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.090304Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:1d19d971070d00dd00c76694ab6198b10a0c2683d5f307572a269d668d32fcd3","observation_id":"21cda32f-45a3-4ae8-a22a-8e9a1467a8ae","resolution":{"observed_at":"2026-08-05T10:58:29.734566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:29.239963Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"f31b56b5-9a28-4e19-89bd-94972023ee84","year":2021},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.182650Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:449cd8c7359551095617ecd8eb50786178bed181aa193b43731dcebfbfb0c632","observation_id":"db8b1beb-fc4e-4632-af2d-dd2f0be45627","resolution":{"observed_at":"2026-08-05T10:58:29.392629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:28.887034Z","title":"Video question answer- ing via gradually refined attention over appearance and mo- tion","venue":null,"work_id":"c743a8b3-c040-4653-a50e-db2a644343aa","year":2017},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.254568Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:94c872eecc05771ceda334b6361f20f8e1526d90fd9ef439fdef05a45c1e55e6","observation_id":"fca9e09a-1ba7-4a9c-b092-17d5b39bb4d6","resolution":{"observed_at":"2026-08-05T10:58:29.082214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:28.455203Z","title":"Pixel- aligned language model","venue":null,"work_id":"3d105732-c8ee-4ea4-9999-a2e2d60603df","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.284425Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:7a5221c0eb3c5540d85df38c6788fa2b43e5e18bc5604e82ae64d54f07902a1f","observation_id":"29bf663a-d044-4637-a009-e82cb6143f55","resolution":{"observed_at":"2026-08-05T10:58:28.697397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-05T10:58:18.338413Z","title":"Pllava: Parameter-free llava extension from images to videos for video dense captioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.338413Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:c1227a99c070fea5cb223c782cd7cd903bd1fdfa5ea4656ed1780d15e1c75ecf","observation_id":"fec93ee3-44fe-4d4c-95c9-34e70626e67a","resolution":{"observed_at":"2026-08-05T10:58:18.338413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-08-08T03:13:28.000968Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-05T10:58:18.397448Z","title":"Slowfast-llava: A strong training-free base- line for video large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.397448Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:91a212e76b8350b4cb51bb92d1089d984433fa4757029f83152777316f103a78","observation_id":"e49916f7-c529-4d14-8a85-dbf10b8ef374","resolution":{"observed_at":"2026-08-05T10:58:18.397448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:18.453965Z","title":"xgen-mm (blip-3): A family of open large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.453965Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:9f9f42445010bc9d92940c2c2cadb9e98e25faa647a3ed82ab182dcd7f721115","observation_id":"823bc005-5621-42bd-9c92-70213b7cdcbd","resolution":{"observed_at":"2026-08-05T10:58:18.453965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16375","last_updated":"2025-01-20T00:29:19Z","snapshot_observed_at":"2026-08-06T04:34:36.950710Z","submitted_at":"2024-04-25T07:29:17Z","title":"List Items One by One: A New Data Source and Learning Paradigm for Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16375","snapshot_observed_at":"2026-08-05T10:58:18.485504Z","title":"List items one by one: A new data source and learning paradigm for multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.485504Z"},"links":{"cited_paper":"/paper/2404.16375","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:9c2d398a829081171360e14b035225ac1bc71e0706b4b73c56429842077aab7c","observation_id":"77f0dbfa-cd10-4f3b-99b6-b97953a77307","resolution":{"observed_at":"2026-08-05T10:58:18.485504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11441","last_updated":"2023-11-06T07:39:49Z","snapshot_observed_at":"2026-08-04T03:29:49.409446Z","submitted_at":"2023-10-17T17:51:31Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.11441","snapshot_observed_at":"2026-08-05T10:58:18.599712Z","title":"Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.599712Z"},"links":{"cited_paper":"/paper/2310.11441","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:f95ff310c98002520f4b39a7d7a38d3cf177a098a4c33e2a1b7a7e489e7193c6","observation_id":"98767322-0937-462c-b96f-9cabf0103a4f","resolution":{"observed_at":"2026-08-05T10:58:18.599712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-05T10:58:18.651851Z","title":"Ferret: Refer and ground anything anywhere at any granularity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.651851Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:302ffaaa2dd2203e3bb4493d641e39cadd8756b16dbb8cd142857b7a008e49d1","observation_id":"93f24843-fe72-45a5-b580-731de430261e","resolution":{"observed_at":"2026-08-05T10:58:18.651851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:28.079049Z","title":"Merlin: Empowering multimodal llms with foresight minds","venue":null,"work_id":"c27652f4-79e2-4c3f-a2da-2786e05d0b09","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.742480Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:b26988998444a11488bd07ec8e41cb5cbcfaa4a82ba4bae0b593c6646d603804","observation_id":"ce8cfc46-6dae-45ea-98e2-9f2899656ef4","resolution":{"observed_at":"2026-08-05T10:58:28.238086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:27.766839Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"03344644-6150-401c-a999-9590a85a2917","year":2019},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.800425Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:43ac5cb9d1d14947670091af09d2fe0c3978940a739077bc164a1bc73adf6071","observation_id":"2ebddabe-83ff-4b38-bf40-e465dcc7d3d2","resolution":{"observed_at":"2026-08-05T10:58:27.920430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:27.368153Z","title":"Osprey: Pixel un- derstanding with visual instruction tuning","venue":null,"work_id":"738314dc-46af-40e5-83b7-ad427e58291e","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.873688Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:9a9beb0d94b15491a325bd18945212bf0a38e8e8a05c92cab51a7bade431306b","observation_id":"9aece6be-3583-454e-8cbd-a93e978b42d7","resolution":{"observed_at":"2026-08-05T10:58:27.591043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:26.979576Z","title":"Videorefer suite: Advancing spatial- temporal object understanding with video llm","venue":null,"work_id":"d0eac80c-907f-486c-aae8-fd5268a79edf","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:18.929032Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:c4e225898156dac7067af5aee26ccc4e420c20efc3c98a1f8a6de75cd26c6504","observation_id":"3a73b83d-ca66-4222-8e4d-56efb146d259","resolution":{"observed_at":"2026-08-05T10:58:27.131183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:26.581954Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"8bf2223a-d107-4e9f-9890-960082c9905d","year":2023},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.027799Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:e575c580286755c0755f1af148419b7044e721a4900671da5e952c0d618b42f0","observation_id":"d5d9aa8a-38c1-4266-bf08-5ad6c73145e3","resolution":{"observed_at":"2026-08-05T10:58:26.785238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-05T10:58:19.096136Z","title":"Videollama 3: Frontier multi- modal foundation models for image and video understand- ing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.096136Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:20d4d839ec63e7057858623463338c7b0ea3470794facd54bfead2e55d0938e0","observation_id":"9d81e749-85c4-4ba8-ae83-d53265e10d7d","resolution":{"observed_at":"2026-08-05T10:58:19.096136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:26.143626Z","title":"Llava-grounding: Grounded visual chat with large multimodal models","venue":null,"work_id":"1ef49a34-21aa-4a06-9382-a2b4a2691623","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.153012Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:ab48ad432b7eb70b6d6da42c0e37d78f93a06f4361bcc4f7ea9d4e62d8aeec70","observation_id":"8918d51d-37d9-45ad-be60-d56aad392477","resolution":{"observed_at":"2026-08-05T10:58:26.325015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07973","last_updated":"2024-04-11T17:56:05Z","snapshot_observed_at":"2026-08-07T09:16:48.366534Z","submitted_at":"2024-04-11T17:56:05Z","title":"Ferret-v2: An Improved Baseline for Referring and Grounding with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07973","snapshot_observed_at":"2026-08-05T10:58:19.257522Z","title":"Ferret- v2: An improved baseline for referring and grounding with large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.257522Z"},"links":{"cited_paper":"/paper/2404.07973","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:508eaa0f958cc0ef7dc772b6e6d9f6b9821e35d6dfc7d4df5ed4d4803c0281e3","observation_id":"65a71161-201a-4948-826e-c30a6127c282","resolution":{"observed_at":"2026-08-05T10:58:19.257522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:25.712618Z","title":"Gpt4roi: Instruction tuning large language model on region- of-interest","venue":null,"work_id":"01257e23-c89c-4c2f-ac99-ad7018e14d89","year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.342543Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:d9f79efb3c7791326e59436325b530819be21425ff963e77c3d4d0955e15df71","observation_id":"103af01e-08d4-495a-a86f-d49e1f4b426e","resolution":{"observed_at":"2026-08-05T10:58:25.935731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:25.315571Z","title":"Video instruction tuning with synthetic data, 2024","venue":null,"work_id":"45b3908f-37bd-4a16-8d4e-24e1e86fe9bd","year":2024},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.436000Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:ddd22dac10dcf5f589a1aceea1548cede68d32bad5ccc0caaef9376912f9da83","observation_id":"0db6e86d-0d55-44b4-8dbc-7acf2127e4cf","resolution":{"observed_at":"2026-08-05T10:58:25.520312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07519","last_updated":"2025-04-10T07:33:39Z","snapshot_observed_at":"2026-08-07T16:06:57.444749Z","submitted_at":"2025-04-10T07:33:39Z","title":"VideoExpert: Augmented LLM for Temporal-Sensitive Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07519","snapshot_observed_at":"2026-08-05T10:58:19.515582Z","title":"active” scene entities can you iden- tify from the video? An entity refers to an object, and “active","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.515582Z"},"links":{"cited_paper":"/paper/2504.07519","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:a360d96f20fd1b4761db4b985a69acfc899f0cb6d8b11c55eb15312f72654f32","observation_id":"59a300d9-b8e6-4ef6-a34b-318b14e219a4","resolution":{"observed_at":"2026-08-05T10:58:19.515582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:24.893949Z","title":"Frames are extracted only from the seg- ment of the video","venue":null,"work_id":"b82a51ea-20fa-4a22-8a4c-f76aee980220","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.612113Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:1f78fd285780e03b89c6fa55221575bfb1efb0817f63c8e0ea54d57f4bc0bf13","observation_id":"29fb73ab-44d5-4544-b861-1cfecb24fc66","resolution":{"observed_at":"2026-08-05T10:58:25.072468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:24.521346Z","title":"Sorry, I’m not sure","venue":null,"work_id":"1ad6cbca-9cff-45b4-af04-8086dcc60784","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.706926Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:223e1a800be63de1da1b4932df47a49f3c8248290013ff0ff8055825006fbb04","observation_id":"796b7f92-13a3-4030-89cb-11fb6d2ec278","resolution":{"observed_at":"2026-08-05T10:58:24.720638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:24.169309Z","title":"Frames are extracted only from the seg- ment of the video","venue":null,"work_id":"29248c44-ed58-43b2-9e6c-3b3f2ab224c9","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.711705Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:c1fd4bbcda36af9032c3f46664014d4bd6367a04b7ce975337d881cb6a96fb53","observation_id":"278c6d5f-1929-483f-8e31-c7a64564ef6a","resolution":{"observed_at":"2026-08-05T10:58:24.365748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:23.840006Z","title":"Yes” or “No","venue":null,"work_id":"3727b94f-d939-4a2e-a18e-a690e62741be","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:19.893683Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:43ceab05d287e3e28e026fa80baaf013017806262aee1bf699cedd367b4b04e8","observation_id":"7ad677b9-c848-4f99-9c64-189edb0320df","resolution":{"observed_at":"2026-08-05T10:58:23.998339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:23.501331Z","title":"Frames are extracted from the full video","venue":null,"work_id":"1da91570-53cb-4d24-b50b-a0757b3b424e","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:20.028670Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:658c33aa6b569849968a1bc38984948f670d3330b41df5ee670df217a718328c","observation_id":"7d300569-30cc-4e3d-842d-d40c885d5598","resolution":{"observed_at":"2026-08-05T10:58:23.669493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:23.147257Z","title":"Frames are extracted from the full video","venue":null,"work_id":"bb6fa2cb-434d-46eb-9815-df50a12b2ff2","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:20.152919Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:246e7663ac3da9938df8009274094c68e91fc5d18a2219d562105d0f895a496e","observation_id":"6bbfbe7e-2d92-4072-90ff-90d6db7ad08a","resolution":{"observed_at":"2026-08-05T10:58:23.298483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:22.781406Z","title":"Frames are extracted from the full video","venue":null,"work_id":"c5c963b3-49f1-4382-829c-f93b8b011393","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:20.210594Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:9226133bd4132e107a283fe87b5d91100dbade4da2cdc3588ea428c3d5878089","observation_id":"2f61eb7b-21ac-44fd-b0ec-b3351f8cb620","resolution":{"observed_at":"2026-08-05T10:58:22.941919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:22.442688Z","title":"Frames are extracted from the full video","venue":null,"work_id":"122413ae-8eec-46bf-8cf7-ff33657fa0b2","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:20.340722Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:7005e8a728b587ada67e9d58785652cef5aef8a5f67e0dd75de99ff88c764cdb","observation_id":"5fa46932-f1db-4106-b288-55198e7607f5","resolution":{"observed_at":"2026-08-05T10:58:22.613750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:22.148192Z","title":"Frames are extracted from the full video","venue":null,"work_id":"9ef6e1e1-019b-4c24-9dfa-96454693aadd","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:20.543748Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:79da00bafea81114bcf69689457d36f528a9d0957057619916f62e9b1c86f67c","observation_id":"c01a5e71-d7c6-47ad-811f-86fc083866b2","resolution":{"observed_at":"2026-08-05T10:58:22.332818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:21.921560Z","title":"Frames are extracted from the full video","venue":null,"work_id":"56f07527-89ff-49db-b79f-96447378745a","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:20.629580Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:7a29b60215c3ea9aae62de726a5b6c6bddd3092cb1170b1ec056422c27b787da","observation_id":"0c0d1e74-60de-41c2-bae6-e8f3673ee118","resolution":{"observed_at":"2026-08-05T10:58:22.033684Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:58:21.677915Z","title":"Frames are extracted from the full video","venue":null,"work_id":"38986418-56d1-4332-8020-a3b3a6ad26a5","year":null},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:20.690107Z"},"links":{"citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:ec311741849d8e2b7c6014424aa72273b40eb2b99650a4643d440475d835408c","observation_id":"ac81c5fd-500b-46bf-a0ae-bd3438dac2a7","resolution":{"observed_at":"2026-08-05T10:58:21.791400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data"},"reference_resolution":{"displayed":85,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":2,"verified_fuzzy":44},"total_outbound_references":85},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 85 of 85 outbound references and 1 inbound Pith citation observation for arXiv:2509.03501."}