{"as_of":"2026-08-11T02:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:80bfb1fca373e4c847e776caa156e69f9a3de1aad818a0d46a9f7bf1aaa4a696","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T22:12:05.365596Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:22:31.939898Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-02T02:46:28.010713Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":"2605.25979","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-07-02T02:46:28.010713Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","venue":"cs.CV","work_id":"4647c652-b82d-436f-a74d-ee82a38349ee","year":2026},"citing_paper":{"arxiv_id":"2606.03920","last_updated":"2026-06-02T17:12:05Z","snapshot_observed_at":"2026-08-02T12:14:20.077142Z","submitted_at":"2026-06-02T17:12:05Z","title":"Benchmarking Visual State Tracking in Multimodal Video Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T10:43:06.811228Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2606.03920"},"observation_digest":"sha256:2cc2b84bc90edf862712fa01eb018d535f6a6101a916a524088d6b90a69f6ea2","observation_id":"840434c4-03d3-4765-814a-a88dc39941f2","resolution":{"observed_at":"2026-07-02T02:46:28.012629Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":"2605.25979","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-07-02T02:46:28.010713Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","venue":"cs.CV","work_id":"4647c652-b82d-436f-a74d-ee82a38349ee","year":2026},"citing_paper":{"arxiv_id":"2606.07639","last_updated":"2026-06-01T09:07:15Z","snapshot_observed_at":"2026-07-06T23:47:15.041928Z","submitted_at":"2026-06-01T09:07:15Z","title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T15:22:31.310003Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2606.07639"},"observation_digest":"sha256:940028e418481f3a091ee50329971361283eab5cb21852e51a9f7234e2058111","observation_id":"a584e931-e40a-462b-bbde-022ef2d05f03","resolution":{"observed_at":"2026-07-01T22:26:17.892593Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-07-12T02:24:07.555850Z","title":"Llava-onevision-2: Towards next-generation perceptual intelligence.arXiv preprint arXiv:2605.25979, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.03451","last_updated":"2026-07-03T16:07:06Z","snapshot_observed_at":"2026-08-06T23:16:45.752886Z","submitted_at":"2026-07-03T16:07:06Z","title":"SkillOpt-Lite: Better and Faster Agent Self-evolution via One Line of Vibe","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-12T02:24:07.555850Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2607.03451"},"observation_digest":"sha256:a7fae37c760eaa8e0a0da4d8d744e77d9d9d556498ce2bc8b2cc5636c3475585","observation_id":"1fc63c2c-7128-4903-beb1-6f086b693d84","resolution":{"observed_at":"2026-07-12T02:24:07.555850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-08-02T00:44:49.651285Z","title":"Llava-onevision-2: Towards next- generation perceptual intelligence, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14935","last_updated":"2026-07-16T12:47:59Z","snapshot_observed_at":"2026-08-09T10:08:14.840784Z","submitted_at":"2026-07-16T12:47:59Z","title":"VideoChat3: Fully Open Video MLLM for Efficient and Generalist Video Understanding","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-02T00:44:49.651285Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2607.14935"},"observation_digest":"sha256:63d122260f400df8ab21d90b6c95e130b903ae4407384bf950c49e7eccf5daf6","observation_id":"916cc133-aeb9-4a69-85be-5014ca2f787c","resolution":{"observed_at":"2026-08-02T00:44:49.651285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-08-01T18:04:11.814953Z","title":"Llava-onevision-2: Towards next-generation perceptual intelligence.arXiv preprint arXiv:2605.25979, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.17423","last_updated":"2026-07-19T22:04:33Z","snapshot_observed_at":"2026-08-05T19:15:04.907010Z","submitted_at":"2026-07-19T22:04:33Z","title":"TimeLens2: Generalist Video Temporal Grounding with Multimodal LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T18:04:11.814953Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2607.17423"},"observation_digest":"sha256:2bdab26b8d906dfb3fa30a2592ab0ac5f7b44add9178b519ef67642033de1a8d","observation_id":"4fbf2bb5-3bfd-4746-9479-38aa8d899787","resolution":{"observed_at":"2026-08-01T18:04:11.814953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-08-01T08:37:44.391888Z","title":"Llava-onevision-2: Towards next-generation perceptual intelligence,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.21061","last_updated":"2026-07-23T08:52:11Z","snapshot_observed_at":"2026-08-08T01:13:21.444754Z","submitted_at":"2026-07-23T08:52:11Z","title":"MVEI & EmObserver: Empowering MLLM-Oriented Visual Emotional Intelligence via Emotion Statement Judgement","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-01T08:37:44.391888Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2607.21061"},"observation_digest":"sha256:18f350d81d4432249f0298ce1848e5ab0f0cb8e03c6d17bbd07c31b8eb34d45c","observation_id":"a1c55d83-6932-4859-8cb1-43d89de6ded3","resolution":{"observed_at":"2026-08-01T08:37:44.391888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-07-31T06:20:13.650085Z","title":"LLaVA-OneVision-2: Towards next-generation perceptual intelligence.arXiv preprint arXiv:2605.25979, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-10T20:15:55.135228Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:13.650085Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:2661267bb2ec98d5dc259a2c8447b28b242e245c8ced1940ffbef6bc6436d58d","observation_id":"3bde2716-bd37-44f1-a6b5-078edd0816e2","resolution":{"observed_at":"2026-07-31T06:20:13.650085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-07-31T05:08:19.737340Z","title":"LLaV A-OneVision-2: Towards next-generation perceptual intelligence.arXiv preprint arXiv:2605.25979, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28509","last_updated":"2026-07-30T16:48:28Z","snapshot_observed_at":"2026-08-09T21:08:35.257576Z","submitted_at":"2026-07-30T16:48:28Z","title":"RefCaptioner: Multi-Reference Image-Grounded Video Captioning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-31T05:08:19.737340Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2607.28509"},"observation_digest":"sha256:4eb8d5f7ee2b470feaccd7305a5eb30de8f13ac0632a8f66884e464d4e728c2a","observation_id":"9e5ef122-d5b2-4e1c-ac12-b90a3088b2d5","resolution":{"observed_at":"2026-07-31T05:08:19.737340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25979","snapshot_observed_at":"2026-08-06T17:22:31.939898Z","title":"arXiv preprint arXiv:2605.25979 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04759","last_updated":"2026-08-05T12:26:23Z","snapshot_observed_at":"2026-08-10T02:12:26.574076Z","submitted_at":"2026-08-05T12:26:23Z","title":"Trace, Verify, and Correct: A Training-Free Framework for Spatial Reasoning in Multimodal LLMs","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:31.939898Z"},"links":{"cited_paper":"/paper/2605.25979","citing_paper":"/paper/2608.04759"},"observation_digest":"sha256:7e6c4c4e4f1e43c13c21db0722ca36465d795393c13249bf6cc641e89cabcd38","observation_id":"e5460fb8-f207-4ee5-bc9f-a1a267c7fb62","resolution":{"observed_at":"2026-08-06T17:22:31.939898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.25979/citation-record","integrity":"/paper/2605.25979/integrity","json":"/paper/2605.25979/citation-record.json","paper":"/paper/2605.25979"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.12382","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:13:59.592044Z","title":"SPARROW: Learning spatial precision and temporal referential consistency in pixel-grounded video MLLMs.arXiv:2603.12382,","venue":null,"work_id":"b976825a-0a38-4aa3-b314-6d21b7b8849e","year":null},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:03708df0ee82dc751af063724fd192b889f6d6d786779d93dcd74146482ee05a","observation_id":"f1b74493-07db-4f49-a4fe-3cec07367994","resolution":{"observed_at":"2026-06-29T22:13:59.593890Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.23661","last_updated":"2025-12-14T14:18:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-28T05:52:55Z","title":"LLaVA-OneVision-1.5: Fully Open Framework for Democratized Multimodal Training","version":3},"cited_work":{"arxiv_id":"2509.23661","doi":"10.48550/arxiv.2509.23661","metadata_source":"pith","pith_arxiv_id":"2509.23661","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision-1.5: Fully Open Framework for Democratized Multimodal Training","venue":"cs.CV","work_id":"41c2802e-aff9-482f-b506-10955ff0838d","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2509.23661","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:5153491846e6e1e9544a5e0a81e9ef8416b4533bb210b1a326d0a1a3e37d48e4","observation_id":"581aa147-7652-47d4-b5e1-341d2bfcc9db","resolution":{"observed_at":"2026-06-29T22:13:59.610224Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:16.817018+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:16.817018+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:b55032c4fdfc6cf764301e82a7771d0457983e862a1b560e1a8ab6cab6e71351","observation_id":"c72ecbfe-6a27-46ce-95c5-c64057214303","resolution":{"observed_at":"2026-06-29T22:13:59.602205Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.13091","last_updated":"2026-04-17T08:06:30Z","snapshot_observed_at":"2026-07-06T22:48:57.474272Z","submitted_at":"2026-03-13T15:40:42Z","title":"Reasoning over Video: Evaluating How MLLMs Extract, Integrate, and Reconstruct Spatiotemporal Evidence","version":2},"cited_work":{"arxiv_id":"2603.13091","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.13091","snapshot_observed_at":"2026-06-29T22:13:59.532829Z","title":"Reasoning over Video: Evaluating How MLLMs Extract, Integrate, and Reconstruct Spatiotemporal Evidence","venue":"cs.CV","work_id":"b00b2dd2-2060-49a0-8bdb-2b1b6dde8d21","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2603.13091","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:e5dbce7449fbb7704fc1f02474f985c37790edb129572fc63ffbaf62aea9f69f","observation_id":"1cfba95e-4081-4c37-b912-7bb7ee65101e","resolution":{"observed_at":"2026-06-29T22:13:59.534505Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09461","last_updated":"2023-03-01T19:45:11Z","snapshot_observed_at":"2026-07-06T14:06:56.291161Z","submitted_at":"2022-10-17T22:23:40Z","title":"Token Merging: Your ViT But Faster","version":3},"cited_work":{"arxiv_id":"2210.09461","doi":"10.48550/arxiv.2210.09461","metadata_source":"pith","pith_arxiv_id":"2210.09461","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Token Merging: Your ViT But Faster","venue":"cs.CV","work_id":"528509bc-2611-4e7f-a772-ea14d25b6dae","year":2022},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2210.09461","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:451b35a59e57876919924cceb519ecaf725de623ddf6a461f3a1ddfd9963e25d","observation_id":"7ab2eb34-103c-4f07-8f40-33ff95e5f780","resolution":{"observed_at":"2026-06-29T22:13:59.539418Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-09T16:57:03.077389Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:d64321e0d89a18a82506f2d884c65aba9a8591472c39d5a8334a36d658a1287e","observation_id":"016b2ce0-c946-4469-add7-ec1b9b10e22c","resolution":{"observed_at":"2026-06-29T22:13:59.560592Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.18702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:39:58.315197Z","title":"Think with Grounding: Curriculum Reinforced Reasoning with Video Grounding for Long Video Understanding","venue":null,"work_id":"133ac2c4-4c1d-4fcd-9c09-00ce73bb6708","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:5f06dfb290589c6dc5efe3cb901ba955b419d236d80b459fa590644bd4a3c308","observation_id":"22ddb1ee-6978-4b5b-b3a2-2accac6591c3","resolution":{"observed_at":"2026-06-29T22:13:59.585208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.17555","last_updated":"2026-05-13T14:09:57Z","snapshot_observed_at":"2026-08-02T08:53:32.433698Z","submitted_at":"2026-02-19T17:09:30Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","version":3},"cited_work":{"arxiv_id":"2602.17555","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.17555","snapshot_observed_at":"2026-06-29T22:13:59.633393Z","title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","venue":"cs.CV","work_id":"34223ca6-ceac-45d4-98ad-4c6d94723a11","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2602.17555","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:147cdceb963b36cc144c2e2a0d2f674c6ab0a671468a97574b0b4f2a544e480b","observation_id":"ca9de3d6-bab6-4863-899b-95acdc4c3104","resolution":{"observed_at":"2026-06-29T22:13:59.634880Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.10611","last_updated":"2026-04-02T16:01:02Z","snapshot_observed_at":"2026-08-02T06:45:30.387180Z","submitted_at":"2026-01-15T17:27:44Z","title":"Molmo2: Open Weights and Data for Vision-Language Models with Video Understanding and Grounding","version":4},"cited_work":{"arxiv_id":"2601.10611","doi":"10.48550/arxiv.2601.10611","metadata_source":"pith","pith_arxiv_id":"2601.10611","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Molmo2: Open Weights and Data for Vision-Language Models with Video Understanding and Grounding","venue":"cs.CV","work_id":"f0dc2969-bd90-4a5d-81df-d557352690b6","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2601.10611","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:def9e41b18a0307cf5a4b0f3b329fd2cb31f8bec71d3267183c77fac5d0e188d","observation_id":"f0b9d4ca-36d0-4cd7-9035-4a3688a5c578","resolution":{"observed_at":"2026-06-29T22:13:59.651251Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2409.17146","doi":"10.48550/arxiv.2409.17146","metadata_source":"pith","pith_arxiv_id":"2409.17146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","venue":"cs.CV","work_id":"b6841af7-2b88-4597-9d0e-54e96aa7a368","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2409.17146","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:2a730d8690415e64d8da039192bbf5932519a7a2f1fae7f0d2927fcce049545a","observation_id":"09a2392a-73ee-43b9-86d8-e8694f9d8ab5","resolution":{"observed_at":"2026-06-29T22:13:59.623529Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.08120","last_updated":"2026-04-09T11:40:25Z","snapshot_observed_at":"2026-07-06T22:57:18.202215Z","submitted_at":"2026-04-09T11:40:25Z","title":"Small Vision-Language Models are Smart Compressors for Long Video Understanding","version":1},"cited_work":{"arxiv_id":"2604.08120","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.08120","snapshot_observed_at":"2026-07-04T16:39:58.308940Z","title":"Small Vision-Language Models are Smart Compressors for Long Video Understanding","venue":"cs.CV","work_id":"a3999756-0681-41aa-9034-8d79de207846","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2604.08120","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:26c3aa594665143c8e438a4d8ce44686e8e0c2e1c9ebcc929a439bfd4937a7cc","observation_id":"11f6c408-8efc-44f9-9362-086b0115b8c7","resolution":{"observed_at":"2026-06-29T22:13:59.629109Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":"2503.21776","doi":"10.48550/arxiv.2503.21776","metadata_source":"pith","pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","venue":"cs.CV","work_id":"0ce88332-564c-4361-8e2a-3850eb1ace9c","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:257c68df8860b0ca89327f5c9965c0944614115f7808e35f4dbbbd9c4f7b3215","observation_id":"ccd5db0d-2e2c-4f72-8418-558f3dae9183","resolution":{"observed_at":"2026-06-29T22:13:59.632058Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.05015","last_updated":"2026-04-06T17:59:56Z","snapshot_observed_at":"2026-07-06T22:53:51.418227Z","submitted_at":"2026-04-06T17:59:56Z","title":"Video-MME-v2: Towards the Next Stage in Benchmarks for Comprehensive Video Understanding","version":1},"cited_work":{"arxiv_id":"2604.05015","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.05015","snapshot_observed_at":"2026-07-04T03:49:30.277574Z","title":"Video-MME-v2: Towards the Next Stage in Benchmarks for Comprehensive Video Understanding","venue":"cs.CV","work_id":"79edc1c3-c8fc-42ee-b2b8-e2f477b88399","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2604.05015","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:162bfeb21c54b3178c3234c6832c4c7c016023feb20015cdfcfd837043332b32","observation_id":"170f529f-5c12-4941-b8a8-b291450df40b","resolution":{"observed_at":"2026-06-29T22:13:59.637794Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.12262","last_updated":"2026-07-17T07:23:56Z","snapshot_observed_at":"2026-08-09T09:42:44.258435Z","submitted_at":"2026-03-12T17:59:51Z","title":"Video Streaming Thinking: VideoLLMs Can Watch and Think Simultaneously","version":2},"cited_work":{"arxiv_id":"2603.12262","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2603.12262","snapshot_observed_at":"2026-07-20T02:18:27.127261Z","title":"Video streaming thinking: VideoLLMs can watch and think simultaneously.arXiv:2603.12262,","venue":null,"work_id":"c165da8e-9c8e-42dc-82c0-27b6305b027f","year":null},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2603.12262","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:f22c297bcfec238b5899b8b40adffc1156403b08b02d27eeb54417225673e38b","observation_id":"393a6d2c-8ec2-4bae-a1c3-04800cbabd82","resolution":{"observed_at":"2026-07-20T02:18:27.127261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05643","last_updated":"2025-03-03T10:28:30Z","snapshot_observed_at":"2026-08-07T05:23:25.250251Z","submitted_at":"2024-10-08T02:46:30Z","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","version":3},"cited_work":{"arxiv_id":"2410.05643","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05643","snapshot_observed_at":"2026-07-02T12:16:57.678370Z","title":"Trace: Temporal grounding video llm via causal event modeling","venue":null,"work_id":"b2af5095-2950-48b5-af5e-270218b8d041","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2410.05643","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:be631897c088ff072adfa0ddb53bdfc8014e3f80ab710716eb7933e5e81f333e","observation_id":"fe821ea5-a2a0-4274-9b99-441da40cc261","resolution":{"observed_at":"2026-06-29T22:13:59.619875Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.21186","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:49:46.480425Z","title":"Spa3R: Predictive spatial field modeling for 3D visual reasoning","venue":null,"work_id":"29c656c5-bb67-4737-98c5-729d283d4645","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:f2f0ce90e546f19e9129a30d14569c09bb622e72afe370f69c0008a044767676","observation_id":"4803f7e1-22e8-4c1a-bb14-5600f3539762","resolution":{"observed_at":"2026-06-29T22:13:59.613532Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.04130","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T05:57:41.574986Z","title":"Token-efficient long video understanding for multimodal llms","venue":null,"work_id":"6e91a1ce-2091-4791-aa88-f9b2ad2bdedb","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:69160cbd9430b1482fa8b27620f4936c704d2b625158c32f1842757ec112943a","observation_id":"e20e8960-f15a-4301-9536-38f5411cd05d","resolution":{"observed_at":"2026-06-29T22:13:59.617077Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.23489","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:13:59.625281Z","title":"Agentrvos: Reasoning over object tracks for zero-shot referring video object segmentation","venue":null,"work_id":"ce43c3a8-6a82-4965-bfba-c61237290d5e","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:a06bc19244f050064071202b0c2e1220497ace1572b1c71b8b83404c92b8a85c","observation_id":"6165672e-831e-4e3c-80a7-753e4d78d22c","resolution":{"observed_at":"2026-06-29T22:13:59.627023Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05993","last_updated":"2025-01-10T08:35:13Z","snapshot_observed_at":"2026-08-10T13:50:34.044232Z","submitted_at":"2024-10-08T12:44:57Z","title":"Aria: An Open Multimodal Native Mixture-of-Experts Model","version":4},"cited_work":{"arxiv_id":"2410.05993","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.05993","snapshot_observed_at":"2026-07-04T07:49:39.565893Z","title":"Aria: An open multimodal native mixture-of- experts model","venue":null,"work_id":"8ed8984e-244e-4c69-9fd5-c0cef9869673","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2410.05993","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:c3854e2f08aab50143361f57ba41eab7c6aa32fffdf1d5a65f7f5fc521d0f4d8","observation_id":"f31884c4-788a-4c9a-a17b-e330fee1ded4","resolution":{"observed_at":"2026-06-29T22:13:59.644704Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:378da1a367019eeb09d6a64fda4c8a5a06eea5d0ce19f9ab153b9d990725b1e9","observation_id":"ae5d3360-d2fa-4279-b130-01683db20b90","resolution":{"observed_at":"2026-06-29T22:13:59.604775Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"cited_work":{"arxiv_id":"2501.00574","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.00574","snapshot_observed_at":"2026-07-04T16:49:57.265026Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","venue":"cs.CV","work_id":"52ec7cb6-1ef6-4366-a43f-d4c13fce23ad","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2501.00574","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:59e7d2db3385192bd421394cb44b9503e3f446d3bd2cbd97624961a1ab874f88","observation_id":"fb4fd70d-585c-48b0-9d6e-22d13ada3ef0","resolution":{"observed_at":"2026-06-29T22:13:59.641616Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18478","last_updated":"2025-04-27T10:39:51Z","snapshot_observed_at":"2026-08-07T16:41:53.661756Z","submitted_at":"2025-03-24T09:21:48Z","title":"Video-XL-Pro: Reconstructive Token Compression for Extremely Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2503.18478","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.18478","snapshot_observed_at":"2026-07-03T09:47:59.499640Z","title":"Video-xl-pro: Reconstructive token compres- sion for extremely long video understanding","venue":null,"work_id":"e244516b-463d-478b-8a4c-afa1acbdc7ee","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2503.18478","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:fa2614fab745488abf6a86d53b2304277a367733c4b9eba553c433d99ec11c17","observation_id":"764b775d-4ec5-44ac-8fa6-1402452ebb2e","resolution":{"observed_at":"2026-06-29T22:13:59.625729Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12961","last_updated":"2025-02-27T06:09:46Z","snapshot_observed_at":"2026-07-06T19:18:17.523057Z","submitted_at":"2024-09-19T17:59:51Z","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","version":4},"cited_work":{"arxiv_id":"2409.12961","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.12961","snapshot_observed_at":"2026-07-04T13:09:50.847114Z","title":"Oryx mllm: On- demand spatial-temporal understanding at arbitrary resolution","venue":null,"work_id":"4a905ca2-7426-40db-a509-3452624d3ae5","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2409.12961","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:c0ff173e3b83bed56170778334882050034ff330ba587fb056aba3441c7ab981","observation_id":"9fc1f3ba-b6a6-45d6-87c6-c59c5d4c378d","resolution":{"observed_at":"2026-06-29T22:13:59.596700Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01805","last_updated":"2025-05-21T09:38:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-02T15:12:17Z","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","version":2},"cited_work":{"arxiv_id":"2504.01805","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.01805","snapshot_observed_at":"2026-07-05T11:41:02.483857Z","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","venue":"cs.CV","work_id":"df3c8f70-c0d0-469a-b74b-b32ceb50d29d","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2504.01805","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:155ccb178dcafbedd6627a68ea5af98c28d483eef33fd34c34e68a1ff670544f","observation_id":"fe30cdd5-5070-40a4-b718-f28b247069ae","resolution":{"observed_at":"2026-06-29T22:13:59.550815Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05336","last_updated":"2025-07-05T11:38:26Z","snapshot_observed_at":"2026-08-10T11:30:31.815084Z","submitted_at":"2025-06-05T17:59:29Z","title":"VideoMolmo: Spatio-Temporal Grounding Meets Pointing","version":2},"cited_work":{"arxiv_id":"2506.05336","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05336","snapshot_observed_at":"2026-07-03T15:48:34.853226Z","title":"Videomolmo: Spatio-temporal grounding meets pointing","venue":null,"work_id":"11238605-33ed-4866-a7b0-1df17e186797","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2506.05336","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:a95002f9cfaa51590cb0544ea4fc00216a3a51b70d94899d16a02c394e36c576","observation_id":"d36b3184-77a9-48be-9423-6b0cd57d2bdd","resolution":{"observed_at":"2026-06-29T22:13:59.563621Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-08-10T13:36:40.776714Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:faf3dd151d23f09292daabbd7e3e6384f05515353096ab31b810eae653a8dad7","observation_id":"62853dd5-b767-49ba-a9a9-4a13f37bcfab","resolution":{"observed_at":"2026-06-29T22:13:59.539308Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2410.17434","doi":"10.48550/arxiv.2410.17434","metadata_source":"pith","pith_arxiv_id":"2410.17434","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","venue":"cs.CV","work_id":"82f3cbed-22bb-41b5-b9c9-67304154cd52","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2410.17434","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:90134b22c61f87523b8bdca2f8febcaacb6ac72a3d470f8e0951c09dbc388596","observation_id":"3910d08d-800d-4c97-afac-42aeba8eb419","resolution":{"observed_at":"2026-06-29T22:13:59.621151Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2604.02317","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T00:57:30.607600Z","title":"A Simple Baseline for Streaming Video Understanding","venue":null,"work_id":"fdb77ee7-ee71-4919-a8c0-0de4924801f5","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:974bed6cabb8a886b7e8977bf94d2b704862bb0cfa303debfe9563bcad342589","observation_id":"51c68fed-389e-4425-9ecc-5cf38efdc3c0","resolution":{"observed_at":"2026-06-29T22:13:59.549967Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.19225","last_updated":"2025-06-24T01:19:56Z","snapshot_observed_at":"2026-08-07T21:07:46.308380Z","submitted_at":"2025-06-24T01:19:56Z","title":"Video-XL-2: Towards Very Long-Video Understanding Through Task-Aware KV Sparsification","version":1},"cited_work":{"arxiv_id":"2506.19225","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.19225","snapshot_observed_at":"2026-07-04T16:39:58.312479Z","title":"Video-XL-2: Towards Very Long-Video Understanding Through Task-Aware KV Sparsification.arXiv preprint arXiv:2506.19225, 2025","venue":null,"work_id":"93f911f1-9747-4144-9975-027df0ac6da7","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2506.19225","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:928b795c1e7cefde4952d9160a684ac75a2172ca9c0aafb106b0d0dfd05f8cb2","observation_id":"b87dd21e-68ef-4313-9a82-2ef1a7024db7","resolution":{"observed_at":"2026-06-29T22:13:59.566701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.17087","last_updated":"2026-04-18T17:52:02Z","snapshot_observed_at":"2026-07-06T23:04:14.688737Z","submitted_at":"2026-04-18T17:52:02Z","title":"EvoComp: Learning Visual Token Compression for Multimodal Large Language Models via Semantic-Guided Evolutionary Labeling","version":1},"cited_work":{"arxiv_id":"2604.17087","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.17087","snapshot_observed_at":"2026-06-29T22:13:59.597820Z","title":"EvoComp: Learning Visual Token Compression for Multimodal Large Language Models via Semantic-Guided Evolutionary Labeling","venue":"cs.CV","work_id":"1e70d3cb-f597-4dca-93db-c19042bbf524","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2604.17087","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:b0f594fec18f39e63dd3a09088b5eb752b6cc5c99c0c01f4789aacab508e0345","observation_id":"aedd76ea-039d-426a-83ce-833ec7a451ed","resolution":{"observed_at":"2026-06-29T22:13:59.599233Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.08683","doi":"10.48550/arxiv.2602.08683","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Onevision-encoder: Codec-aligned sparsity as a foundational principle for multimodal intelligence","venue":"Open MIND","work_id":"c56c5c97-488b-4708-b8c8-07fabfb29f41","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:673867eab7ca006d1e5820019cf2ad2a497221fc432e04f40d65b689d24790f1","observation_id":"e28f33d9-354b-49a1-b24f-34aeae931da4","resolution":{"observed_at":"2026-06-29T22:13:59.524140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14786","last_updated":"2025-02-20T18:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-20T18:08:29Z","title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","version":1},"cited_work":{"arxiv_id":"2502.14786","doi":"10.48550/arxiv.2502.14786","metadata_source":"pith","pith_arxiv_id":"2502.14786","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","venue":"cs.CV","work_id":"50eec732-2d41-432f-9dcf-ac7fff235ea5","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2502.14786","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:8482cf44101d8dd1bc38c3e789e15994e1c90e83d64aa660bb02d1c8b456897d","observation_id":"7ecee9cc-9e28-48e1-bfca-af410ad6f48d","resolution":{"observed_at":"2026-06-29T22:13:59.634667Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-07-10T23:49:08.777694+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T23:49:08.777694+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03290","last_updated":"2025-08-21T05:15:19Z","snapshot_observed_at":"2026-08-02T03:09:09.488305Z","submitted_at":"2024-10-04T10:04:37Z","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2410.03290","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03290","snapshot_observed_at":"2026-07-04T15:09:55.249321Z","title":"Grounded-videollm: Sharpening fine-grained temporal grounding in video large language models","venue":null,"work_id":"6cda7c97-9a94-4872-b214-cddce67fea61","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2410.03290","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:699ad428e97ffef01d9e207972b748f638d06629ecb28d166f82b39964ec8a2d","observation_id":"e8e66c53-1aa6-462f-8eae-33e353ef4e39","resolution":{"observed_at":"2026-06-29T22:13:59.520753Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"cited_work":{"arxiv_id":"2508.18265","doi":"10.48550/arxiv.2508.18265","metadata_source":"pith","pith_arxiv_id":"2508.18265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","venue":"cs.CV","work_id":"b8f5e260-fff5-444e-bcf5-2c42cfefd83d","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2508.18265","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:8a7a6cbffde43406fddd78687caf50260bf74b769822b0636aff9039367af5df","observation_id":"1289afdf-8daa-48c5-81c9-110436f9cd37","resolution":{"observed_at":"2026-06-29T22:13:59.631862Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T17:38:12.771444+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T17:38:12.771444+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-10T17:38:12.771444+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01328","last_updated":"2025-04-02T03:24:58Z","snapshot_observed_at":"2026-08-07T16:15:53.305377Z","submitted_at":"2025-04-02T03:24:58Z","title":"Slow-Fast Architecture for Video Multi-Modal Large Language Models","version":1},"cited_work":{"arxiv_id":"2504.01328","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.01328","snapshot_observed_at":"2026-06-29T22:13:59.574945Z","title":"Slow-fast architecture for video multi-modal large language models","venue":null,"work_id":"6885223e-2904-428e-ae58-0c36e078cff9","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2504.01328","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:2a6ea14cba9fd00c3877275cb7e0defe828d9026c80ca56f524a4aec6abd8baf","observation_id":"9502d1c9-2655-4300-9f10-c4a927bcd63f","resolution":{"observed_at":"2026-06-29T22:13:59.576656Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.16557","last_updated":"2026-07-30T13:33:20Z","snapshot_observed_at":"2026-08-09T11:29:22.831265Z","submitted_at":"2026-04-17T08:39:07Z","title":"S-GRPO: Unified Post-Training for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2604.16557","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16557","snapshot_observed_at":"2026-06-29T22:13:59.571736Z","title":"S-GRPO: Unified Post-Training for Large Vision-Language Models","venue":"cs.LG","work_id":"516a8a11-d3bb-4de3-861b-eabb966cd40d","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2604.16557","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:969b10a56c51c285a32b600ad00d424fd225749e628aca1d208e4c347271cb60","observation_id":"d5465249-d1cd-46c3-bc73-74d1dd4d8379","resolution":{"observed_at":"2026-06-29T22:13:59.573692Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.01563","last_updated":"2025-09-07T13:40:44Z","snapshot_observed_at":"2026-08-08T11:15:53.631141Z","submitted_at":"2025-09-01T15:46:58Z","title":"Kwai Keye-VL 1.5 Technical Report","version":3},"cited_work":{"arxiv_id":"2509.01563","doi":"10.48550/arxiv.2509.01563","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.01563","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kwai keye-vl 1.5 technical report","venue":"arXiv (Cornell University)","work_id":"87836623-1e5f-4d18-babd-0613eddc4b6e","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2509.01563","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:4d22da4646162a535a9a62831407773795252d864c1b36100acf381505a2fab2","observation_id":"5b48ea6b-4a0b-444a-a5ff-4f6ef7bdb83c","resolution":{"observed_at":"2026-06-29T22:13:59.514435Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23764","last_updated":"2026-05-23T03:42:56Z","snapshot_observed_at":"2026-08-07T12:36:19.326245Z","submitted_at":"2025-05-29T17:59:52Z","title":"MMSI-Bench: A Benchmark for Multi-Image Spatial Intelligence","version":3},"cited_work":{"arxiv_id":"2505.23764","doi":"10.48550/arxiv.2505.23764","metadata_source":"pith","pith_arxiv_id":"2505.23764","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmsi-bench: A benchmark for multi-image spatial intelligence","venue":"cs.CV","work_id":"3c86b01a-2b7e-4a70-94c5-92bf4d05ad52","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2505.23764","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:08acfdb0fba4db889bfd5a2ef6dbffaf4afb85dc228dab667b610dadd885aa17","observation_id":"d07ce9e9-344a-4537-b425-749d832bfd2c","resolution":{"observed_at":"2026-06-29T22:13:59.579379Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-08T01:58:42.644918Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":"2501.04001","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-07-10T03:06:43.598138Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","venue":"cs.CV","work_id":"fbffda10-106c-432c-b84e-5d0908666921","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:68ddbff0865bee6c8124dbee1f12394d9f0239a9f6f1d7b23bde928ff74e0fa4","observation_id":"d4292cc4-d4b9-41b1-b0f5-c9f077a36c83","resolution":{"observed_at":"2026-06-29T22:13:59.587708Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07888","last_updated":"2025-01-24T05:16:36Z","snapshot_observed_at":"2026-08-10T20:27:34.844428Z","submitted_at":"2025-01-14T06:54:39Z","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","version":3},"cited_work":{"arxiv_id":"2501.07888","doi":"10.48550/arxiv.2501.07888","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07888","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tarsier2: Advancing large vision-language models from detailed video description to comprehensive video understanding","venue":"ArXiv.org","work_id":"77605c65-9b44-4521-ad8c-79d9c78cf33a","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2501.07888","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:ccde455da6054965a5227c91ef37a0b90d66ace7fb20fd7c3965c26421cbe51d","observation_id":"e825296b-f6c7-4712-8e04-fc4d80848624","resolution":{"observed_at":"2026-06-29T22:13:59.509917Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":"2501.13106","doi":"10.48550/arxiv.2501.13106","metadata_source":"pith","pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","venue":"cs.CV","work_id":"38f52461-37fd-4266-bc46-9dea31be2824","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:edc5114f3e21788d7c01d35afdd05bfb975703d378356fc9948a39fc47b84ce2","observation_id":"f09a84f8-cd03-4435-b572-40820617678a","resolution":{"observed_at":"2026-06-29T22:13:59.517193Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24300","last_updated":"2026-05-05T23:43:25Z","snapshot_observed_at":"2026-08-01T02:51:18.898730Z","submitted_at":"2026-04-27T10:45:51Z","title":"ReVSI: Rebuilding Visual Spatial Intelligence Evaluation for Accurate Assessment of VLM 3D Reasoning","version":2},"cited_work":{"arxiv_id":"2604.24300","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.24300","snapshot_observed_at":"2026-07-04T03:29:31.551884Z","title":"ReVSI: Rebuilding Visual Spatial Intelligence Evaluation for Accurate Assessment of VLM 3D Reasoning","venue":"cs.CV","work_id":"0cc75688-7e39-46df-89c4-71f5fa9522a7","year":2026},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2604.24300","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:07fcaa5971f2186e0ac012acee87ec68e35bf1f410269edb6983cdca06638102","observation_id":"9299af09-cb1b-48c1-9478-4ee03cd0ebf6","resolution":{"observed_at":"2026-06-29T22:13:59.618491Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.04308","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T19:30:08.026359Z","title":"RoboRefer: Towards spatial referring with rea- soning in vision-language models for robotics","venue":null,"work_id":"fda95042-555b-4a64-9d35-7978eb50b269","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:75a730d0ec62bf6aa3d619285854f3a92d9b476473d7e0933e8e8f9ff560d275","observation_id":"c8914bf5-4cef-41c4-9da6-e2f06246096e","resolution":{"observed_at":"2026-06-29T22:13:59.570597Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-10T18:37:57.419939Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":"2504.10479","doi":"10.48550/arxiv.2504.10479","metadata_source":"pith","pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","venue":"cs.CV","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","year":2025},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:d059b30fdfad54c45174f67399438de50ecf29564f11d189e48024d7e216462c","observation_id":"a8fe1062-99fb-4d1e-8d07-2ddd18337f18","resolution":{"observed_at":"2026-06-29T22:13:59.553695Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-06T08:59:28.938249Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"cited_work":{"arxiv_id":"2412.10360","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.10360","snapshot_observed_at":"2026-07-03T10:48:02.918791Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","venue":null,"work_id":"a2514001-657a-44cc-8c33-0562df7ac008","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2412.10360","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:91477aa8c21dcbbf9c40639ad21016ba25589fd7145adddfda404866b6597a11","observation_id":"0a37124e-a73b-4485-b966-0c8d1371122d","resolution":{"observed_at":"2026-06-29T22:13:59.596352Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":0,"verified_exact":42,"verified_fuzzy":0},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 9 inbound Pith citation observations for arXiv:2605.25979."}