{"as_of":"2026-08-11T07:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ea5b57250f36db5102793d2533030fdd285b61b5b25a2c5518766ebdcc769e49","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":16,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":16,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":16,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T00:47:12.501355Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T15:09:55.195600Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"reference_index":116,"source":"pdf_text","source_observed_at":"2026-05-10T21:07:31.387726Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2408.01800"},"observation_digest":"sha256:ae047bf02b7da5075a58e97cbff580bf4bf95eeb5a6c490be53465de3b7d41cc","observation_id":"905fa20d-3f0a-40a2-a892-68407df11eb5","resolution":{"observed_at":"2026-05-10T21:07:32.136962Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-11T00:47:12.501355Z","title":"Next-chat: An lmm for chat, detec- tion and segmentation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.19326","last_updated":"2025-06-30T13:15:13Z","snapshot_observed_at":"2026-08-11T00:39:56.306346Z","submitted_at":"2024-12-26T18:56:05Z","title":"Task Preference Optimization: Improving Multimodal Large Language Models with Vision Task Alignment","version":2},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-11T00:47:12.501355Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2412.19326"},"observation_digest":"sha256:ab5f7e886d2b5dc8a3c1e871e921e60b929dcdb37baa43264c499893cffeb213","observation_id":"b66f6c78-d26d-44b3-95be-f2a4daf22cb6","resolution":{"observed_at":"2026-08-11T00:47:12.501355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-10T23:40:45.252649Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.20070","last_updated":"2025-05-31T11:45:12Z","snapshot_observed_at":"2026-08-10T23:33:09.116776Z","submitted_at":"2024-12-28T07:50:00Z","title":"Exploring Compositional Generalization of Multimodal LLMs for Medical Imaging","version":2},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-08-10T23:40:45.252649Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2412.20070"},"observation_digest":"sha256:f4600472905687981368f4c93c470279c1d271fab8953021d0f0edc65f295f51","observation_id":"b6a900c4-f51b-41f0-b3c0-e35bfa0503ab","resolution":{"observed_at":"2026-08-10T23:40:45.252649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-10T22:08:09.252943Z","title":"Next- chat: An lmm for chat, detection and segmentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-11T04:56:13.579583Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-10T22:08:09.252943Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2501.02765"},"observation_digest":"sha256:15882d2afba8af9d9f5749297ba94f361b41f3c3f37aa8dfa7236eb461883d8b","observation_id":"36949f8a-0830-4cdb-9115-913f2a23aded","resolution":{"observed_at":"2026-08-10T22:08:09.252943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-10T20:16:33.505694Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.08982","last_updated":"2026-07-20T09:21:17Z","snapshot_observed_at":"2026-08-10T21:57:37.675916Z","submitted_at":"2025-01-15T17:59:32Z","title":"CityLoc: 6DoF Pose Distributional Localization for Text Descriptions in Large-Scale Scenes with Gaussian Representation","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:16:33.505694Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2501.08982"},"observation_digest":"sha256:827b2d663a7c24bcd5a3d43bc6fe8d87ba967438c9e31e50d15b927ba2ccde6e","observation_id":"3470eee5-4f32-451f-96d3-38bd064a9e36","resolution":{"observed_at":"2026-08-10T20:16:33.505694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:c5d052c7c7e762257e891a73cff9b3c52dc0d3b0d076d806f3e0893ad99aea29","observation_id":"64f6bba2-b9e5-4be4-a078-cbdb17883279","resolution":{"observed_at":"2026-05-17T02:52:20.818808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-06T16:43:56.096102Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12883","last_updated":"2025-08-13T05:27:53Z","snapshot_observed_at":"2026-08-09T12:18:34.812978Z","submitted_at":"2025-07-17T08:09:31Z","title":"HRSeg: High-Resolution Visual Perception and Enhancement for Reasoning Segmentation","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T16:43:56.096102Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2507.12883"},"observation_digest":"sha256:67acb8c18830bf38d30c03fbecac26edac2a34403567d9048021362c7693b55d","observation_id":"8df77128-ee58-44d8-88f3-598d45721f14","resolution":{"observed_at":"2026-08-06T16:43:56.096102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-06T15:16:25.400535Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.16877","last_updated":"2025-07-22T11:23:48Z","snapshot_observed_at":"2026-08-09T21:44:11.890499Z","submitted_at":"2025-07-22T11:23:48Z","title":"ReMeREC: Relation-aware and Multi-entity Referring Expression Comprehension","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T15:16:25.400535Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2507.16877"},"observation_digest":"sha256:b1ff29af279748cdb9ffe311f4a10b058f9bce4cb21ce15af5d129d11d8fe745","observation_id":"ee2978c9-438b-4601-bb5e-388ffae6413d","resolution":{"observed_at":"2026-08-06T15:16:25.400535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-03T22:08:38.655224Z","title":"Next-chat: An lmm for chat, detection and segmenta- tion.arXiv preprint arXiv:2311.04498, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.12110","last_updated":"2026-07-01T13:10:45Z","snapshot_observed_at":"2026-08-10T22:29:50.104210Z","submitted_at":"2025-11-15T08:59:21Z","title":"MediRound: Multi-Round Entity-Level Reasoning Segmentation in Medical Images","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-03T22:08:38.655224Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2511.12110"},"observation_digest":"sha256:e7f95dcf7942d90e19ecfa12f9d86003b734e658073ee59005f2d056dca441af","observation_id":"0b5d19e3-4542-4384-b022-8553bc7d6b45","resolution":{"observed_at":"2026-08-03T22:08:38.655224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2512.10554","last_updated":"2026-04-02T03:14:28Z","snapshot_observed_at":"2026-07-30T11:23:08.233438Z","submitted_at":"2025-12-11T11:38:50Z","title":"Grounding Everything in Tokens for Multimodal Large Language Models","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-16T23:31:05.422935Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2512.10554"},"observation_digest":"sha256:ed924ac73c6ff2d8dabb88617e1ade17f83a43bf05a16e29844cd8be086ba042","observation_id":"a988e8a7-e882-4a9d-969d-d4f191512d13","resolution":{"observed_at":"2026-05-16T23:31:21.914672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2604.07765","last_updated":"2026-04-12T05:49:10Z","snapshot_observed_at":"2026-08-11T00:45:28.817631Z","submitted_at":"2026-04-09T03:40:46Z","title":"RemoteAgent: Bridging Vague Human Intents and Earth Observation with RL-based Agentic MLLMs","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-10T18:00:20.216268Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2604.07765"},"observation_digest":"sha256:f0435530bb7460677519588d03b5ec5c3042e68b2c9b73fcc76c45695acdd14d","observation_id":"00f6c03f-552e-4c09-9edf-3245116f0303","resolution":{"observed_at":"2026-05-11T05:40:59.609219Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2604.11789","last_updated":"2026-04-20T14:38:53Z","snapshot_observed_at":"2026-08-09T05:10:13.009841Z","submitted_at":"2026-04-13T17:55:02Z","title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","version":2},"reference_index":223,"source":"pdf_text","source_observed_at":"2026-05-10T15:35:37.095627Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2604.11789"},"observation_digest":"sha256:daca005ede6fc8a5ac44831687c600985dc2de616555c69b40bb9d7ed2a01670","observation_id":"1adf90e1-9e2c-4522-a5a5-ff1336077c44","resolution":{"observed_at":"2026-05-11T10:11:08.095451Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2604.18562","last_updated":"2026-04-22T02:31:50Z","snapshot_observed_at":"2026-08-04T10:02:00.731719Z","submitted_at":"2026-04-20T17:49:22Z","title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","version":3},"reference_index":196,"source":"arxiv_source","source_observed_at":"2026-05-10T05:10:44.608959Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2604.18562"},"observation_digest":"sha256:c5d9b469f5ad1fbbc59728adb3d2966402a1ea4a800c956a17d65a2aa0d6196f","observation_id":"7389e0fe-44dd-4cf4-962e-228af96f60a8","resolution":{"observed_at":"2026-05-10T09:43:49.462541Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:0fd66db87d2e2581272e06aaf9b2900f0467c0af0a8768107d0a8ddc0bcd1843","observation_id":"94e2fd50-8d09-4487-a8e6-b171afce61fb","resolution":{"observed_at":"2026-07-03T10:48:03.076990Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2606.26196","last_updated":"2026-06-24T15:20:32Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T15:20:32Z","title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-06-26T01:50:54.242508Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2606.26196"},"observation_digest":"sha256:1560f29c6b11d0898d21ff6bfd635f25be92008fe086ecdb9303110e893f05f9","observation_id":"f3c278d6-bd05-46b6-9d72-b4fdd40d153a","resolution":{"observed_at":"2026-07-04T15:09:55.197378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-08-01T10:20:57.384780Z","title":"arXiv preprint arXiv:2311.04498 , primaryclass =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.20284","last_updated":"2026-07-22T15:30:08Z","snapshot_observed_at":"2026-08-09T18:33:19.700255Z","submitted_at":"2026-07-22T15:30:08Z","title":"Multimodal Large Language Models for Remote Sensing Image Understanding: Domain-Specific or General-Purpose?","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-01T10:20:57.384780Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2607.20284"},"observation_digest":"sha256:431f61cc4aa35537d2b211ca3b83a5e8281d0a72527c987df03064990f37d34c","observation_id":"210014b3-c990-4050-8b71-41f38297f8e5","resolution":{"observed_at":"2026-08-01T10:20:57.384780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.04498/citation-record","integrity":"/paper/2311.04498/integrity","json":"/paper/2311.04498/citation-record.json","paper":"/paper/2311.04498"},"outbound":[],"paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 16 inbound Pith citation observations for arXiv:2311.04498."}