{"as_of":"2026-08-08T06:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8fbd2e6272d81d94a19ca1ef936fe002155c54891a76e9f75f317b61a6d6d8e8","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T20:06:36.667965Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T10:48:03.080928Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":"2406.08394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-07-03T10:48:03.080928Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model","venue":null,"work_id":"9f5a686c-3950-4ee0-b961-85f882649e32","year":2024},"citing_paper":{"arxiv_id":"2410.23262","last_updated":"2025-09-23T04:19:59Z","snapshot_observed_at":"2026-08-05T06:31:46.069300Z","submitted_at":"2024-10-30T17:46:31Z","title":"EMMA: End-to-End Multimodal Model for Autonomous Driving","version":3},"reference_index":196,"source":"arxiv_source","source_observed_at":"2026-05-15T05:08:54.368109Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2410.23262"},"observation_digest":"sha256:a8ea81ce4b7c50dd08d4406a7af25e829b38d6dda3695495f7adadece8b13ae4","observation_id":"c2d4b389-214a-44b0-8bcd-325deeaa4856","resolution":{"observed_at":"2026-05-15T05:08:54.481513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":"2406.08394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-07-03T10:48:03.080928Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model","venue":null,"work_id":"9f5a686c-3950-4ee0-b961-85f882649e32","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:e324f83d452a9a3164b22d303ba58e6e247e8dfb58e5f048d8653cbd1166cce1","observation_id":"d5166116-14a6-4c38-b05d-f5764bbb5c74","resolution":{"observed_at":"2026-05-17T02:52:20.796481Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-07T20:06:36.667965Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.09925","last_updated":"2025-02-14T05:32:46Z","snapshot_observed_at":"2026-08-07T19:59:35.811665Z","submitted_at":"2025-02-14T05:32:46Z","title":"TaskGalaxy: Scaling Multi-modal Instruction Fine-tuning with Tens of Thousands Vision Task Types","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T20:06:36.667965Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2502.09925"},"observation_digest":"sha256:a9881e1b4db91b20cd84037bcce7c2ca745d50f9ee8f2cd7aad96ce29358849e","observation_id":"65a650b0-96d6-417c-aeb3-4f6ac7e412d7","resolution":{"observed_at":"2026-08-07T20:06:36.667965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-07T05:34:56.427639Z","title":"Visionllm v2: An end- to-end generalist multimodal large language model for hundreds of vision-language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07643","last_updated":"2025-06-09T11:09:10Z","snapshot_observed_at":"2026-08-07T05:26:32.056348Z","submitted_at":"2025-06-09T11:09:10Z","title":"Synthetic Visual Genome","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:56.427639Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2506.07643"},"observation_digest":"sha256:6b95c82969f6ad72a35e2798c18813532b5f271a0b6b2dd57282bd879d01d2dc","observation_id":"eff2ee90-61f0-4ad2-88af-dc5d121c97ec","resolution":{"observed_at":"2026-08-07T05:34:56.427639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-07T04:44:03.000950Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09954","last_updated":"2025-06-11T17:23:41Z","snapshot_observed_at":"2026-08-07T21:45:10.885875Z","submitted_at":"2025-06-11T17:23:41Z","title":"Vision Generalist Model: A Survey","version":1},"reference_index":182,"source":"arxiv_source","source_observed_at":"2026-08-07T04:44:03.000950Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2506.09954"},"observation_digest":"sha256:8b84b82f1d8a70b606e5ea6bec499478b00e637fc4e699e89de5dad9bad112bd","observation_id":"395da69c-2c56-49e8-bf51-6110e7352f7f","resolution":{"observed_at":"2026-08-07T04:44:03.000950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T21:52:23.743835Z","title":"Visionllm v2: An end-to-end general- ist multimodal large language model for hundreds of vision- language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.743835Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:14bdb29bce716635435b898ae95e67dc885198874082a6991106175a7cd3a129","observation_id":"bf746aaa-b2a3-428e-94c4-d7883a7ee9c8","resolution":{"observed_at":"2026-08-06T21:52:23.743835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T19:25:02.926461Z","title":"Visionllm v2: An end-to-end general- ist multimodal large language model for hundreds of vision- language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06272","last_updated":"2025-08-09T05:40:33Z","snapshot_observed_at":"2026-08-07T12:25:32.445608Z","submitted_at":"2025-07-08T07:46:26Z","title":"LIRA: Inferring Segmentation in Large Multi-modal Models with Local Interleaved Region Assistance","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:02.926461Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2507.06272"},"observation_digest":"sha256:de19452e48951ced0105d38d7937a9b05a7c1dc5fe3ae67fc040065f0ba0fe93","observation_id":"a8036437-e011-48ec-9078-abfa965eb13e","resolution":{"observed_at":"2026-08-06T19:25:02.926461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T17:22:13.366819Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.366819Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9ca980b5f7028580262834c38723bfb101a828ee17b9b3170d4a06848bcda32b","observation_id":"d43a7054-ee52-4ca0-8c0d-5cc729736004","resolution":{"observed_at":"2026-08-06T17:22:13.366819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T16:43:54.904999Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12883","last_updated":"2025-08-13T05:27:53Z","snapshot_observed_at":"2026-08-06T16:33:30.327495Z","submitted_at":"2025-07-17T08:09:31Z","title":"HRSeg: High-Resolution Visual Perception and Enhancement for Reasoning Segmentation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T16:43:54.904999Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2507.12883"},"observation_digest":"sha256:4f99148275afa02df55c07f5d765af96b76511700e6acd246a47a6f766087150","observation_id":"d3964afb-e158-4a4a-aa01-3eb146b337d8","resolution":{"observed_at":"2026-08-06T16:43:54.904999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":"2406.08394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-07-03T10:48:03.080928Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model","venue":null,"work_id":"9f5a686c-3950-4ee0-b961-85f882649e32","year":2024},"citing_paper":{"arxiv_id":"2604.10527","last_updated":"2026-04-12T08:43:28Z","snapshot_observed_at":"2026-07-06T22:59:09.778842Z","submitted_at":"2026-04-12T08:43:28Z","title":"STORM: End-to-End Referring Multi-Object Tracking in Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T16:25:31.777907Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2604.10527"},"observation_digest":"sha256:c061974648520148c09ecc668735b2a3303f2e4c69ddee086c1b2de711ce4bbf","observation_id":"3de9bd74-2cfd-4ca6-9297-c43e1d710c2e","resolution":{"observed_at":"2026-05-11T08:56:00.514788Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":"2406.08394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-07-03T10:48:03.080928Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model","venue":null,"work_id":"9f5a686c-3950-4ee0-b961-85f882649e32","year":2024},"citing_paper":{"arxiv_id":"2605.15997","last_updated":"2026-05-15T14:27:07Z","snapshot_observed_at":"2026-07-06T23:27:11.118592Z","submitted_at":"2026-05-15T14:27:07Z","title":"Segmentation, Detection and Explanation: A Unified Framework for CT Appearance Reasoning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-20T18:20:38.720544Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2605.15997"},"observation_digest":"sha256:af829d80ab31710ca00219ac65c5609b5699eb9553c12c1075983f8025b55153","observation_id":"8eb46f7e-5f49-4aef-9311-4c5ae542773f","resolution":{"observed_at":"2026-05-20T18:23:37.692557Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":"2406.08394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-07-03T10:48:03.080928Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model","venue":null,"work_id":"9f5a686c-3950-4ee0-b961-85f882649e32","year":2024},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:75f704b85b9f9ee8500bd459907114e4be01f09e62f1791c53bba4626d72cdea","observation_id":"d5b68e07-a65a-41c0-aa12-e8d670aae0ec","resolution":{"observed_at":"2026-07-03T10:48:03.082361Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2406.08394/citation-record","integrity":"/paper/2406.08394/integrity","json":"/paper/2406.08394/citation-record.json","paper":"/paper/2406.08394"},"outbound":[],"paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2406.08394."}