{"as_of":"2026-08-07T10:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:749918c51f99d896551f6f8f819568d7256f3392f5312d231d59a51a688fcf74","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:43:51.207678Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:08:55.658612Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:0f59ba6f9d9c2278ad7d5df1be1f318a813d63e8e89512f97a2d09084fbe3ebc","observation_id":"dd337dbf-d56d-4135-a28d-da0da5248840","resolution":{"observed_at":"2026-05-13T23:30:00.598826Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:f0cb1e9e500d466e3df9dd06495aa31e83bf20174d7e73f1589012a87fda39dc","observation_id":"feddb060-ac08-41fd-85b3-b92b79efb313","resolution":{"observed_at":"2026-05-15T06:30:22.516674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2309.07864","last_updated":"2023-09-19T08:29:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-14T17:12:03Z","title":"The Rise and Potential of Large Language Model Based Agents: A Survey","version":3},"reference_index":299,"source":"pdf_text","source_observed_at":"2026-05-11T10:47:44.152066Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2309.07864"},"observation_digest":"sha256:c9c200c06580e179ff9653fe936e7480440f5a785411e25d66695db6939f6ec4","observation_id":"b34065cb-728f-40bc-9c7d-ea7dd6ff454b","resolution":{"observed_at":"2026-05-11T10:47:54.533970Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-13T22:46:09.693156Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2312.14238"},"observation_digest":"sha256:55c89f3f65fdd3a269bc80a93a046761fb1851554fe6adb27ad6f138e8226adf","observation_id":"ee97f41b-1822-4edc-85b2-da5fcc40a46f","resolution":{"observed_at":"2026-05-13T22:46:10.030518Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2401.02458","last_updated":"2026-04-29T07:43:10Z","snapshot_observed_at":"2026-07-29T23:57:44.096283Z","submitted_at":"2024-01-04T08:00:32Z","title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","version":3},"reference_index":181,"source":"pdf_text","source_observed_at":"2026-05-24T04:13:05.328492Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2401.02458"},"observation_digest":"sha256:6e5fe3507f00a759d55d4dce8f958026b1e65f15c750d71c1c172f5122a52ff1","observation_id":"91f3c253-72ec-41f9-855e-271bfc0e9c18","resolution":{"observed_at":"2026-05-24T04:13:52.961508Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-08-06T02:31:58.372974Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T02:33:30.143907Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2401.15947"},"observation_digest":"sha256:b5324cb0b7e7b6d85145a84fd4655fc92928fde396348acc2212b57a9fdb9635","observation_id":"7336206c-68d3-48c4-b7a7-934d169d5c05","resolution":{"observed_at":"2026-05-16T02:33:30.353423Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:4f6cb496477c680ce21d33b6e439c0f77f6615f9e95189202a5f2667b276574b","observation_id":"7d452ab8-3858-4dea-a383-87f323cf3352","resolution":{"observed_at":"2026-05-12T20:58:59.117021Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2411.10442","last_updated":"2025-04-07T09:09:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-15T18:59:27Z","title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-16T09:16:17.150383Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2411.10442"},"observation_digest":"sha256:24d35e814d8430e29f9c196ab422184509c7408e6efbd19e9e652522fbfde944","observation_id":"6fd60986-4138-4328-9308-0a4cbd49ab73","resolution":{"observed_at":"2026-05-16T09:16:17.302720Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-07T04:43:51.207678Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond language","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09954","last_updated":"2025-06-11T17:23:41Z","snapshot_observed_at":"2026-08-07T09:15:25.343290Z","submitted_at":"2025-06-11T17:23:41Z","title":"Vision Generalist Model: A Survey","version":1},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-08-07T04:43:51.207678Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2506.09954"},"observation_digest":"sha256:5c8662e577099dbfdef9a8a37ee6a9b71f39b895b6252c3b2c2eee754b1a63b8","observation_id":"d21d2863-bf31-45e2-87fa-a91a5f5dbc6a","resolution":{"observed_at":"2026-08-07T04:43:51.207678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-06T19:25:02.773169Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond language","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06272","last_updated":"2025-08-09T05:40:33Z","snapshot_observed_at":"2026-08-07T08:24:37.121819Z","submitted_at":"2025-07-08T07:46:26Z","title":"LIRA: Inferring Segmentation in Large Multi-modal Models with Local Interleaved Region Assistance","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:02.773169Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2507.06272"},"observation_digest":"sha256:cb512b4163f48212fe80b397a54b311effd0a02cf0ffca23cddd4c5f46d90422","observation_id":"4b577851-b868-4a87-a58d-e891afe63e52","resolution":{"observed_at":"2026-08-06T19:25:02.773169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-06T15:57:02.802194Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-06T15:47:53.180347Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:02.802194Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:0efba1d6054f54d5cb9c2a5f6b0fb40eb36bbf95e2bd416410266d56c1040b40","observation_id":"3c10a5ff-4100-4602-8d76-33c8e97745b3","resolution":{"observed_at":"2026-08-06T15:57:02.802194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-06T00:59:52.580855Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04088","last_updated":"2025-08-07T03:52:48Z","snapshot_observed_at":"2026-08-06T11:50:51.532315Z","submitted_at":"2025-08-06T05:10:29Z","title":"GM-PRM: A Generative Multimodal Process Reward Model for Multimodal Mathematical Reasoning","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T00:59:52.580855Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2508.04088"},"observation_digest":"sha256:8942bf10dc9db24af3e4c8963bcd0ce4eb7482342d94b4f8b9be347ad91200b5","observation_id":"2e4ab4fc-1b0e-4b94-b490-7cca333b0fa0","resolution":{"observed_at":"2026-08-06T00:59:52.580855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2604.03231","last_updated":"2026-04-03T17:59:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T17:59:51Z","title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T20:28:30.864143Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2604.03231"},"observation_digest":"sha256:49185434985e4307cbe748544662aae10ce94c5b454b104e0b2506c40c1710f1","observation_id":"2f6371ee-f53a-4db8-9243-894cc83ee150","resolution":{"observed_at":"2026-05-13T20:33:17.105582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2604.15670","last_updated":"2026-07-24T04:48:23Z","snapshot_observed_at":"2026-08-02T16:09:56.976755Z","submitted_at":"2026-04-17T03:48:56Z","title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T09:20:54.635375Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2604.15670"},"observation_digest":"sha256:94c18973af2cac9789ccd6299664189cabb060d3ae1238f1fe13f9b132c554b3","observation_id":"64140c31-6391-473c-8d4b-e3a21d9a9c6d","resolution":{"observed_at":"2026-05-10T09:23:37.122437Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-02T16:10:01.893221Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond language.arXiv preprint arXiv:2305.05662, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.15670","last_updated":"2026-07-24T04:48:23Z","snapshot_observed_at":"2026-08-02T16:09:56.976755Z","submitted_at":"2026-04-17T03:48:56Z","title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T16:10:01.893221Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2604.15670"},"observation_digest":"sha256:e652cc1b117c2c09566799779edb971829ea2397c64e3c8ad5dda848df97fd58","observation_id":"6cac5e5d-667c-4c06-a6b4-08b10b62f271","resolution":{"observed_at":"2026-08-02T16:10:01.893221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2606.12555","last_updated":"2026-07-02T19:10:03Z","snapshot_observed_at":"2026-07-12T14:16:32.755376Z","submitted_at":"2026-06-10T18:06:27Z","title":"AudioX-Turbo: A Unified Framework for Efficient Anything-to-Audio Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T08:04:48.283908Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2606.12555"},"observation_digest":"sha256:39a4b4bacec35a98c3f965b52d1a2380a64e8838e93c38b3f58c74578bcc17dd","observation_id":"1529a806-b412-4f95-b41e-e8133415e65a","resolution":{"observed_at":"2026-07-03T13:28:18.704967Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2606.17539","last_updated":"2026-06-16T05:32:39Z","snapshot_observed_at":"2026-07-06T23:53:07.149182Z","submitted_at":"2026-06-16T05:32:39Z","title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","version":1},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-06-27T01:42:30.005911Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2606.17539"},"observation_digest":"sha256:350b89cfe36b18d9181aa02e6464edec6a4e9187c4fd1afab927afb8b35b8aa3","observation_id":"2724805f-f09a-4d6c-bea4-08dd5cc7b783","resolution":{"observed_at":"2026-07-03T20:08:55.660752Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2305.05662/citation-record","integrity":"/paper/2305.05662/integrity","json":"/paper/2305.05662/citation-record.json","paper":"/paper/2305.05662"},"outbound":[],"paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2305.05662."}