{"as_of":"2026-08-12T10:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a0bfb62ca36191d3f1b8922235755124162284fd123ca15af73c4fc35524607c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T16:10:28.973202Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T15:09:55.058979Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2303.16199","last_updated":"2024-09-18T23:54:36Z","snapshot_observed_at":"2026-08-06T06:36:02.994951Z","submitted_at":"2023-03-28T17:59:12Z","title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","version":3},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-05-14T23:07:42.245641Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2303.16199"},"observation_digest":"sha256:52ef06af28301f9be4d89b47656f72a7387692170e105c02ed5fc260226ba3da","observation_id":"3bd4fa02-295b-4fc1-a701-98eb8693ee29","resolution":{"observed_at":"2026-05-14T23:07:42.641601Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-17T02:46:16.632743Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2403.00476"},"observation_digest":"sha256:4047cfba759f66a57410419caa41b532b122bde16fb5965e594de132024750e9","observation_id":"8f99e238-87b7-449b-a9b6-b90bdbdbe51d","resolution":{"observed_at":"2026-05-17T02:46:16.755206Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T04:09:36.019146Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2403.09611"},"observation_digest":"sha256:e3b0c2893bee9f3e5b89331e3e38ce49085b50f0582bdc114fb701c31818c12a","observation_id":"af8ae0c2-4493-4ead-8c91-db74540d08e0","resolution":{"observed_at":"2026-05-16T04:09:36.109213Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2403.14624","last_updated":"2024-08-18T08:10:16Z","snapshot_observed_at":"2026-08-04T04:40:00.850270Z","submitted_at":"2024-03-21T17:59:50Z","title":"MathVerse: Does Your Multi-modal LLM Truly See the Diagrams in Visual Math Problems?","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T01:29:30.032408Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2403.14624"},"observation_digest":"sha256:7da0ea3c03e273645eafda95ec3a116b36fb2d91267dd7cf3877a6c927b2fcb3","observation_id":"c0f9db05-ce6e-42d4-a7e3-872c866349bf","resolution":{"observed_at":"2026-05-17T01:29:30.108729Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2403.20330","last_updated":"2024-04-09T15:17:50Z","snapshot_observed_at":"2026-08-07T12:15:30.838846Z","submitted_at":"2024-03-29T17:59:34Z","title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T19:41:44.263663Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2403.20330"},"observation_digest":"sha256:a714ab2d9606ce76e9089264a0905e04731fe051935fe752bce48a900f836dfe","observation_id":"29366462-c276-43d9-aec2-81dab9c59aa7","resolution":{"observed_at":"2026-05-12T19:41:44.501498Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-17T00:05:03.547664Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2406.16860"},"observation_digest":"sha256:de90db42c490745774e962e348452fc29bbaad60dc6537b8b7e3fd57359fe3db","observation_id":"55871160-18e5-499d-9433-355b10e849f1","resolution":{"observed_at":"2026-05-17T00:05:03.779425Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-07-06T18:44:24.873040Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T06:01:53.730356Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2407.07895"},"observation_digest":"sha256:76da272101c904357a1ec3b20be09d12a84a80bd7e8296691901179ad4bb0543","observation_id":"7be3e54b-3137-42fd-b0d5-754a9281ea62","resolution":{"observed_at":"2026-05-11T06:01:54.007023Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-11T16:10:28.973202Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10342","last_updated":"2025-02-03T15:23:02Z","snapshot_observed_at":"2026-08-12T00:02:16.364157Z","submitted_at":"2024-12-13T18:40:10Z","title":"Iris: Breaking GUI Complexity with Adaptive Focus and Self-Refining","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T16:10:28.973202Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2412.10342"},"observation_digest":"sha256:3a553fdc904d9cd96dee59d98b4e4071bf60cef6ae679062bc1e87036ac7c2d1","observation_id":"289ae9cd-18e5-408e-a7a0-98ad1fee55ee","resolution":{"observed_at":"2026-08-11T16:10:28.973202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-11T12:46:59.501091Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13871","last_updated":"2025-03-19T10:04:22Z","snapshot_observed_at":"2026-08-11T14:38:15.849933Z","submitted_at":"2024-12-18T14:07:46Z","title":"LLaVA-UHD v2: an MLLM Integrating High-Resolution Semantic Pyramid via Hierarchical Window Transformer","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T12:46:59.501091Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2412.13871"},"observation_digest":"sha256:7977d4b623a5a6cec2c352330d5c9e1ae05f4cf02e03d630e457374996462eb3","observation_id":"5f912bcb-b483-4bd9-8eef-d73b36f1e3da","resolution":{"observed_at":"2026-08-11T12:46:59.501091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2412.14164","last_updated":"2024-12-18T18:58:50Z","snapshot_observed_at":"2026-08-08T15:05:21.947334Z","submitted_at":"2024-12-18T18:58:50Z","title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","version":1},"reference_index":175,"source":"arxiv_source","source_observed_at":"2026-05-17T07:51:12.953777Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2412.14164"},"observation_digest":"sha256:7ce933da67db5b25c987d46f31c0068dc86ab0be610b36e1968252645fdec03a","observation_id":"63b36243-8350-4724-85a8-431ab9f2679a","resolution":{"observed_at":"2026-05-17T07:51:13.313149Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-10T20:39:35.622849Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07783","last_updated":"2025-01-14T01:57:41Z","snapshot_observed_at":"2026-08-11T01:32:14.336933Z","submitted_at":"2025-01-14T01:57:41Z","title":"Parameter-Inverted Image Pyramid Networks for Visual Perception and Multimodal Understanding","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-10T20:39:35.622849Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2501.07783"},"observation_digest":"sha256:cc14580118b2335b7d6881bacc73b0a5c55316db8757b95a24edc4b770e40722","observation_id":"2cbadcd9-caee-4e34-9511-f6bea7f32226","resolution":{"observed_at":"2026-08-10T20:39:35.622849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-10T11:26:29.178954Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16688","last_updated":"2025-01-28T03:56:17Z","snapshot_observed_at":"2026-08-10T13:25:41.324961Z","submitted_at":"2025-01-28T03:56:17Z","title":"MME-Industry: A Cross-Industry Multimodal Evaluation Benchmark","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T11:26:29.178954Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2501.16688"},"observation_digest":"sha256:7d90f2522441948b653e7057cf66ab963b8d6aa8fe9bd7ef3e50bdb15f44fb60","observation_id":"7b675fb5-e226-4e93-853a-ee05c37b0b8b","resolution":{"observed_at":"2026-08-10T11:26:29.178954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2503.16549","last_updated":"2026-04-18T11:44:37Z","snapshot_observed_at":"2026-08-08T08:24:44.422314Z","submitted_at":"2025-03-19T11:46:19Z","title":"MathFlow: Enhancing the Perceptual Flow of MLLMs for Visual Mathematical Problems","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-22T22:55:34.238427Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2503.16549"},"observation_digest":"sha256:9b2b09cc547fcada10ad46f52b9f687b649077f22cdbbc5bcc7fe31e6f414bb9","observation_id":"617b104a-6ce5-43d2-8271-dfbca4478006","resolution":{"observed_at":"2026-05-22T22:57:13.216249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-07T12:25:06.518053Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24541","last_updated":"2025-05-30T12:48:07Z","snapshot_observed_at":"2026-08-09T16:02:42.942836Z","submitted_at":"2025-05-30T12:48:07Z","title":"Mixpert: Mitigating Multimodal Learning Conflicts with Efficient Mixture-of-Vision-Experts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:25:06.518053Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2505.24541"},"observation_digest":"sha256:1587ed621058a732c73ce6a0d6ced5173469b09e6bd6f6ab9f914a4c6139607a","observation_id":"a4ca0389-5ce2-4125-898c-fd9dac8f4f8c","resolution":{"observed_at":"2026-08-07T12:25:06.518053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-07T10:28:11.917393Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-10T19:28:49.876683Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.917393Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:bd634e08d1ab0b2c26e2c8761124bf04a1b6b5fd8a77032db0c39de499580f70","observation_id":"9626591a-cf6b-4f0e-85b6-962987fec048","resolution":{"observed_at":"2026-08-07T10:28:11.917393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-07T05:43:39.543588Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07202","last_updated":"2025-06-08T15:52:38Z","snapshot_observed_at":"2026-08-11T11:34:16.404599Z","submitted_at":"2025-06-08T15:52:38Z","title":"Reasoning Multimodal Large Language Model: Data Contamination and Dynamic Evaluation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:43:39.543588Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2506.07202"},"observation_digest":"sha256:58b5134e41bc9bb1243c058ab629d3821c3e9ea9f499c91aa3b9ccbf68d48c0b","observation_id":"dc258f49-a426-4b23-bc29-558dde71b26f","resolution":{"observed_at":"2026-08-07T05:43:39.543588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-06T23:42:13.746024Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16691","last_updated":"2025-06-20T02:25:33Z","snapshot_observed_at":"2026-08-11T00:54:41.191381Z","submitted_at":"2025-06-20T02:25:33Z","title":"LaVi: Efficient Large Vision-Language Models via Internal Feature Modulation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T23:42:13.746024Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2506.16691"},"observation_digest":"sha256:70fbbb88530ee46cd22e6b521cc5604bd5c888d8b75d7bba8d65e54e99996ae9","observation_id":"ce0f5d7b-3421-45af-b0a9-5c6c2810d65c","resolution":{"observed_at":"2026-08-06T23:42:13.746024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-05T21:11:51.363684Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.14080","last_updated":"2025-08-12T19:43:44Z","snapshot_observed_at":"2026-08-09T22:37:03.821950Z","submitted_at":"2025-08-12T19:43:44Z","title":"KnowDR-REC: A Benchmark for Referring Expression Comprehension with Real-World Knowledge","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T21:11:51.363684Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2508.14080"},"observation_digest":"sha256:6dcf4308ccb27b3ebfba45edbeff11d88cde5c6b5a66de301923b83764de9a2e","observation_id":"019be010-9436-40f4-8fa9-182185e4c18a","resolution":{"observed_at":"2026-08-05T21:11:51.363684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2510.21122","last_updated":"2026-04-07T14:13:33Z","snapshot_observed_at":"2026-08-11T13:59:17.998991Z","submitted_at":"2025-10-24T03:23:34Z","title":"NoisyGRPO: Incentivizing Multimodal CoT Reasoning via Noise Injection and Bayesian Estimation","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-18T04:39:58.296388Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2510.21122"},"observation_digest":"sha256:e55056585332477c68daf07da4018d7c62cde4c22b0aa60c20ee85b6e18eaf9a","observation_id":"11692ed6-db86-4248-a5f4-961a1285f591","resolution":{"observed_at":"2026-05-18T04:40:52.962013Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-03T20:23:05.200466Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.20272","last_updated":"2026-07-03T06:27:17Z","snapshot_observed_at":"2026-08-03T20:22:57.328566Z","submitted_at":"2025-11-25T12:58:32Z","title":"VKnowU: Evaluating Visual Knowledge Understanding in Multimodal LLMs","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-03T20:23:05.200466Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2511.20272"},"observation_digest":"sha256:1766ed487bb03149842a5bfaaf477373998d2d6d27f4a8d35a6450a3cdbbf4b0","observation_id":"9038ad2b-7b01-4e72-b0ee-82aab3cd54fa","resolution":{"observed_at":"2026-08-03T20:23:05.200466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-02T19:58:16.978097Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.00461","last_updated":"2026-06-10T06:57:00Z","snapshot_observed_at":"2026-08-05T02:58:46.928689Z","submitted_at":"2026-02-28T04:42:34Z","title":"ReMoT: Reinforcement Learning with Motion Contrast Triplets","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T19:58:16.978097Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2603.00461"},"observation_digest":"sha256:419aba44628a84d28be95fecbb44b6869cc59d63a3620a80c2f7fc38a1b632c9","observation_id":"7ed6f792-9916-4963-8016-0e0cb6490cef","resolution":{"observed_at":"2026-08-02T19:58:16.978097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2603.09921","last_updated":"2026-07-02T16:25:34Z","snapshot_observed_at":"2026-07-14T23:55:23.688065Z","submitted_at":"2026-03-10T17:18:53Z","title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T13:11:54.384284Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2603.09921"},"observation_digest":"sha256:891e1551c6a99ea3f9d2a6e0f448654a8c13f4d2135cfa48e672a698bf8cbe6f","observation_id":"c2171398-8dff-454d-8f16-48ea58911e64","resolution":{"observed_at":"2026-05-15T13:15:50.606098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-14T23:55:24.006436Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.ArXiv, abs/2402.05935,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.09921","last_updated":"2026-07-02T16:25:34Z","snapshot_observed_at":"2026-07-14T23:55:23.688065Z","submitted_at":"2026-03-10T17:18:53Z","title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-14T23:55:24.006436Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2603.09921"},"observation_digest":"sha256:f7d0de46eea5e9f1aee2c9bf8eda0f455e9c3734e51f51f973230f8e800bbbf2","observation_id":"da617aea-d05e-4d36-83ed-82ac78656311","resolution":{"observed_at":"2026-07-14T23:55:24.006436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2604.08719","last_updated":"2026-04-09T19:13:14Z","snapshot_observed_at":"2026-07-06T22:57:45.084987Z","submitted_at":"2026-04-09T19:13:14Z","title":"LMGenDrive: Bridging Multimodal Understanding and Generative World Modeling for End-to-End Driving","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T17:36:38.627415Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2604.08719"},"observation_digest":"sha256:27618379648189863d897d620caaa3ab016879641ce723c6a99dc351323ff734","observation_id":"507128b0-d48a-44cc-bd47-42f6a73732b9","resolution":{"observed_at":"2026-05-11T06:31:01.255346Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2604.14629","last_updated":"2026-04-16T05:13:57Z","snapshot_observed_at":"2026-08-10T21:30:42.156932Z","submitted_at":"2026-04-16T05:13:57Z","title":"Switch-KD: Visual-Switch Knowledge Distillation for Vision-Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T11:23:46.371799Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2604.14629"},"observation_digest":"sha256:e133b055372f1e01368bae460156dceb889125ae47ccc0517e98299d49dc2d13","observation_id":"65867ad8-9a18-4f36-bbb4-ad9155a03ef7","resolution":{"observed_at":"2026-05-10T11:25:18.565430Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2606.00275","last_updated":"2026-05-29T19:08:20Z","snapshot_observed_at":"2026-08-08T17:56:30.968996Z","submitted_at":"2026-05-29T19:08:20Z","title":"Hyperbolic and Evidence-Prioritized Experts for Large Vision-Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-28T22:43:33.929871Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2606.00275"},"observation_digest":"sha256:4493cc77c70c647fec0b6ed0e2e4b6a99d57c9d0e115e13fbc30e30580cad2e0","observation_id":"cccfad59-f34e-4e61-bdfd-7697f1a14d00","resolution":{"observed_at":"2026-07-01T19:26:00.087658Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2606.21337","last_updated":"2026-07-31T19:36:30Z","snapshot_observed_at":"2026-08-07T09:45:00.435849Z","submitted_at":"2026-06-19T11:31:43Z","title":"DataClaw0: Agentic Tailoring Multimodal Data from Raw Streams","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-26T14:24:19.702588Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2606.21337"},"observation_digest":"sha256:f507e114e304dd1c66b98af39f119eae5578123b7e8b126ff5e02b06daf3af3d","observation_id":"076ef308-5182-43f4-9b8e-d6bfde10de36","resolution":{"observed_at":"2026-07-04T06:29:37.911591Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-04T02:47:38.306016Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.21337","last_updated":"2026-07-31T19:36:30Z","snapshot_observed_at":"2026-08-07T09:45:00.435849Z","submitted_at":"2026-06-19T11:31:43Z","title":"DataClaw0: Agentic Tailoring Multimodal Data from Raw Streams","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T02:47:38.306016Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2606.21337"},"observation_digest":"sha256:276f6a37930553eef3983cda5ee559adad15d0b0662941228c3539be3c7eed9a","observation_id":"964e3653-86d9-4bd9-a086-fc91ddd3fe2d","resolution":{"observed_at":"2026-08-04T02:47:38.306016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":289,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:8e2f2373a31b7e0eea7ecbdb924d7f91e6cf199164e72491ea284341632d0c75","observation_id":"3ca1043b-cd7d-4edf-994b-34da34f2cbe3","resolution":{"observed_at":"2026-07-04T06:39:37.649753Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.05935","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-07-04T15:09:55.058979Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models","venue":null,"work_id":"dbfa4c97-4580-4acc-b396-499d7de46399","year":2024},"citing_paper":{"arxiv_id":"2606.26196","last_updated":"2026-06-24T15:20:32Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T15:20:32Z","title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-26T01:50:54.242508Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2606.26196"},"observation_digest":"sha256:d99183e16198e603867dc816e1612b8d1dadf9e27b450fc9fffbe1684483ecb4","observation_id":"6c1e1df4-7704-4278-b672-c2868a1fb8b6","resolution":{"observed_at":"2026-07-04T15:09:55.060578Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2402.05935/citation-record","integrity":"/paper/2402.05935/integrity","json":"/paper/2402.05935/citation-record.json","paper":"/paper/2402.05935"},"outbound":[],"paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2402.05935."}