{"as_of":"2026-08-09T04:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4a257e8d9a2ec419dad1d14e50d8d573b8c6defa4e059b1e5a6dfd83ed58af92","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":37,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T00:12:36.209585Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":74,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2307.05663","last_updated":"2023-07-11T17:57:40Z","snapshot_observed_at":"2026-07-06T15:52:52.180267Z","submitted_at":"2023-07-11T17:57:40Z","title":"Objaverse-XL: A Universe of 10M+ 3D Objects","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T13:02:11.512409Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2307.05663"},"observation_digest":"sha256:1fdeff3418e8a1ae893617b7d02dbb66c8443ff63a7a0026c7b0dd2d80f093f7","observation_id":"53e4f70b-9858-4a6d-889c-4aa69b2da6f1","resolution":{"observed_at":"2026-05-17T13:02:11.604349Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:11e395e0343951d60d394514e6908191e498945df74c280f0db5c47dc8f08260","observation_id":"5b0763c5-fe42-473d-ae4b-031e297c053a","resolution":{"observed_at":"2026-05-15T06:30:22.712246Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2308.01390","last_updated":"2023-08-07T17:53:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-02T19:10:23Z","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T01:52:01.163900Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2308.01390"},"observation_digest":"sha256:345528f1900db94cb8970f0e37145640dcdfe3936f8b04ec4ac0b2fc35503dd0","observation_id":"b7853a04-19fe-49b5-bcb6-5fa3b35955f9","resolution":{"observed_at":"2026-05-14T01:52:01.415728Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2311.04257","last_updated":"2023-11-09T01:56:51Z","snapshot_observed_at":"2026-08-06T04:08:15.266897Z","submitted_at":"2023-11-07T14:21:29Z","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-18T03:18:51.582340Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2311.04257"},"observation_digest":"sha256:7718817ca86beb1e893a5070425491c5a8cd299d2b3dd3beed0a6b68c52bffa5","observation_id":"50537838-c88e-44dd-add6-32810a4947d5","resolution":{"observed_at":"2026-05-18T03:18:51.844203Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-04T08:17:54.774738Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T17:08:12.727773Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2311.12793"},"observation_digest":"sha256:387637a2c0eaec37621324a9eea1034b8240cb681cf6b67348a2385a49770562","observation_id":"dbd967c6-4515-481d-8b9a-4f0905432019","resolution":{"observed_at":"2026-05-13T17:08:12.842273Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-17T22:58:16.523267Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2406.11794"},"observation_digest":"sha256:cda2e376d07e9fac15602727c39469f9a450f2068d466f230a553f565a400d0b","observation_id":"16a27061-3fdc-4a50-b658-94fc2332560b","resolution":{"observed_at":"2026-05-17T22:58:16.961906Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":211,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:58830e8d69b4a0345150829c9563941601aa60343b81c1f9e1b60f9ef50eee2b","observation_id":"26e6ffb9-7845-4953-af59-dcce8109799e","resolution":{"observed_at":"2026-05-20T06:20:36.377313Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-09T00:12:36.209585Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.03950","last_updated":"2025-05-18T22:20:47Z","snapshot_observed_at":"2026-08-09T00:05:17.355061Z","submitted_at":"2025-02-06T10:40:42Z","title":"LR0.FM: Low-Res Benchmark and Improving Robustness for Zero-Shot Classification in Foundation Models","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-09T00:12:36.209585Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2502.03950"},"observation_digest":"sha256:7367825ae9c5c7435e18c1a5b7b24e9c363f652ae9c930e5050824a60e115c85","observation_id":"7dc61a48-3420-4b62-82db-fccc3d0d19ae","resolution":{"observed_at":"2026-08-09T00:12:36.209585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2504.09925","last_updated":"2026-04-29T06:12:36Z","snapshot_observed_at":"2026-08-02T07:57:37.201421Z","submitted_at":"2025-04-14T06:33:29Z","title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T19:49:00.961388Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2504.09925"},"observation_digest":"sha256:c2105d9c3a150c62929fce8259063509c54564d2e5d7892ce694a0d6bfda931b","observation_id":"30ac765f-d39d-43fa-9d45-b57bc71a3f6f","resolution":{"observed_at":"2026-05-22T19:52:01.977529Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T15:42:32.345850Z","title":"Y .et al.Datacomp: In search of the next generation of multimodal datasets (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14204","last_updated":"2025-05-20T11:04:14Z","snapshot_observed_at":"2026-08-07T23:52:33.058884Z","submitted_at":"2025-05-20T11:04:14Z","title":"Beginning with You: Perceptual-Initialization Improves Vision-Language Representation and Alignment","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:32.345850Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2505.14204"},"observation_digest":"sha256:5a3bcbd5419c2ee8dc8ca714d878af647b234b9f434bd533689b2cbf21df0754","observation_id":"aadfe9c0-a28f-4092-b195-fc2eae3e4cd4","resolution":{"observed_at":"2026-08-07T15:42:32.345850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T08:34:36.824053Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:4f66e2e919756017fe64dec6be74d81817d7260f1673c20352cdde8292e1e801","observation_id":"e6df2df8-30ed-41f0-b903-f650583f6d3b","resolution":{"observed_at":"2026-05-16T08:34:36.875949Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T00:59:13.826054Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:1c02051205bbb7824570a21b471508b1e88479b7b0e3c5bc627013551ad9bdf3","observation_id":"38170498-c1f5-4e46-8004-1602482eff3c","resolution":{"observed_at":"2026-05-22T01:00:51.274211Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T11:26:56.445045Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02698","last_updated":"2025-06-06T03:14:26Z","snapshot_observed_at":"2026-08-08T17:53:50.140226Z","submitted_at":"2025-06-03T09:47:22Z","title":"Smoothed Preference Optimization via ReNoise Inversion for Aligning Diffusion Models with Varied Human Preferences","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:56.445045Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.02698"},"observation_digest":"sha256:c0c46ac8fd66fca28dbdc0292b4df6fca1044426a488ce1fdcc49c26f925a9b8","observation_id":"fceae8d0-9b28-4509-a9a7-255461e1da22","resolution":{"observed_at":"2026-08-07T11:26:56.445045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T05:27:16.025685Z","title":"Dat- acomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08071","last_updated":"2025-06-09T17:54:41Z","snapshot_observed_at":"2026-08-07T23:16:48.649505Z","submitted_at":"2025-06-09T17:54:41Z","title":"CuRe: Cultural Gaps in the Long Tail of Text-to-Image Systems","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:27:16.025685Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.08071"},"observation_digest":"sha256:86bb3aa8a7b338dc91264976f480a39047ab243850caaa8ac6a960c7a2f3c042","observation_id":"1a413808-640f-485c-a0d7-4112960d33e7","resolution":{"observed_at":"2026-08-07T05:27:16.025685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T04:59:40.265102Z","title":"Gallou´edec, Q., Beeching, E., Romac, C., and Dellandr´ea, E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-09T02:05:58.972748Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.265102Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:bcdddfa945332fd97ba76ebcebce9c3d360fa63717aa42e670960bbfbee5e12d","observation_id":"28da12cc-c86a-4b77-a6b4-f5ab6e92c7cb","resolution":{"observed_at":"2026-08-07T04:59:40.265102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T05:01:13.736890Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10038","last_updated":"2025-06-10T22:37:39Z","snapshot_observed_at":"2026-08-08T07:43:39.610396Z","submitted_at":"2025-06-10T22:37:39Z","title":"Ambient Diffusion Omni: Training Good Models with Bad Data","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:01:13.736890Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.10038"},"observation_digest":"sha256:1be07da3458e46f0d2fbe206afd86cd01fae394bec7c199b6e969e2a92301730","observation_id":"12352945-800f-405c-90d6-e72bb1f3bac9","resolution":{"observed_at":"2026-08-07T05:01:13.736890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-06T22:03:46.175499Z","title":"Datacomp: In search of the next generation of multimodal datasets.arXiv preprint arXiv:2304.14108, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22881","last_updated":"2026-05-31T13:32:45Z","snapshot_observed_at":"2026-08-06T21:53:10.592056Z","submitted_at":"2025-06-28T13:36:44Z","title":"CLIP-like Model as a Foundational Density Ratio Estimator","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:03:46.175499Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.22881"},"observation_digest":"sha256:8db9d7318abdd3bb688516c1b60d80342c1b23e92e7ff1bf158c919fcb7e0464","observation_id":"8ca22854-6d9e-43e6-a931-eeb949699ba6","resolution":{"observed_at":"2026-08-06T22:03:46.175499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T14:59:19.528028Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.20691","last_updated":"2025-08-28T11:50:22Z","snapshot_observed_at":"2026-08-06T13:33:55.665527Z","submitted_at":"2025-08-28T11:50:22Z","title":"MobileCLIP2: Improving Multi-Modal Reinforced Training","version":1},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-05T14:59:19.528028Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2508.20691"},"observation_digest":"sha256:043a6205dafc59cca6bb825e476b4a022e98ae4f52e0249191d6a206fde3d4c1","observation_id":"6823ab5d-aa99-4459-8da0-55367a9087ae","resolution":{"observed_at":"2026-08-05T14:59:19.528028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T12:22:33.659440Z","title":"Dat- acomp: In search of the next generation of multimodal datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01644","last_updated":"2025-09-01T17:38:21Z","snapshot_observed_at":"2026-08-08T14:55:06.188124Z","submitted_at":"2025-09-01T17:38:21Z","title":"OpenVision 2: A Family of Generative Pretrained Visual Encoders for Multimodal Learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T12:22:33.659440Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2509.01644"},"observation_digest":"sha256:13d04f1899bf57773807e93eb6d4e578b43bab04bf2777d974452565ad613f1f","observation_id":"98faa756-30e0-4b35-a4e4-807594854e0b","resolution":{"observed_at":"2026-08-05T12:22:33.659440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2510.04056","last_updated":"2026-05-12T17:38:57Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T06:34:30Z","title":"QuiLL: An LLM-Based Vulnerability Assessment Framework for the Wild","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-18T10:56:28.973065Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2510.04056"},"observation_digest":"sha256:779e3abf633405655236fee2fcf87a65117515c692ca6f89994d700a5527c0fb","observation_id":"ba5b2e99-95be-4cbb-8b2c-e65ec2e8cb55","resolution":{"observed_at":"2026-05-18T11:01:17.173725Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2604.14198","last_updated":"2026-04-03T04:26:43Z","snapshot_observed_at":"2026-08-06T11:58:56.492844Z","submitted_at":"2026-04-03T04:26:43Z","title":"MixAtlas: Uncertainty-aware Data Mixture Optimization for Multimodal LLM Midtraining","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T20:07:29.544613Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2604.14198"},"observation_digest":"sha256:4f42a1f4ca014f0ac074f1fa54500508628500b9b1ea1faf1bd9995931f93e22","observation_id":"36a23a66-588a-4566-86fd-137999944328","resolution":{"observed_at":"2026-05-13T20:08:12.623943Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2604.25154","last_updated":"2026-04-28T02:56:17Z","snapshot_observed_at":"2026-07-06T23:11:02.043135Z","submitted_at":"2026-04-28T02:56:17Z","title":"Prior-Aligned Data Cleaning for Tabular Foundation Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-07T16:47:31.905149Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2604.25154"},"observation_digest":"sha256:4cc96cc5413fcd4916749ea6e43ef12735a17015ba352cde82029c76f4200524","observation_id":"354b2574-26db-433b-b021-4ab0150da4c1","resolution":{"observed_at":"2026-05-11T23:31:17.041197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.05416","last_updated":"2026-05-06T20:20:17Z","snapshot_observed_at":"2026-08-03T01:45:26.187986Z","submitted_at":"2026-05-06T20:20:17Z","title":"From Cradle to Cloud: A Life Cycle Review of AI's Environmental Footprint","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-08T15:43:50.422887Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.05416"},"observation_digest":"sha256:ec0c33bda83108f0dc005f9b19d623e0b1cbb6a2fd97de0993d9305fa8aff648","observation_id":"44239207-3e69-40cd-b826-4e2a54fcd34f","resolution":{"observed_at":"2026-05-11T18:31:12.927636Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.09433","last_updated":"2026-05-10T09:13:40Z","snapshot_observed_at":"2026-07-06T23:21:30.897592Z","submitted_at":"2026-05-10T09:13:40Z","title":"Offline Preference Optimization for Rectified Flow with Noise-Tracked Pairs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T02:10:27.595446Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.09433"},"observation_digest":"sha256:df44f2fd43ed80d2a82f8d5ddba6fe2de144ad9be1e5d1c78912d68ce6be7eff","observation_id":"dd4dc2db-6d56-4ac3-be94-13ed77e75d64","resolution":{"observed_at":"2026-05-12T02:11:15.568450Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.20278","last_updated":"2026-05-24T12:22:09Z","snapshot_observed_at":"2026-07-06T23:30:53.900592Z","submitted_at":"2026-05-19T04:39:28Z","title":"ClaimDiff-RL: Fine-Grained Caption Reinforcement Learning through Visual Claim Comparison","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T08:36:27.676888Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.20278"},"observation_digest":"sha256:5cef9080627c44b8c73f523e2fcf4d17622f564f27d15d5f0f9a37391d671650","observation_id":"fb96e050-6f1b-4a26-8252-11400aca9b15","resolution":{"observed_at":"2026-05-21T08:39:53.732465Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.20278","last_updated":"2026-05-24T12:22:09Z","snapshot_observed_at":"2026-07-06T23:30:53.900592Z","submitted_at":"2026-05-19T04:39:28Z","title":"ClaimDiff-RL: Fine-Grained Caption Reinforcement Learning through Visual Claim Comparison","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-30T17:57:47.409741Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.20278"},"observation_digest":"sha256:b51a7cf3df419c94fe4aa38f00fcc124b4aa7c11e5244bf54bc62de02a4e84b2","observation_id":"cc80062b-c6f6-46c5-be8d-15b48b4c6989","resolution":{"observed_at":"2026-06-30T18:04:58.323685Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2605.30341","last_updated":"2026-05-28T17:59:26Z","snapshot_observed_at":"2026-07-06T23:39:38.845840Z","submitted_at":"2026-05-28T17:59:26Z","title":"GPIC: A Giant Permissive Image Corpus for Visual Generation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T07:36:21.262064Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2605.30341"},"observation_digest":"sha256:4bb514a198b86cf768600a362dfb200b40966b282f4c9848e2b025c49534f7fe","observation_id":"bffc5f5e-6bfb-4ee6-a4e5-d39898cd29a4","resolution":{"observed_at":"2026-06-29T07:43:14.139724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.06624","last_updated":"2026-06-08T15:12:03Z","snapshot_observed_at":"2026-08-02T09:46:22.331867Z","submitted_at":"2026-06-04T18:21:03Z","title":"Principles and Practice of Deep Representation Learning: or a Mathematical Theory of Memory","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T03:07:52.730713Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.06624"},"observation_digest":"sha256:dc9684d589b6cb0dee858375777dd4254bc437e51f6add40030def681dc60aa9","observation_id":"7fab5c0d-92cc-4308-b2b2-1ece77cb957d","resolution":{"observed_at":"2026-07-02T11:46:55.266557Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.07865","last_updated":"2026-06-05T21:53:39Z","snapshot_observed_at":"2026-07-06T23:47:29.065892Z","submitted_at":"2026-06-05T21:53:39Z","title":"Instrumented data for causal scientific machine learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T22:24:28.643956Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.07865"},"observation_digest":"sha256:cc9183ee18081aaeacc3b57b641d165bd8db02393b5d59b66c2f49e921bb16e5","observation_id":"8becee99-1461-4517-9165-db60f18cc0b9","resolution":{"observed_at":"2026-06-27T22:31:21.232799Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.17030","last_updated":"2026-06-17T13:54:57Z","snapshot_observed_at":"2026-08-01T21:48:31.832288Z","submitted_at":"2026-06-15T17:52:31Z","title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","version":3},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-06-27T04:19:26.332718Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.17030"},"observation_digest":"sha256:71bc797a824dfbaeeccc548f344e7de8627b176e15fcc6bf64d755fc8af81013","observation_id":"02ffbe7a-5687-4cdd-a96d-6257e73f5924","resolution":{"observed_at":"2026-07-03T17:18:43.855348Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.24253","last_updated":"2026-06-26T09:37:54Z","snapshot_observed_at":"2026-08-08T09:42:45.507846Z","submitted_at":"2026-06-23T07:42:22Z","title":"TuringViT: Making SOTA Vision Transformers Accessible to All","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-26T00:29:41.291832Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.24253"},"observation_digest":"sha256:b045163be97c8229e8028b5e17a07dd16cba9894d5e73e575c480ac8e0003d16","observation_id":"1ed58a5c-454a-49fe-9e34-bd224feed00c","resolution":{"observed_at":"2026-07-04T16:29:57.904493Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.24253","last_updated":"2026-06-26T09:37:54Z","snapshot_observed_at":"2026-08-08T09:42:45.507846Z","submitted_at":"2026-06-23T07:42:22Z","title":"TuringViT: Making SOTA Vision Transformers Accessible to All","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T05:32:26.746776Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.24253"},"observation_digest":"sha256:09b2f726331c54c5508dcdbb64bd5e35e5b88001dc65a7e0579087c55320ad6b","observation_id":"60f42309-0369-402a-a0f7-a289360b21fd","resolution":{"observed_at":"2026-06-29T15:03:32.239700Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.26199","last_updated":"2026-06-26T02:47:28Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T16:23:10Z","title":"MIRAGE: Protecting against Malicious Image Editing via False Moderation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-26T01:52:12.291420Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.26199"},"observation_digest":"sha256:f05eac0629769ffa1c34af74dbfee266155a1a6ef6b4d72530b43da9967c700c","observation_id":"06c80a0a-6a21-4901-9a71-0ac0cf997d2b","resolution":{"observed_at":"2026-07-04T15:09:54.596740Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":"2304.14108","doi":"10.48550/arxiv.2304.14108","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":"arXiv (Cornell University)","work_id":"a309097d-9350-42a1-9232-7b765166f2da","year":2023},"citing_paper":{"arxiv_id":"2606.26199","last_updated":"2026-06-26T02:47:28Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T16:23:10Z","title":"MIRAGE: Protecting against Malicious Image Editing via False Moderation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T04:46:56.552601Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2606.26199"},"observation_digest":"sha256:5460a4993483cbb0d665a402b947bbe78a0bbed938a7a5f9022c5c9388defc39","observation_id":"ad1a47ec-70b8-44bb-9cfa-7ce7b22c1616","resolution":{"observed_at":"2026-06-29T19:13:53.521084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-07-12T01:07:20.766474Z","title":"Efros, Jenia Jitsev, Yair Carmon, Lud- wig Schmidt, and Vaishaal Shankar","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03624","last_updated":"2026-07-03T22:58:54Z","snapshot_observed_at":"2026-08-01T21:59:16.066897Z","submitted_at":"2026-07-03T22:58:54Z","title":"RADIO1D: Elastic Representations for Condensed Vision Modeling","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-12T01:07:20.766474Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2607.03624"},"observation_digest":"sha256:5ae25fe4e0fb9cd62efc366457e3ebc0b1f044f34621fc95b8d4659e30efcac1","observation_id":"64f9987d-9b2e-495e-97c9-e375d3f97e63","resolution":{"observed_at":"2026-07-12T01:07:20.766474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-07-14T03:31:19.309532Z","title":"arXiv:2304.14108 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11738","last_updated":"2026-07-13T16:00:03Z","snapshot_observed_at":"2026-08-06T12:36:13.363835Z","submitted_at":"2026-07-13T16:00:03Z","title":"Qwen-Audio-VAE Technical Report","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-07-14T03:31:19.309532Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2607.11738"},"observation_digest":"sha256:8f467f558af2fac4e940a677fd5d9955506e14e3fac72a2f64873afdddc3cd65","observation_id":"139999be-369b-4d71-8716-c21c3f3d0e95","resolution":{"observed_at":"2026-07-14T03:31:19.309532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-01T01:15:07.149546Z","title":"2304.14108 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25886","last_updated":"2026-07-28T15:46:41Z","snapshot_observed_at":"2026-08-07T09:45:34.637759Z","submitted_at":"2026-07-28T15:46:41Z","title":"RSIBench-Data: Benchmarking Data-Centric Research for Recursive Self-Improvement","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-01T01:15:07.149546Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2607.25886"},"observation_digest":"sha256:e16b90aa7292199b833b44fb0db37f08ea2899639a62a2b46f0a03fe691ef15b","observation_id":"cf286f86-7e59-4e71-9d37-597deed39d0f","resolution":{"observed_at":"2026-08-01T01:15:07.149546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2304.14108/citation-record","integrity":"/paper/2304.14108/integrity","json":"/paper/2304.14108/citation-record.json","paper":"/paper/2304.14108"},"outbound":[],"paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T15:20:41.322682Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 37 inbound Pith citation observations for arXiv:2304.14108."}