{"as_of":"2026-08-08T08:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e6d6417e67335cbd6b9e76789e658fd385f5de1b18d8c71ec9b4cc5936f1af82","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:50:19.855755Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":98,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2305.14299","last_updated":"2026-04-12T00:03:25Z","snapshot_observed_at":"2026-07-06T15:31:41.985646Z","submitted_at":"2023-05-23T17:40:41Z","title":"Template-assisted Contrastive Learning of Task-oriented Dialogue Sentence Embeddings","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-24T09:11:37.220487Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2305.14299"},"observation_digest":"sha256:4e42200100c527643188f2e814d0e069a977a12cebac453896c0bdce3396a4d9","observation_id":"ad9eeb77-3851-4f76-8702-3702171fdb4a","resolution":{"observed_at":"2026-05-24T09:14:16.094675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-06T19:50:19.855755Z","title":"Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.13364","last_updated":"2025-07-06T18:51:22Z","snapshot_observed_at":"2026-08-06T19:43:47.990764Z","submitted_at":"2025-07-06T18:51:22Z","title":"OmniVec2 -- A Novel Transformer based Network for Large Scale Multimodal and Multitask Learning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:50:19.855755Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2507.13364"},"observation_digest":"sha256:2c09d8dcb2a90617475ab232a9e3fbd604f6aaba7182b507d925866f297ba6a3","observation_id":"9a8f0407-498c-4677-afa6-5800b0e5c46b","resolution":{"observed_at":"2026-08-06T19:50:19.855755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-04T09:37:23.778509Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2510.14546","last_updated":"2026-08-03T08:20:35Z","snapshot_observed_at":"2026-08-08T06:39:49.614098Z","submitted_at":"2025-10-16T10:41:31Z","title":"QuASH: Using Natural-Language Heuristics to Query Visual-Language Robotic Maps","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T09:37:23.778509Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2510.14546"},"observation_digest":"sha256:f9d36982ac4b4250d46aa548937a3e6b65d72ee0dea2ea4032415492b9b8579c","observation_id":"4ea027f2-e3fe-4a8d-8cd9-a931b167e2b1","resolution":{"observed_at":"2026-08-04T09:37:23.778509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-03T19:22:50.461183Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2512.00918","last_updated":"2026-06-20T15:23:57Z","snapshot_observed_at":"2026-08-07T00:49:23.123068Z","submitted_at":"2025-11-30T14:52:11Z","title":"Sparse Neuron Ablation Triggers Catastrophic Collapse of the Language Core in Large Vision-Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T19:22:50.461183Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2512.00918"},"observation_digest":"sha256:63cb513767bab44faa3ef2ec169766a4b5c5d878a1e76bf2c8ef7920e12bbb2a","observation_id":"fcc47bb4-a808-48f3-84ad-483bb1702b32","resolution":{"observed_at":"2026-08-03T19:22:50.461183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2604.14684","last_updated":"2026-05-26T13:10:47Z","snapshot_observed_at":"2026-07-12T20:05:14.844082Z","submitted_at":"2026-04-16T06:40:44Z","title":"DETR-ViP: Detection Transformer with Robust Discriminative Visual Prompts","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T12:07:21.203513Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2604.14684"},"observation_digest":"sha256:81843909d32585723f364e9df71e6a60c87829129bb189c5061ac6012e405993","observation_id":"eb9a4f0d-d496-4c51-98a7-aed0c5d3577c","resolution":{"observed_at":"2026-05-10T12:10:21.947358Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-07-12T20:05:20.702092Z","title":"Grounded language-image pre- training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.14684","last_updated":"2026-05-26T13:10:47Z","snapshot_observed_at":"2026-07-12T20:05:14.844082Z","submitted_at":"2026-04-16T06:40:44Z","title":"DETR-ViP: Detection Transformer with Robust Discriminative Visual Prompts","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-12T20:05:20.702092Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2604.14684"},"observation_digest":"sha256:010d8f427fdc6b4935b43eb3535f2e5d6922b6fe807e1b6dec1573f846615f71","observation_id":"31fade60-4c4a-48ce-9548-16a5c26c91dd","resolution":{"observed_at":"2026-07-12T20:05:20.702092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2605.06477","last_updated":"2026-05-07T16:01:59Z","snapshot_observed_at":"2026-07-06T23:18:55.724335Z","submitted_at":"2026-05-07T16:01:59Z","title":"GeoStack: A Framework for Quasi-Abelian Knowledge Composition in VLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T13:17:48.295630Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2605.06477"},"observation_digest":"sha256:41b8d6739b498c67aef9b209392744837e6a36fc44588943d0d4ad4d3110bf83","observation_id":"65e85d52-f2ec-425b-92e6-29a8222780e3","resolution":{"observed_at":"2026-05-11T18:56:06.934858Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2605.08384","last_updated":"2026-06-05T21:57:48Z","snapshot_observed_at":"2026-08-02T16:50:30.597437Z","submitted_at":"2026-05-08T18:45:15Z","title":"jina-embeddings-v5-omni: Geometry-preserving Embeddings via Locked Aligned Towers","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-12T01:28:48.722301Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2605.08384"},"observation_digest":"sha256:64cf039beb0a0822c733c009805419d34e6778edba863dcce52875b15182f972","observation_id":"90dd2a2a-699c-4669-a914-e563b100455d","resolution":{"observed_at":"2026-05-12T07:56:28.955356Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2605.08384","last_updated":"2026-06-05T21:57:48Z","snapshot_observed_at":"2026-08-02T16:50:30.597437Z","submitted_at":"2026-05-08T18:45:15Z","title":"jina-embeddings-v5-omni: Geometry-preserving Embeddings via Locked Aligned Towers","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T06:57:25.015358Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2605.08384"},"observation_digest":"sha256:1c7c4fc271cc5c60cc7fde8d4e1a7acda4b5d9e35bccac0c68a87a0dc37ac2fe","observation_id":"672f9bdc-1ebe-4665-a0c0-58a83e959e86","resolution":{"observed_at":"2026-05-13T07:02:27.740739Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2605.08384","last_updated":"2026-06-05T21:57:48Z","snapshot_observed_at":"2026-08-02T16:50:30.597437Z","submitted_at":"2026-05-08T18:45:15Z","title":"jina-embeddings-v5-omni: Geometry-preserving Embeddings via Locked Aligned Towers","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-30T22:56:43.298141Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2605.08384"},"observation_digest":"sha256:616688355cdc0b83ad8ff668d4e6ec4d712561de1a1f8ec654873d6f4a136464","observation_id":"4ed090c5-521d-49d9-8a7c-e078e60cf360","resolution":{"observed_at":"2026-07-01T13:35:46.736252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2605.11237","last_updated":"2026-05-11T20:54:42Z","snapshot_observed_at":"2026-08-02T20:27:49.610199Z","submitted_at":"2026-05-11T20:54:42Z","title":"DeconDTN-Toolkit: A Library for Evaluation and Enhancement of Robustness to Provenance Shift","version":1},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-05-13T02:12:54.796350Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2605.11237"},"observation_digest":"sha256:8ac1739b4ef18eff5b34f9ffe5ca880de1b5cee6ae4c8882447da555a83f18de","observation_id":"65b23aaa-f7d2-4ab2-a270-9cc6ba0fc6c0","resolution":{"observed_at":"2026-05-13T02:17:05.858809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2605.12872","last_updated":"2026-05-13T01:36:43Z","snapshot_observed_at":"2026-08-04T16:43:43.487806Z","submitted_at":"2026-05-13T01:36:43Z","title":"SMA: Submodular Modality Aligner For Data Efficient Multimodal Learning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-14T20:32:00.361991Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2605.12872"},"observation_digest":"sha256:c0611da8f3445ee1b5d589cc73c962f543921ded8a73858d31dcfe453dd90c22","observation_id":"77e2541c-9271-478f-9107-975d68981210","resolution":{"observed_at":"2026-05-14T20:32:56.665073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2606.11190","last_updated":"2026-06-10T19:46:16Z","snapshot_observed_at":"2026-07-06T23:50:16.299969Z","submitted_at":"2026-06-09T17:59:58Z","title":"When to Align, When to Predict: A Phase Diagram for Multimodal Learning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T14:08:33.542700Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2606.11190"},"observation_digest":"sha256:09c18833d489605c1a3c98e2bc9212cf001b4e111c0a2474500a34bc5597924f","observation_id":"c5ef43e4-90d6-4834-a917-68e5c3113c5b","resolution":{"observed_at":"2026-07-03T04:07:37.050488Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2606.11190","last_updated":"2026-06-10T19:46:16Z","snapshot_observed_at":"2026-07-06T23:50:16.299969Z","submitted_at":"2026-06-09T17:59:58Z","title":"When to Align, When to Predict: A Phase Diagram for Multimodal Learning","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T14:08:33.542700Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2606.11190"},"observation_digest":"sha256:beae9f629aa4235813ebf96e98aa7bf7220d62c462ceaafa6576927ec3e2883f","observation_id":"40b2cdd8-71d1-4a78-b8e7-790c42dd38ce","resolution":{"observed_at":"2026-06-27T14:10:57.796470Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":"2203.02053","doi":"10.48550/arxiv.2203.02053","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Information Processing Systems, Vol","venue":"arXiv (Cornell University)","work_id":"fd718874-b9ab-4026-830c-0c5ba317d12e","year":2021},"citing_paper":{"arxiv_id":"2606.30625","last_updated":"2026-06-29T17:55:40Z","snapshot_observed_at":"2026-07-30T07:15:48.840477Z","submitted_at":"2026-06-29T17:55:40Z","title":"Optimization Dynamics Imprint Semantic Specificity in Contrastive Embedding Norms","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-30T03:27:31.177489Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2606.30625"},"observation_digest":"sha256:fa24ccb7805a0c711f7cd41f6404d7b94dc398ba4cede2270ca7a8f12dcba8ac","observation_id":"d65a7949-bf34-45f5-aecf-fbe5eee2f8a7","resolution":{"observed_at":"2026-06-30T03:34:13.357363Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-02T04:33:53.287239Z","title":"Yifei Ming, Ziyang Cai, Jiuxiang Gu, Yiyou Sun, Wei Li, and Yixuan Li","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13660","last_updated":"2026-07-15T10:05:53Z","snapshot_observed_at":"2026-08-06T00:19:12.467067Z","submitted_at":"2026-07-15T10:05:53Z","title":"The Hyperspherical Geometry of CLIP Latent Space: A Semantic Mixture Model","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T04:33:53.287239Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2607.13660"},"observation_digest":"sha256:65ca60131aa840d6a23625d924275221fcc8045e5a1ac16a0d2f68da5fe72cab","observation_id":"58328478-9e2b-4265-9cf4-83b2d7999dc3","resolution":{"observed_at":"2026-08-02T04:33:53.287239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02053","snapshot_observed_at":"2026-08-04T04:31:48.890280Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning.arXiv e-prints, page arXiv:2203.02053, March 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.02573","last_updated":"2026-08-03T17:49:40Z","snapshot_observed_at":"2026-08-06T23:39:11.550280Z","submitted_at":"2026-08-03T17:49:40Z","title":"Foundation Models for Astrophysics","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-04T04:31:48.890280Z"},"links":{"cited_paper":"/paper/2203.02053","citing_paper":"/paper/2608.02573"},"observation_digest":"sha256:2364927991e0e54c1202e0a9f37282820ef26c084ac285c0d59f34297fb90b3c","observation_id":"fdc3b59e-ffdc-4230-b0ab-23375b4f97d9","resolution":{"observed_at":"2026-08-04T04:31:48.890280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2203.02053/citation-record","integrity":"/paper/2203.02053/integrity","json":"/paper/2203.02053/citation-record.json","paper":"/paper/2203.02053"},"outbound":[],"paper":{"arxiv_id":"2203.02053","last_updated":"2022-10-19T20:39:13Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-04T04:38:21.674548Z","submitted_at":"2022-03-03T22:53:54Z","title":"Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2203.02053."}