{"as_of":"2026-08-09T06:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:90d2382cd7f5fa531305b03bcf2d4bd339d18199582378ce735f8b1afdf9d43d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T05:54:15.993546Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T15:49:57.631122Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-08T05:54:15.993546Z","title":"Mm-embed: Universal multimodal retrieval with multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08254","last_updated":"2025-02-12T09:49:43Z","snapshot_observed_at":"2026-08-08T05:46:51.015561Z","submitted_at":"2025-02-12T09:49:43Z","title":"UniCoRN: Unified Commented Retrieval Network with LMMs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-08T05:54:15.993546Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2502.08254"},"observation_digest":"sha256:52ff72caa4a4e3c5746cacc58f6f3436d34ec0b6298422d4e24dd2555649d703","observation_id":"591ce93e-7e28-434c-9f98-8987dc53e7b8","resolution":{"observed_at":"2026-08-08T05:54:15.993546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-07T14:14:41.299337Z","title":"Mm-embed: Universal multimodal retrieval with multimodal llms.arXiv preprint arXiv:2411.02571, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19650","last_updated":"2025-05-27T11:57:17Z","snapshot_observed_at":"2026-08-07T14:07:13.987973Z","submitted_at":"2025-05-26T08:09:44Z","title":"Modality Curation: Building Universal Embeddings for Advanced Multimodal Information Retrieval","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:14:41.299337Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2505.19650"},"observation_digest":"sha256:de9196f250b284a297a6bdefbcb9ff7ad98b6cc907e8199c99365a5a9a018308","observation_id":"a6817cad-c2f1-499f-afc9-97037fc12468","resolution":{"observed_at":"2026-08-07T14:14:41.299337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-07T12:42:25.168442Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24073","last_updated":"2025-08-26T16:42:37Z","snapshot_observed_at":"2026-08-07T21:48:31.035838Z","submitted_at":"2025-05-29T23:32:03Z","title":"mRAG: Elucidating the Design Space of Multi-modal Retrieval-Augmented Generation","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T12:42:25.168442Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2505.24073"},"observation_digest":"sha256:1d8127ff213acbb254a7c2ecead91b7458ec894abb492d517e8af64cbf875466","observation_id":"e0561101-cd0f-4b08-b7a2-6713c5757cd8","resolution":{"observed_at":"2026-08-07T12:42:25.168442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-06T21:52:24.419284Z","title":"Mm-embed: Universal multimodal retrieval with multimodal llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23115","last_updated":"2025-06-29T06:41:00Z","snapshot_observed_at":"2026-08-07T01:05:19.522559Z","submitted_at":"2025-06-29T06:41:00Z","title":"MoCa: Modality-aware Continual Pre-training Makes Better Bidirectional Multimodal Embeddings","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.419284Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2506.23115"},"observation_digest":"sha256:38c3377d23b9ca6ee2633214ec0ce8ceeb9d247b37e5f9d3754c479cd07378ea","observation_id":"b54f5c2c-b25a-4369-8aeb-aafe3724a387","resolution":{"observed_at":"2026-08-06T21:52:24.419284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2507.04590","last_updated":"2025-07-07T00:51:57Z","snapshot_observed_at":"2026-08-08T15:48:31.665148Z","submitted_at":"2025-07-07T00:51:57Z","title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-18T14:10:14.929207Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2507.04590"},"observation_digest":"sha256:a98bbbef5231c8617fceb9c06781e059c52edc79a11646c4a23babac30ec4673","observation_id":"2d954e84-55b7-4d5a-acaa-8464a1b88474","resolution":{"observed_at":"2026-05-18T14:10:15.181457Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-05T20:28:53.509159Z","title":"Lin et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10955","last_updated":"2025-08-14T07:25:45Z","snapshot_observed_at":"2026-08-07T18:23:08.889281Z","submitted_at":"2025-08-14T07:25:45Z","title":"Empowering Multimodal LLMs with External Tools: A Comprehensive Survey","version":1},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-08-05T20:28:53.509159Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2508.10955"},"observation_digest":"sha256:2671acbd37c2009444872543e5adbe4d995f3329a278446ff01090dd24429934","observation_id":"8e347a1a-dff9-4f3b-93bf-c328d78ab3fd","resolution":{"observed_at":"2026-08-05T20:28:53.509159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-05T18:43:33.749092Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.14280","last_updated":"2025-08-19T21:28:12Z","snapshot_observed_at":"2026-08-08T13:05:48.967809Z","submitted_at":"2025-08-19T21:28:12Z","title":"Multi-Rationale Explainable Object Recognition via Contrastive Conditional Inference","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T18:43:33.749092Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2508.14280"},"observation_digest":"sha256:e82904ca0b60a28c1aa54186754f57ed4e49c098849c433f5bf4e045994c1843","observation_id":"8ff187b7-6219-4f13-b0fc-4bc36a32b6d5","resolution":{"observed_at":"2026-08-05T18:43:33.749092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2509.24621","last_updated":"2026-05-25T10:53:36Z","snapshot_observed_at":"2026-08-08T21:41:13.842209Z","submitted_at":"2025-09-29T11:28:42Z","title":"FreeRet: MLLMs as Training-Free Retrievers","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T13:00:31.952588Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2509.24621"},"observation_digest":"sha256:268bca62e9c50672cffe7531512ed5e7af75b63fe72b2eeb5f164fbe38d75262","observation_id":"23f78e3b-2ba3-4a7f-81d8-b13912b962a4","resolution":{"observed_at":"2026-05-18T13:01:23.497573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-04T13:51:45.715775Z","title":"Mm-embed: Universal multimodal retrieval with multimodal llms.arXiv preprint arXiv:2411.02571,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.24621","last_updated":"2026-05-25T10:53:36Z","snapshot_observed_at":"2026-08-08T21:41:13.842209Z","submitted_at":"2025-09-29T11:28:42Z","title":"FreeRet: MLLMs as Training-Free Retrievers","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T13:51:45.715775Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2509.24621"},"observation_digest":"sha256:7c3ca12db277fcbf7bc5357449d7cdbe6691571938e34d14e49f4b7d26b6e48c","observation_id":"b9d4bee7-5631-4a32-8c17-ee87f78270b7","resolution":{"observed_at":"2026-08-04T13:51:45.715775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-03T23:35:18.907951Z","title":"Haotian Liu, Chunyuan Li, Yuheng Li, and Yong Jae Lee","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.05017","last_updated":"2026-06-07T15:01:12Z","snapshot_observed_at":"2026-08-07T05:11:19.374607Z","submitted_at":"2025-11-07T06:39:54Z","title":"Towards Mitigating Hallucinations in Large Vision-Language Models by Refining Textual Embeddings","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T23:35:18.907951Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2511.05017"},"observation_digest":"sha256:e775f8a9f2397592110460a518953ce59670dd92455af02645f40d091997d38b","observation_id":"066a815b-5d7f-4b47-b465-4e705632a3ae","resolution":{"observed_at":"2026-08-03T23:35:18.907951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-03T22:05:16.854105Z","title":"Mm-embed: Universal multimodal retrieval with multimodal llms.arXiv preprint arXiv:2411.02571, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.12449","last_updated":"2026-07-29T13:03:53Z","snapshot_observed_at":"2026-08-08T05:09:09.199330Z","submitted_at":"2025-11-16T04:29:35Z","title":"MOON2.0: Dynamic Modality-balanced Multimodal Representation Learning for E-commerce Product Understanding","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T22:05:16.854105Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2511.12449"},"observation_digest":"sha256:629f144105245bae59b5c05d6d4b0ba64e2e05736a299b0e7829cde56af2f9a9","observation_id":"7602a1db-8b2f-45aa-be7e-5ab2a868d988","resolution":{"observed_at":"2026-08-03T22:05:16.854105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2512.13511","last_updated":"2026-05-01T11:23:38Z","snapshot_observed_at":"2026-07-06T22:39:04.164003Z","submitted_at":"2025-12-15T16:38:59Z","title":"Adapting MLLMs for Nuanced Video Retrieval","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T22:20:09.051957Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2512.13511"},"observation_digest":"sha256:8d4e36bae063a0b0768050e744a0d789e6d16c3d664a345f00f1032b2a6e4f0c","observation_id":"571ba26f-cb8a-498b-948e-ff8195dc5e6c","resolution":{"observed_at":"2026-05-16T22:21:18.872651Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-03T04:20:54.793364Z","title":"Mm- embed: Universal multimodal retrieval with multimodal llms.arXiv preprint arXiv:2411.02571, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05275","last_updated":"2026-07-01T09:38:37Z","snapshot_observed_at":"2026-08-08T17:38:17.395786Z","submitted_at":"2026-02-05T04:01:01Z","title":"Magic-MM-Embedding: Towards Visual-Token-Efficient Universal Multimodal Embedding with MLLMs","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-03T04:20:54.793364Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2602.05275"},"observation_digest":"sha256:92878231b1baef3dc9d784f47518dde33327c217cebe4e4d0898bb3c36dbfdcc","observation_id":"dd0e4e57-7762-4cbd-815f-a76f08928956","resolution":{"observed_at":"2026-08-03T04:20:54.793364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2604.02073","last_updated":"2026-04-20T03:16:36Z","snapshot_observed_at":"2026-07-06T22:51:40.181565Z","submitted_at":"2026-04-02T14:04:53Z","title":"PLUME: Latent Reasoning Based Universal Multimodal Embedding","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T21:48:40.722921Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.02073"},"observation_digest":"sha256:15a2050e962ffbeaf923bd652032cd3eec391f43c96172e062e6142fe3bdc49c","observation_id":"bdbc48e4-9afb-4b04-b7c2-854017371b57","resolution":{"observed_at":"2026-05-13T21:53:19.976365Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2604.11095","last_updated":"2026-04-13T07:12:12Z","snapshot_observed_at":"2026-07-06T22:59:36.571641Z","submitted_at":"2026-04-13T07:12:12Z","title":"Bottleneck Tokens for Unified Multimodal Retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T15:14:00.615638Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.11095"},"observation_digest":"sha256:89c3c8b0a8c14ccec8f497ac6116d3be13e2e0404b11f0b0585a85d635a73f44","observation_id":"f741e51f-173c-49fe-adad-56c022dd8aa5","resolution":{"observed_at":"2026-05-11T11:01:03.405897Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2604.12148","last_updated":"2026-04-13T23:54:58Z","snapshot_observed_at":"2026-08-03T13:03:27.832426Z","submitted_at":"2026-04-13T23:54:58Z","title":"ViLL-E: Video LLM Embeddings for Retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T15:00:43.573409Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.12148"},"observation_digest":"sha256:f8bd534e497843f6bdd9df2c48dc1e553b05056be8a243b9cb8cf4c69159278f","observation_id":"7872be79-f826-4695-9471-b56e8ceac2c5","resolution":{"observed_at":"2026-05-11T11:21:02.017173Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2604.13710","last_updated":"2026-05-09T08:14:59Z","snapshot_observed_at":"2026-08-03T00:43:33.977595Z","submitted_at":"2026-04-15T10:39:42Z","title":"SLQ: Bridging Modalities via Shared Latent Queries for Retrieval with Frozen MLLMs","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T14:03:23.807710Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.13710"},"observation_digest":"sha256:f686e0a5ff3e57ca99a6c9049b70f603547f78a45e1e46d74bd87bbaaf0e8d06","observation_id":"edd7da31-3353-4b1b-ab14-541ecba97f9e","resolution":{"observed_at":"2026-05-10T14:05:29.280756Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2604.13710","last_updated":"2026-05-09T08:14:59Z","snapshot_observed_at":"2026-08-03T00:43:33.977595Z","submitted_at":"2026-04-15T10:39:42Z","title":"SLQ: Bridging Modalities via Shared Latent Queries for Retrieval with Frozen MLLMs","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T00:54:25.022963Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.13710"},"observation_digest":"sha256:e175801d1acb57af913a2bcc08e13d1a3e11596c9599e4741e39bae5c4a5594a","observation_id":"f8fee9d3-182e-4f19-b4e4-531bd55d0151","resolution":{"observed_at":"2026-05-12T00:56:14.404261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2604.22280","last_updated":"2026-07-15T17:09:27Z","snapshot_observed_at":"2026-08-02T15:40:27.791487Z","submitted_at":"2026-04-24T06:50:11Z","title":"Beyond Chain-of-Thought: Rewrite as a Universal Interface for Generative Multimodal Embeddings","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-08T12:46:30.827346Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.22280"},"observation_digest":"sha256:1b0b3685e183069496d17ef6fdcc66336aad479f869a8891296a9b07d6a1a935","observation_id":"817840ff-853c-410a-8776-972f69a02598","resolution":{"observed_at":"2026-05-11T19:06:08.280546Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-12T18:31:37.144968Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.22280","last_updated":"2026-07-15T17:09:27Z","snapshot_observed_at":"2026-08-02T15:40:27.791487Z","submitted_at":"2026-04-24T06:50:11Z","title":"Beyond Chain-of-Thought: Rewrite as a Universal Interface for Generative Multimodal Embeddings","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-12T18:31:37.144968Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.22280"},"observation_digest":"sha256:5e44f02a0bc55e167ac3b590a1b3065550aad338dc7fcc43d0e852c48917626c","observation_id":"3a35c670-b516-4a38-b881-f27ea1a16359","resolution":{"observed_at":"2026-07-12T18:31:37.144968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-02T15:40:29.196574Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.22280","last_updated":"2026-07-15T17:09:27Z","snapshot_observed_at":"2026-08-02T15:40:27.791487Z","submitted_at":"2026-04-24T06:50:11Z","title":"Beyond Chain-of-Thought: Rewrite as a Universal Interface for Generative Multimodal Embeddings","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T15:40:29.196574Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.22280"},"observation_digest":"sha256:6ce68fc0f07e3ff513df5cca013e52208c04113154b212d25a1b9c75b02b3c09","observation_id":"a727a064-ce40-48d5-a21b-5c6cda706a59","resolution":{"observed_at":"2026-08-02T15:40:29.196574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2604.23321","last_updated":"2026-07-29T16:19:23Z","snapshot_observed_at":"2026-08-02T15:37:56.825812Z","submitted_at":"2026-04-25T14:15:05Z","title":"MMEB-V3: Measuring the Performance Gaps of Omni-Modality Embedding Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-08T07:28:09.725591Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.23321"},"observation_digest":"sha256:53307b7348489c29f896daec3e2a4677bf3a1b60b0d1e46d2e05b06f9e4c1103","observation_id":"be6b8f37-bd82-45f4-b0a1-eb69e0d897c0","resolution":{"observed_at":"2026-05-11T21:01:11.445346Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-02T15:38:00.262447Z","title":"Mm-embed: Universal multimodal retrieval with multimodal llms, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.23321","last_updated":"2026-07-29T16:19:23Z","snapshot_observed_at":"2026-08-02T15:37:56.825812Z","submitted_at":"2026-04-25T14:15:05Z","title":"MMEB-V3: Measuring the Performance Gaps of Omni-Modality Embedding Models","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-02T15:38:00.262447Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2604.23321"},"observation_digest":"sha256:d24791e069b1dbfbf3984527a4b6fc85b581f9b1490404b0dba22a2c5179afad","observation_id":"d30ecb05-8785-4015-a30b-1b2b6393d385","resolution":{"observed_at":"2026-08-02T15:38:00.262447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2605.13277","last_updated":"2026-05-13T09:54:31Z","snapshot_observed_at":"2026-07-31T16:56:58.325856Z","submitted_at":"2026-05-13T09:54:31Z","title":"Utility-Oriented Visual Evidence Selection for Multimodal Retrieval-Augmented Generation","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-14T19:09:18.975682Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2605.13277"},"observation_digest":"sha256:8a56d5c426f37c336fa6f965e3a9e166d388fc1516305cbbbe93a8cdae168270","observation_id":"ba782c43-126d-40dd-91b5-65a3ff2c48c8","resolution":{"observed_at":"2026-05-14T19:09:22.929992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2605.16638","last_updated":"2026-05-15T21:10:56Z","snapshot_observed_at":"2026-08-03T04:32:55.741433Z","submitted_at":"2026-05-15T21:10:56Z","title":"TTE-Flash: Accelerating Reasoning-based Multimodal Representations via Think-Then-Embed Tokens","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T18:00:18.315737Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2605.16638"},"observation_digest":"sha256:b88766e3e463e32b7af0312d5786c5fa06713c2d2098782cb88d299c0b48abad","observation_id":"fefd53b2-54b2-4e8c-88e6-cfe8f9e72098","resolution":{"observed_at":"2026-05-20T18:03:36.966857Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2605.18434","last_updated":"2026-05-18T14:07:20Z","snapshot_observed_at":"2026-08-03T18:04:01.369059Z","submitted_at":"2026-05-18T14:07:20Z","title":"TIGER-FG: Text-Guided Implicit Fine-Grained Grounding for E-commerce Retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-19T23:54:32.646978Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2605.18434"},"observation_digest":"sha256:3371cd04e16b34d9646bfa7403acc52c862d9841880d886f57cf716a8b85db09","observation_id":"36421437-351c-4ded-bc38-5d7fe16c31fe","resolution":{"observed_at":"2026-05-19T23:57:53.255359Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2605.24530","last_updated":"2026-05-23T11:48:28Z","snapshot_observed_at":"2026-08-07T01:21:42.748527Z","submitted_at":"2026-05-23T11:48:28Z","title":"Unveil: Unified Visual-Textual Integration and Distillation for Multi-modal Document Retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-30T13:17:04.441743Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2605.24530"},"observation_digest":"sha256:19646e54c37cb6b3439ef6655b446cc63535a0ba7ba51e9232db25e688dd9f43","observation_id":"68e3f404-c0df-4047-b61b-98556488ebec","resolution":{"observed_at":"2026-06-30T13:24:40.415138Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2605.27295","last_updated":"2026-05-26T17:07:55Z","snapshot_observed_at":"2026-08-03T02:31:44.388877Z","submitted_at":"2026-05-26T17:07:55Z","title":"Gemini Embedding 2: A Native Multimodal Embedding Model from Gemini","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T18:23:30.681253Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2605.27295"},"observation_digest":"sha256:9052a3fdd46ee76be39e8670f747e3df652aa840c25e64e14ba10a8dc60899b4","observation_id":"dc643c07-c8c5-47a8-a898-d3d1f5ace4d0","resolution":{"observed_at":"2026-06-29T18:23:50.323587Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2605.29606","last_updated":"2026-05-28T08:42:21Z","snapshot_observed_at":"2026-08-07T08:04:07.008644Z","submitted_at":"2026-05-28T08:42:21Z","title":"HiKEY: Hierarchical Multimodal Retrieval for Open-Domain Document Question Answering","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T07:39:14.379668Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2605.29606"},"observation_digest":"sha256:2f4147ce5f5098c71f95a336b37e36f77b63ec7414e537dffb7aacfb3c0e373d","observation_id":"05bd93df-79c5-445d-a3e1-8e26b5cc654e","resolution":{"observed_at":"2026-06-29T07:43:13.821615Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2605.30027","last_updated":"2026-05-28T14:50:53Z","snapshot_observed_at":"2026-08-06T11:47:39.586448Z","submitted_at":"2026-05-28T14:50:53Z","title":"DocRetriever: A Plug-and-Play Framework for Multimodal Document Retrieval with Comprehensive Benchmark","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T08:09:41.068000Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2605.30027"},"observation_digest":"sha256:f4e4ad376d4fbaccaebeaf712cc51fa85f6fbeb06e1b4d4d01c7a3cd84fe3540","observation_id":"0a1b5cb4-a5ae-4ee1-b253-7cab5ddb9914","resolution":{"observed_at":"2026-06-29T08:13:15.090280Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2606.12215","last_updated":"2026-06-10T15:29:45Z","snapshot_observed_at":"2026-07-06T23:51:13.191639Z","submitted_at":"2026-06-10T15:29:45Z","title":"MLT-Dedup: Efficient Large-Scale Online Video Deduplication via Multi-Level Representations and Spatial-Temporal Matching","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T09:43:22.054789Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2606.12215"},"observation_digest":"sha256:9386c162031f024c3eca6cf63f031240ffe79868c82458d46808b911f253fc54","observation_id":"60feab1f-1f0d-4d6c-aa5e-9c5c62c4187f","resolution":{"observed_at":"2026-07-03T10:58:03.542138Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2606.13141","last_updated":"2026-06-11T10:05:49Z","snapshot_observed_at":"2026-07-06T23:51:55.044440Z","submitted_at":"2026-06-11T10:05:49Z","title":"Rethinking RAG in Long Videos: What to Retrieve and How to Use It?","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T06:30:33.428489Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2606.13141"},"observation_digest":"sha256:ad0382e89208c4231d7cc809c042f3a6e57f672e195f334925dce2a20085f8b7","observation_id":"c124fc11-5e34-4994-a06c-07c25adecec3","resolution":{"observed_at":"2026-07-03T15:28:34.021407Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2606.20280","last_updated":"2026-07-04T09:32:50Z","snapshot_observed_at":"2026-07-12T13:16:44.503259Z","submitted_at":"2026-06-18T14:23:23Z","title":"ELVA: Exploring Ranking-Driven Universal Multimodal Retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-26T15:34:55.062016Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2606.20280"},"observation_digest":"sha256:10d34db61b804da5727933e9f195b478f42c2a5bfddbfce2cba59cda505e7f5f","observation_id":"b99378b5-4ddd-4d88-9031-cb92beb5cea5","resolution":{"observed_at":"2026-07-04T05:49:36.643419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2606.24094","last_updated":"2026-06-23T03:20:09Z","snapshot_observed_at":"2026-08-06T22:07:11.614337Z","submitted_at":"2026-06-23T03:20:09Z","title":"Universal Guideline-Driven Image Clustering via a Hybrid LLM Agent","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-26T01:19:19.253076Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2606.24094"},"observation_digest":"sha256:efa1927238b812cc5959dfff7852d78667b50d19d1d86c689616dccca7ed9e9d","observation_id":"0e40d168-2c80-4f17-9044-2d424136352b","resolution":{"observed_at":"2026-07-04T15:49:57.633180Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.02571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-07-04T15:49:57.631122Z","title":"arXiv preprint arXiv:2411.02571 , year=","venue":null,"work_id":"da61907d-fb9d-4f7b-a9bc-4750420fb05e","year":2024},"citing_paper":{"arxiv_id":"2606.28329","last_updated":"2026-05-19T15:18:23Z","snapshot_observed_at":"2026-07-07T00:02:29.000757Z","submitted_at":"2026-05-19T15:18:23Z","title":"$M^3 QuestionIng$: Multi-modal Multi-span Medical Question Answering","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-30T18:09:46.506953Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2606.28329"},"observation_digest":"sha256:9da6bbb441d2c37524b3f6737224e804edea328a986b84fe27d6639e34b4088c","observation_id":"7212669b-c4ee-4a8d-b674-f06b960e0043","resolution":{"observed_at":"2026-06-30T18:15:00.027444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-03T00:35:21.397349Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28751","last_updated":"2026-07-30T18:18:03Z","snapshot_observed_at":"2026-08-07T20:26:31.560997Z","submitted_at":"2026-07-30T18:18:03Z","title":"ReLoop-UME: Recurrent Depth with Learnable Retrieval Registers for Universal Multimodal Embedding","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-03T00:35:21.397349Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2607.28751"},"observation_digest":"sha256:7cc71dbbc22fb6d8fba953353cbb31d397bcdd6dbdcef665dfbcade629e44068","observation_id":"4d40886e-5dfc-4178-86e3-e3a93d8769e7","resolution":{"observed_at":"2026-08-03T00:35:21.397349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2411.02571/citation-record","integrity":"/paper/2411.02571/integrity","json":"/paper/2411.02571/citation-record.json","paper":"/paper/2411.02571"},"outbound":[],"paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-05T15:40:56.242909Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2411.02571."}