{"as_of":"2026-08-12T18:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5866f485378484baf5dcfe63f03e9593a2a60efb748878452af3db4918c9651c","coverage":[{"denominator":63,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":63,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T15:20:47.304539Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T19:31:53.371412Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-10T22:50:50.048388Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"cited_work":{"arxiv_id":"2412.11087","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.11087","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5837f9d8-33f1-4fcb-86c6-df24c88f3c4f","year":2024},"citing_paper":{"arxiv_id":"2604.05583","last_updated":"2026-04-07T08:23:39Z","snapshot_observed_at":"2026-07-30T03:36:30.860630Z","submitted_at":"2026-04-07T08:23:39Z","title":"WRF4CIR: Weight-Regularized Fine-Tuning Network for Composed Image Retrieval","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T19:31:53.371412Z"},"links":{"cited_paper":"/paper/2412.11087","citing_paper":"/paper/2604.05583"},"observation_digest":"sha256:9a11b7e7fd4413cb7443a78bfe1481dff225edbbc45864a2de62ebcfe2d609f9","observation_id":"068c18a3-4b06-4d85-a674-58f6890816df","resolution":{"observed_at":"2026-05-10T22:50:50.051194Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.11087/citation-record","integrity":"/paper/2412.11087/integrity","json":"/paper/2412.11087/citation-record.json","paper":"/paper/2412.11087"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.072504Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.072504Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:64c2b9adbf7999a35c50deb5f1e433ea60575aa4d175b9ad40e6d61a63f371ee","observation_id":"5fb24540-56e8-4360-9f8f-e6986dcef930","resolution":{"observed_at":"2026-08-11T15:20:47.072504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-11T15:20:47.077372Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.077372Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:7de755d41278db74444569cd8d0e7fad01a5fc8539dc8653261f97ca9e4f9cf1","observation_id":"a9f5b0d6-ec8b-466c-a352-abc8c34e8974","resolution":{"observed_at":"2026-08-11T15:20:47.077372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05473","last_updated":"2023-10-09T07:31:44Z","snapshot_observed_at":"2026-07-06T16:29:39.982733Z","submitted_at":"2023-10-09T07:31:44Z","title":"Sentence-level Prompts Benefit Composed Image Retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05473","snapshot_observed_at":"2026-08-11T15:20:47.081805Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.081805Z"},"links":{"cited_paper":"/paper/2310.05473","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:7181d2985ae539d2fc1bd5fa35f41c029c03cbbaad3c87988df2060fb0ad8745","observation_id":"727a0145-d411-48a7-ad65-39c497ca8cbf","resolution":{"observed_at":"2026-08-11T15:20:47.081805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.977409Z","title":null,"venue":null,"work_id":"9ce1bc0b-7524-4193-8e70-b20ac3115d9a","year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.086242Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:7b9b08bce1e566993e26ce7dc2ea0998f45c426af520169707422be6eb717312","observation_id":"064b6ace-4224-4561-be2d-2e86603ad21f","resolution":{"observed_at":"2026-08-11T15:20:47.980959Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.966556Z","title":"L.; Berg, A","venue":null,"work_id":"d7e5c857-465c-4bf9-9449-d42c361dfb53","year":2010},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.089784Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:9224e6669b3c4d3e75e919e3ed64117f9b55c1c7d4a5cd81dc0fd51864018bf1","observation_id":"dd9be402-77fe-4010-b660-7eeeec471912","resolution":{"observed_at":"2026-08-11T15:20:47.969858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.093309Z","title":"D.; Dhariwal, P.; Neelakantan, A.; Shyam, P.; Sastry, G.; Askell, A.; et al","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.093309Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:e8f5e15dc641c137a43ea5466e325a494844d67b1fc4e8f274d9f0b88d5d3466","observation_id":"6ce4b15a-ff37-4188-a439-13b8ca563391","resolution":{"observed_at":"2026-08-11T15:20:47.093309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.946669Z","title":null,"venue":null,"work_id":"b91b53bb-7ee7-4a4a-bea1-d4d47b71353d","year":2020},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.097963Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:61b3344606e2817a4c60f85c0be349ee618237bbda62f01b3decf28709f134ef","observation_id":"ea39360f-46e6-4b14-b7ae-72b02ed00b32","resolution":{"observed_at":"2026-08-11T15:20:47.951530Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.101264Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.101264Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:404b4e4374fb32810fde1c54249e1420880c5a8e00a5960dc305971235d58692","observation_id":"9249f51c-0d35-48e9-a670-ef8862423a9a","resolution":{"observed_at":"2026-08-11T15:20:47.101264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-08-11T15:20:47.104874Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.104874Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:3b4e65d53fe2abdd15ca5d47879ac0829b7aa1dd59d85e0278653952697efe31","observation_id":"92e8164d-392a-4a45-8e4b-4b200b236fe0","resolution":{"observed_at":"2026-08-11T15:20:47.104874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.08101","last_updated":"2022-05-16T15:20:04Z","snapshot_observed_at":"2026-08-09T05:12:05.320269Z","submitted_at":"2022-03-15T17:29:20Z","title":"ARTEMIS: Attention-based Retrieval with Text-Explicit Matching and Implicit Similarity","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.08101","snapshot_observed_at":"2026-08-11T15:20:47.108405Z","title":"S.; Csurka, G.; and Larlus, D","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.108405Z"},"links":{"cited_paper":"/paper/2203.08101","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:fb114d0b515043fe41c7261806ef4886b9a7774118eb7d1f6ba5c17b65826fc1","observation_id":"631d5c9e-d0aa-439e-985b-88480d072afb","resolution":{"observed_at":"2026-08-11T15:20:47.108405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.926879Z","title":"S.; Shlens, J.; Bengio, S.; Dean, J.; Ranzato, M.; and Mikolov, T","venue":null,"work_id":"14e94069-ca87-4464-946e-f23eace12ad1","year":2013},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.112664Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:a15ebfa46392da2db74f58d3d76f9eea97f733254b4e856590cf78149d9336c6","observation_id":"4973a0fb-8a17-47c5-bdf0-48ff6745052e","resolution":{"observed_at":"2026-08-11T15:20:47.931167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.01618","last_updated":"2022-08-02T17:50:36Z","snapshot_observed_at":"2026-08-02T23:40:32.342515Z","submitted_at":"2022-08-02T17:50:36Z","title":"An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.01618","snapshot_observed_at":"2026-08-11T15:20:47.116387Z","title":"H.; Chechik, G.; and Cohen-Or, D","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.116387Z"},"links":{"cited_paper":"/paper/2208.01618","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:763b48663c0cf00f3e5e8d242ab95a75829c838d47196371c8a7b7cd82c6edfe","observation_id":"27f9bddc-f97c-4171-8b23-0714a9ce819d","resolution":{"observed_at":"2026-08-11T15:20:47.116387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.913673Z","title":null,"venue":null,"work_id":"1ae2b358-a5a4-463b-adc8-56071b0fe112","year":2020},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.120566Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:dd15561d4ec4a12457f13bbdaf40db5b493537b98b20b16c2d45b4fcb2fce40f","observation_id":"c1974dd6-5e0e-46f7-afe5-c803db7f9efe","resolution":{"observed_at":"2026-08-11T15:20:47.917791Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.902367Z","title":null,"venue":null,"work_id":"7b8e7bab-b8b8-41c4-b7cd-424773ab6c69","year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.124246Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:985d46a683451d7e1090f256740e77d0504e3e5eb4c1726536ffb788d3d5b65b","observation_id":"5b237d09-9245-4d6b-80fb-235cab2dcb04","resolution":{"observed_at":"2026-08-11T15:20:47.906077Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.892029Z","title":null,"venue":null,"work_id":"af053a29-d4c5-4456-83d2-8c2835015237","year":2016},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.128141Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:a50f4eb206c26e62d6f69679318a552d64280ebe76f88fc6171c3b6f2a02ae20","observation_id":"fce0905e-7f11-4e6f-9f28-6774c2fd175e","resolution":{"observed_at":"2026-08-11T15:20:47.895282Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.881130Z","title":null,"venue":null,"work_id":"60f74fc5-2ec7-4c9c-a0ae-2bf990081b1c","year":2018},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.131595Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:8451fea9762c267069ae75693b7e0ad1ea1611aed1eb282756231aa90dec4ded","observation_id":"ef0a1cfb-6a81-498f-aaa9-de20bb8b4924","resolution":{"observed_at":"2026-08-11T15:20:47.884739Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.869669Z","title":null,"venue":null,"work_id":"841695fd-38d3-40eb-af58-45332cfc60aa","year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.135577Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:5adf80a34e909f8b1423ea443a8a46827d3a31aab773dcb44c7be54658cf6135","observation_id":"8c8290c0-af88-45f2-abd0-a459b4ecd52a","resolution":{"observed_at":"2026-08-11T15:20:47.873396Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-11T08:20:29.798517Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-11T15:20:47.139250Z","title":"J.; Shen, Y.; Wallis, P.; Allen-Zhu, Z.; Li, Y.; Wang, S.; Wang, L.; and Chen, W","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.139250Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:5c5ee8316315efc03a4644aab83e681ccf7da0b6a9ee0b761a96e4711bd4b377","observation_id":"4c176027-f57b-43fe-9489-4e11646f569c","resolution":{"observed_at":"2026-08-11T15:20:47.139250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.142767Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.142767Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:351ec79afac125022446636416eabf0ae127ac211ac35160813e9ac2d4db1a1a","observation_id":"cc39824b-a43e-44aa-8c39-2eeb3f6eaa6f","resolution":{"observed_at":"2026-08-11T15:20:47.142767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.145792Z","title":"A.; and Manning, C","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.145792Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:54f24aab8be44b9ffa1a630fc2d2dce6a1a150596cf383a900edfa184db27bd0","observation_id":"509bffd9-bbe0-425d-a872-31402c8ee968","resolution":{"observed_at":"2026-08-11T15:20:47.145792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.08484","last_updated":"2022-03-15T01:52:38Z","snapshot_observed_at":"2026-08-10T02:40:19.998968Z","submitted_at":"2021-10-16T06:07:59Z","title":"A Good Prompt Is Worth Millions of Parameters: Low-resource Prompt-based Learning for Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.08484","snapshot_observed_at":"2026-08-11T15:20:47.148781Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.148781Z"},"links":{"cited_paper":"/paper/2110.08484","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:24a355c6187e0884fb92f56dfd9d2386f98efb24a9c27a338be546e2be3f610f","observation_id":"dab740d2-3bb6-4637-9ecb-e65e974755ef","resolution":{"observed_at":"2026-08-11T15:20:47.148781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09291","last_updated":"2024-02-26T18:59:49Z","snapshot_observed_at":"2026-08-12T09:56:51.055987Z","submitted_at":"2023-10-13T17:59:38Z","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.09291","snapshot_observed_at":"2026-08-11T15:20:47.152427Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.152427Z"},"links":{"cited_paper":"/paper/2310.09291","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:fbe757046094bf3c08cd2a1258f6f9495c963876cb94b4ec19fefc0c7709fda5","observation_id":"0b0799d0-d39e-4c7e-8208-1ab26c86643a","resolution":{"observed_at":"2026-08-11T15:20:47.152427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.843085Z","title":null,"venue":null,"work_id":"72c68571-4f96-498e-bb13-fe247ce1fcbe","year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.155913Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:b5661483385d54cd8e06e23df16b85d2d38437fa92bf930b4ff85b3e76adbb6b","observation_id":"b82718db-b4c4-4818-91ec-5d3d25385df1","resolution":{"observed_at":"2026-08-11T15:20:47.847330Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-11T15:20:47.159571Z","title":"P.; and Ba, J","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.159571Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:7f61a6b2ec50c89b273ef4d1f1f60930a6654926bcdd1c5fae8cccd86e096660","observation_id":"7d86ae43-c7fc-479b-bc15-e9d1593981ae","resolution":{"observed_at":"2026-08-11T15:20:47.159571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.831454Z","title":null,"venue":null,"work_id":"886bb5bb-02d0-44e8-a43c-b231321cd1c5","year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.164035Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:a852bc1173495f0d91c500afd25145a2bda196a870a892f162d6ebcb5c2cbd74","observation_id":"91f68c5f-8c88-4ee8-960a-f9b9dc32b077","resolution":{"observed_at":"2026-08-11T15:20:47.835166Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.09429","last_updated":"2023-12-20T11:07:57Z","snapshot_observed_at":"2026-08-09T06:35:51.544751Z","submitted_at":"2023-03-16T16:02:24Z","title":"Data Roaming and Quality Assessment for Composed Image Retrieval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.09429","snapshot_observed_at":"2026-08-11T15:20:47.167708Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.167708Z"},"links":{"cited_paper":"/paper/2303.09429","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:e99e63558aa6443bb475526ce1b9d0cc5d85a161abc19b11f0f43127601d4081","observation_id":"0132dc7e-1cec-41a4-9c4e-fdfcc16c05ad","resolution":{"observed_at":"2026-08-11T15:20:47.167708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16125","last_updated":"2023-08-02T08:02:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-30T04:25:16Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16125","snapshot_observed_at":"2026-08-11T15:20:47.171646Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.171646Z"},"links":{"cited_paper":"/paper/2307.16125","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:0613021e040eda483a8cffc190b860735316c6b7f5ff45b9859488c166effa6e","observation_id":"69f869d3-b441-4c75-9e42-c5642a8396c0","resolution":{"observed_at":"2026-08-11T15:20:47.171646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.175513Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.175513Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:83ff5015dd89f539ca3a090831dcfec176fc6d90de6736683f6978df3e12e505","observation_id":"a65b9e6d-9794-4c1b-8b79-7825e177d0e4","resolution":{"observed_at":"2026-08-11T15:20:47.175513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.179120Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.179120Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:8ede6fc41257e46c2f651056f573634922b3a93e6e3090a536df072eee15eba2","observation_id":"041e9df5-d825-44df-8817-753de58a5356","resolution":{"observed_at":"2026-08-11T15:20:47.179120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10355","last_updated":"2023-10-26T02:52:40Z","snapshot_observed_at":"2026-08-12T18:48:30.326248Z","submitted_at":"2023-05-17T16:34:01Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10355","snapshot_observed_at":"2026-08-11T15:20:47.183013Z","title":"X.; and Wen, J.-R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.183013Z"},"links":{"cited_paper":"/paper/2305.10355","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:d21398aa24445d06eedd3c430308cd046c880330f3df4ed0392a6e5f64fb2670","observation_id":"894eadbe-dadf-4997-9037-3ca42f8f56fe","resolution":{"observed_at":"2026-08-11T15:20:47.183013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-11T15:20:47.186912Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.186912Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:0841b9124ba7430d61a2ed6d4ac0728f948dfa653bf3b9b7b41d098baf02e147","observation_id":"09255caa-b43c-4572-b1b4-756ea2c990c7","resolution":{"observed_at":"2026-08-11T15:20:47.186912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.06281","last_updated":"2024-08-20T03:56:03Z","snapshot_observed_at":"2026-07-06T15:53:19.485466Z","submitted_at":"2023-07-12T16:23:09Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.06281","snapshot_observed_at":"2026-08-11T15:20:47.190182Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.190182Z"},"links":{"cited_paper":"/paper/2307.06281","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:4ae7966abf9966a2c4f55ed07eb73fb0af6414632bec5780e537784627e5e86a","observation_id":"db129af9-232b-421b-a51c-ffc2462569ea","resolution":{"observed_at":"2026-08-11T15:20:47.190182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.803199Z","title":null,"venue":null,"work_id":"4b946e1f-60d6-44e4-95ed-7885abd1159a","year":2016},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.193804Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:19eddeddfe8e268b8c11443e24729f5127315f2ac85718ce1e0053c96e1c2b5d","observation_id":"9c6279a2-7dbd-4356-a9fd-1a300c492206","resolution":{"observed_at":"2026-08-11T15:20:47.807648Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.792425Z","title":null,"venue":null,"work_id":"0d1eb3b6-abeb-4a42-a947-6e43c8cbb537","year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.196669Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:5a7a508f26f1cc3e832697dca42515dd649933d24f93bbb13df957512474f641","observation_id":"0d246f28-a1f9-487b-a393-eb2c7b077dff","resolution":{"observed_at":"2026-08-11T15:20:47.795602Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16304","last_updated":"2024-01-29T05:03:55Z","snapshot_observed_at":"2026-08-01T20:24:08.035012Z","submitted_at":"2023-05-25T17:56:24Z","title":"Candidate Set Re-ranking for Composed Image Retrieval with Dual Multi-modal Encoder","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16304","snapshot_observed_at":"2026-08-11T15:20:47.199484Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.199484Z"},"links":{"cited_paper":"/paper/2305.16304","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:1adc88fae7d3c34b68b383ed1bfb7e10168c89fac2cee4f8617f7524379aa729","observation_id":"822d1bec-c915-4244-88cb-fa5d1786eab1","resolution":{"observed_at":"2026-08-11T15:20:47.199484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08319","last_updated":"2023-10-12T13:32:35Z","snapshot_observed_at":"2026-07-06T16:31:52.204065Z","submitted_at":"2023-10-12T13:32:35Z","title":"Fine-Tuning LLaMA for Multi-Stage Text Retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08319","snapshot_observed_at":"2026-08-11T15:20:47.202636Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.202636Z"},"links":{"cited_paper":"/paper/2310.08319","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:ada92a278b7a7d7f17816be41261642ffe75eb212f5dd9d56206784895bad9ff","observation_id":"2beb4b4a-b894-4e0e-b389-5f4ab2c24db5","resolution":{"observed_at":"2026-08-11T15:20:47.202636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.206294Z","title":"K.; and Chakraborty, A","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.206294Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:f08bb28b6e6cd39723c914b2617274cd0ada6447189169876218744c9864d5f3","observation_id":"d5a39774-fd91-441e-9753-a11741d1c624","resolution":{"observed_at":"2026-08-11T15:20:47.206294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.08904","last_updated":"2022-08-05T09:33:10Z","snapshot_observed_at":"2026-08-11T07:35:49.888577Z","submitted_at":"2022-02-17T21:35:56Z","title":"SGPT: GPT Sentence Embeddings for Semantic Search","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.08904","snapshot_observed_at":"2026-08-11T15:20:47.209922Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.209922Z"},"links":{"cited_paper":"/paper/2202.08904","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:70ff75d43b80230ee2ecc4446ba2ffda393a7dc649d6f6e2d7352352fbfd82ed","observation_id":"8cbbff1d-dbc6-4e95-84c2-f9d74f256452","resolution":{"observed_at":"2026-08-11T15:20:47.209922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.09906","last_updated":"2025-03-03T04:28:49Z","snapshot_observed_at":"2026-07-06T17:30:34.153488Z","submitted_at":"2024-02-15T12:12:19Z","title":"Generative Representational Instruction Tuning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.09906","snapshot_observed_at":"2026-08-11T15:20:47.213998Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.213998Z"},"links":{"cited_paper":"/paper/2402.09906","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:d1f4a019c21952909ebd7de3a9a95fa56ca7e9bff4e0976ad76861c6b196135d","observation_id":"7a2ecf54-2ea2-4fca-9c19-b297e9ac1ba9","resolution":{"observed_at":"2026-08-11T15:20:47.213998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.217824Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.217824Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:b2b5c74f52187592140c2e7e6ab0e58e6c95b8875e0fbc996167a9f174912887","observation_id":"205da4d2-3dce-411c-8f3e-daf13d0f8dc2","resolution":{"observed_at":"2026-08-11T15:20:47.217824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.221844Z","title":"W.; Hallacy, C.; Ramesh, A.; Goh, G.; Agarwal, S.; Sastry, G.; Askell, A.; Mishkin, P.; Clark, J.; et al","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.221844Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:613d2060f4e7bcb04d1a2c913964123cbce07ab1fbe84b2f69131ce0155cdef8","observation_id":"fdd48168-0ce0-4c6e-9570-987871562290","resolution":{"observed_at":"2026-08-11T15:20:47.221844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.760351Z","title":null,"venue":null,"work_id":"dff9e2ae-5954-4442-918c-8a9d902bdf7f","year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.225620Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:e8df42e7e8f16f16ea84fd89a863f2b5405a91674be38a3f1a9cb83a179ad543","observation_id":"95fa12b0-5b56-4198-9901-d81d658bcb4d","resolution":{"observed_at":"2026-08-11T15:20:47.763751Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.748653Z","title":"G.; Malinowski, M.; Pascanu, R.; Battaglia, P.; and Lillicrap, T","venue":null,"work_id":"32b61938-0807-4215-8b06-cf6dd181ead1","year":2017},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.229206Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:e9e005fb625c73cd228cdb3e46a113dda67c23e8aa039848a6e0c9e5738ab1cc","observation_id":"b0aaf17e-3454-4555-849a-b062c4180004","resolution":{"observed_at":"2026-08-11T15:20:47.753157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.15980","last_updated":"2020-11-07T05:33:35Z","snapshot_observed_at":"2026-08-05T14:17:34.026911Z","submitted_at":"2020-10-29T22:54:00Z","title":"AutoPrompt: Eliciting Knowledge from Language Models with Automatically Generated Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.15980","snapshot_observed_at":"2026-08-11T15:20:47.232840Z","title":"L.; Wallace, E.; and Singh, S","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.232840Z"},"links":{"cited_paper":"/paper/2010.15980","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:c6ba2334aff0f8d8904d81325433dbbe758a8be645ee564d9ab610907b384789","observation_id":"0c56815c-e770-4609-9eec-1287024e33ba","resolution":{"observed_at":"2026-08-11T15:20:47.232840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1811.00491","last_updated":"2019-07-21T05:26:36Z","snapshot_observed_at":"2026-08-01T22:56:26.162617Z","submitted_at":"2018-11-01T16:47:44Z","title":"A Corpus for Reasoning About Natural Language Grounded in Photographs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.00491","snapshot_observed_at":"2026-08-11T15:20:47.236367Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.236367Z"},"links":{"cited_paper":"/paper/1811.00491","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:d813ac0ddcdf8689009cd5534d6e421ab7dbc7dd0e86b959dad9031b1346f788","observation_id":"3ac7e409-e237-4800-83db-7be3e5894c67","resolution":{"observed_at":"2026-08-11T15:20:47.236367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08924","last_updated":"2024-03-24T14:23:59Z","snapshot_observed_at":"2026-07-06T17:01:40.744034Z","submitted_at":"2023-12-14T13:31:01Z","title":"Training-free Zero-shot Composed Image Retrieval with Local Concept Reranking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08924","snapshot_observed_at":"2026-08-11T15:20:47.239606Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.239606Z"},"links":{"cited_paper":"/paper/2312.08924","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:98f38d81aee3f98b52dec76c2cf66e78c7f76b104a9557cea6ca96ac32efa784","observation_id":"4fa180c8-e150-48d4-ad96-8008474f3364","resolution":{"observed_at":"2026-08-11T15:20:47.239606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.735582Z","title":null,"venue":null,"work_id":"dc8209bd-45de-46aa-9d63-93457122c60b","year":2024},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.243146Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:262b50939638958ec0bea64a0fb19df90064a5f470bc80eb96dd49f84df61e3f","observation_id":"3645044f-2070-4130-88db-9d6714d33017","resolution":{"observed_at":"2026-08-11T15:20:47.740836Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-11T15:20:47.246222Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.246222Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:05d4bf6858b0f820f56687fbc2a6da518739422a6d6eadce26923594c3539460","observation_id":"cb588802-6fb7-48b3-a0af-1a7328346ce8","resolution":{"observed_at":"2026-08-11T15:20:47.246222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.724010Z","title":null,"venue":null,"work_id":"824e5ecc-cd1c-4426-8ad3-9be89380321b","year":2019},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.250381Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:06e90df7c9b051b67b756449904c7018de63e8cae4ee439330e7610e38cdfef3","observation_id":"ab887efd-101c-41a6-9015-109824a3b76f","resolution":{"observed_at":"2026-08-11T15:20:47.727791Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15126","last_updated":"2023-10-10T11:57:26Z","snapshot_observed_at":"2026-08-10T10:45:21.733481Z","submitted_at":"2023-08-29T08:51:24Z","title":"Evaluation and Analysis of Hallucination in Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15126","snapshot_observed_at":"2026-08-11T15:20:47.254339Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.254339Z"},"links":{"cited_paper":"/paper/2308.15126","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:239f33b463c8c900938235842cd4f0819e42acbd7d8839d1d055690ddaf7f3eb","observation_id":"b799e0cf-2820-45d6-991e-a2dfde406b13","resolution":{"observed_at":"2026-08-11T15:20:47.254339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.258153Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.258153Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:2baa9669e70330d15dd2c41552bca66ff20d99c23e04fd6017cfbf48a14fe193","observation_id":"5b2309af-10e4-462b-9d65-9c5a6205a06b","resolution":{"observed_at":"2026-08-11T15:20:47.258153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.706061Z","title":null,"venue":null,"work_id":"11eb8093-b157-4929-9340-8fd7ab66dd3d","year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.261872Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:ed5cafb906feede69d0e0f8e14f83b9c96d7e5611dbb647e86fd48ecd598350f","observation_id":"60d3d41a-88a4-4690-937b-ebf4e679913a","resolution":{"observed_at":"2026-08-11T15:20:47.709440Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.695216Z","title":null,"venue":null,"work_id":"baf74f04-f8c9-4644-9b7a-431f20058227","year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.265736Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:3805e3e309feb27902cebdf08a7d9c38f66499324f800610ec5a93ecd3ebf606","observation_id":"51ad0a63-dcc2-4c5f-83cc-1caffdad4311","resolution":{"observed_at":"2026-08-11T15:20:47.698621Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.684903Z","title":null,"venue":null,"work_id":"4e784df1-5466-47be-9507-5307700d591a","year":2021},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.269897Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:d51e8ce8824a88f6cb58343ccf5f04b2e7fc0a5a5a33aac7934549289f1c1c98","observation_id":"804d1bc3-1f7f-4aec-9fbb-823628419a0f","resolution":{"observed_at":"2026-08-11T15:20:47.688103Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.673168Z","title":null,"venue":null,"work_id":"14572477-9250-41c9-9183-0ab43375f714","year":2024},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.273631Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:9bc618ba6e6f5ef15ef28b8ad814d388941a1aaad6d1d71d30ac09f853c8d899","observation_id":"6e473cf3-9d1f-45bf-a40b-2ad714ab883f","resolution":{"observed_at":"2026-08-11T15:20:47.677071Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02490","last_updated":"2024-12-01T05:46:03Z","snapshot_observed_at":"2026-08-08T03:31:37.699253Z","submitted_at":"2023-08-04T17:59:47Z","title":"MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.02490","snapshot_observed_at":"2026-08-11T15:20:47.276893Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.276893Z"},"links":{"cited_paper":"/paper/2308.02490","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:cd5d787714ce2f01c9ea336232a84fe6c95776e1d1e38b7e7486084b88640119","observation_id":"9c3ff6a2-6eba-485f-a0fa-a705fcdeb841","resolution":{"observed_at":"2026-08-11T15:20:47.276893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.07225","last_updated":"2022-10-13T17:50:24Z","snapshot_observed_at":"2026-08-09T12:39:06.378102Z","submitted_at":"2022-10-13T17:50:24Z","title":"Unified Vision and Language Prompt Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.07225","snapshot_observed_at":"2026-08-11T15:20:47.281351Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.281351Z"},"links":{"cited_paper":"/paper/2210.07225","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:5d53c69bb448b505700ac47a555c7b88ac0caf248c917475340230daae106776","observation_id":"5004d9f0-efa5-4d6c-956a-19b78d0446eb","resolution":{"observed_at":"2026-08-11T15:20:47.281351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.11212","last_updated":"2022-04-24T08:10:06Z","snapshot_observed_at":"2026-08-04T13:46:56.686917Z","submitted_at":"2022-04-24T08:10:06Z","title":"Progressive Learning for Image Retrieval with Hybrid-Modality Queries","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.11212","snapshot_observed_at":"2026-08-11T15:20:47.285162Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.285162Z"},"links":{"cited_paper":"/paper/2204.11212","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:e5459f38fdb65678aced30553f9f7c304912e92f0a3975b735dd3d1c3ff8041a","observation_id":"42aa9c8d-351b-4ef2-be75-b2fd0a4dc6b2","resolution":{"observed_at":"2026-08-11T15:20:47.285162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.288651Z","title":"C.; and Liu, Z","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.288651Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:7c77f8dffcff86ba269877d1688fcb8be17f57ce97ec90e62094612eb8456f0e","observation_id":"ddc6c121-5f93-40bb-b94f-f1e32b4e0bc6","resolution":{"observed_at":"2026-08-11T15:20:47.288651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-11T15:20:47.291989Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.291989Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:9caeb4b1496933da5f76c0e4d5abac927d3213c8d25572e8667c958c06c6d3dd","observation_id":"372f6699-56bc-4c03-8978-d3f62914bdb5","resolution":{"observed_at":"2026-08-11T15:20:47.291989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.654667Z","title":null,"venue":null,"work_id":"064cdc2c-9b96-492a-bb48-3cec7d88a0b6","year":2023},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.296031Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:42a74d58163098447cbb04a14b3936cc01de7ca526237e2f0273d04ba6800f51","observation_id":"ace799f6-7218-4536-8016-cce571eb8dbc","resolution":{"observed_at":"2026-08-11T15:20:47.659404Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.299661Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.299661Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:72efdc8d7ac82504d33c4461060b2f453b7fd3f2e4b9915db7582a8200d509ca","observation_id":"d25233e3-11bc-4165-b1de-a7de402374fb","resolution":{"observed_at":"2026-08-11T15:20:47.299661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:20:47.304539Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-11T15:20:47.304539Z"},"links":{"citing_paper":"/paper/2412.11087"},"observation_digest":"sha256:013d11196eb2a9bc5d72caf8a3318597b127d68a02757ddca78cef3877706934","observation_id":"22567bc8-5e40-4428-8c68-e7cac127861f","resolution":{"observed_at":"2026-08-11T15:20:47.304539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.11087","last_updated":"2024-12-15T07:09:02Z","latest_version":1,"primary_category":"cs.IR","snapshot_observed_at":"2026-08-11T19:43:25.170912Z","submitted_at":"2024-12-15T07:09:02Z","title":"Leveraging Large Vision-Language Model as User Intent-aware Encoder for Composed Image Retrieval"},"reference_resolution":{"displayed":63,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":60,"verified_exact":0,"verified_fuzzy":3},"total_outbound_references":63},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 63 of 63 outbound references and 1 inbound Pith citation observation for arXiv:2412.11087."}