{"as_of":"2026-08-07T00:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d356359a947bd9f91a97e0fb3660991b0b670f9eb0255c765ccf4c55eb7a52f4","coverage":[{"denominator":78,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":78,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T22:44:36.059582Z","state":"measured"},{"denominator":79,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":79,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T01:12:46.295455Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-03T20:38:56.147115Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"cited_work":{"arxiv_id":"2606.07032","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.07032","snapshot_observed_at":"2026-07-03T20:38:56.147115Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","venue":"cs.CV","work_id":"de29a1b9-7e72-469a-aa47-64ddd3ed1710","year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2606.07032","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:972a834571bfdf7c6bf98c7e418d4d3eacb0c820cca4792256e176b6881a6f63","observation_id":"3062e5b6-9dfd-4c90-876c-a06af031dc4d","resolution":{"observed_at":"2026-07-03T20:38:56.149055Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2606.07032/citation-record","integrity":"/paper/2606.07032/integrity","json":"/paper/2606.07032/citation-record.json","paper":"/paper/2606.07032"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Heterogeneous feature fusion and cross-modal alignment for composed image retrieval,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:c7e125c96b896ab55bb614e057e0a6b24ea75611f4581abca9fb10f08ded8421","observation_id":"5eb7b69d-7cea-4f65-b4b5-bf98e3f4e976","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Effective conditioned and composed image retrieval combining clip-based features,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:080821a409825f0a5ac2495631635de111dbed8e4700f56308212ea9f2880df8","observation_id":"1f39d5ed-8f5d-46c4-a85f-b6547350f427","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Target-guided composed image retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:c27d621df22ba11ceb3eb5bd137dcfec0756eb290956d901e118772ec93e945f","observation_id":"446c99c6-ef44-4505-83ae-9550143215d7","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Self-training boosted multi-factor matching network for composed image retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:57e0faf960e300c1c701715e8bacb4ef35749d7d18903afaf34547e654817fa5","observation_id":"f86d1e4c-ec94-49f9-8b56-91f138383314","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Pic2word: Mapping pictures to words for zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:8f82d9eb825af9386cba554c1c69aad15ba0c2f703db2bcd443ec44528fc8d86","observation_id":"098c753e-95f1-451a-aa96-33012075331a","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15247","last_updated":"2023-08-19T14:04:41Z","snapshot_observed_at":"2026-07-06T15:08:26.568788Z","submitted_at":"2023-03-27T14:31:25Z","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","version":2},"cited_work":{"arxiv_id":"2303.15247","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2303.15247","snapshot_observed_at":"2026-07-02T16:27:09.003857Z","title":"Zero- shot composed image retrieval with textual inversion,","venue":null,"work_id":"d7eddd38-25c3-4122-ba17-4ced245c953c","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2303.15247","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:503a86448a024f36ec4fe63212dff7f1b731eb395396074fec55fb83bfa23cae","observation_id":"c698c4d4-8b46-4981-baea-43f4377dd961","resolution":{"observed_at":"2026-07-02T16:27:09.005749Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16137","last_updated":"2023-12-15T09:32:37Z","snapshot_observed_at":"2026-08-02T09:09:00.390118Z","submitted_at":"2023-09-28T03:35:25Z","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","version":2},"cited_work":{"arxiv_id":"2309.16137","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16137","snapshot_observed_at":"2026-07-02T16:27:09.016427Z","title":"Context- i2w: Mapping images to context-dependent words for accurate zero-shot composed image retrieval,","venue":null,"work_id":"851483aa-f20f-4603-bef1-ce9c83c979e9","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2309.16137","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:f00cf69ae0d456bd85e644ece23589b2f849695c45ada6caa246b4c3a4e5ce16","observation_id":"811d0319-d9fc-425e-a80b-aeedb1bce59a","resolution":{"observed_at":"2026-07-02T16:27:09.017796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Image retrieval on real-life images with pre-trained vision-and-language models,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:520211a27dcef47ae83edbd23e07cd64538584debcbfc914e677140d55cd1a26","observation_id":"61df25ed-969e-491e-b733-e76afef4b15d","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Genecis: A benchmark for general con- ditional image similarity,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:d45222df900095fb90739c187d520a26039a67f865a5b829aa61873de96b7989","observation_id":"6c004942-a749-468e-b7d6-9f9e9f5b7529","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Fashion iq: A new dataset towards retrieving images by natural language feedback,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:0cbe5cdafb702ba3f79a88dda729cc0f4bb61088893377e38180a167a85d5f08","observation_id":"3fbfeff2-130c-4258-82aa-563179dc996a","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Data roaming and quality assessment for composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:a1741ee4c9d7ae68fc4b2666aeb92948c1081892034cd893cc6b4b0a5c8622db","observation_id":"6f368c76-baa5-4926-a2e3-c0cd4e035b9e","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Dual compositional learning in interactive image retrieval,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:048040baffaba27b3596ef021cbdf34ba3026b61e08e708d7d68f2d15d041812","observation_id":"13eba261-1956-4688-82c0-79d8b4230741","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Visual compositional learning for human-object interaction detection,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:d43906f65acb6ad883d3af4d0f333c4cee4393b12259f0be1f86021387b89d64","observation_id":"8e658440-3cd7-4d36-a488-ac70a1765f23","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Leveraging large vision-language model as user intent-aware encoder for composed image retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:94daa5c6a00c2a80ad72f0c72cff0c186e3b41da7682182fee209310615c97ff","observation_id":"35531127-c7e7-43c0-bc4c-5fcfe1a569be","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Ccin: Compositional conflict identification and neutralization for composed image retrieval","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:0a0755a2deb4143bdec97b001c9ef93de52b7bd3b3b3c67b8583866498382d8e","observation_id":"80abc98f-ad59-43c9-953e-697fe69a88f7","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Covr: Learning composed video retrieval from web video captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:04154c6ca53769bdd55ef0145724a1d742da0102a530ba1bcf3f48bc7bfe2954","observation_id":"223b5389-6303-4ff6-87a3-d48cbf8a8c98","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Bi-directional training for composed image retrieval via text prompt learning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:d673cc3d9c8a20ce2b36d1c423dcd3a18317372da7935be3ff67e25d2049bd1a","observation_id":"9a300e96-ae04-4fdc-bbea-44ed1606ac90","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Sentence-level prompts benefit composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:fac78feb8f32a09440ac8994f6e5322c62cddc3b632fc71ead4cb4eb2caf1333","observation_id":"4a57f7e5-dfe0-4d86-8bfc-d7cf4f10665d","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Dynamic bit-wise semantic transformer hashing for multi-modal retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:0aaa60a24bcf4c98e9326796ab384bea9b5c686f441a1081dd45695e162c5488","observation_id":"af264afd-b0bf-4239-a6dd-2b3daa670635","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Fashion retrieval via graph reasoning networks on a similarity pyramid,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:5603e628706a6f6c365c9079d87197a5993cd4e3502b397dd2b8c4c925096a6c","observation_id":"c70e21b3-44d3-4e86-9122-ce985088fa6e","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Multilin- gual text-to-image person retrieval via bidirectional relation reasoning and aligning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:e0fdc7520c2f2ada050aa3c1cb2d0fb122a2c4f6aa8330c877661610d84fbeb1","observation_id":"d0606636-f7bc-4fbf-924b-cb049d0fcb3e","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"A corpus of natural language for visual reasoning,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:7c03bf64d347fbdb88897e5b812f8c7b6b47fda86e3c28ebff8c85c709c0ef4b","observation_id":"45f5d826-d653-4e7f-a43c-75a1eb053701","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Microsoft coco: Common objects in context,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:5b82e6741abe1f181391d224d14abd46c153774d99636abbe594cc04e0e78d06","observation_id":"8b87c6a8-a8eb-46b3-a188-ca79681bc9e6","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Imagenet: A large-scale hierarchical image database,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:2472a29444121578fd8366908f039adeb291a019ced78f9df0037839c507d686","observation_id":"58dcc51d-ebf8-481b-9c47-3f1d29371f49","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"“this is my unicorn, fluffy","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:18fd7856a5f5add4b352287860e041c70e3642512ead033d8b78150e40d63249","observation_id":"02f864c4-5c23-4855-af8d-c9afd093d007","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01998","last_updated":"2024-03-31T22:58:09Z","snapshot_observed_at":"2026-07-06T16:56:38.112343Z","submitted_at":"2023-12-04T16:22:06Z","title":"Language-only Efficient Training of Zero-shot Composed Image Retrieval","version":2},"cited_work":{"arxiv_id":"2312.01998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.01998","snapshot_observed_at":"2026-07-02T16:27:08.988160Z","title":"Language-only efficient training of zero-shot composed image retrieval,","venue":null,"work_id":"925bb785-0637-41b3-92d8-8a0106d00006","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2312.01998","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:386daad0094073a611ff303d73c7c3690fbde5bacee37da10d84ca490e2d2a2f","observation_id":"90f49ea8-c0b9-492f-879b-1ade5371772b","resolution":{"observed_at":"2026-07-02T16:27:08.989476Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09291","last_updated":"2024-02-26T18:59:49Z","snapshot_observed_at":"2026-08-06T04:16:13.638840Z","submitted_at":"2023-10-13T17:59:38Z","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","version":2},"cited_work":{"arxiv_id":"2310.09291","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.09291","snapshot_observed_at":"2026-07-02T16:27:09.023654Z","title":"Vision-by-language for training-free com- positional image retrieval.arXiv preprint arXiv:2310.09291","venue":null,"work_id":"6ade8e03-ba6e-46ea-b01f-88fe15324fd9","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2310.09291","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:edad1d5b8893742ed067d62bb74de85014940457d0ad1e70df9a0fb30c57eb47","observation_id":"cac05558-6818-457b-b095-d2c2dfe52d60","resolution":{"observed_at":"2026-07-02T16:27:09.025375Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Ldre: Llm-based divergent reasoning and ensemble for zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:867fa92208e28aa6c731415c77f8587b66c7b96bdf12ccf9089437b89703879b","observation_id":"40b226c5-07cf-4ef0-93f1-ad464089aa88","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Semantic editing increment benefits zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:78183d8d801e24545df1468a10df85a650b2af051ff1771ef710f2a2d39c83bf","observation_id":"5e28f45c-55e3-42c7-8413-2a545eb79560","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Mllm-i2w: Harnessing multimodal large language model for zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:57c2635ce6e6a628b5de2275542aae52470b2c18f2c9d4b3af673431651b6fbb","observation_id":"8e8c9a2e-76f0-4497-9393-7f2a70470a7e","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Generative zero-shot composed image retrieval","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:6633ee2ca098fb84766afdc3743c0d2d66f6dc472ae6139984af4fa3a85ebe77","observation_id":"5d391179-443e-4514-9012-8a18f3bcd8ae","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Active supervised cross- modal retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:d11d399910017a30bc3bfdec2032a61b3479c07b79c72d0bf9dbe5b798fe2a3e","observation_id":"b1474ca5-80ca-45b4-b404-be260001f46e","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Prvr: Partially relevant video retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:4aec95e540279ddf9f362a465247b626a073a31de2e04e9a260ffc65b4f673f5","observation_id":"67a2aa83-5855-40ed-820d-2001c897fb77","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:5575e6f88081cd3b53a9f9eb7998b914cba79ba042b14396d841d2a0e9c28c8f","observation_id":"bfca25d9-5ee4-400d-81d8-d4350352e56b","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Laion- 5b: An open large-scale dataset for training next generation image-text models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:d42812945a060f1995652ea08ccc04707f1a14a8f607dd2ccb4d9ee864c902f7","observation_id":"4c693f89-9606-40b9-93d4-5df3c99a1f6a","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Datacomp: In search of the next generation of multimodal datasets,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:7e1098adced5ffed50f9eb7f01be3fa1729e9b48f972abde39771f3956a35add","observation_id":"59eccd9a-9721-4058-90e5-90e42f438a10","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Wit: Wikipedia-based image text dataset for multimodal multilingual machine learning,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:52694ea19dc0afffbc3eda41d9507ce6a9e8602e8f23eb2f7068009ab0322f7e","observation_id":"c60950f5-d858-4187-b3ff-ca3b937b1ef4","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.06794","last_updated":"2023-06-05T17:55:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-14T17:24:07Z","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","version":4},"cited_work":{"arxiv_id":"2209.06794","doi":null,"metadata_source":"pith","pith_arxiv_id":"2209.06794","snapshot_observed_at":"2026-07-04T19:30:07.491958Z","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","venue":"cs.CV","work_id":"29921cff-29c1-4aad-a27d-adac346027ec","year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2209.06794","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:b72867857e1df9d97e78f0bee3d500e78e06d86e55434aea6fa3f4b49b90a61e","observation_id":"4febbffb-4a82-474f-925e-42fa4de6a4a5","resolution":{"observed_at":"2026-07-02T16:27:08.991831Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17425","last_updated":"2023-11-06T02:47:51Z","snapshot_observed_at":"2026-07-06T16:25:38.571377Z","submitted_at":"2023-09-29T17:37:29Z","title":"Data Filtering Networks","version":3},"cited_work":{"arxiv_id":"2309.17425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17425","snapshot_observed_at":"2026-07-04T20:50:12.674098Z","title":"arXiv preprint arXiv:2309.17425 (2023)","venue":null,"work_id":"9922937e-9c54-4233-8816-d729dfaf746e","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2309.17425","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:911004d4275b65005a8cc5418d86c2e2fa002bc7db1d603c63f6aafeaff712e2","observation_id":"854a9759-a307-45cc-97f0-b1ff1b61f5eb","resolution":{"observed_at":"2026-07-02T16:27:09.020161Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Composing text and image for image retrieval-an empirical odyssey,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:2f46a240a6d0e3882c942b400acf84173be71b14444b1e3d19eff8d13ac73bbd","observation_id":"4699784f-86a7-439d-a3f1-5f274009e3bb","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Image search with text feedback by visiolinguistic attention learning,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:3f15858d423112f4cb9dc5b3fd128a531e47d75f698c7df673199da2051230eb","observation_id":"1b5b1f11-ad27-49c3-b91b-328a77b0155c","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Image search with text feedback by deep hierarchical attention mutual information maximization,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:02dca5a3ba390fee7e284f1800ef82bd103a4040fefc4c8fc45abdceabe267d7","observation_id":"4b8bcb31-f9e5-437c-a337-78f0d9f49ebf","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Cosmo: Content-style modulation for image retrieval with text feedback,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:e4a6627eb25e127d415f66170f18085084f4e247437e1e06faf401ba152ba3cd","observation_id":"339a7394-8da7-4230-b304-ac52ccb1f705","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.07272","last_updated":"2024-03-06T07:16:06Z","snapshot_observed_at":"2026-08-05T18:22:20.470650Z","submitted_at":"2023-06-12T17:56:01Z","title":"Zero-shot Composed Text-Image Retrieval","version":2},"cited_work":{"arxiv_id":"2306.07272","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.07272","snapshot_observed_at":"2026-07-02T16:27:08.984772Z","title":"Zero-shot composed text-image retrieval,","venue":null,"work_id":"53cd5c91-679d-463c-9a37-edd75132152f","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2306.07272","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:ff6e53b383e6db2e65162bd3907bd4d7e97e14ba1e9a9c22124ffd23076a9815","observation_id":"ca95bf01-3fb3-4801-bd50-208b661f02a8","resolution":{"observed_at":"2026-07-02T16:27:08.987164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"isearle: Improving textual inversion for zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:f9ac25157829f4dd1a0b7e90f55bdf0ad8d14abc244c0bb2f462f2f75fdc60db","observation_id":"353a1d1d-8fc0-4e9d-a00d-9ac6ee0b83a5","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Towards codebook-free deep probabilistic quantization for image retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:769c1c2e90fa6fc14ca75e0450dcc8b84d25520dfb291b7ba799eb6f2b550c7f","observation_id":"ee94aa73-edf9-4c8a-ac25-b50617eeda2b","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Mixture of subspaces image representation and compact coding for large-scale image retrieval,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:0d10a36e3fb37d23eee7646c9dc8f4b88452145c14d6de2b80ac79913c8465af","observation_id":"7b6676c1-f11d-444d-94fa-b86d58417b1c","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":"1810.04805","doi":"10.1111/jofi.12885","metadata_source":"pith","pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"cs.CL","work_id":"ed240a10-5b19-406c-baa5-30803f465785","year":2018},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:c992a2ae43ac4b2401131d454fd279f512c23e136aacd20ba702eef29a86af46","observation_id":"ed908753-9182-4225-8342-ceb476273c57","resolution":{"observed_at":"2026-07-02T16:27:08.996267Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Uniter: Universal image-text representation learning,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:de85aa1af7da8bc5d5891b11c7b62be5ab14f2ea69b1971de4fbb66218d5dbed","observation_id":"d534b49c-fdb6-4a5c-9f93-29288e968081","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03557","last_updated":"2019-08-09T17:57:13Z","snapshot_observed_at":"2026-07-06T08:13:26.248817Z","submitted_at":"2019-08-09T17:57:13Z","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","version":1},"cited_work":{"arxiv_id":"1908.03557","doi":null,"metadata_source":"pith","pith_arxiv_id":"1908.03557","snapshot_observed_at":"2026-07-04T09:09:43.648844Z","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","venue":"cs.CV","work_id":"5a9f7670-5c21-4203-8fe1-cc6eb9bc26d5","year":2019},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/1908.03557","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:96e4ef107770befe068ae361cd4edebc04e13fa4394f78b0a59cfbb2120bbd22","observation_id":"8daaf6b4-1826-4e49-bfb4-74f98aee3e18","resolution":{"observed_at":"2026-07-02T16:27:09.000692Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Oscar: Object-semantics aligned pre-training for vision-language tasks,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:32bc388100e55a179b8301661e1caad25f72a0e813a847f10722a723f51ff1b8","observation_id":"59e1d280-1f91-4d23-a280-b6030a469e67","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:385e2974337273f9f130a13a05395f55ae7eb3b10a1a81489fc6f9f978a99e02","observation_id":"d0c9f4f9-3bd0-4ea7-8314-066962fa14d6","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:3833b6a6921a6bbb24e66e6249088c18b9c6ecf56feaface97d4c7df95c8f519","observation_id":"4ad469ad-cf4b-40aa-a0e3-ab7a4bf8d44e","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.07490","last_updated":"2019-12-03T19:30:19Z","snapshot_observed_at":"2026-08-06T03:44:36.669916Z","submitted_at":"2019-08-20T17:05:18Z","title":"LXMERT: Learning Cross-Modality Encoder Representations from Transformers","version":3},"cited_work":{"arxiv_id":"1908.07490","doi":null,"metadata_source":"pith","pith_arxiv_id":"1908.07490","snapshot_observed_at":"2026-07-11T00:37:42.291267Z","title":"Lxmert: Learning cross- modality encoder representations from transformers","venue":"cs.CL","work_id":"a328317e-ee7d-44a5-b290-dbafa77f31f7","year":2019},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/1908.07490","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:4e18aaff71283df058c2bcfa7d30601669b5f62910dfb39a46b87fde5bc7e5fa","observation_id":"e1d4918e-fecc-46a0-9a86-32c4c830a764","resolution":{"observed_at":"2026-07-02T16:27:09.010267Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Optimization of rank losses for image retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:4f252486824b0feafd748262e2e7cd287157c527e8305dc1234c30dca3723bd1","observation_id":"0641a4dc-e4b0-4809-9c1d-44519274eb43","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Attack as defense: Proactive adversarial multi-modal learning to evade retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:1198b322a0bee562005789b2977b0dc8499cf325b66c018ac35e592f9b756483","observation_id":"e57909c8-9423-4715-aa00-b864e8c239d7","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:735fe1cb163e26e051096f6f002573c8c92d4c1352ffc10a18ae83cef8a0fd50","observation_id":"ede8a08f-34e0-428a-ab8d-4d63e2aee859","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Conditioned and composed image retrieval combining and partially fine-tuning clip-based features,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:1fb44b45d05967c1dd20ff193cf936b0d0372e214345f2c68c9a156dc3ac46c1","observation_id":"894ce067-9aae-4741-b2de-00a27a7d88da","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Fame-vil: Multi-tasking vision-language model for heterogeneous fashion tasks,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:463a3352ee468297563df81a081f3a12bc45433bf7ac2493f3991018177263a0","observation_id":"7dd07608-5bad-41cd-ade2-fc06ea6d55f3","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:27310ab7980d16055addb87cc3673d86179545734b719334a46c82daf8811b69","observation_id":"e5899d17-ab65-4ff5-a9aa-06b38d89f85b","resolution":{"observed_at":"2026-07-02T16:27:09.007916Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:9dfedbdb83204c56fc4c9f52f7e351e95a611479cf2ab03315f07f9b63e95d53","observation_id":"8fc72112-78cd-4984-a35f-f23dfde16900","resolution":{"observed_at":"2026-07-02T16:27:08.998421Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Svbench: A benchmark with temporal multi-turn dialogues for streaming video understanding,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:40eb92e7760f7ec44b25a3ee3700034b434820e536122a430f788dae09ec9b02","observation_id":"ea7d97e4-1a6c-4a93-8a8b-b65bd210e62b","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Livestar: Live streaming assistant for real-world online video understanding,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:7a8a8e8a7b3289f842ba45be95a3849d6d3b511bd6ce07e133cdaa145a0e7d86","observation_id":"84da6844-188f-481e-8e8b-d348186503b3","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Querystream: Advancing streaming video understanding with query-aware pruning and proactive response,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:c9e47f036ea12b24dda5e17ddc99498b4eb929083545ef7eded34bb755e865fe","observation_id":"e322133c-fdce-411d-a25c-5b0bdb5004bc","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:2eb8a2f008aade82e0288732068117305d3e2063e9a67d50b43c7f742259882f","observation_id":"87837274-f1b6-4a2a-a0d5-fce1a8b971cc","resolution":{"observed_at":"2026-07-02T16:27:09.002889Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Reason-before-retrieve: One-stage reflective chain- of-thoughts for training-free zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:5a615126f636e557733a32a7b44ad9f5683f5bd84f3dc913bedb9fb32501e1b2","observation_id":"849c4da3-674c-47c1-aedf-dc96647ad78d","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Merlot reserve: Neural script knowledge through vision and language and sound,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:3f99344ee56eb19f4b32a5704da2b750dde36c3b7762593d5a57159e9d24449e","observation_id":"32752517-fb52-4f96-9490-f87197612446","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":"2010.11929","doi":"10.1175/jcli-d-22-0357.1","metadata_source":"pith","pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","venue":"cs.CV","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","year":2020},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:9df1f0c256682ae06a788c5aae60d431a79d9cdb770aa24ee1cf23fc4602a654","observation_id":"dd0dc6ae-f94d-4901-986a-e936aa0474f9","resolution":{"observed_at":"2026-07-02T16:27:09.012450Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Vision-by-language for training-free compositional image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:612f51bd978ca7f8d6dfdcbdac358bd714410b29df16fd97579af3f41e3b1680","observation_id":"45819137-670d-4572-be5a-8ff5c0445cea","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"\"this is my unicorn, fluffy","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:0baa61142b7a23cb5aef7d68675b88168b2f71bf4d00f54ec1a5a8338336a762","observation_id":"286c6393-1da7-4f5c-9a2d-ff7a455c6348","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02951","last_updated":"2025-07-26T10:50:24Z","snapshot_observed_at":"2026-07-06T18:09:59.472962Z","submitted_at":"2024-05-05T14:39:06Z","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","version":2},"cited_work":{"arxiv_id":"2405.02951","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.02951","snapshot_observed_at":"2026-07-03T15:48:34.974875Z","title":"isearle: Improving textual inversion for zero-shot composed image retrieval","venue":null,"work_id":"f67f5d54-a40d-4d81-a74b-c9ba4e6c6007","year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2405.02951","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:3204aabc32eda6a6aba32cf5b06e280aee575c4b84e930d2cd2402e0d8166654","observation_id":"e188e77e-1f62-4589-b813-835bb1d6960e","resolution":{"observed_at":"2026-07-02T16:27:08.994177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Language-only training of zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:a1b8085f2111d63856f4cc07d6db3a6af2ea53717e58105786f2d1e40b632be2","observation_id":"0d8b53f1-430a-4377-b492-5bd2ece3deec","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.08101","last_updated":"2022-05-16T15:20:04Z","snapshot_observed_at":"2026-07-06T12:48:13.087800Z","submitted_at":"2022-03-15T17:29:20Z","title":"ARTEMIS: Attention-based Retrieval with Text-Explicit Matching and Implicit Similarity","version":2},"cited_work":{"arxiv_id":"2203.08101","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.08101","snapshot_observed_at":"2026-07-02T16:27:09.013903Z","title":"Artemis: Attention-based retrieval with text-explicit matching and implicit similarity","venue":null,"work_id":"380f8586-bd79-4008-a8dd-2730c0f6b8b6","year":2022},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2203.08101","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:95958715bab7b81146b2749ac8b43d9e03696e7acc896cbe1b55c2645a3f873f","observation_id":"22f97239-b564-4257-886d-d9d58a6ce561","resolution":{"observed_at":"2026-07-02T16:27:09.015326Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Amc: Adaptive multi-expert collaborative network for text-guided image retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:1a7bf45ea8dc5bc8ba264ea8f4168f2985032fe4d25f80c1f450e8cb408f0543","observation_id":"d48b8729-0fc2-40d1-855b-1ad3ad73701a","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00145","last_updated":"2020-06-30T22:55:02Z","snapshot_observed_at":"2026-07-06T09:34:15.665695Z","submitted_at":"2020-06-30T22:55:02Z","title":"Modality-Agnostic Attention Fusion for visual search with text feedback","version":1},"cited_work":{"arxiv_id":"2007.00145","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00145","snapshot_observed_at":"2026-07-03T23:49:02.522138Z","title":"Doddset al","venue":null,"work_id":"4c6ba9b2-6a9b-455d-9c17-f259de0fb86e","year":2007},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"cited_paper":"/paper/2007.00145","citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:5a13af637c0a3a422acb371a0829c77cfd88f27cf465057a1a1795054017addb","observation_id":"b87902bf-c34b-49a5-9146-4bcc2b93b382","resolution":{"observed_at":"2026-07-02T16:27:09.022557Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Dynamic weighted combiner for mixed-modal image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:54fb09ef2cd2a4a01ea35b104fa8a462529fd70b4a50d8c15b8d696317e8fcae","observation_id":"bbd71be3-e5c7-4cca-a8c9-c14b10453cc5","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Composed image retrieval using contrastive learning and task-oriented clip-based features,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:44fbf7e9c1667c35ba01b84ecd52d78076cf96e2c304c4aa453c0fececd8812d","observation_id":"b8cc411e-8009-467a-9582-1e02c021364b","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:44:36.059582Z","title":"Blip-2: Bootstrapping language- image pre-training with frozen image encoders and large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-27T22:44:36.059582Z"},"links":{"citing_paper":"/paper/2606.07032"},"observation_digest":"sha256:050974dfef7c705f1b9f9e87dcfd5e6d13d12e33a273d53c9bfbcff1b02517ec","observation_id":"280e856e-9043-4b41-bb6a-e0c918432f1f","resolution":{"observed_at":"2026-06-27T22:44:36.059582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.07032","last_updated":"2026-06-05T08:23:25Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:46:46.925980Z","submitted_at":"2026-06-05T08:23:25Z","title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets"},"reference_resolution":{"displayed":78,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":61,"verified_exact":15,"verified_fuzzy":0},"total_outbound_references":78},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 78 of 78 outbound references and 1 inbound Pith citation observation for arXiv:2606.07032."}