{"as_of":"2026-08-05T10:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:952dcdf0552780112fc4d9f9df7fcd64eb1522928459dd8762f2d5b6a0a1ef26","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-20T13:31:45.891242Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.17365/citation-record","integrity":"/paper/2605.17365/integrity","json":"/paper/2605.17365/citation-record.json","paper":"/paper/2605.17365"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning semantic structure-preserved embeddings for cross-modal retrieval","venue":null,"work_id":"c782f250-06f9-4f54-acf9-54394ea7c455","year":2018},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:55235ee2e2de99a8a36f91fb1532b2cbf44a5e50a1613aa75fe99aac6ce1d78f","observation_id":"163046f1-5153-4a00-a6da-8802b648520e","resolution":{"observed_at":"2026-05-20T13:33:20.036622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fine-grained visual textual alignment for cross- modal retrieval using transformer encoders","venue":null,"work_id":"72821647-0846-45fb-a832-8f44c7b615eb","year":2021},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:8e8e7efb7dcaa62b98e07c40491494b2d833195f6608b3f959cff01236bd1ffc","observation_id":"35bb7b2a-9781-4558-bb50-f179cc9a1326","resolution":{"observed_at":"2026-05-20T13:33:20.044375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fine-grained image-text matching by cross-modal hard aligning network","venue":null,"work_id":"7ebc8a53-a636-4392-bdfe-62a62ff199df","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:9b008a26ffefe059e4cc2aa04c4310f065f455e8dd91682a867b646df3ea2985","observation_id":"0cdc4157-2f5f-4fc7-87c8-2f76ee8ee5b7","resolution":{"observed_at":"2026-05-20T13:33:20.032886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An end-to-end graph attention network hashing for cross-modal retrieval","venue":null,"work_id":"0e84b4c4-03af-4e6f-9ca6-30730948ca51","year":2024},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:dd05ada67219a512404be05794bb0b11892aceef313be17d685fda74abefddff","observation_id":"67cb7748-326a-4745-89ef-363810759940","resolution":{"observed_at":"2026-05-20T13:33:20.049850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Achieving ensemble-like performance in a single model: A feature diversification framework for image-text matching","venue":null,"work_id":"b977ac57-9d35-4f38-8524-62a4d51442ff","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:eb8608f5933b325ada93e25505c5f840adaf657d1accc81ee6bf6dc0a9f3a09c","observation_id":"0a94d6c4-02e9-4b13-abef-a4258391f4f9","resolution":{"observed_at":"2026-05-20T13:33:20.051583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient token-guided image-text retrieval with consistent multimodal contrastive training","venue":null,"work_id":"dfee6569-3e7e-4bf4-a3fb-f3e5153c91ce","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:3d6afe0639efb4ba2901ed1cc01d2ef5baa31831299a44d758a4fad7f743fbb3","observation_id":"a75f7a07-42bc-4acf-9909-3b23e380501c","resolution":{"observed_at":"2026-05-20T13:33:20.048070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ucpm: Uncertainty-guided cross-modal retrieval with partially mis- matched pairs","venue":null,"work_id":"e0f641b1-d3e1-43d3-9cee-f14365b6ee84","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:940e3011c89927aca128989ebd7de6a416ff35a5680a8234b3752707cf1a0d12","observation_id":"d62981d7-d5ab-4dd1-bbed-ac54bf6e8c95","resolution":{"observed_at":"2026-05-20T13:33:20.042515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dual uncertainty-aware correspondence adapting and retaining for continual composed image retrieval","venue":null,"work_id":"d733042e-6987-4a42-a255-74ed56e0c769","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:c0c6b33ea71790cedfe7d87461f4c1595983db9c469a8cc9aabb4002e3760976","observation_id":"660a2e16-5b40-4849-88ad-3291240fbefd","resolution":{"observed_at":"2026-05-20T13:33:20.034682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T13:44:55.388145Z","title":"Rebalanced vision-language retrieval considering structure-aware distillation","venue":null,"work_id":"5c3551f1-5f15-473e-86a5-69d515625c72","year":2024},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:15901c4b1ad01253cba7e2f2f1dd80caf1b1d2e03b21db9a716e1f6384ab7210","observation_id":"d2e82eff-4f3f-4083-9da4-76fddc9cdc18","resolution":{"observed_at":"2026-05-20T13:33:20.038888Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chatting makes perfect: Chat-based image retrieval","venue":null,"work_id":"bf55effd-ef02-4336-bf87-81084a2be4e7","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:b78f6ebcea1d363034484b7a19ccaed5f11cb05c81d81fae4848d010cfd7c0fd","observation_id":"2b397f60-6b52-4885-aed6-92bc454c65e3","resolution":{"observed_at":"2026-05-20T13:33:20.031083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Interactive text-to-image retrieval with large language models: A plug-and-play approach","venue":null,"work_id":"c4ace528-f2cc-45bd-aa59-a4e9453d1ea0","year":2024},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:736af1afac4842def0384502e1a32f9399b1453e1248e327e12d33aa00c9f02d","observation_id":"8be3b75e-1e98-4de7-b480-7f04302dd575","resolution":{"observed_at":"2026-05-20T13:33:20.040721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enhancing inter- active image retrieval with query rewriting using large language models and vision language models","venue":null,"work_id":"ee2a6f6e-3081-4e20-856b-d8786aa36721","year":2024},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:7d1bdd8500a9f9b36dcb7bcde7d68a9ed3409f500875772c2fd4d18da329b1cf","observation_id":"ac922863-eeea-44f0-81e3-4b02a1fb012a","resolution":{"observed_at":"2026-05-20T13:33:20.021712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Diffusion augmented retrieval: A training-free approach to inter- active text-to-image retrieval","venue":null,"work_id":"18811ccb-11b4-4f19-a1db-3ecb5c8729e4","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:479d51359c8529e8990bf116ce0f9845d5434ae3f99fa2a02aa8eb7fca208640","observation_id":"dc767dae-163d-431c-8658-84a09442a265","resolution":{"observed_at":"2026-05-20T13:33:20.025358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chat-based person retrieval via dialogue-refined cross-modal alignment","venue":null,"work_id":"fc8316c3-b218-4e9b-b811-6f3cf9d793c2","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:6bd73b0e7b3018ace8dd0e342d4b01ca4f2d2591416350ab55156912324564a8","observation_id":"0c32558e-f8b2-403d-900b-a7599fb82b39","resolution":{"observed_at":"2026-05-20T13:33:20.019916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:24:51.265660Z","title":"Mai: A multi-turn aggregation- iteration model for composed image retrieval","venue":null,"work_id":"6a9816b7-f3e1-46ae-b7bd-c930e67430e0","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:fb100cbbd2f2ed2ceafecbd0d705197e049525a86774c5ee7ab62ae75f5cbadb","observation_id":"9a012c30-be8b-4581-8a19-685202c23fc5","resolution":{"observed_at":"2026-05-20T13:33:20.023488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:24:51.258048Z","title":"Imagescope: Unifying language-guided image retrieval via large multimodal model collective reasoning","venue":null,"work_id":"6c386afe-847a-4232-b5cd-f0ce0bd5170f","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:f58d783d71d4f6f620f30a7211c24b3fff652e069fb15c6b94f161c5c98e31bc","observation_id":"7e1c213c-f971-4115-96f4-3dc644f9d95b","resolution":{"observed_at":"2026-05-20T13:33:20.027231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:9edb9c25cc37910804c0c04a6ff582126e07182931754f1ba56d731fad7929ae","observation_id":"3202849f-9b0c-401f-951b-ddbeec4664ca","resolution":{"observed_at":"2026-05-20T13:33:19.112592Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T23:05:44.579640Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"6de2065a-afd9-45af-8961-27be26fbd586","year":2022},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:ca997c12d13d718e3cd3f7404f7e5ee16ce068723074cb1ad023388822876f24","observation_id":"49d56b30-2d50-43f4-9523-e7e8959ad649","resolution":{"observed_at":"2026-05-20T13:33:20.014574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:6cb950a9261924839f63fadfa3018e5913f58f5f7410963cc3804e4ee8b29337","observation_id":"bf1149ce-fe4a-41d5-8a22-7c616aad005f","resolution":{"observed_at":"2026-05-20T13:33:19.121810Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pairwise rela- tionship guided deep hashing for cross-modal retrieval","venue":null,"work_id":"629fbfbe-0a3a-4534-964d-d2938b8ff032","year":2017},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:9e3538ebcbe25cefc5c0214b7b81c194cfc538b98c3977eb69ad68c68ba7844c","observation_id":"0cdf328e-3c94-45f0-bef7-5dec8c119d08","resolution":{"observed_at":"2026-05-20T13:33:20.010694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Context-aware attention network for image-text retrieval","venue":null,"work_id":"91383bcc-09a7-4ef4-8991-bd8e3040092d","year":2020},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:02a805d2b9290cb51ca216d9ef51891abed6b4523203446a1d46573d57f1544a","observation_id":"47bc705d-c3ce-4d7f-9567-4aed3ec7ee76","resolution":{"observed_at":"2026-05-20T13:33:20.008904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Align before fuse: Vision and language representation learning with momentum distillation","venue":null,"work_id":"ec145f3f-9103-401b-ba95-67504228526b","year":2021},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:f993c7e070f68af81558d291c61f5b698f8da1b4014249f5c4f4f69048353bac","observation_id":"c1a0c0c1-2ed2-4647-a250-9686ee8ddd96","resolution":{"observed_at":"2026-05-20T13:33:20.012690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":"b6c2ace6-5c1e-447a-8759-a568df3ce15e","year":2021},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:629a400a8353bff704e2d7cd04a8bc5c72bbed24987170d2c1a57958e7f08e71","observation_id":"de809191-9bc9-4069-8e84-6c468e911698","resolution":{"observed_at":"2026-05-20T13:33:20.016442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gssf: Generalized structural sparse function for deep cross-modal metric learning","venue":null,"work_id":"f9ef119d-979d-4c2d-bbed-88ca2fdf8d89","year":2024},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:3153ee64e302c747ac50e589ca39fdcaec6964d92ca5fa94bdd0d00a72447ad3","observation_id":"8858231a-36ec-4533-b011-1053ea3becc1","resolution":{"observed_at":"2026-05-20T13:33:20.005111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Multi-relational deep hash- ing for cross-modal search","venue":null,"work_id":"6de97eb7-1c22-4195-8a61-7e1114bd8061","year":2024},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:188cd5943d170d591db519f0786be0ed055da36bdc59450c745cf9ebda49c8e2","observation_id":"3e3cbea2-6323-4f0e-8bc6-957dcb056bf0","resolution":{"observed_at":"2026-05-20T13:33:20.001249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Composed image retrieval via cross relation network with hierarchical aggregation transformer","venue":null,"work_id":"ab188d23-b601-4d74-9286-0b1c8637ba9b","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:76200f43f564b3dad1fa473c02909b0303612f65098869411186e011bab76082","observation_id":"d82e29d0-fb33-4f63-95e4-9bfc3dd63c3d","resolution":{"observed_at":"2026-05-20T13:33:19.997588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transformer-xl: Attentive language models beyond a fixed-length context","venue":null,"work_id":"17309658-a8c3-4cf6-ba6b-2ca129b134f5","year":2019},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:5cd91d0ae67da54074ec880d9df54feb2f541c5cf99d5fcff6ad6fb5a74e78c5","observation_id":"6770103f-dedd-42c2-b919-fde21eec835b","resolution":{"observed_at":"2026-05-20T13:33:19.999437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T09:46:13.472673Z","title":"Retrieval- augmented generation for knowledge-intensive nlp tasks","venue":null,"work_id":"ff6dfa2c-1f47-4244-8202-f5a3b57070af","year":2020},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:c3e2a4bb2440815e125b67ceb16704a49e47cb788fac79161fb58281ac8c5866","observation_id":"7dbb41aa-52a0-4b50-a10f-fcfb7b340cf9","resolution":{"observed_at":"2026-05-20T13:33:20.003303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Augmenting language models with long-term memory","venue":null,"work_id":"564697a3-f47e-4438-820a-cf4981e3ac0b","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:08657e3223e45b89aa4b89ff34fd11473b08b37aae511348267396ea66ee234d","observation_id":"405c7b02-4622-4a2d-be14-f4c896d211f9","resolution":{"observed_at":"2026-05-20T13:33:20.046237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prompting video-language foundation models with domain-specific fine-grained heuristics for video question answering","venue":null,"work_id":"24dd9b03-74d7-44ea-8bed-b4708b8d4385","year":2024},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:029cf6bd230923150027416b74317675b0e46d28fa4db424980692635363aadb","observation_id":"a77c8b2d-e878-4872-8c30-3375d60eb73b","resolution":{"observed_at":"2026-05-20T13:33:19.995728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hmt: Hierarchical memory transformer for efficient long context language processing","venue":null,"work_id":"6672b561-9b8e-4b02-8c60-54d7b81e8152","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:bab080ace4da21963bc739f498d7360c7c8471cf8beff70b4fc028bfc6c7dcb1","observation_id":"e3f97b3d-788f-40a5-9be9-82d3c870c322","resolution":{"observed_at":"2026-05-20T13:33:19.993937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.07605","last_updated":"2026-04-27T13:49:08Z","snapshot_observed_at":"2026-08-03T08:11:38.794368Z","submitted_at":"2026-02-07T16:16:51Z","title":"Fine-R1: Make Multi-modal LLMs Excel in Fine-Grained Visual Recognition by Chain-of-Thought Reasoning","version":3},"cited_work":{"arxiv_id":"2602.07605","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.07605","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fine-R1: Make Multi-modal LLMs Excel in Fine-Grained Visual Recognition by Chain-of-Thought Reasoning","venue":"cs.CV","work_id":"b64fb34c-d9af-447b-8da3-b665bbf09e54","year":2026},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"cited_paper":"/paper/2602.07605","citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:1735d5fbde60502134b7f81d6c86f23daf87775c22e65a6d70fff338d23a8536","observation_id":"57b59ecd-da60-4a05-ab63-0fd367ee6b12","resolution":{"observed_at":"2026-05-20T13:33:19.116270Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mitigating hallucinations in large vision-language models via reasoning uncertainty- guided refinement","venue":null,"work_id":"1563ebda-680b-4706-a4c1-ea5b28a52319","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:b3f99a39b91b0c0e0d636428aa8dce39de95dbdd9be3e360ce5d1e8934262e68","observation_id":"58660d82-ae09-45cc-bc40-f5542b421940","resolution":{"observed_at":"2026-05-20T13:33:19.990132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prompt learning with knowledge regularization for pre-trained vision-language models","venue":null,"work_id":"d7888732-3db2-4800-a982-51577d346102","year":2025},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:2d8745c5eb93dccf5c6928ba8e91ca91b0c9c74fe4bdba98224a63bb404a20a6","observation_id":"b4fe9085-6be5-44e3-9cdb-88062f06dbd0","resolution":{"observed_at":"2026-05-20T13:33:19.992012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.09281","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Star: Sensitive trajectory regulation for unlearning in large reasoning models","venue":null,"work_id":"6fc95cad-fad7-48ba-846c-8bd4b260b33a","year":2026},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:0407df62994b20a58afac21567133eb6b2cd734dbd18a6f06a3f3207f343c130","observation_id":"e996234a-e82e-4d45-84ed-546e6505f554","resolution":{"observed_at":"2026-05-20T13:33:19.124609Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improving language models by retrieving from trillions of tokens","venue":null,"work_id":"1b15143d-1278-4d7b-bfc3-00077d3db9a3","year":2022},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:9943d538ecfc3361b1b98edc289561cc48ede8c93f2b89d2db662bcaaa80c8f3","observation_id":"d744c5f1-188d-4bce-bcc7-3d4d279b5b00","resolution":{"observed_at":"2026-05-20T13:33:20.007153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Atlas: Few-shot learning with retrieval augmented language models","venue":null,"work_id":"9bc9b494-294c-434f-9567-61d67208bdb8","year":2023},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:5dc291c05ed1d2026a245471d4859047adb096474bcc228c9404e863ff0fe61e","observation_id":"86c31a20-ff09-4d9c-9984-7fad1a674336","resolution":{"observed_at":"2026-05-20T13:33:20.018192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1911.05507","last_updated":"2019-11-13T14:36:01Z","snapshot_observed_at":"2026-07-06T08:36:43.643240Z","submitted_at":"2019-11-13T14:36:01Z","title":"Compressive Transformers for Long-Range Sequence Modelling","version":1},"cited_work":{"arxiv_id":"1911.05507","doi":"10.48550/arxiv.1911.05507","metadata_source":"pith","pith_arxiv_id":"1911.05507","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Compressive Transformers for Long-Range Sequence Modelling","venue":"cs.LG","work_id":"70572d78-130b-4a13-908f-0702dcf423df","year":2019},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"cited_paper":"/paper/1911.05507","citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:90926637d001a7e05301ffde80a8a84238cbfb36a493b32295c16f3af2dc4d05","observation_id":"9f22a9a7-4c73-450d-b969-c2053966c400","resolution":{"observed_at":"2026-05-20T13:33:19.119023Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T14:07:06.849417Z","title":"Attention is all you need","venue":null,"work_id":"4f585692-8ef8-4f20-8f75-5e0e32647d76","year":2017},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:6d6afc0cdf7429241269a87670f8aa19629701bb227f2b38a37608d6fdf5384c","observation_id":"7f8259a1-b62e-4c6e-871c-419b16669026","resolution":{"observed_at":"2026-05-20T13:33:19.982426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:17:04.534365Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":"9119843a-a3b3-4652-bc58-24e1ff08660f","year":2022},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:fd2a25aa675085504ec8996a85aac6e16e151786ec0b00a59eaf634715a8f7b0","observation_id":"256eda04-b163-46a2-a388-d197cdacb5fc","resolution":{"observed_at":"2026-05-20T13:33:19.984526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"VSE++: Improving visual-semantic embeddings with hard negatives","venue":null,"work_id":"1c264eb7-09db-4e6c-8e6e-648268cb2f25","year":2018},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:fb3f4b8f4cf55aa243d2b5d9049ad1ceaa90d49caa7b70af74959b9c1548f6af","observation_id":"17e3fd6b-99f3-4410-aef3-e2f7f8440bee","resolution":{"observed_at":"2026-05-20T13:33:20.029004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"91d4583f-e8c5-4389-a82d-11b70a273629","year":2021},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:e120ce4bd3c11028bbdb6ad64f40fda0782e082849696709cc6b103c2ed94b68","observation_id":"31784e8d-a797-4e69-9ad9-57c825604a90","resolution":{"observed_at":"2026-05-20T13:33:19.986392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual dialog","venue":null,"work_id":"56c0871d-fa34-4b4d-bf5b-2522e1d89968","year":2017},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:ca011bbf59cd2d25cb7315dd0585eab9455bd38c968e8f1cd30492b044696f8a","observation_id":"272d7617-ec5e-46ed-9090-59fac2e71888","resolution":{"observed_at":"2026-05-20T13:33:19.988095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Large-scale pretraining for visual dialog: A simple state-of-the-art baseline","venue":null,"work_id":"c98f9cfc-2f09-458e-b0b5-a027a263d13b","year":2020},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:4ca0c7b828c2be77d8f577b02ee6753e4e6ec8b6f93aa58af3be983b8d28d0be","observation_id":"1c8e1b02-f5d0-4041-84a2-e71fcb0ddc01","resolution":{"observed_at":"2026-05-20T13:33:19.980524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"His research interests include multimedia understanding, retrieval, and recommendation","venue":null,"work_id":"2ddd32ab-a988-4bb1-af20-33dab8291855","year":null},"citing_paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-20T13:31:45.891242Z"},"links":{"citing_paper":"/paper/2605.17365"},"observation_digest":"sha256:94a6e7a80d5dbfcb6b299cf3d34ec924e7146c1c157c93b84bc33dd1783e1655","observation_id":"95c99856-b363-481a-8e53-b1381843a27a","resolution":{"observed_at":"2026-05-20T13:33:19.978531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.17365","last_updated":"2026-05-17T10:17:41Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:17:41Z","title":"Memory-Augmented Query Intent Understanding for Efficient Chat-based Image Retrieval"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":5,"verified_fuzzy":40},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 0 inbound Pith citation observations for arXiv:2605.17365."}