{"as_of":"2026-08-09T06:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4e11c241c984ed26a8af9bd6002aecefa8b2ec5f4ddf45a153722f99df240dba","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T12:21:15.520982Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T11:13:58.950328Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T18:23:51.303795Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":"2507.21917","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-06-29T18:23:51.303795Z","title":"& Castellano, G","venue":null,"work_id":"bff5d480-9b89-40b1-9426-a1b05a46e3a1","year":2025},"citing_paper":{"arxiv_id":"2603.18472","last_updated":"2026-04-09T02:35:56Z","snapshot_observed_at":"2026-07-06T22:49:37.944352Z","submitted_at":"2026-03-19T04:08:20Z","title":"Cognitive Mismatch in Multimodal Large Language Models for Discrete Symbol Understanding","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-15T09:11:31.870441Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2603.18472"},"observation_digest":"sha256:05d180a9ec4d735e77005e298c8e00a0424fc98f6a774dd23195a7c6b2efa877","observation_id":"40f2e3ed-7984-40f9-bb3f-3843d57451db","resolution":{"observed_at":"2026-05-15T09:15:21.084154Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":"2507.21917","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-06-29T18:23:51.303795Z","title":"& Castellano, G","venue":null,"work_id":"bff5d480-9b89-40b1-9426-a1b05a46e3a1","year":2025},"citing_paper":{"arxiv_id":"2604.07338","last_updated":"2026-04-08T17:53:26Z","snapshot_observed_at":"2026-07-06T22:55:37.787732Z","submitted_at":"2026-04-08T17:53:26Z","title":"Appear2Meaning: A Cross-Cultural Benchmark for Structured Cultural Metadata Inference from Images","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T17:49:25.227568Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2604.07338"},"observation_digest":"sha256:c370b92a014a6e55f135b65186cc70b6684ace8a08d2d177c02be1f595f58392","observation_id":"e6393261-fd1e-4261-a288-b4a7d816a19c","resolution":{"observed_at":"2026-05-11T06:05:55.467704Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":"2507.21917","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-06-29T18:23:51.303795Z","title":"& Castellano, G","venue":null,"work_id":"bff5d480-9b89-40b1-9426-a1b05a46e3a1","year":2025},"citing_paper":{"arxiv_id":"2606.27947","last_updated":"2026-06-26T10:42:43Z","snapshot_observed_at":"2026-08-08T10:08:09.608301Z","submitted_at":"2026-06-26T10:42:43Z","title":"Understanding How MLLMs Describe Artworks Using Token Activation Maps","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T05:06:35.085802Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2606.27947"},"observation_digest":"sha256:bfe458e3a7a1d9c3bbfec26075da6fafa9a5fca241ff8f82dd945c1a9ed75a2c","observation_id":"710e3f64-12c0-4227-a319-519d0174c297","resolution":{"observed_at":"2026-06-29T18:23:51.305082Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-08-06T11:13:58.950328Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05026","last_updated":"2026-08-05T16:30:12Z","snapshot_observed_at":"2026-08-08T23:13:48.283206Z","submitted_at":"2026-08-05T16:30:12Z","title":"ArtAnno: Annotating Implicit Semantics in Artworks through LLM Agent-Driven Bidirectional Human-AI Augmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T11:13:58.950328Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2608.05026"},"observation_digest":"sha256:3ef72f1809df47d8ba51f35a9c78c530f7df8bf61781c4e3c1d66a2b6f8a2d45","observation_id":"6879c8b8-a8b5-4248-bb49-3e95baadafc6","resolution":{"observed_at":"2026-08-06T11:13:58.950328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.21917/citation-record","integrity":"/paper/2507.21917/integrity","json":"/paper/2507.21917/citation-record.json","paper":"/paper/2507.21917"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.943480Z","title":null,"venue":null,"work_id":"8fae7dc8-10b7-4882-a3d7-b04e1d0f773d","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.376287Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:3b426d8e1cbbf079f63c64fdbaffec94f9b4644890c3495e23cc193a1b71e1b7","observation_id":"e58c71a6-42b4-4ffa-b0b1-921d762e9479","resolution":{"observed_at":"2026-08-06T12:21:15.946480Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.934719Z","title":"Leveraging knowledge graphs and deep learning for automatic art analysis,","venue":null,"work_id":"9f2bd266-b74f-4be8-b407-a561ef520c77","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.379980Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:358ead09883b3f5adf6bd9132f7264e177ba7cccb6ba20e1d6db2e6aa5514ddc","observation_id":"bb92f3e5-ee8b-4339-9da7-04b63b97b603","resolution":{"observed_at":"2026-08-06T12:21:15.937813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.926394Z","title":"GraphCLIP: Image-graph contrastive learning for multimodal artwork classification,","venue":null,"work_id":"cec8c821-14b9-4b26-80b4-238056c47074","year":2025},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.383516Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:ece39415b175575707ee9c34597ee0d29521ab05351a75c263a270fe3dc3550e","observation_id":"85581239-a6e7-49b6-b7fe-088a5ce8c4e9","resolution":{"observed_at":"2026-08-06T12:21:15.929473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.918061Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"4f142f88-c7f2-40c0-9de3-58f2ebcbce50","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.386819Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d654a5423cddec0afc7963d9672dbc2ca46390835d3a213620f6378c00ae268f","observation_id":"09f67382-5b4e-41f5-b76b-b8608701fff6","resolution":{"observed_at":"2026-08-06T12:21:15.921080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T12:21:15.389939Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.389939Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:975e863724498ea4764d9eb498c414d707119916462760385ec630244923ad33","observation_id":"89069abc-1406-4f9d-9b92-372b40e2c1cf","resolution":{"observed_at":"2026-08-06T12:21:15.389939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07490","last_updated":"2024-04-04T18:55:18Z","snapshot_observed_at":"2026-08-03T13:45:12.497823Z","submitted_at":"2023-05-12T14:04:30Z","title":"ArtGPT-4: Towards Artistic-understanding Large Vision-Language Models with Enhanced Adapter","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07490","snapshot_observed_at":"2026-08-06T12:21:15.393243Z","title":"ArtGPT-4: Towards artistic-understanding large vision-language models with enhanced adapter,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.393243Z"},"links":{"cited_paper":"/paper/2305.07490","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:9c26d9396d9e1d67ca3f628a36a6933bdd787b679847a882d7d96cefedc0b6fe","observation_id":"82ddbe2f-079b-4080-8e8e-f674a9773db3","resolution":{"observed_at":"2026-08-06T12:21:15.393243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.908814Z","title":"Gallerygpt: Analyzing paintings with large multimodal models,","venue":null,"work_id":"fd9643bc-1b2e-4cdb-932b-181e4cec7cdf","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.396782Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:302c593019110d7b9c4b872e5d33fa73374c1a46ba0176919afd03886cd96da6","observation_id":"303579f4-8f4d-4066-ae43-bd80e62d7ad1","resolution":{"observed_at":"2026-08-06T12:21:15.912785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10921","last_updated":"2024-09-17T06:39:18Z","snapshot_observed_at":"2026-08-07T13:44:45.880775Z","submitted_at":"2024-09-17T06:39:18Z","title":"KALE: An Artwork Image Captioning System Augmented with Heterogeneous Graph","version":1},"cited_work":{"arxiv_id":"2409.10921","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.10921","snapshot_observed_at":"2026-08-06T12:21:15.592463Z","title":"KALE: An Artwork Image Captioning System Augmented with Heterogeneous Graph","venue":"cs.CV","work_id":"8d13328a-731c-4703-97b7-a2e5af3e011f","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.399409Z"},"links":{"cited_paper":"/paper/2409.10921","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:fa6345ec75bba739fbee555d75ec15011cb6f9b04755628bf73b5e1608572cd3","observation_id":"96df4660-a14e-427b-b6f7-17c4164ac575","resolution":{"observed_at":"2026-08-06T12:21:15.597754Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.900491Z","title":"Colbert: Efficient and effective passage search via contextualized late interaction over bert,","venue":null,"work_id":"ac422665-3445-471f-9291-82983ed65695","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.402733Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e02ed6df706317b1bab83ddb1a625c81a4e2d3b1f22fb0af034966338b0946ea","observation_id":"d17427ac-7cd6-4fd3-a2c1-15891a390e1a","resolution":{"observed_at":"2026-08-06T12:21:15.903528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.892289Z","title":"Artpedia: A new visual-semantic dataset with visual and contextual sentences in the artistic domain,","venue":null,"work_id":"9ba6b050-6558-42f0-aca9-68f2ab7f0bda","year":2019},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.405602Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:409cc0ed7f490b1da55f199c429f7a5bd1c0a63021557428717b021c1572e9a5","observation_id":"5218236b-b32b-4c5f-ae03-183c3e8f9dfb","resolution":{"observed_at":"2026-08-06T12:21:15.895336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.883078Z","title":"Deep learning approaches to pattern extraction and recognition in paintings and drawings: An overview,","venue":null,"work_id":"1014c4c1-37fd-498d-ac0f-ce91244fd427","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.408392Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:7f609fb3d45ee5d6be5f439e163c367bcba665b1163c5f7566eb7d5790cea803","observation_id":"cb8a1aa8-1494-4cab-973d-395f0f91ca35","resolution":{"observed_at":"2026-08-06T12:21:15.886284Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.874809Z","title":"Machine learning for cultural heritage: A survey,","venue":null,"work_id":"e7521fc7-b3e0-45e9-8d8e-b1451d21f803","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.411512Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:79ce21b8059fc58e9158728bebd803764ade507c8def62250f5b51383d658824","observation_id":"bf5868df-3321-4f3c-991a-9a2cd3b0770f","resolution":{"observed_at":"2026-08-06T12:21:15.877773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.866524Z","title":"Fine-tuning convolutional neural networks for fine art classification,","venue":null,"work_id":"a008f925-5a42-4198-aa01-99ff49805f3e","year":2018},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.415129Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d42eb0bc98336145ca48d2c8c9aeb5b270144cd1b631c4aca46cf6474f920078","observation_id":"fd1a1c4a-be3f-4538-8249-2b852a36360e","resolution":{"observed_at":"2026-08-06T12:21:15.869536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1311.3715","last_updated":"2014-07-23T07:56:20Z","snapshot_observed_at":"2026-07-06T03:28:14.741672Z","submitted_at":"2013-11-15T03:37:50Z","title":"Recognizing Image Style","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1311.3715","snapshot_observed_at":"2026-08-06T12:21:15.417940Z","title":"Recognizing image style,","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.417940Z"},"links":{"cited_paper":"/paper/1311.3715","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:71131c91fbcad6a86998fd97a361767a8ea3566591e8d017d4f24629c25873b2","observation_id":"92586de4-cc32-4b0c-957d-e42eef8eea45","resolution":{"observed_at":"2026-08-06T12:21:15.417940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1505.00855","last_updated":"2015-05-05T01:25:26Z","snapshot_observed_at":"2026-07-06T04:16:54.272660Z","submitted_at":"2015-05-05T01:25:26Z","title":"Large-scale Classification of Fine-Art Paintings: Learning The Right Metric on The Right Feature","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1505.00855","snapshot_observed_at":"2026-08-06T12:21:15.421188Z","title":"Large-scale classification of fine-art paintings: Learning the right metric on the right feature,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.421188Z"},"links":{"cited_paper":"/paper/1505.00855","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:125e7de7682ce9d42b7d6af0f0e3c0bbba803494ed33864842e9a22138202948","observation_id":"f80128db-6fba-4cf4-a641-c745b61c6bf9","resolution":{"observed_at":"2026-08-06T12:21:15.421188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.858112Z","title":"Toward Discovery of the Artist’s Style: Learning to recognize artists by their artworks,","venue":null,"work_id":"2eea0b59-2db5-49fb-9d2c-5d0ee61731d3","year":2015},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.424471Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:34f26505f8f32c3453379fd15c10b34e73151bdd56677eb496b42ac808e5e052","observation_id":"1ff222b3-d6e5-4844-9d06-8e57cf17d884","resolution":{"observed_at":"2026-08-06T12:21:15.861235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.849928Z","title":"A deep learning approach to clustering visual arts,","venue":null,"work_id":"8804c547-c91b-430d-b97e-e4ec6c721e66","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.428015Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:c7deed258d110438068d7ead646f893d84a999ff98cfdb70d9beb3cb3edb3d22","observation_id":"1f9b41be-672e-4017-8e92-b83db479d8b6","resolution":{"observed_at":"2026-08-06T12:21:15.852876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.841627Z","title":"Toward automated discovery of artistic influence,","venue":null,"work_id":"d8a4f743-3e7a-4539-b77c-0546225b4b30","year":2016},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.431094Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:5c6e8f4545f2a31f86ef79902431641886758efa94705709d50a0c3001fb1755","observation_id":"f4e985b6-a115-4f2d-8495-4c3b7d2d6ce4","resolution":{"observed_at":"2026-08-06T12:21:15.844686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.832955Z","title":"Wasielewski, Computational formalism: Art history and machine learning","venue":null,"work_id":"0ad475c6-b015-4288-8e2b-4c962f6263ee","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.433853Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:2b1f35e641e646cb6183af750a3181252dc577833df0c36d709a40c97825564a","observation_id":"21280f40-d304-4669-b5fa-44b502e038a3","resolution":{"observed_at":"2026-08-06T12:21:15.835852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.824638Z","title":"ContextNet: representation and exploration for painting classification and retrieval in context,","venue":null,"work_id":"af808e5f-8dcb-4316-9b90-4783a3346cad","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.437235Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:a94e8c8dc6ae9b19d16e8c8796a1ef3813d92d9ea4eec0f8a2b36450de827b86","observation_id":"cb2c3577-f29e-4ff7-ba22-6848451ddf9e","resolution":{"observed_at":"2026-08-06T12:21:15.827653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.815411Z","title":"How to read paintings: semantic art understanding with multi-modal retrieval,","venue":null,"work_id":"2005b5c1-d654-415f-b70f-cc5a292bbda6","year":2018},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.440087Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d247d43a8ec62880dd2714396ab59a1033275611bcfbbbedb129be4a1852c6fc","observation_id":"227ce487-ca6a-4f85-b30b-e37473388c43","resolution":{"observed_at":"2026-08-06T12:21:15.818505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.807352Z","title":"Generating captions for images of ancient artworks,","venue":null,"work_id":"7c16e063-e59c-459e-97e5-b496b121d634","year":2019},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.442949Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:15ed74f971fafe36a9d9965630eaa625da69f228db2b7e0f4a56ddaa8d1a84d5","observation_id":"b8fde18c-aff2-42f4-b86d-791f7c01a484","resolution":{"observed_at":"2026-08-06T12:21:15.810228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.799009Z","title":"A dataset and baselines for visual question answering on art,","venue":null,"work_id":"0e6c2f2b-e74c-491a-8919-ced488808dce","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.445701Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:c7ced670adaf6b0efbd1ca96870fa56ad017374af22d53963b63f31db5fbd2cc","observation_id":"803fa69d-df7a-40d3-95bb-2278452f278f","resolution":{"observed_at":"2026-08-06T12:21:15.801853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.790811Z","title":"Iconographic image captioning for artworks,","venue":null,"work_id":"5da9b98c-05a2-427b-a61b-8955a58c606c","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.448635Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:be2ad8c8992c2089cb102f6a668583c5ae2b2b2d78b85601d96258495a23514d","observation_id":"2528bd1d-25c3-4ea5-97de-16a86f7f7764","resolution":{"observed_at":"2026-08-06T12:21:15.793638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.781635Z","title":"Iconclass: an iconographic classification system,","venue":null,"work_id":"1a0ba35d-8c00-4d5b-ab52-a0b25e3a975f","year":1983},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.451450Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:061e9ff36d9aaee7d6e9982688253a3ee74dca1e9d802b6126d8c232a5663b69","observation_id":"f6a71688-807f-4219-9c59-382d2bb1c4fe","resolution":{"observed_at":"2026-08-06T12:21:15.785017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.771952Z","title":"Explain me the painting: Multi-topic knowledgeable art description generation,","venue":null,"work_id":"9572b2ec-bc88-4288-99a2-aaefb9983d01","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.454256Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:8570c7e2376a3fae09429a48b06809ec9fb4c04edc2d350765e561a7bfd66149","observation_id":"634affea-ec92-4ce7-874b-3a3714ce6a93","resolution":{"observed_at":"2026-08-06T12:21:15.775091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00051","last_updated":"2017-04-28T03:53:14Z","snapshot_observed_at":"2026-07-06T05:36:07.769337Z","submitted_at":"2017-03-31T20:39:10Z","title":"Reading Wikipedia to Answer Open-Domain Questions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00051","snapshot_observed_at":"2026-08-06T12:21:15.457109Z","title":"Reading wikipedia to answer open-domain questions,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.457109Z"},"links":{"cited_paper":"/paper/1704.00051","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:c3851858bbd79190ce7db2b454875c0a0c7f30b436611cc741ab9b3fd4950eb9","observation_id":"b6e747ae-a0df-4e17-a0f2-a48a7065f2b0","resolution":{"observed_at":"2026-08-06T12:21:15.457109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.762817Z","title":"Is GPT-3 all you need for visual question answering in cultural heritage?","venue":null,"work_id":"b9df6b51-c29a-4721-8764-6a17d95b95a4","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.460567Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:bcd49da8d24a2e89ab49dd3b45a9b9825887d4c02f4fc2addb3c996401959a7c","observation_id":"0582f23c-164b-4137-8317-8aa4fe6e72a9","resolution":{"observed_at":"2026-08-06T12:21:15.765957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.753610Z","title":"Exploring the Synergy Between Vision-Language Pretrain- ing and ChatGPT for Artwork Captioning: A Preliminary Study,","venue":null,"work_id":"ff921752-e507-4474-aa78-fc52b400d185","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.463245Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d8d99aa84fe55f0189f81190ff642b473386fc1cb93e142f806116d8c67aaf27","observation_id":"cb6d3a00-b468-43c0-9457-842b4bf6981c","resolution":{"observed_at":"2026-08-06T12:21:15.756816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.744627Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale,","venue":null,"work_id":"d03089f5-5dea-4e56-874c-d7785f6e9941","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.466019Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:bf2a943ff1c281f36259f9c155bffdfdd720c655b406be7e86e059ce027bf1c3","observation_id":"854e98c0-f9c8-42c8-af33-75ac959ec5d1","resolution":{"observed_at":"2026-08-06T12:21:15.747936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.736161Z","title":"Flamingo: a visual language model for few-shot learning,","venue":null,"work_id":"3f33d0e2-8d66-455c-b25b-735967aef280","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.468917Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e88273d376a15e2fb63da851c66563e786197183749b64308f06be376f116bdc","observation_id":"150cff47-c133-499e-b3d4-4e4f410708fe","resolution":{"observed_at":"2026-08-06T12:21:15.739098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.471718Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.471718Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:52bfa4e4419abcc8db1393d8a907cfb752ba67d87ec2f921dbd051b75fcf75ca","observation_id":"c9a50e0a-707c-40ec-aa9b-29fad0d34079","resolution":{"observed_at":"2026-08-06T12:21:15.471718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.722328Z","title":"InstructBLIP: Towards General- purpose Vision-Language Models with Instruction Tuning,","venue":null,"work_id":"4c0347ec-f1ad-4a81-a29a-a74d6200a7d6","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.474586Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:295a6cdac0e5e0c8d00d1fc328fd295902e1a8a2e677fa4c0c963b9211bac4d4","observation_id":"6d0664b4-9995-4025-96fc-bc41d9ab6673","resolution":{"observed_at":"2026-08-06T12:21:15.725332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.713938Z","title":"Reveal: Retrieval- augmented visual-language pre-training with multi-source multimodal knowledge memory,","venue":null,"work_id":"81bf7e04-9f5d-4230-a32c-b217f4481ab6","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.477353Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:7b3bfa3bc5afa5a0a7f8d202bf6fcccf66e4cb5c267ba5b1fce4ba87cd2ffff0","observation_id":"796d7235-184e-4229-9233-793acc8f5ea4","resolution":{"observed_at":"2026-08-06T12:21:15.717080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.705390Z","title":"EchoSight: Advancing Visual-Language Models with Wiki Knowledge,","venue":null,"work_id":"b60a59cf-e679-4a0b-b15a-82ee24b120ea","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.480040Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:c2fa8ab1345fdfc9b8ee273b44ffb545aa4868ff9bec34af2009a72adb0f88cf","observation_id":"e0c4461a-5074-4713-aac2-d48110b8557c","resolution":{"observed_at":"2026-08-06T12:21:15.708566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.696403Z","title":"Wiki-llava: Hierarchical retrieval-augmented generation for multimodal llms,","venue":null,"work_id":"a6be181b-2642-4502-8722-03e4ac7854f2","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.482995Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:8f392784dd296b2956cc6e5b4eb0339dccc856ea3b3ea87001a98d2be36d96dd","observation_id":"d4765d3a-bea7-46bb-823d-108c18c0938c","resolution":{"observed_at":"2026-08-06T12:21:15.699418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-06T12:21:15.486031Z","title":"Qwen2. 5-vl technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.486031Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d206b4534f95d11d70f734c8cca30c45465c8c403350ae5c9768d3fce7485730","observation_id":"60a1f0ce-cbdd-4d57-87b8-6a4c87f7f94b","resolution":{"observed_at":"2026-08-06T12:21:15.486031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.687514Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools,","venue":null,"work_id":"8c3d074d-35ab-488c-8261-a5666fb81d38","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.489064Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e520d504268f54dc1b4ffeced69d988225cb78eaec6db97558d0dbe3950ebd75","observation_id":"b1ff6eda-cc9f-404b-bbe8-e9a45320b9d4","resolution":{"observed_at":"2026-08-06T12:21:15.690765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.491934Z","title":"React: Synergizing reasoning and acting in language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.491934Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:1767fc7e5e58883eb5d0c5190045766954603f18bbf578f1b28d24ae2a0d51d9","observation_id":"fcc00fb5-76f2-47b9-bf60-adc2a0cb7931","resolution":{"observed_at":"2026-08-06T12:21:15.491934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.673118Z","title":"Colpali: Efficient document retrieval with vision language models,","venue":null,"work_id":"fe6c1336-bdda-4670-b3e3-352bbc6ae00e","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.494631Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:6f635b1cbb0b3652baa052825132c9e46c9bd5444c675c05e20d51bc4a707448","observation_id":"6b1053e5-fd02-4592-9b53-b6675d9f4ca3","resolution":{"observed_at":"2026-08-06T12:21:15.676205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.664209Z","title":"Wikiextractor,","venue":null,"work_id":"65cbee2a-2e77-4ff1-9ec2-2ebce1cc5a2b","year":2012},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.497512Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d0738bb78bce1c7215d2bb5f6c4a333b645f9807ca9471eed79804f047f789f3","observation_id":"ad6f203a-8928-413e-a0fb-bbc6b3324ca3","resolution":{"observed_at":"2026-08-06T12:21:15.667440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14683","last_updated":"2024-09-23T03:12:43Z","snapshot_observed_at":"2026-08-07T06:44:36.523865Z","submitted_at":"2024-09-23T03:12:43Z","title":"Reducing the Footprint of Multi-Vector Retrieval with Minimal Performance Impact via Token Pooling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14683","snapshot_observed_at":"2026-08-06T12:21:15.500544Z","title":"Reducing the Footprint of Multi-Vector Retrieval with Minimal Performance Impact via Token Pooling,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.500544Z"},"links":{"cited_paper":"/paper/2409.14683","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:f201b6154494b53580da1452ae0161549197007dd790e8bde9bc00df12d62216","observation_id":"a4726a7f-1eef-41f6-918b-9b5744df52f2","resolution":{"observed_at":"2026-08-06T12:21:15.500544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.503536Z","title":"Sigmoid loss for language image pre-training,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.503536Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:05339f8e265f10c819b4bb200ca1b574f932d927170565d95e7c30130447ab9c","observation_id":"26aa50d7-e60d-4a0c-b09f-33329d07c8c3","resolution":{"observed_at":"2026-08-06T12:21:15.503536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.506829Z","title":"Multi-task learning using uncertainty to weigh losses for scene geometry and semantics,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.506829Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:289ae1ed5be2b53333adf4055c782ab7be4c3f0c3ca2db03caf0bd3e89245a03","observation_id":"8e9eafa2-5e31-410c-a71a-abea7a1ca900","resolution":{"observed_at":"2026-08-06T12:21:15.506829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.645965Z","title":"Art History: A Preliminary Handbook (1996)","venue":null,"work_id":"d2bf38b4-362a-49d5-9669-9b5097627b31","year":1996},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.509740Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:b10c7665938f2033fc65a298308c5b2777fda163a5f11505ad62ad5131f44542","observation_id":"c7b3448b-451d-4bf3-9902-a94a24d34c47","resolution":{"observed_at":"2026-08-06T12:21:15.648818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T12:21:15.512431Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.512431Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:13c4f99d8afc8780f4dd98106c1d98e6faf44e82c953ec9c1432c2b176c04db9","observation_id":"659a6a7a-cd51-4859-9372-5b7321fd9989","resolution":{"observed_at":"2026-08-06T12:21:15.512431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.637108Z","title":"Composed image retrieval using contrastive learning and task-oriented CLIP-based features,","venue":null,"work_id":"63768869-d75f-4509-9cb0-fb1ee88c66f2","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.515457Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:ecd1a4d22b93c0af7286d3b27a3d21cded5820953ea79cd1296606f6137b822b","observation_id":"e1c506db-022a-43b2-bcb1-58e6f8c08359","resolution":{"observed_at":"2026-08-06T12:21:15.640222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.628643Z","title":"Artwork interpretation,","venue":null,"work_id":"4b6c5a2a-f2fe-4fd6-8b43-9e1848f009df","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.518248Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:af427a220c919e721c21d5bf2e50bf49a6771dd23ad4b1a412573ebbdc63c55f","observation_id":"baa7618f-90fc-4fe1-b022-3a3206541662","resolution":{"observed_at":"2026-08-06T12:21:15.631502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.619881Z","title":"Artquest: Countering hidden language biases in artvqa,","venue":null,"work_id":"8f3669ce-b962-4653-a2b3-ada330d5b725","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.520982Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:ebb6151e240f4ce0c6b1082a13fb27ccbb05cc7d4ee51f91a88fe9966e269443","observation_id":"c841696b-c349-4518-a218-08b8bee8ffa7","resolution":{"observed_at":"2026-08-06T12:21:15.623100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T03:27:38.759723Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":1,"verified_fuzzy":35},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 4 inbound Pith citation observations for arXiv:2507.21917."}