{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:AKXIBHV76YHKOB3JKTMTLAAS7G","short_pith_number":"pith:AKXIBHV7","schema_version":"1.0","canonical_sha256":"02ae809ebff60ea7076954d9358012f9b20e717696ec127ca0261e9cd5c6ce80","source":{"kind":"arxiv","id":"2203.08101","version":2},"attestation_state":"computed","paper":{"title":"ARTEMIS: Attention-based Retrieval with Text-Explicit Matching and Implicit Similarity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CV","authors_text":"Diane Larlus, Gabriela Csurka, Ginger Delmas, Rafael Sampaio de Rezende","submitted_at":"2022-03-15T17:29:20Z","abstract_excerpt":"An intuitive way to search for images is to use queries composed of an example image and a complementary text. While the first provides rich and implicit context for the search, the latter explicitly calls for new traits, or specifies how some elements of the example image should be changed to retrieve the desired target image. Current approaches typically combine the features of each of the two elements of the query into a single representation, which can then be compared to the ones of the potential target images. Our work aims at shedding new light on the task by looking at it through the p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.08101","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-15T17:29:20Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"853e64b639b54d314feb63c228cf671eb7a5228e3605d3fc2bb69a5b794bb371","abstract_canon_sha256":"8d3ca1103c8534a509fb54cee0f1c7a3665b6e1e74ffc96b79d79644f0fa2c23"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:23:25.845266Z","signature_b64":"tLRQGJ49DnqRSywLIKsi9WAODHQBvQAinw45qahodyMN0BEzyULGibXCXTDjW1lVwiuzAxTOapTi7sDv8plbAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"02ae809ebff60ea7076954d9358012f9b20e717696ec127ca0261e9cd5c6ce80","last_reissued_at":"2026-07-05T04:23:25.844849Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:23:25.844849Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ARTEMIS: Attention-based Retrieval with Text-Explicit Matching and Implicit Similarity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CV","authors_text":"Diane Larlus, Gabriela Csurka, Ginger Delmas, Rafael Sampaio de Rezende","submitted_at":"2022-03-15T17:29:20Z","abstract_excerpt":"An intuitive way to search for images is to use queries composed of an example image and a complementary text. While the first provides rich and implicit context for the search, the latter explicitly calls for new traits, or specifies how some elements of the example image should be changed to retrieve the desired target image. Current approaches typically combine the features of each of the two elements of the query into a single representation, which can then be compared to the ones of the potential target images. Our work aims at shedding new light on the task by looking at it through the p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.08101","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.08101/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.08101","created_at":"2026-07-05T04:23:25.844905+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.08101v2","created_at":"2026-07-05T04:23:25.844905+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.08101","created_at":"2026-07-05T04:23:25.844905+00:00"},{"alias_kind":"pith_short_12","alias_value":"AKXIBHV76YHK","created_at":"2026-07-05T04:23:25.844905+00:00"},{"alias_kind":"pith_short_16","alias_value":"AKXIBHV76YHKOB3J","created_at":"2026-07-05T04:23:25.844905+00:00"},{"alias_kind":"pith_short_8","alias_value":"AKXIBHV7","created_at":"2026-07-05T04:23:25.844905+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07032","citing_title":"Never Seen Before: Benchmarking Genuine Zero-Shot Composed Image Retrieval with Consistent Video-Sourced Datasets","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00374","citing_title":"Learning to Compose: Revisiting Proxy Task Design for Zero-Shot Composed Image Retrieval","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03470","citing_title":"Mixed-Modality Dual Face-Hair Retrieval","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21261","citing_title":"STiTch: Semantic Transition and Transportation in Collaboration for Training-Free Zero-Shot Composed Image Retrieval","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G","json":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G.json","graph_json":"https://pith.science/api/pith-number/AKXIBHV76YHKOB3JKTMTLAAS7G/graph.json","events_json":"https://pith.science/api/pith-number/AKXIBHV76YHKOB3JKTMTLAAS7G/events.json","paper":"https://pith.science/paper/AKXIBHV7"},"agent_actions":{"view_html":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G","download_json":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G.json","view_paper":"https://pith.science/paper/AKXIBHV7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.08101&json=true","fetch_graph":"https://pith.science/api/pith-number/AKXIBHV76YHKOB3JKTMTLAAS7G/graph.json","fetch_events":"https://pith.science/api/pith-number/AKXIBHV76YHKOB3JKTMTLAAS7G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G/action/storage_attestation","attest_author":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G/action/author_attestation","sign_citation":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G/action/citation_signature","submit_replication":"https://pith.science/pith/AKXIBHV76YHKOB3JKTMTLAAS7G/action/replication_record"}},"created_at":"2026-07-05T04:23:25.844905+00:00","updated_at":"2026-07-05T04:23:25.844905+00:00"}