{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:GBKIQFU3HOGRVRY4NESIIJ5WWQ","short_pith_number":"pith:GBKIQFU3","schema_version":"1.0","canonical_sha256":"305488169b3b8d1ac71c69248427b6b40f624680c159e26de8f11f6e89bc6f46","source":{"kind":"arxiv","id":"2203.16778","version":1},"attestation_state":"computed","paper":{"title":"ViSTA: Vision and Scene Text Aggregation for Cross-Modal Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Errui Ding, Guoli Song, Jie Chen, Jingdong Wang, Jingtuo Liu, Junyu Han, Kun Yao, Longchao Wang, Mengjun Cheng, Xiongwei Zhu, Yipeng Sun","submitted_at":"2022-03-31T03:40:21Z","abstract_excerpt":"Visual appearance is considered to be the most important cue to understand images for cross-modal retrieval, while sometimes the scene text appearing in images can provide valuable information to understand the visual semantics. Most of existing cross-modal retrieval approaches ignore the usage of scene text information and directly adding this information may lead to performance degradation in scene text free scenarios. To address this issue, we propose a full transformer architecture to unify these cross-modal retrieval scenarios in a single $\\textbf{Vi}$sion and $\\textbf{S}$cene $\\textbf{T}"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.16778","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-31T03:40:21Z","cross_cats_sorted":[],"title_canon_sha256":"cafdd5ae3b8009d93bd2ff17cb6c3de979156f4977a62cca047851c7c66b4d82","abstract_canon_sha256":"b600e2d188ef3c3371b06591561ff20920002640574c40304c2341db5bb76f5f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:27.652609Z","signature_b64":"ogp5KV3T4hAm+FIxfV32Js9Vb9+BjoxCqQyhYGVyjQS4lc5P/BWFnRPaYmcbERqPTVzCr9kNIcGi7Y3rCo2FCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"305488169b3b8d1ac71c69248427b6b40f624680c159e26de8f11f6e89bc6f46","last_reissued_at":"2026-07-05T04:10:27.652192Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:27.652192Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViSTA: Vision and Scene Text Aggregation for Cross-Modal Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Errui Ding, Guoli Song, Jie Chen, Jingdong Wang, Jingtuo Liu, Junyu Han, Kun Yao, Longchao Wang, Mengjun Cheng, Xiongwei Zhu, Yipeng Sun","submitted_at":"2022-03-31T03:40:21Z","abstract_excerpt":"Visual appearance is considered to be the most important cue to understand images for cross-modal retrieval, while sometimes the scene text appearing in images can provide valuable information to understand the visual semantics. Most of existing cross-modal retrieval approaches ignore the usage of scene text information and directly adding this information may lead to performance degradation in scene text free scenarios. To address this issue, we propose a full transformer architecture to unify these cross-modal retrieval scenarios in a single $\\textbf{Vi}$sion and $\\textbf{S}$cene $\\textbf{T}"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.16778","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.16778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.16778","created_at":"2026-07-05T04:10:27.652252+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.16778v1","created_at":"2026-07-05T04:10:27.652252+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.16778","created_at":"2026-07-05T04:10:27.652252+00:00"},{"alias_kind":"pith_short_12","alias_value":"GBKIQFU3HOGR","created_at":"2026-07-05T04:10:27.652252+00:00"},{"alias_kind":"pith_short_16","alias_value":"GBKIQFU3HOGRVRY4","created_at":"2026-07-05T04:10:27.652252+00:00"},{"alias_kind":"pith_short_8","alias_value":"GBKIQFU3","created_at":"2026-07-05T04:10:27.652252+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ","json":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ.json","graph_json":"https://pith.science/api/pith-number/GBKIQFU3HOGRVRY4NESIIJ5WWQ/graph.json","events_json":"https://pith.science/api/pith-number/GBKIQFU3HOGRVRY4NESIIJ5WWQ/events.json","paper":"https://pith.science/paper/GBKIQFU3"},"agent_actions":{"view_html":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ","download_json":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ.json","view_paper":"https://pith.science/paper/GBKIQFU3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.16778&json=true","fetch_graph":"https://pith.science/api/pith-number/GBKIQFU3HOGRVRY4NESIIJ5WWQ/graph.json","fetch_events":"https://pith.science/api/pith-number/GBKIQFU3HOGRVRY4NESIIJ5WWQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ/action/storage_attestation","attest_author":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ/action/author_attestation","sign_citation":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ/action/citation_signature","submit_replication":"https://pith.science/pith/GBKIQFU3HOGRVRY4NESIIJ5WWQ/action/replication_record"}},"created_at":"2026-07-05T04:10:27.652252+00:00","updated_at":"2026-07-05T04:10:27.652252+00:00"}