{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BBGYH46DZ4OLTJ5ZW7U75UW5L3","short_pith_number":"pith:BBGYH46D","schema_version":"1.0","canonical_sha256":"084d83f3c3cf1cb9a7b9b7e9fed2dd5eef96e1f367fd96431dfaab6c3c04246f","source":{"kind":"arxiv","id":"2504.07718","version":1},"attestation_state":"computed","paper":{"title":"Multi-modal Reference Learning for Fine-grained Text-to-Image Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hao Chen, Limin Su, Shiliang Zhang, Wei Zeng, Zehong Ma","submitted_at":"2025-04-10T13:09:52Z","abstract_excerpt":"Fine-grained text-to-image retrieval aims to retrieve a fine-grained target image with a given text query. Existing methods typically assume that each training image is accurately depicted by its textual descriptions. However, textual descriptions can be ambiguous and fail to depict discriminative visual details in images, leading to inaccurate representation learning. To alleviate the effects of text ambiguity, we propose a Multi-Modal Reference learning framework to learn robust representations. We first propose a multi-modal reference construction module to aggregate all visual and textual "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07718","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-10T13:09:52Z","cross_cats_sorted":[],"title_canon_sha256":"d1b520b5e9e021ca4fa1ab87b55ebd39660a16618ec0204dae693e0b6f1eb2eb","abstract_canon_sha256":"d94f202c36b83420f15b9900c3f6b6f747905354dc69ad54eb6ab525cbd4b355"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:13.782285Z","signature_b64":"uQospeCtoco9ukha5vvGRvFR0V3S8xrsqlqX1XP0+DscYPjEOD8mquzi2IuIYtGeDpq/8rH5zJTUucV3eKboCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"084d83f3c3cf1cb9a7b9b7e9fed2dd5eef96e1f367fd96431dfaab6c3c04246f","last_reissued_at":"2026-07-05T10:47:13.781820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:13.781820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-modal Reference Learning for Fine-grained Text-to-Image Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hao Chen, Limin Su, Shiliang Zhang, Wei Zeng, Zehong Ma","submitted_at":"2025-04-10T13:09:52Z","abstract_excerpt":"Fine-grained text-to-image retrieval aims to retrieve a fine-grained target image with a given text query. Existing methods typically assume that each training image is accurately depicted by its textual descriptions. However, textual descriptions can be ambiguous and fail to depict discriminative visual details in images, leading to inaccurate representation learning. To alleviate the effects of text ambiguity, we propose a Multi-Modal Reference learning framework to learn robust representations. We first propose a multi-modal reference construction module to aggregate all visual and textual "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07718","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07718","created_at":"2026-07-05T10:47:13.781875+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07718v1","created_at":"2026-07-05T10:47:13.781875+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07718","created_at":"2026-07-05T10:47:13.781875+00:00"},{"alias_kind":"pith_short_12","alias_value":"BBGYH46DZ4OL","created_at":"2026-07-05T10:47:13.781875+00:00"},{"alias_kind":"pith_short_16","alias_value":"BBGYH46DZ4OLTJ5Z","created_at":"2026-07-05T10:47:13.781875+00:00"},{"alias_kind":"pith_short_8","alias_value":"BBGYH46D","created_at":"2026-07-05T10:47:13.781875+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3","json":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3.json","graph_json":"https://pith.science/api/pith-number/BBGYH46DZ4OLTJ5ZW7U75UW5L3/graph.json","events_json":"https://pith.science/api/pith-number/BBGYH46DZ4OLTJ5ZW7U75UW5L3/events.json","paper":"https://pith.science/paper/BBGYH46D"},"agent_actions":{"view_html":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3","download_json":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3.json","view_paper":"https://pith.science/paper/BBGYH46D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07718&json=true","fetch_graph":"https://pith.science/api/pith-number/BBGYH46DZ4OLTJ5ZW7U75UW5L3/graph.json","fetch_events":"https://pith.science/api/pith-number/BBGYH46DZ4OLTJ5ZW7U75UW5L3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3/action/storage_attestation","attest_author":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3/action/author_attestation","sign_citation":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3/action/citation_signature","submit_replication":"https://pith.science/pith/BBGYH46DZ4OLTJ5ZW7U75UW5L3/action/replication_record"}},"created_at":"2026-07-05T10:47:13.781875+00:00","updated_at":"2026-07-05T10:47:13.781875+00:00"}