{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AHLOUWR7WHMJGJTSTNWOB4QOM6","short_pith_number":"pith:AHLOUWR7","schema_version":"1.0","canonical_sha256":"01d6ea5a3fb1d89326729b6ce0f20e67bc0291a96b2c193e4a778303a16b891d","source":{"kind":"arxiv","id":"2302.11352","version":1},"attestation_state":"computed","paper":{"title":"X-TRA: Improving Chest X-ray Tasks with Cross-Modal Retrieval Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Marcel Worring, Tom van Sonsbeek","submitted_at":"2023-02-22T12:53:33Z","abstract_excerpt":"An important component of human analysis of medical images and their context is the ability to relate newly seen things to related instances in our memory. In this paper we mimic this ability by using multi-modal retrieval augmentation and apply it to several tasks in chest X-ray analysis. By retrieving similar images and/or radiology reports we expand and regularize the case at hand with additional knowledge, while maintaining factual knowledge consistency. The method consists of two components. First, vision and language modalities are aligned using a pre-trained CLIP model. To enforce that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.11352","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-02-22T12:53:33Z","cross_cats_sorted":[],"title_canon_sha256":"9f582812e9ae61b85e79bee44748ce8281a7415545b5230e6f4172ceecb9ff30","abstract_canon_sha256":"b53ad303710695325f00dd21a9d621c2a15844540f3853605c7e2a6b08e1d083"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:44:42.757116Z","signature_b64":"ZqPoDYtlGBjU/5KWYHAN6caO8aKHmeVRoZ8I2TZqreHSZdLvcZR2r554uynSSrsQBb/4Kx6AiWEsbDlRW9PPCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"01d6ea5a3fb1d89326729b6ce0f20e67bc0291a96b2c193e4a778303a16b891d","last_reissued_at":"2026-07-05T05:44:42.756687Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:44:42.756687Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"X-TRA: Improving Chest X-ray Tasks with Cross-Modal Retrieval Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Marcel Worring, Tom van Sonsbeek","submitted_at":"2023-02-22T12:53:33Z","abstract_excerpt":"An important component of human analysis of medical images and their context is the ability to relate newly seen things to related instances in our memory. In this paper we mimic this ability by using multi-modal retrieval augmentation and apply it to several tasks in chest X-ray analysis. By retrieving similar images and/or radiology reports we expand and regularize the case at hand with additional knowledge, while maintaining factual knowledge consistency. The method consists of two components. First, vision and language modalities are aligned using a pre-trained CLIP model. To enforce that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.11352","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.11352/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.11352","created_at":"2026-07-05T05:44:42.756745+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.11352v1","created_at":"2026-07-05T05:44:42.756745+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.11352","created_at":"2026-07-05T05:44:42.756745+00:00"},{"alias_kind":"pith_short_12","alias_value":"AHLOUWR7WHMJ","created_at":"2026-07-05T05:44:42.756745+00:00"},{"alias_kind":"pith_short_16","alias_value":"AHLOUWR7WHMJGJTS","created_at":"2026-07-05T05:44:42.756745+00:00"},{"alias_kind":"pith_short_8","alias_value":"AHLOUWR7","created_at":"2026-07-05T05:44:42.756745+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.15929","citing_title":"Meta-Entity Driven Triplet Mining for Aligning Medical Vision-Language Models","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6","json":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6.json","graph_json":"https://pith.science/api/pith-number/AHLOUWR7WHMJGJTSTNWOB4QOM6/graph.json","events_json":"https://pith.science/api/pith-number/AHLOUWR7WHMJGJTSTNWOB4QOM6/events.json","paper":"https://pith.science/paper/AHLOUWR7"},"agent_actions":{"view_html":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6","download_json":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6.json","view_paper":"https://pith.science/paper/AHLOUWR7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.11352&json=true","fetch_graph":"https://pith.science/api/pith-number/AHLOUWR7WHMJGJTSTNWOB4QOM6/graph.json","fetch_events":"https://pith.science/api/pith-number/AHLOUWR7WHMJGJTSTNWOB4QOM6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6/action/storage_attestation","attest_author":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6/action/author_attestation","sign_citation":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6/action/citation_signature","submit_replication":"https://pith.science/pith/AHLOUWR7WHMJGJTSTNWOB4QOM6/action/replication_record"}},"created_at":"2026-07-05T05:44:42.756745+00:00","updated_at":"2026-07-05T05:44:42.756745+00:00"}