{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LBLDKNTBXDVYS2YIRRIPNQWQTZ","short_pith_number":"pith:LBLDKNTB","schema_version":"1.0","canonical_sha256":"5856353661b8eb896b088c50f6c2d09e5e3fb857de7ac526268d214c3caa4aa6","source":{"kind":"arxiv","id":"2502.20509","version":1},"attestation_state":"computed","paper":{"title":"CoCa-CXR: Contrastive Captioners Learn Strong Temporal Structures for Chest X-Ray Vision-Language Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Andrew Sellergren, Avinatan Hassidim, Daniel Golden, Lin Yang, Shawn Xu, Shravya Shetty, Yixiong Chen, Yossi Matias","submitted_at":"2025-02-27T20:39:03Z","abstract_excerpt":"Vision-language models have proven to be of great benefit for medical image analysis since they learn rich semantics from both images and reports. Prior efforts have focused on better alignment of image and text representations to enhance image understanding. However, though explicit reference to a prior image is common in Chest X-Ray (CXR) reports, aligning progression descriptions with the semantics differences in image pairs remains under-explored. In this work, we propose two components to address this issue. (1) A CXR report processing pipeline to extract temporal structure. It processes "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20509","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-27T20:39:03Z","cross_cats_sorted":[],"title_canon_sha256":"e07f43df9ccf4a5025dd6a30ed0f3beb1f2a05339fcbcf935aa4c6e33ef2158e","abstract_canon_sha256":"db3c60dc8e59c270b882295ae457f055f965cc96fc5122e25242b1697198cdd5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:34.393792Z","signature_b64":"n+tRQeOLhYRKpyPLhWaF3IuaiM1G2bhslc7stRwCAa1L10mBUcnlnQd/8RoCTsRxezin3Uii8uP5i9EJRojKBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5856353661b8eb896b088c50f6c2d09e5e3fb857de7ac526268d214c3caa4aa6","last_reissued_at":"2026-07-05T10:21:34.393265Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:34.393265Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoCa-CXR: Contrastive Captioners Learn Strong Temporal Structures for Chest X-Ray Vision-Language Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Andrew Sellergren, Avinatan Hassidim, Daniel Golden, Lin Yang, Shawn Xu, Shravya Shetty, Yixiong Chen, Yossi Matias","submitted_at":"2025-02-27T20:39:03Z","abstract_excerpt":"Vision-language models have proven to be of great benefit for medical image analysis since they learn rich semantics from both images and reports. Prior efforts have focused on better alignment of image and text representations to enhance image understanding. However, though explicit reference to a prior image is common in Chest X-Ray (CXR) reports, aligning progression descriptions with the semantics differences in image pairs remains under-explored. In this work, we propose two components to address this issue. (1) A CXR report processing pipeline to extract temporal structure. It processes "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20509","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20509/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20509","created_at":"2026-07-05T10:21:34.393332+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20509v1","created_at":"2026-07-05T10:21:34.393332+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20509","created_at":"2026-07-05T10:21:34.393332+00:00"},{"alias_kind":"pith_short_12","alias_value":"LBLDKNTBXDVY","created_at":"2026-07-05T10:21:34.393332+00:00"},{"alias_kind":"pith_short_16","alias_value":"LBLDKNTBXDVYS2YI","created_at":"2026-07-05T10:21:34.393332+00:00"},{"alias_kind":"pith_short_8","alias_value":"LBLDKNTB","created_at":"2026-07-05T10:21:34.393332+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05810","citing_title":"CXR-ContraBench: Benchmarking Negated-Option Attraction in Medical VLMs","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ","json":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ.json","graph_json":"https://pith.science/api/pith-number/LBLDKNTBXDVYS2YIRRIPNQWQTZ/graph.json","events_json":"https://pith.science/api/pith-number/LBLDKNTBXDVYS2YIRRIPNQWQTZ/events.json","paper":"https://pith.science/paper/LBLDKNTB"},"agent_actions":{"view_html":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ","download_json":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ.json","view_paper":"https://pith.science/paper/LBLDKNTB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20509&json=true","fetch_graph":"https://pith.science/api/pith-number/LBLDKNTBXDVYS2YIRRIPNQWQTZ/graph.json","fetch_events":"https://pith.science/api/pith-number/LBLDKNTBXDVYS2YIRRIPNQWQTZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ/action/storage_attestation","attest_author":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ/action/author_attestation","sign_citation":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ/action/citation_signature","submit_replication":"https://pith.science/pith/LBLDKNTBXDVYS2YIRRIPNQWQTZ/action/replication_record"}},"created_at":"2026-07-05T10:21:34.393332+00:00","updated_at":"2026-07-05T10:21:34.393332+00:00"}