{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HBH6RTMR5RVCBWUJAGYDO6252M","short_pith_number":"pith:HBH6RTMR","schema_version":"1.0","canonical_sha256":"384fe8cd91ec6a20da8901b0377b5dd3248f5dbada173bf27044177ee7a00e92","source":{"kind":"arxiv","id":"2306.08832","version":4},"attestation_state":"computed","paper":{"title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aishwarya Agrawal, Le Zhang, Rabiul Awal","submitted_at":"2023-06-15T03:26:28Z","abstract_excerpt":"Vision-Language Models (VLMs), such as CLIP, exhibit strong image-text comprehension abilities, facilitating advances in several downstream tasks such as zero-shot image classification, image-text retrieval, and text-to-image generation. However, the compositional reasoning abilities of existing VLMs remains subpar. The root of this limitation lies in the inadequate alignment between the images and captions in the pretraining datasets. Additionally, the current contrastive learning objective fails to focus on fine-grained grounding components like relations, actions, and attributes, resulting "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.08832","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-06-15T03:26:28Z","cross_cats_sorted":[],"title_canon_sha256":"55e84b8de8cda700f06f6ddcd42c85e00e473435c1c77bb4c0d1497c930000b8","abstract_canon_sha256":"f56e695e8dda9bbba2acb38af03b316a2caa98cc94a365b5b36341913fcef134"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:11:53.206366Z","signature_b64":"r7B3HKnYmU1g4tefiV9pZPM5srh8ayd5vcUQweZ/R6c0IinMj33M/aqMv9xfinL/ZZi9Xx9xncYliOf+udVrDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"384fe8cd91ec6a20da8901b0377b5dd3248f5dbada173bf27044177ee7a00e92","last_reissued_at":"2026-07-05T08:11:53.205893Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:11:53.205893Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aishwarya Agrawal, Le Zhang, Rabiul Awal","submitted_at":"2023-06-15T03:26:28Z","abstract_excerpt":"Vision-Language Models (VLMs), such as CLIP, exhibit strong image-text comprehension abilities, facilitating advances in several downstream tasks such as zero-shot image classification, image-text retrieval, and text-to-image generation. However, the compositional reasoning abilities of existing VLMs remains subpar. The root of this limitation lies in the inadequate alignment between the images and captions in the pretraining datasets. Additionally, the current contrastive learning objective fails to focus on fine-grained grounding components like relations, actions, and attributes, resulting "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.08832","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.08832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.08832","created_at":"2026-07-05T08:11:53.205955+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.08832v4","created_at":"2026-07-05T08:11:53.205955+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.08832","created_at":"2026-07-05T08:11:53.205955+00:00"},{"alias_kind":"pith_short_12","alias_value":"HBH6RTMR5RVC","created_at":"2026-07-05T08:11:53.205955+00:00"},{"alias_kind":"pith_short_16","alias_value":"HBH6RTMR5RVCBWUJ","created_at":"2026-07-05T08:11:53.205955+00:00"},{"alias_kind":"pith_short_8","alias_value":"HBH6RTMR","created_at":"2026-07-05T08:11:53.205955+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.17267","citing_title":"CF-VLM:CounterFactual Vision-Language Fine-tuning","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M","json":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M.json","graph_json":"https://pith.science/api/pith-number/HBH6RTMR5RVCBWUJAGYDO6252M/graph.json","events_json":"https://pith.science/api/pith-number/HBH6RTMR5RVCBWUJAGYDO6252M/events.json","paper":"https://pith.science/paper/HBH6RTMR"},"agent_actions":{"view_html":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M","download_json":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M.json","view_paper":"https://pith.science/paper/HBH6RTMR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.08832&json=true","fetch_graph":"https://pith.science/api/pith-number/HBH6RTMR5RVCBWUJAGYDO6252M/graph.json","fetch_events":"https://pith.science/api/pith-number/HBH6RTMR5RVCBWUJAGYDO6252M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M/action/storage_attestation","attest_author":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M/action/author_attestation","sign_citation":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M/action/citation_signature","submit_replication":"https://pith.science/pith/HBH6RTMR5RVCBWUJAGYDO6252M/action/replication_record"}},"created_at":"2026-07-05T08:11:53.205955+00:00","updated_at":"2026-07-05T08:11:53.205955+00:00"}