{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4EGGB7AZ2477SHH3KNWNVCBSDR","short_pith_number":"pith:4EGGB7AZ","schema_version":"1.0","canonical_sha256":"e10c60fc19d73ff91cfb536cda88321c59423d770aa1036fe8dbf64206027ca4","source":{"kind":"arxiv","id":"2505.15576","version":2},"attestation_state":"computed","paper":{"title":"Visual Perturbation and Adaptive Hard Negative Contrastive Learning for Compositional Reasoning in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Ruibin Li, Tong Jia, Wei Zheng, Xin Huang, Ya Wang","submitted_at":"2025-05-21T14:28:43Z","abstract_excerpt":"Vision-Language Models (VLMs) are essential for multimodal tasks, especially compositional reasoning (CR) tasks, which require distinguishing fine-grained semantic differences between visual and textual embeddings. However, existing methods primarily fine-tune the model by generating text-based hard negative samples, neglecting the importance of image-based negative samples, which results in insufficient training of the visual encoder and ultimately impacts the overall performance of the model. Moreover, negative samples are typically treated uniformly, without considering their difficulty lev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15576","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-21T14:28:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a3847549588c8de3eab09fff1e86e143808ce2f65cac7dc6c853445bf526839b","abstract_canon_sha256":"958eeb51eef121f7554524a3dd9325edd6c1d26e0eaf024b4415fc64621705ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:00:27.929419Z","signature_b64":"s02G9M27wXjEAx5PVmFnJ/KLlgVsJ6yvvGouxNiM80UynE0Cuo3cZDujXdqgFRwoc+FiZaJgV/qSUWf2wE8zBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e10c60fc19d73ff91cfb536cda88321c59423d770aa1036fe8dbf64206027ca4","last_reissued_at":"2026-07-05T12:00:27.928923Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:00:27.928923Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Perturbation and Adaptive Hard Negative Contrastive Learning for Compositional Reasoning in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Ruibin Li, Tong Jia, Wei Zheng, Xin Huang, Ya Wang","submitted_at":"2025-05-21T14:28:43Z","abstract_excerpt":"Vision-Language Models (VLMs) are essential for multimodal tasks, especially compositional reasoning (CR) tasks, which require distinguishing fine-grained semantic differences between visual and textual embeddings. However, existing methods primarily fine-tune the model by generating text-based hard negative samples, neglecting the importance of image-based negative samples, which results in insufficient training of the visual encoder and ultimately impacts the overall performance of the model. Moreover, negative samples are typically treated uniformly, without considering their difficulty lev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15576","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15576/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15576","created_at":"2026-07-05T12:00:27.928989+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15576v2","created_at":"2026-07-05T12:00:27.928989+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15576","created_at":"2026-07-05T12:00:27.928989+00:00"},{"alias_kind":"pith_short_12","alias_value":"4EGGB7AZ2477","created_at":"2026-07-05T12:00:27.928989+00:00"},{"alias_kind":"pith_short_16","alias_value":"4EGGB7AZ2477SHH3","created_at":"2026-07-05T12:00:27.928989+00:00"},{"alias_kind":"pith_short_8","alias_value":"4EGGB7AZ","created_at":"2026-07-05T12:00:27.928989+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR","json":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR.json","graph_json":"https://pith.science/api/pith-number/4EGGB7AZ2477SHH3KNWNVCBSDR/graph.json","events_json":"https://pith.science/api/pith-number/4EGGB7AZ2477SHH3KNWNVCBSDR/events.json","paper":"https://pith.science/paper/4EGGB7AZ"},"agent_actions":{"view_html":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR","download_json":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR.json","view_paper":"https://pith.science/paper/4EGGB7AZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15576&json=true","fetch_graph":"https://pith.science/api/pith-number/4EGGB7AZ2477SHH3KNWNVCBSDR/graph.json","fetch_events":"https://pith.science/api/pith-number/4EGGB7AZ2477SHH3KNWNVCBSDR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR/action/storage_attestation","attest_author":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR/action/author_attestation","sign_citation":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR/action/citation_signature","submit_replication":"https://pith.science/pith/4EGGB7AZ2477SHH3KNWNVCBSDR/action/replication_record"}},"created_at":"2026-07-05T12:00:27.928989+00:00","updated_at":"2026-07-05T12:00:27.928989+00:00"}