{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2GQD4WFAPCLFIR743KCRORC4WM","short_pith_number":"pith:2GQD4WFA","schema_version":"1.0","canonical_sha256":"d1a03e58a078965447fcda8517445cb323dd072505c49df3dfa5a921b252b6a9","source":{"kind":"arxiv","id":"2410.06912","version":2},"attestation_state":"computed","paper":{"title":"Compositional Entailment Learning for Hyperbolic Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alessandro Flaborea, Avik Pal, Fabio Galasso, Guido Maria D'Amely di Melendugno, Max van Spengler, Pascal Mettes","submitted_at":"2024-10-09T14:12:50Z","abstract_excerpt":"Image-text representation learning forms a cornerstone in vision-language models, where pairs of images and textual descriptions are contrastively aligned in a shared embedding space. Since visual and textual concepts are naturally hierarchical, recent work has shown that hyperbolic space can serve as a high-potential manifold to learn vision-language representation with strong downstream performance. In this work, for the first time we show how to fully leverage the innate hierarchical nature of hyperbolic embeddings by looking beyond individual image-text pairs. We propose Compositional Enta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.06912","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-09T14:12:50Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"61de5fb2c16e8a9bfb14a8c07a06d52ef6cc38dc57a87b212ca7b4b86a7b54d6","abstract_canon_sha256":"84ab28abf71adf08f311bedff7bd79269ff049941ec14719a440da21997d500a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:06.889976Z","signature_b64":"TdBz6tFowfiGagb5OuDAMj2/C74HkdbSzQFY9dmthvSwEEKzlm4ZPoK5XWIruV1/+dAiofyzRhBgMsn9hPEPBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1a03e58a078965447fcda8517445cb323dd072505c49df3dfa5a921b252b6a9","last_reissued_at":"2026-07-05T10:22:06.889464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:06.889464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compositional Entailment Learning for Hyperbolic Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alessandro Flaborea, Avik Pal, Fabio Galasso, Guido Maria D'Amely di Melendugno, Max van Spengler, Pascal Mettes","submitted_at":"2024-10-09T14:12:50Z","abstract_excerpt":"Image-text representation learning forms a cornerstone in vision-language models, where pairs of images and textual descriptions are contrastively aligned in a shared embedding space. Since visual and textual concepts are naturally hierarchical, recent work has shown that hyperbolic space can serve as a high-potential manifold to learn vision-language representation with strong downstream performance. In this work, for the first time we show how to fully leverage the innate hierarchical nature of hyperbolic embeddings by looking beyond individual image-text pairs. We propose Compositional Enta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.06912","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.06912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.06912","created_at":"2026-07-05T10:22:06.889530+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.06912v2","created_at":"2026-07-05T10:22:06.889530+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.06912","created_at":"2026-07-05T10:22:06.889530+00:00"},{"alias_kind":"pith_short_12","alias_value":"2GQD4WFAPCLF","created_at":"2026-07-05T10:22:06.889530+00:00"},{"alias_kind":"pith_short_16","alias_value":"2GQD4WFAPCLFIR74","created_at":"2026-07-05T10:22:06.889530+00:00"},{"alias_kind":"pith_short_8","alias_value":"2GQD4WFA","created_at":"2026-07-05T10:22:06.889530+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21838","citing_title":"Beyond Flat Labels: Level-Restricted Contrastive Learning for Hierarchical Fine-Grained Vision Classification","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31245","citing_title":"HyperVLP: Enhancing Hierarchical Surgical Video-Language Pre-training in Hyperbolic Space","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23058","citing_title":"GeoWorld: Geometric World Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06440","citing_title":"Hyperbolic Concept Bottleneck Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06440","citing_title":"Hyperbolic Concept Bottleneck Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17454","citing_title":"HSG: Hyperbolic Scene Graph","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM","json":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM.json","graph_json":"https://pith.science/api/pith-number/2GQD4WFAPCLFIR743KCRORC4WM/graph.json","events_json":"https://pith.science/api/pith-number/2GQD4WFAPCLFIR743KCRORC4WM/events.json","paper":"https://pith.science/paper/2GQD4WFA"},"agent_actions":{"view_html":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM","download_json":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM.json","view_paper":"https://pith.science/paper/2GQD4WFA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.06912&json=true","fetch_graph":"https://pith.science/api/pith-number/2GQD4WFAPCLFIR743KCRORC4WM/graph.json","fetch_events":"https://pith.science/api/pith-number/2GQD4WFAPCLFIR743KCRORC4WM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM/action/storage_attestation","attest_author":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM/action/author_attestation","sign_citation":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM/action/citation_signature","submit_replication":"https://pith.science/pith/2GQD4WFAPCLFIR743KCRORC4WM/action/replication_record"}},"created_at":"2026-07-05T10:22:06.889530+00:00","updated_at":"2026-07-05T10:22:06.889530+00:00"}