{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E7PSPX7P4YMU52FKLTYUOTLBDZ","short_pith_number":"pith:E7PSPX7P","schema_version":"1.0","canonical_sha256":"27df27dfefe6194ee8aa5cf1474d611e46d3802198d2174b59100f7df4b38408","source":{"kind":"arxiv","id":"2403.15378","version":3},"attestation_state":"computed","paper":{"title":"Long-CLIP: Unlocking the Long-Text Capability of CLIP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Beichen Zhang, Jiaqi Wang, Pan Zhang, Xiaoyi Dong, Yuhang Zang","submitted_at":"2024-03-22T17:58:16Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) has been the cornerstone for zero-shot classification, text-image retrieval, and text-image generation by aligning image and text modalities. Despite its widespread adoption, a significant limitation of CLIP lies in the inadequate length of text input. The length of the text token is restricted to 77, and an empirical study shows the actual effective length is even less than 20. This prevents CLIP from handling detailed descriptions, limiting its applications for image retrieval and text-to-image generation with extensive prerequisites. To this en"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.15378","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-22T17:58:16Z","cross_cats_sorted":[],"title_canon_sha256":"3468a7570f550013abf8b02387ec2d71f301822381920284fb50aea6540e338c","abstract_canon_sha256":"aefbcafcdd368a1d8b54c56445bd456c1a398476dd37147721481423cfd0aab8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:23.191556Z","signature_b64":"b2zvamLOLrRC1kH4afk/1zjvny3GIOq0j0JG6CaHJCUoFMxhmHq38bnhObzAZ0i2USafM3p/PZ3b4HnYzXTBAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27df27dfefe6194ee8aa5cf1474d611e46d3802198d2174b59100f7df4b38408","last_reissued_at":"2026-07-05T08:46:23.191065Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:23.191065Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long-CLIP: Unlocking the Long-Text Capability of CLIP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Beichen Zhang, Jiaqi Wang, Pan Zhang, Xiaoyi Dong, Yuhang Zang","submitted_at":"2024-03-22T17:58:16Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) has been the cornerstone for zero-shot classification, text-image retrieval, and text-image generation by aligning image and text modalities. Despite its widespread adoption, a significant limitation of CLIP lies in the inadequate length of text input. The length of the text token is restricted to 77, and an empirical study shows the actual effective length is even less than 20. This prevents CLIP from handling detailed descriptions, limiting its applications for image retrieval and text-to-image generation with extensive prerequisites. To this en"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.15378","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.15378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.15378","created_at":"2026-07-05T08:46:23.191123+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.15378v3","created_at":"2026-07-05T08:46:23.191123+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.15378","created_at":"2026-07-05T08:46:23.191123+00:00"},{"alias_kind":"pith_short_12","alias_value":"E7PSPX7P4YMU","created_at":"2026-07-05T08:46:23.191123+00:00"},{"alias_kind":"pith_short_16","alias_value":"E7PSPX7P4YMU52FK","created_at":"2026-07-05T08:46:23.191123+00:00"},{"alias_kind":"pith_short_8","alias_value":"E7PSPX7P","created_at":"2026-07-05T08:46:23.191123+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24253","citing_title":"TuringViT: Making SOTA Vision Transformers Accessible to All","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28421","citing_title":"JuZhou 1.0 Technical Report: The First Edge-Native Text-to-Image Foundation Model Trained Entirely on China-Developed AI Accelerators","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24253","citing_title":"TuringViT: Making SOTA Vision Transformers Accessible to All","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16241","citing_title":"Offline Semantic Guidance for Efficient Vision-Language-Action Policy Distillation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22378","citing_title":"Zero-Effort Image-to-Music Generation: An Interpretable RAG-based VLM Approach","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":171,"is_internal_anchor":false},{"citing_arxiv_id":"2407.12580","citing_title":"E5-V: Universal Embeddings with Multimodal Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14594","citing_title":"LFS: Learnable Frame Selector for Event-Aware and Temporally Diverse Video Captioning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02834","citing_title":"VideoNet: A Large-Scale Dataset for Domain-Specific Action Recognition","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ","json":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ.json","graph_json":"https://pith.science/api/pith-number/E7PSPX7P4YMU52FKLTYUOTLBDZ/graph.json","events_json":"https://pith.science/api/pith-number/E7PSPX7P4YMU52FKLTYUOTLBDZ/events.json","paper":"https://pith.science/paper/E7PSPX7P"},"agent_actions":{"view_html":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ","download_json":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ.json","view_paper":"https://pith.science/paper/E7PSPX7P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.15378&json=true","fetch_graph":"https://pith.science/api/pith-number/E7PSPX7P4YMU52FKLTYUOTLBDZ/graph.json","fetch_events":"https://pith.science/api/pith-number/E7PSPX7P4YMU52FKLTYUOTLBDZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ/action/storage_attestation","attest_author":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ/action/author_attestation","sign_citation":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ/action/citation_signature","submit_replication":"https://pith.science/pith/E7PSPX7P4YMU52FKLTYUOTLBDZ/action/replication_record"}},"created_at":"2026-07-05T08:46:23.191123+00:00","updated_at":"2026-07-05T08:46:23.191123+00:00"}