{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:43ZLMAFEM646P4M4OXKWNLGZ6Q","short_pith_number":"pith:43ZLMAFE","schema_version":"1.0","canonical_sha256":"e6f2b600a467b9e7f19c75d566acd9f43ff2dd140d9193ffe80944291d655226","source":{"kind":"arxiv","id":"2410.10034","version":2},"attestation_state":"computed","paper":{"title":"TULIP: Token-length Upgraded CLIP","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cees G. M. Snoek, Ivona Najdenkoska, Marcel Worring, Mohammad Mahdi Derakhshani, Nanne van Noord, Yuki M. Asano","submitted_at":"2024-10-13T22:34:15Z","abstract_excerpt":"We address the challenge of representing long captions in vision-language models, such as CLIP. By design these models are limited by fixed, absolute positional encodings, restricting inputs to a maximum of 77 tokens and hindering performance on tasks requiring longer descriptions. Although recent work has attempted to overcome this limit, their proposed approaches struggle to model token relationships over longer distances and simply extend to a fixed new token length. Instead, we propose a generalizable method, named TULIP, able to upgrade the token length to any length for CLIP-like models."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10034","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-13T22:34:15Z","cross_cats_sorted":[],"title_canon_sha256":"44c9898521b53815fa2c2803cdff1e7ddf37b7d0385a0560981988ce886ab133","abstract_canon_sha256":"dc8f1452b1c6a317ff1ea4d4845c6ad2a576bb6d2b5321ab64fc7c6bfd8b0eae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:58.830445Z","signature_b64":"toWTXqdL2wsdVVGLpGKkzCnJNiRxJpsmZA62AAZ+sdqpyfUKZD+uRs8j2psnCQM0I0jgIPFHnLMvsc1riWXaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6f2b600a467b9e7f19c75d566acd9f43ff2dd140d9193ffe80944291d655226","last_reissued_at":"2026-07-05T10:40:58.829932Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:58.829932Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TULIP: Token-length Upgraded CLIP","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cees G. M. Snoek, Ivona Najdenkoska, Marcel Worring, Mohammad Mahdi Derakhshani, Nanne van Noord, Yuki M. Asano","submitted_at":"2024-10-13T22:34:15Z","abstract_excerpt":"We address the challenge of representing long captions in vision-language models, such as CLIP. By design these models are limited by fixed, absolute positional encodings, restricting inputs to a maximum of 77 tokens and hindering performance on tasks requiring longer descriptions. Although recent work has attempted to overcome this limit, their proposed approaches struggle to model token relationships over longer distances and simply extend to a fixed new token length. Instead, we propose a generalizable method, named TULIP, able to upgrade the token length to any length for CLIP-like models."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10034","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10034/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10034","created_at":"2026-07-05T10:40:58.829991+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10034v2","created_at":"2026-07-05T10:40:58.829991+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10034","created_at":"2026-07-05T10:40:58.829991+00:00"},{"alias_kind":"pith_short_12","alias_value":"43ZLMAFEM646","created_at":"2026-07-05T10:40:58.829991+00:00"},{"alias_kind":"pith_short_16","alias_value":"43ZLMAFEM646P4M4","created_at":"2026-07-05T10:40:58.829991+00:00"},{"alias_kind":"pith_short_8","alias_value":"43ZLMAFE","created_at":"2026-07-05T10:40:58.829991+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24253","citing_title":"TuringViT: Making SOTA Vision Transformers Accessible to All","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25012","citing_title":"Learning from Semantic Dictionaries: Discriminative Codebook Contrastive Learning for Unified Visual Representation and Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24253","citing_title":"TuringViT: Making SOTA Vision Transformers Accessible to All","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q","json":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q.json","graph_json":"https://pith.science/api/pith-number/43ZLMAFEM646P4M4OXKWNLGZ6Q/graph.json","events_json":"https://pith.science/api/pith-number/43ZLMAFEM646P4M4OXKWNLGZ6Q/events.json","paper":"https://pith.science/paper/43ZLMAFE"},"agent_actions":{"view_html":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q","download_json":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q.json","view_paper":"https://pith.science/paper/43ZLMAFE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10034&json=true","fetch_graph":"https://pith.science/api/pith-number/43ZLMAFEM646P4M4OXKWNLGZ6Q/graph.json","fetch_events":"https://pith.science/api/pith-number/43ZLMAFEM646P4M4OXKWNLGZ6Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q/action/storage_attestation","attest_author":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q/action/author_attestation","sign_citation":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q/action/citation_signature","submit_replication":"https://pith.science/pith/43ZLMAFEM646P4M4OXKWNLGZ6Q/action/replication_record"}},"created_at":"2026-07-05T10:40:58.829991+00:00","updated_at":"2026-07-05T10:40:58.829991+00:00"}