{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O2F4IW5GIP6ISKY7QTNFUIUX4A","short_pith_number":"pith:O2F4IW5G","schema_version":"1.0","canonical_sha256":"768bc45ba643fc892b1f84da5a2297e03597eb16c51a80c09469f2daa1dde64d","source":{"kind":"arxiv","id":"2405.19315","version":2},"attestation_state":"computed","paper":{"title":"Matryoshka Query Transformer for Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Amita Kamath, Kai-Wei Chang, Liunian Harold Li, Nanyun Peng, Wenbo Hu, Zi-Yi Dou","submitted_at":"2024-05-29T17:39:42Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) typically encode an image into a fixed number of visual tokens (e.g., 576) and process these tokens with a language model. Despite their strong performance, LVLMs face challenges in adapting to varying computational constraints. This raises the question: can we achieve flexibility in the number of visual tokens to suit different tasks and computational resources? We answer this with an emphatic yes. Inspired by Matryoshka Representation Learning, we introduce the Matryoshka Query Transformer (MQT), capable of encoding an image into m visual tokens during in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19315","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-29T17:39:42Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"e7500f381870bcec8a215fc4659c05210a3a8076311dab75aef21c1e46737122","abstract_canon_sha256":"0fdf3819399afa0f0bfc28b795dd792bfc54c6bc3a7cd432eaaeb6e2c73b3f9e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:41.566748Z","signature_b64":"xTmW38DR1IoWTK3TfZPM1VKHlzVo0abLqUcW+thbd1ZM/N5F5CjyUvLezOXAT7FyJnERBF3x6fNfbZne5uHVCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"768bc45ba643fc892b1f84da5a2297e03597eb16c51a80c09469f2daa1dde64d","last_reissued_at":"2026-07-05T08:28:41.566286Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:41.566286Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Matryoshka Query Transformer for Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Amita Kamath, Kai-Wei Chang, Liunian Harold Li, Nanyun Peng, Wenbo Hu, Zi-Yi Dou","submitted_at":"2024-05-29T17:39:42Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) typically encode an image into a fixed number of visual tokens (e.g., 576) and process these tokens with a language model. Despite their strong performance, LVLMs face challenges in adapting to varying computational constraints. This raises the question: can we achieve flexibility in the number of visual tokens to suit different tasks and computational resources? We answer this with an emphatic yes. Inspired by Matryoshka Representation Learning, we introduce the Matryoshka Query Transformer (MQT), capable of encoding an image into m visual tokens during in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19315","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19315/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19315","created_at":"2026-07-05T08:28:41.566355+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19315v2","created_at":"2026-07-05T08:28:41.566355+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19315","created_at":"2026-07-05T08:28:41.566355+00:00"},{"alias_kind":"pith_short_12","alias_value":"O2F4IW5GIP6I","created_at":"2026-07-05T08:28:41.566355+00:00"},{"alias_kind":"pith_short_16","alias_value":"O2F4IW5GIP6ISKY7","created_at":"2026-07-05T08:28:41.566355+00:00"},{"alias_kind":"pith_short_8","alias_value":"O2F4IW5G","created_at":"2026-07-05T08:28:41.566355+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06158","citing_title":"Adaptive Tokenisation Via Temporal Redundancy Masking And Latent Inpainting","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06038","citing_title":"Fourier Compressor: Frequency-Domain Visual Token Compression for Vision-Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24374","citing_title":"MIPIC: Matryoshka Representation Learning via Self-Distilled Intra-Relational and Progressive Information Chaining","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A","json":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A.json","graph_json":"https://pith.science/api/pith-number/O2F4IW5GIP6ISKY7QTNFUIUX4A/graph.json","events_json":"https://pith.science/api/pith-number/O2F4IW5GIP6ISKY7QTNFUIUX4A/events.json","paper":"https://pith.science/paper/O2F4IW5G"},"agent_actions":{"view_html":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A","download_json":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A.json","view_paper":"https://pith.science/paper/O2F4IW5G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19315&json=true","fetch_graph":"https://pith.science/api/pith-number/O2F4IW5GIP6ISKY7QTNFUIUX4A/graph.json","fetch_events":"https://pith.science/api/pith-number/O2F4IW5GIP6ISKY7QTNFUIUX4A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A/action/storage_attestation","attest_author":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A/action/author_attestation","sign_citation":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A/action/citation_signature","submit_replication":"https://pith.science/pith/O2F4IW5GIP6ISKY7QTNFUIUX4A/action/replication_record"}},"created_at":"2026-07-05T08:28:41.566355+00:00","updated_at":"2026-07-05T08:28:41.566355+00:00"}