{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:MDQBQAOS4IKTTFL44XYUUG4BXE","short_pith_number":"pith:MDQBQAOS","schema_version":"1.0","canonical_sha256":"60e01801d2e21539957ce5f14a1b81b92f1287cbae5c7a9ddf04ba01dce54e16","source":{"kind":"arxiv","id":"2603.15685","version":2},"attestation_state":"computed","paper":{"title":"DASH: Dynamic Audio-Driven Semantic Chunking for Efficient Omnimodal Token Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.SD"],"primary_cat":"cs.MM","authors_text":"Bingzhou Li, Tao Huang","submitted_at":"2026-03-15T15:22:06Z","abstract_excerpt":"Omnimodal large language models (OmniLLMs) jointly process audio and visual streams, but the resulting long multimodal token sequences make inference prohibitively expensive. Existing compression methods typically rely on fixed window partitioning and attention-based pruning, which overlook the piecewise semantic structure of audio-visual signals and become fragile under aggressive token reduction. We propose Dynamic Audio-driven Semantic cHunking (DASH), a training-free framework that aligns token compression with semantic structure. DASH treats audio embeddings as a semantic anchor and detec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.15685","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.MM","submitted_at":"2026-03-15T15:22:06Z","cross_cats_sorted":["cs.AI","cs.CV","cs.SD"],"title_canon_sha256":"c8db9f49970a24b5f221f035101e4770ecb1bcb9928503d18d6b7f4339c385a9","abstract_canon_sha256":"6f5011e12dad64905dea0050f8277da3d38f0b5fc6de53349d3bbecffb1a9ad6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-09T01:19:52.942647Z","signature_b64":"vx5QT6PIg3tDZHH9vEjoMS8fdY6TX/b/l2CzS7KuC6suubUqS2A84ygEqw38aHF3R1N0ZGnwtY2H3Ng1XPKEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60e01801d2e21539957ce5f14a1b81b92f1287cbae5c7a9ddf04ba01dce54e16","last_reissued_at":"2026-07-09T01:19:52.942124Z","signature_status":"signed_v1","first_computed_at":"2026-07-09T01:19:52.942124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DASH: Dynamic Audio-Driven Semantic Chunking for Efficient Omnimodal Token Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.SD"],"primary_cat":"cs.MM","authors_text":"Bingzhou Li, Tao Huang","submitted_at":"2026-03-15T15:22:06Z","abstract_excerpt":"Omnimodal large language models (OmniLLMs) jointly process audio and visual streams, but the resulting long multimodal token sequences make inference prohibitively expensive. Existing compression methods typically rely on fixed window partitioning and attention-based pruning, which overlook the piecewise semantic structure of audio-visual signals and become fragile under aggressive token reduction. We propose Dynamic Audio-driven Semantic cHunking (DASH), a training-free framework that aligns token compression with semantic structure. DASH treats audio embeddings as a semantic anchor and detec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.15685","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.15685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.15685","created_at":"2026-07-09T01:19:52.942183+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.15685v2","created_at":"2026-07-09T01:19:52.942183+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.15685","created_at":"2026-07-09T01:19:52.942183+00:00"},{"alias_kind":"pith_short_12","alias_value":"MDQBQAOS4IKT","created_at":"2026-07-09T01:19:52.942183+00:00"},{"alias_kind":"pith_short_16","alias_value":"MDQBQAOS4IKTTFL4","created_at":"2026-07-09T01:19:52.942183+00:00"},{"alias_kind":"pith_short_8","alias_value":"MDQBQAOS","created_at":"2026-07-09T01:19:52.942183+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.10147","citing_title":"From Senses to Decisions: The Information Flow of Auditory and Visual Perception in Multimodal LLMs","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2605.12056","citing_title":"OmniRefine: Alignment-Aware Cooperative Compression for Efficient Omnimodal Large Language Models","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE","json":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE.json","graph_json":"https://pith.science/api/pith-number/MDQBQAOS4IKTTFL44XYUUG4BXE/graph.json","events_json":"https://pith.science/api/pith-number/MDQBQAOS4IKTTFL44XYUUG4BXE/events.json","paper":"https://pith.science/paper/MDQBQAOS"},"agent_actions":{"view_html":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE","download_json":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE.json","view_paper":"https://pith.science/paper/MDQBQAOS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.15685&json=true","fetch_graph":"https://pith.science/api/pith-number/MDQBQAOS4IKTTFL44XYUUG4BXE/graph.json","fetch_events":"https://pith.science/api/pith-number/MDQBQAOS4IKTTFL44XYUUG4BXE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE/action/storage_attestation","attest_author":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE/action/author_attestation","sign_citation":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE/action/citation_signature","submit_replication":"https://pith.science/pith/MDQBQAOS4IKTTFL44XYUUG4BXE/action/replication_record"}},"created_at":"2026-07-09T01:19:52.942183+00:00","updated_at":"2026-07-09T01:19:52.942183+00:00"}