{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3PSURHUYIU63ZEQ3PEELFJKOI3","short_pith_number":"pith:3PSURHUY","schema_version":"1.0","canonical_sha256":"dbe5489e98453dbc921b7908b2a54e46fce6f71e48c578ad0951c64cd56542db","source":{"kind":"arxiv","id":"2503.06362","version":2},"attestation_state":"computed","paper":{"title":"Adaptive Audio-Visual Speech Recognition via Matryoshka-Based Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Minsu Kim, Stavros Petridis, Umberto Cappellazzo","submitted_at":"2025-03-09T00:02:10Z","abstract_excerpt":"Audio-Visual Speech Recognition (AVSR) leverages audio and visual modalities to improve robustness in noisy environments. Recent advances in Large Language Models (LLMs) show strong performance in speech recognition, including AVSR. However, the long speech representations lead to high computational costs for LLMs. Prior methods compress inputs before feeding them to LLMs, but high compression often harms accuracy. To address this, we propose Llama-MTSK, the first Matryoshka-based Multimodal LLM for AVSR, which flexibly adapts audio-visual token allocation under varying compute constraints. In"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.06362","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-09T00:02:10Z","cross_cats_sorted":["cs.MM","cs.SD","eess.AS"],"title_canon_sha256":"958102f5b396646d95feb8f7cdc70c9cca1e4c3a739f9bc002acbcf37d0f5dd5","abstract_canon_sha256":"0f4d00bc2ba9b7dbc2aa92fc7c130a4d77cf4266a2a33dc395071eb79e79cd38"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:32.824265Z","signature_b64":"28HBbPiuobn7/UM6ZEpf8bzROOMcsh/CidjeeUAGvU5lPWKXVi1CmLEXz9VdBbC5LddoljhaoASzDtsrP7/cAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dbe5489e98453dbc921b7908b2a54e46fce6f71e48c578ad0951c64cd56542db","last_reissued_at":"2026-07-05T11:49:32.823676Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:32.823676Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Audio-Visual Speech Recognition via Matryoshka-Based Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Minsu Kim, Stavros Petridis, Umberto Cappellazzo","submitted_at":"2025-03-09T00:02:10Z","abstract_excerpt":"Audio-Visual Speech Recognition (AVSR) leverages audio and visual modalities to improve robustness in noisy environments. Recent advances in Large Language Models (LLMs) show strong performance in speech recognition, including AVSR. However, the long speech representations lead to high computational costs for LLMs. Prior methods compress inputs before feeding them to LLMs, but high compression often harms accuracy. To address this, we propose Llama-MTSK, the first Matryoshka-based Multimodal LLM for AVSR, which flexibly adapts audio-visual token allocation under varying compute constraints. In"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.06362","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.06362/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.06362","created_at":"2026-07-05T11:49:32.823756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.06362v2","created_at":"2026-07-05T11:49:32.823756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.06362","created_at":"2026-07-05T11:49:32.823756+00:00"},{"alias_kind":"pith_short_12","alias_value":"3PSURHUYIU63","created_at":"2026-07-05T11:49:32.823756+00:00"},{"alias_kind":"pith_short_16","alias_value":"3PSURHUYIU63ZEQ3","created_at":"2026-07-05T11:49:32.823756+00:00"},{"alias_kind":"pith_short_8","alias_value":"3PSURHUY","created_at":"2026-07-05T11:49:32.823756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.14336","citing_title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3","json":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3.json","graph_json":"https://pith.science/api/pith-number/3PSURHUYIU63ZEQ3PEELFJKOI3/graph.json","events_json":"https://pith.science/api/pith-number/3PSURHUYIU63ZEQ3PEELFJKOI3/events.json","paper":"https://pith.science/paper/3PSURHUY"},"agent_actions":{"view_html":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3","download_json":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3.json","view_paper":"https://pith.science/paper/3PSURHUY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.06362&json=true","fetch_graph":"https://pith.science/api/pith-number/3PSURHUYIU63ZEQ3PEELFJKOI3/graph.json","fetch_events":"https://pith.science/api/pith-number/3PSURHUYIU63ZEQ3PEELFJKOI3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3/action/storage_attestation","attest_author":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3/action/author_attestation","sign_citation":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3/action/citation_signature","submit_replication":"https://pith.science/pith/3PSURHUYIU63ZEQ3PEELFJKOI3/action/replication_record"}},"created_at":"2026-07-05T11:49:32.823756+00:00","updated_at":"2026-07-05T11:49:32.823756+00:00"}