{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EXSBQNCLU7WS7CWVYRSSVXA3MX","short_pith_number":"pith:EXSBQNCL","schema_version":"1.0","canonical_sha256":"25e418344ba7ed2f8ad5c4652adc1b65f4a2da36207dd3608de859f21d38324d","source":{"kind":"arxiv","id":"2512.14008","version":2},"attestation_state":"computed","paper":{"title":"Sparse-LaViDa: Sparse Multimodal Discrete Diffusion Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aditya Grover, Jason Kuen, Jiuxiang Gu, Kangning Liu, Shufan Li, Zhe Lin, Zijun Wei","submitted_at":"2025-12-16T02:06:06Z","abstract_excerpt":"Masked Discrete Diffusion Models (MDMs) have achieved strong performance across a wide range of multimodal tasks, including image understanding, generation, and editing. However, their inference speed remains suboptimal due to the need to repeatedly process redundant masked tokens at every sampling step. In this work, we propose Sparse-LaViDa, a novel modeling framework that dynamically truncates unnecessary masked tokens at each inference step to accelerate MDM sampling. To preserve generation quality, we introduce specialized register tokens that serve as compact representations for the trun"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.14008","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-12-16T02:06:06Z","cross_cats_sorted":[],"title_canon_sha256":"1a3cc9b7e59ef403bfa20a81bd002c8d901d295fc00b1cf5772c0fe9eda2885e","abstract_canon_sha256":"b426aa663d3f34c93f400afc8ea2fc4c1e350f729860a2c8ccfee19bc6cf60da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T00:20:42.713027Z","signature_b64":"VrLCj2DF3vqe6y2a+FRlCuZENTRhGDC4pMCD3pUBI1p5ufqc5YqtzOQyiccwMFWFun+DjT7KMy27eNPOkXUsBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25e418344ba7ed2f8ad5c4652adc1b65f4a2da36207dd3608de859f21d38324d","last_reissued_at":"2026-07-17T00:20:42.712045Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T00:20:42.712045Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparse-LaViDa: Sparse Multimodal Discrete Diffusion Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aditya Grover, Jason Kuen, Jiuxiang Gu, Kangning Liu, Shufan Li, Zhe Lin, Zijun Wei","submitted_at":"2025-12-16T02:06:06Z","abstract_excerpt":"Masked Discrete Diffusion Models (MDMs) have achieved strong performance across a wide range of multimodal tasks, including image understanding, generation, and editing. However, their inference speed remains suboptimal due to the need to repeatedly process redundant masked tokens at every sampling step. In this work, we propose Sparse-LaViDa, a novel modeling framework that dynamically truncates unnecessary masked tokens at each inference step to accelerate MDM sampling. To preserve generation quality, we introduce specialized register tokens that serve as compact representations for the trun"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.14008","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.14008/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.14008","created_at":"2026-07-17T00:20:42.712526+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.14008v2","created_at":"2026-07-17T00:20:42.712526+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.14008","created_at":"2026-07-17T00:20:42.712526+00:00"},{"alias_kind":"pith_short_12","alias_value":"EXSBQNCLU7WS","created_at":"2026-07-17T00:20:42.712526+00:00"},{"alias_kind":"pith_short_16","alias_value":"EXSBQNCLU7WS7CWV","created_at":"2026-07-17T00:20:42.712526+00:00"},{"alias_kind":"pith_short_8","alias_value":"EXSBQNCL","created_at":"2026-07-17T00:20:42.712526+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.01775","citing_title":"Set Diffusion: Interpolating Token Orderings Between Autoregression and Diffusion for Fast and Flexible Decoding","ref_index":134,"is_internal_anchor":true},{"citing_arxiv_id":"2606.29814","citing_title":"Nemotron-Labs-Diffusion-Image: Advancing Masked Discrete Diffusion for High-Resolution Image Synthesis","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2605.25820","citing_title":"Visual-Redundancy-Controlled Parallel Decoding for Diffusion-Based Multimodal Large Language Models","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX","json":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX.json","graph_json":"https://pith.science/api/pith-number/EXSBQNCLU7WS7CWVYRSSVXA3MX/graph.json","events_json":"https://pith.science/api/pith-number/EXSBQNCLU7WS7CWVYRSSVXA3MX/events.json","paper":"https://pith.science/paper/EXSBQNCL"},"agent_actions":{"view_html":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX","download_json":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX.json","view_paper":"https://pith.science/paper/EXSBQNCL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.14008&json=true","fetch_graph":"https://pith.science/api/pith-number/EXSBQNCLU7WS7CWVYRSSVXA3MX/graph.json","fetch_events":"https://pith.science/api/pith-number/EXSBQNCLU7WS7CWVYRSSVXA3MX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX/action/storage_attestation","attest_author":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX/action/author_attestation","sign_citation":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX/action/citation_signature","submit_replication":"https://pith.science/pith/EXSBQNCLU7WS7CWVYRSSVXA3MX/action/replication_record"}},"created_at":"2026-07-17T00:20:42.712526+00:00","updated_at":"2026-07-17T00:20:42.712526+00:00"}