{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UWMAUFWTBY3QUVCRIWJDJZL6T2","short_pith_number":"pith:UWMAUFWT","schema_version":"1.0","canonical_sha256":"a5980a16d30e370a5451459234e57e9e9a0380e5456693a10b2a5b8f58d8f087","source":{"kind":"arxiv","id":"2410.07659","version":2},"attestation_state":"computed","paper":{"title":"MotionAura: Generating High-Quality and Motion Consistent Videos using Discrete Diffusion","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chirag Sehgal, Jishu Sen Gupta, Onkar Susladkar, Rekha Singhal, Sparsh Mittal","submitted_at":"2024-10-10T07:07:56Z","abstract_excerpt":"The spatio-temporal complexity of video data presents significant challenges in tasks such as compression, generation, and inpainting. We present four key contributions to address the challenges of spatiotemporal video processing. First, we introduce the 3D Mobile Inverted Vector-Quantization Variational Autoencoder (3D-MBQ-VAE), which combines Variational Autoencoders (VAEs) with masked token modeling to enhance spatiotemporal video compression. The model achieves superior temporal consistency and state-of-the-art (SOTA) reconstruction quality by employing a novel training strategy with full "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07659","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-10T07:07:56Z","cross_cats_sorted":[],"title_canon_sha256":"51f2a2ef8a4c074c41bd7de1f6a0b9ed5c1dd91733558cda7154e8d40df52a83","abstract_canon_sha256":"ae32a0488ef6f353c5a4748e539fa47f664ab6b09ac93e5d88687627e5acfe4b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:28:24.066517Z","signature_b64":"d8DFPB870raSkEpJZ6dR03d3Ia5cDGbUM9N/ucsIFRnt2g/fSbSmYXSD9EQ6RUOxvgGg251ES/lfBj5K62OMCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a5980a16d30e370a5451459234e57e9e9a0380e5456693a10b2a5b8f58d8f087","last_reissued_at":"2026-07-05T10:28:24.066000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:28:24.066000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MotionAura: Generating High-Quality and Motion Consistent Videos using Discrete Diffusion","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chirag Sehgal, Jishu Sen Gupta, Onkar Susladkar, Rekha Singhal, Sparsh Mittal","submitted_at":"2024-10-10T07:07:56Z","abstract_excerpt":"The spatio-temporal complexity of video data presents significant challenges in tasks such as compression, generation, and inpainting. We present four key contributions to address the challenges of spatiotemporal video processing. First, we introduce the 3D Mobile Inverted Vector-Quantization Variational Autoencoder (3D-MBQ-VAE), which combines Variational Autoencoders (VAEs) with masked token modeling to enhance spatiotemporal video compression. The model achieves superior temporal consistency and state-of-the-art (SOTA) reconstruction quality by employing a novel training strategy with full "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07659","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07659/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07659","created_at":"2026-07-05T10:28:24.066065+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07659v2","created_at":"2026-07-05T10:28:24.066065+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07659","created_at":"2026-07-05T10:28:24.066065+00:00"},{"alias_kind":"pith_short_12","alias_value":"UWMAUFWTBY3Q","created_at":"2026-07-05T10:28:24.066065+00:00"},{"alias_kind":"pith_short_16","alias_value":"UWMAUFWTBY3QUVCR","created_at":"2026-07-05T10:28:24.066065+00:00"},{"alias_kind":"pith_short_8","alias_value":"UWMAUFWT","created_at":"2026-07-05T10:28:24.066065+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.18745","citing_title":"DiffMVR: Diffusion-based Automated Multi-Guidance Video Restoration","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2","json":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2.json","graph_json":"https://pith.science/api/pith-number/UWMAUFWTBY3QUVCRIWJDJZL6T2/graph.json","events_json":"https://pith.science/api/pith-number/UWMAUFWTBY3QUVCRIWJDJZL6T2/events.json","paper":"https://pith.science/paper/UWMAUFWT"},"agent_actions":{"view_html":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2","download_json":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2.json","view_paper":"https://pith.science/paper/UWMAUFWT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07659&json=true","fetch_graph":"https://pith.science/api/pith-number/UWMAUFWTBY3QUVCRIWJDJZL6T2/graph.json","fetch_events":"https://pith.science/api/pith-number/UWMAUFWTBY3QUVCRIWJDJZL6T2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2/action/storage_attestation","attest_author":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2/action/author_attestation","sign_citation":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2/action/citation_signature","submit_replication":"https://pith.science/pith/UWMAUFWTBY3QUVCRIWJDJZL6T2/action/replication_record"}},"created_at":"2026-07-05T10:28:24.066065+00:00","updated_at":"2026-07-05T10:28:24.066065+00:00"}