{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F4TWELCRDRARHWPSDFVBDG2J4H","short_pith_number":"pith:F4TWELCR","schema_version":"1.0","canonical_sha256":"2f27622c511c4113d9f2196a119b49e1c78ec3b8962b5aa7c8d44c3370c7debc","source":{"kind":"arxiv","id":"2506.10915","version":1},"attestation_state":"computed","paper":{"title":"M4V: Multi-Modal Mamba for Text-to-Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Gengwei Zhang, Jiancheng Huang, Ling Chen, Lin Ma, Siyu Jiao, Yinlong Qian, Yunchao Wei, Zequn Jie","submitted_at":"2025-06-12T17:29:40Z","abstract_excerpt":"Text-to-video generation has significantly enriched content creation and holds the potential to evolve into powerful world simulators. However, modeling the vast spatiotemporal space remains computationally demanding, particularly when employing Transformers, which incur quadratic complexity in sequence processing and thus limit practical applications. Recent advancements in linear-time sequence modeling, particularly the Mamba architecture, offer a more efficient alternative. Nevertheless, its plain design limits its direct applicability to multi-modal and spatiotemporal video generation task"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.10915","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-12T17:29:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b0ae898416a9a3ff82607f1ea3a1ea95373b09f90ec4d70ff8cdeaadb6c59ad1","abstract_canon_sha256":"b5e64524e70d6475c4aa594038405458130a0d3cd8c9e6f5a37bf54ffc775c9a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:40.267786Z","signature_b64":"GDFj9RyON2C+jd1bQMuxRHFt2pyIFuEJYbSRZHVoIYOiK+AjUT3kWHl4Iad4vRY7C/wEcNvr5JH8HOb8RHcIDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f27622c511c4113d9f2196a119b49e1c78ec3b8962b5aa7c8d44c3370c7debc","last_reissued_at":"2026-07-05T11:20:40.267302Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:40.267302Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"M4V: Multi-Modal Mamba for Text-to-Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Gengwei Zhang, Jiancheng Huang, Ling Chen, Lin Ma, Siyu Jiao, Yinlong Qian, Yunchao Wei, Zequn Jie","submitted_at":"2025-06-12T17:29:40Z","abstract_excerpt":"Text-to-video generation has significantly enriched content creation and holds the potential to evolve into powerful world simulators. However, modeling the vast spatiotemporal space remains computationally demanding, particularly when employing Transformers, which incur quadratic complexity in sequence processing and thus limit practical applications. Recent advancements in linear-time sequence modeling, particularly the Mamba architecture, offer a more efficient alternative. Nevertheless, its plain design limits its direct applicability to multi-modal and spatiotemporal video generation task"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.10915","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.10915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.10915","created_at":"2026-07-05T11:20:40.267373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.10915v1","created_at":"2026-07-05T11:20:40.267373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.10915","created_at":"2026-07-05T11:20:40.267373+00:00"},{"alias_kind":"pith_short_12","alias_value":"F4TWELCRDRAR","created_at":"2026-07-05T11:20:40.267373+00:00"},{"alias_kind":"pith_short_16","alias_value":"F4TWELCRDRARHWPS","created_at":"2026-07-05T11:20:40.267373+00:00"},{"alias_kind":"pith_short_8","alias_value":"F4TWELCR","created_at":"2026-07-05T11:20:40.267373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06173","citing_title":"MobileWan: Closing the Quality Gap for Mobile Video Diffusion","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2512.12598","citing_title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17685","citing_title":"FutureSightDrive: Thinking Visually with Spatio-Temporal CoT for Autonomous Driving","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11804","citing_title":"OmniShow: Unifying Multimodal Conditions for Human-Object Interaction Video Generation","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H","json":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H.json","graph_json":"https://pith.science/api/pith-number/F4TWELCRDRARHWPSDFVBDG2J4H/graph.json","events_json":"https://pith.science/api/pith-number/F4TWELCRDRARHWPSDFVBDG2J4H/events.json","paper":"https://pith.science/paper/F4TWELCR"},"agent_actions":{"view_html":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H","download_json":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H.json","view_paper":"https://pith.science/paper/F4TWELCR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.10915&json=true","fetch_graph":"https://pith.science/api/pith-number/F4TWELCRDRARHWPSDFVBDG2J4H/graph.json","fetch_events":"https://pith.science/api/pith-number/F4TWELCRDRARHWPSDFVBDG2J4H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H/action/storage_attestation","attest_author":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H/action/author_attestation","sign_citation":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H/action/citation_signature","submit_replication":"https://pith.science/pith/F4TWELCRDRARHWPSDFVBDG2J4H/action/replication_record"}},"created_at":"2026-07-05T11:20:40.267373+00:00","updated_at":"2026-07-05T11:20:40.267373+00:00"}