{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7PP3HCBFGVOMFX4QOKUSUBFLVM","short_pith_number":"pith:7PP3HCBF","schema_version":"1.0","canonical_sha256":"fbdfb38825355cc2df9072a92a04abab3256c1501f0f474db7b1a424cbb5ac15","source":{"kind":"arxiv","id":"2502.07737","version":2},"attestation_state":"computed","paper":{"title":"Next Block Prediction: Video Generation via Semi-Autoregressive Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Furu Wei, Shuhuai Ren, Shuming Ma, Xu Sun","submitted_at":"2025-02-11T17:57:53Z","abstract_excerpt":"Next-Token Prediction (NTP) is a de facto approach for autoregressive (AR) video generation, but it suffers from suboptimal unidirectional dependencies and slow inference speed. In this work, we propose a semi-autoregressive (semi-AR) framework, called Next-Block Prediction (NBP), for video generation. By uniformly decomposing video content into equal-sized blocks (e.g., rows or frames), we shift the generation unit from individual tokens to blocks, allowing each token in the current block to simultaneously predict the corresponding token in the next block. Unlike traditional AR modeling, our "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.07737","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-11T17:57:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d6ce66f55a9d115ce141a8c228882cf60bf6e33902f41209aa9fdd7543d2394c","abstract_canon_sha256":"3e22601445a72a6e5d9690f56832ab99819f445a915e96a5516a75bf054adbe8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:09.738109Z","signature_b64":"j+nQAhUVblvW+HFZXXlgIoEyb5otsGjBXUSCmeY7W33iTGliJ7PCbqZ8ckDXRUg23QxtG2lQKtNRwQucxd4MCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fbdfb38825355cc2df9072a92a04abab3256c1501f0f474db7b1a424cbb5ac15","last_reissued_at":"2026-07-05T10:13:09.737583Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:09.737583Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Next Block Prediction: Video Generation via Semi-Autoregressive Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Furu Wei, Shuhuai Ren, Shuming Ma, Xu Sun","submitted_at":"2025-02-11T17:57:53Z","abstract_excerpt":"Next-Token Prediction (NTP) is a de facto approach for autoregressive (AR) video generation, but it suffers from suboptimal unidirectional dependencies and slow inference speed. In this work, we propose a semi-autoregressive (semi-AR) framework, called Next-Block Prediction (NBP), for video generation. By uniformly decomposing video content into equal-sized blocks (e.g., rows or frames), we shift the generation unit from individual tokens to blocks, allowing each token in the current block to simultaneously predict the corresponding token in the next block. Unlike traditional AR modeling, our "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.07737","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.07737/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.07737","created_at":"2026-07-05T10:13:09.737648+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.07737v2","created_at":"2026-07-05T10:13:09.737648+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.07737","created_at":"2026-07-05T10:13:09.737648+00:00"},{"alias_kind":"pith_short_12","alias_value":"7PP3HCBFGVOM","created_at":"2026-07-05T10:13:09.737648+00:00"},{"alias_kind":"pith_short_16","alias_value":"7PP3HCBFGVOMFX4Q","created_at":"2026-07-05T10:13:09.737648+00:00"},{"alias_kind":"pith_short_8","alias_value":"7PP3HCBF","created_at":"2026-07-05T10:13:09.737648+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24982","citing_title":"Latent Block-Diffusion Temporal Point Processes: A Semi-Autoregressive Framework for Asynchronous Event Sequence Generation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11187","citing_title":"Next Forcing: Causal World Modeling with Multi-Chunk Prediction","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04432","citing_title":"DSA: Dynamic Step Allocation for Fast Autoregressive Video Generation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07775","citing_title":"Rolling Sink: Bridging Limited-Horizon Training and Open-Ended Testing in Autoregressive Video Diffusion","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04592","citing_title":"From Static Inference to Dynamic Interaction: A Survey of Streaming Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14487","citing_title":"Head Forcing: Long Autoregressive Video Generation via Head Heterogeneity","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08009","citing_title":"Self Forcing: Bridging the Train-Test Gap in Autoregressive Video Diffusion","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07402","citing_title":"Accelerating Training of Autoregressive Video Generation Models via Local Optimization with Representation Continuity","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07209","citing_title":"INSPATIO-WORLD: A Real-Time 4D World Simulator via Spatiotemporal Autoregressive Modeling","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04461","citing_title":"Stream-T1: Test-Time Scaling for Streaming Video Generation","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM","json":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM.json","graph_json":"https://pith.science/api/pith-number/7PP3HCBFGVOMFX4QOKUSUBFLVM/graph.json","events_json":"https://pith.science/api/pith-number/7PP3HCBFGVOMFX4QOKUSUBFLVM/events.json","paper":"https://pith.science/paper/7PP3HCBF"},"agent_actions":{"view_html":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM","download_json":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM.json","view_paper":"https://pith.science/paper/7PP3HCBF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.07737&json=true","fetch_graph":"https://pith.science/api/pith-number/7PP3HCBFGVOMFX4QOKUSUBFLVM/graph.json","fetch_events":"https://pith.science/api/pith-number/7PP3HCBFGVOMFX4QOKUSUBFLVM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM/action/storage_attestation","attest_author":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM/action/author_attestation","sign_citation":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM/action/citation_signature","submit_replication":"https://pith.science/pith/7PP3HCBFGVOMFX4QOKUSUBFLVM/action/replication_record"}},"created_at":"2026-07-05T10:13:09.737648+00:00","updated_at":"2026-07-05T10:13:09.737648+00:00"}