{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:LEJ2YQDKJAOQQWDY3DM4X3HFQH","short_pith_number":"pith:LEJ2YQDK","schema_version":"1.0","canonical_sha256":"5913ac406a481d085878d8d9cbece581e5d3a2a7513707c0eb2ac4f6518a1ec7","source":{"kind":"arxiv","id":"1906.02634","version":3},"attestation_state":"computed","paper":{"title":"Scaling Autoregressive Video Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dirk Weissenborn, Jakob Uszkoreit, Oscar T\\\"ackstr\\\"om","submitted_at":"2019-06-06T15:06:21Z","abstract_excerpt":"Due to the statistical complexity of video, the high degree of inherent stochasticity, and the sheer amount of data, generating natural video remains a challenging task. State-of-the-art video generation models often attempt to address these issues by combining sometimes complex, usually video-specific neural network architectures, latent variable models, adversarial training and a range of other methods. Despite their often high complexity, these approaches still fall short of generating high quality video continuations outside of narrow domains and often struggle with fidelity. In contrast, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.02634","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-06-06T15:06:21Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"35bcffc5dac93f353b42276a0ee7de44dcec9dbfc6675e8127643b3bfb227b16","abstract_canon_sha256":"937fb565e5e8dc7d96bd997947a7134b0a4ad003651a89904bbb67423e81dd5c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:39:31.571212Z","signature_b64":"JaCdQK5Qb2u1mEsfIOrPHIapwX0mIr8JJzMo9udnVxT4d+LM7Pel33EGWYpAPv7H8gDPAND0xqgP29LubJQDCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5913ac406a481d085878d8d9cbece581e5d3a2a7513707c0eb2ac4f6518a1ec7","last_reissued_at":"2026-07-05T00:39:31.570725Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:39:31.570725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Autoregressive Video Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dirk Weissenborn, Jakob Uszkoreit, Oscar T\\\"ackstr\\\"om","submitted_at":"2019-06-06T15:06:21Z","abstract_excerpt":"Due to the statistical complexity of video, the high degree of inherent stochasticity, and the sheer amount of data, generating natural video remains a challenging task. State-of-the-art video generation models often attempt to address these issues by combining sometimes complex, usually video-specific neural network architectures, latent variable models, adversarial training and a range of other methods. Despite their often high complexity, these approaches still fall short of generating high quality video continuations outside of narrow domains and often struggle with fidelity. In contrast, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.02634","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.02634/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.02634","created_at":"2026-07-05T00:39:31.570794+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.02634v3","created_at":"2026-07-05T00:39:31.570794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.02634","created_at":"2026-07-05T00:39:31.570794+00:00"},{"alias_kind":"pith_short_12","alias_value":"LEJ2YQDKJAOQ","created_at":"2026-07-05T00:39:31.570794+00:00"},{"alias_kind":"pith_short_16","alias_value":"LEJ2YQDKJAOQQWDY","created_at":"2026-07-05T00:39:31.570794+00:00"},{"alias_kind":"pith_short_8","alias_value":"LEJ2YQDK","created_at":"2026-07-05T00:39:31.570794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09156","citing_title":"OmniGen-AR: AutoRegressive Any-to-Image Generation","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07508","citing_title":"Streaming Video Generation with Streaming Force Control","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15141","citing_title":"Causal Forcing++: Scalable Few-Step Autoregressive Diffusion Distillation for Real-Time Interactive Video Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02214","citing_title":"Causal Forcing: Autoregressive Diffusion Distillation Done Right for High-Quality Real-Time Interactive Video Generation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02214","citing_title":"Causal Forcing: Autoregressive Diffusion Distillation Done Right for High-Quality Real-Time Interactive Video Generation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21996","citing_title":"VRAG: Learning World Models for Interactive Video Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2102.01293","citing_title":"Scaling Laws for Transfer","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"1910.11215","citing_title":"RoboNet: Large-Scale Multi-Robot Learning","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25161","citing_title":"Rolling Forcing: Autoregressive Long Video Diffusion in Real Time","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07775","citing_title":"Rolling Sink: Bridging Limited-Horizon Training and Open-Ended Testing in Autoregressive Video Diffusion","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06114","citing_title":"Learning Interactive Real-World Simulators","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2211.13221","citing_title":"Latent Video Diffusion Models for High-Fidelity Long Video Generation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14487","citing_title":"Head Forcing: Long Autoregressive Video Generation via Head Heterogeneity","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2503.00200","citing_title":"Unified Video Action Model","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2104.10157","citing_title":"VideoGPT: Video Generation using VQ-VAE and Transformers","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2010.14701","citing_title":"Scaling Laws for Autoregressive Generative Modeling","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2205.15868","citing_title":"CogVideo: Large-scale Pretraining for Text-to-Video Generation via Transformers","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2207.05221","citing_title":"Language Models (Mostly) Know What They Know","ref_index":116,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH","json":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH.json","graph_json":"https://pith.science/api/pith-number/LEJ2YQDKJAOQQWDY3DM4X3HFQH/graph.json","events_json":"https://pith.science/api/pith-number/LEJ2YQDKJAOQQWDY3DM4X3HFQH/events.json","paper":"https://pith.science/paper/LEJ2YQDK"},"agent_actions":{"view_html":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH","download_json":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH.json","view_paper":"https://pith.science/paper/LEJ2YQDK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.02634&json=true","fetch_graph":"https://pith.science/api/pith-number/LEJ2YQDKJAOQQWDY3DM4X3HFQH/graph.json","fetch_events":"https://pith.science/api/pith-number/LEJ2YQDKJAOQQWDY3DM4X3HFQH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH/action/storage_attestation","attest_author":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH/action/author_attestation","sign_citation":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH/action/citation_signature","submit_replication":"https://pith.science/pith/LEJ2YQDKJAOQQWDY3DM4X3HFQH/action/replication_record"}},"created_at":"2026-07-05T00:39:31.570794+00:00","updated_at":"2026-07-05T00:39:31.570794+00:00"}