{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OIRTPRZKSASLFKVHKI6PSTV5EY","short_pith_number":"pith:OIRTPRZK","schema_version":"1.0","canonical_sha256":"722337c72a9024b2aaa7523cf94ebd261542186263ae07bdce00c48a71d0cba6","source":{"kind":"arxiv","id":"2504.05298","version":1},"attestation_state":"computed","paper":{"title":"One-Minute Video Generation with Test-Time Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Carlos Guestrin, Daniel Koceja, Gashon Hussein, Jan Kautz, Jiarui Xu, Ka Chun Cheung, Karan Dalal, Sanmi Koyejo, Shihao Han, Tatsunori Hashimoto, Xiaolong Wang, Yejin Choi, Youjin Song, Yue Zhao, Yu Sun","submitted_at":"2025-04-07T17:56:31Z","abstract_excerpt":"Transformers today still struggle to generate one-minute videos because self-attention layers are inefficient for long context. Alternatives such as Mamba layers struggle with complex multi-scene stories because their hidden states are less expressive. We experiment with Test-Time Training (TTT) layers, whose hidden states themselves can be neural networks, therefore more expressive. Adding TTT layers into a pre-trained Transformer enables it to generate one-minute videos from text storyboards. For proof of concept, we curate a dataset based on Tom and Jerry cartoons. Compared to baselines suc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.05298","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-07T17:56:31Z","cross_cats_sorted":[],"title_canon_sha256":"64746e037089aaa39814fe5db5721d3d13e47de3cf54f1f20bdf8ac64ac82f8a","abstract_canon_sha256":"3820a5cecbb31fd002ea8b3df2677ed7ee3a24e3ecbf53b11c7a592618ab9a1c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:37.096151Z","signature_b64":"GLMFtMHTH4+5f9Xyb64Bld9mr+vdb91iqvyNkDVCmyaS4RrIeqg3Cps+UsP6WGTmf761QyLgI5xeJc0Ayt0bAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"722337c72a9024b2aaa7523cf94ebd261542186263ae07bdce00c48a71d0cba6","last_reissued_at":"2026-07-05T10:45:37.095748Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:37.095748Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"One-Minute Video Generation with Test-Time Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Carlos Guestrin, Daniel Koceja, Gashon Hussein, Jan Kautz, Jiarui Xu, Ka Chun Cheung, Karan Dalal, Sanmi Koyejo, Shihao Han, Tatsunori Hashimoto, Xiaolong Wang, Yejin Choi, Youjin Song, Yue Zhao, Yu Sun","submitted_at":"2025-04-07T17:56:31Z","abstract_excerpt":"Transformers today still struggle to generate one-minute videos because self-attention layers are inefficient for long context. Alternatives such as Mamba layers struggle with complex multi-scene stories because their hidden states are less expressive. We experiment with Test-Time Training (TTT) layers, whose hidden states themselves can be neural networks, therefore more expressive. Adding TTT layers into a pre-trained Transformer enables it to generate one-minute videos from text storyboards. For proof of concept, we curate a dataset based on Tom and Jerry cartoons. Compared to baselines suc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.05298","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.05298/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.05298","created_at":"2026-07-05T10:45:37.095806+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.05298v1","created_at":"2026-07-05T10:45:37.095806+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.05298","created_at":"2026-07-05T10:45:37.095806+00:00"},{"alias_kind":"pith_short_12","alias_value":"OIRTPRZKSASL","created_at":"2026-07-05T10:45:37.095806+00:00"},{"alias_kind":"pith_short_16","alias_value":"OIRTPRZKSASLFKVH","created_at":"2026-07-05T10:45:37.095806+00:00"},{"alias_kind":"pith_short_8","alias_value":"OIRTPRZK","created_at":"2026-07-05T10:45:37.095806+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2507.18809","citing_title":"Test-time Offline Reinforcement Learning on Goal-related Experience","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19325","citing_title":"Long-Context Autoregressive Video Modeling with Next-Frame Prediction","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY","json":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY.json","graph_json":"https://pith.science/api/pith-number/OIRTPRZKSASLFKVHKI6PSTV5EY/graph.json","events_json":"https://pith.science/api/pith-number/OIRTPRZKSASLFKVHKI6PSTV5EY/events.json","paper":"https://pith.science/paper/OIRTPRZK"},"agent_actions":{"view_html":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY","download_json":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY.json","view_paper":"https://pith.science/paper/OIRTPRZK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.05298&json=true","fetch_graph":"https://pith.science/api/pith-number/OIRTPRZKSASLFKVHKI6PSTV5EY/graph.json","fetch_events":"https://pith.science/api/pith-number/OIRTPRZKSASLFKVHKI6PSTV5EY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY/action/storage_attestation","attest_author":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY/action/author_attestation","sign_citation":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY/action/citation_signature","submit_replication":"https://pith.science/pith/OIRTPRZKSASLFKVHKI6PSTV5EY/action/replication_record"}},"created_at":"2026-07-05T10:45:37.095806+00:00","updated_at":"2026-07-05T10:45:37.095806+00:00"}