{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ZREX7JIJ6AV5MHGGZLGEJPKFUC","short_pith_number":"pith:ZREX7JIJ","schema_version":"1.0","canonical_sha256":"cc497fa509f02bd61cc6cacc44bd45a0bf0b2c3db73ebf306e79e3a4e29f09a8","source":{"kind":"arxiv","id":"2106.13195","version":1},"attestation_state":"computed","paper":{"title":"FitVid: Overfitting in Pixel-Level Video Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chelsea Finn, Dumitru Erhan, Mohammad Babaeizadeh, Mohammad Taghi Saffar, Sergey Levine, Suraj Nair","submitted_at":"2021-06-24T17:20:21Z","abstract_excerpt":"An agent that is capable of predicting what happens next can perform a variety of tasks through planning with no additional training. Furthermore, such an agent can internally represent the complex dynamics of the real-world and therefore can acquire a representation useful for a variety of visual perception tasks. This makes predicting the future frames of a video, conditioned on the observed past and potentially future actions, an interesting task which remains exceptionally challenging despite many recent advances. Existing video prediction models have shown promising results on simple narr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.13195","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-06-24T17:20:21Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a8d8a27e7a25bd67d91ad1df0307e5c3669f5128b661a49fecdbfaa086e3b19c","abstract_canon_sha256":"ed00f4055c5ee1e7e3f4d06701796387c21c86b2434415ef7c41cc96860ee448"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:52:10.486497Z","signature_b64":"LlBhh93MAsucp+EQW+qrfwGSMfIbgi2TZpLjhEYxMeD56A67IemZJ0Mb8aTHasN6np2YkNCmecWew8zMEty8Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc497fa509f02bd61cc6cacc44bd45a0bf0b2c3db73ebf306e79e3a4e29f09a8","last_reissued_at":"2026-07-05T02:52:10.486050Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:52:10.486050Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FitVid: Overfitting in Pixel-Level Video Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chelsea Finn, Dumitru Erhan, Mohammad Babaeizadeh, Mohammad Taghi Saffar, Sergey Levine, Suraj Nair","submitted_at":"2021-06-24T17:20:21Z","abstract_excerpt":"An agent that is capable of predicting what happens next can perform a variety of tasks through planning with no additional training. Furthermore, such an agent can internally represent the complex dynamics of the real-world and therefore can acquire a representation useful for a variety of visual perception tasks. This makes predicting the future frames of a video, conditioned on the observed past and potentially future actions, an interesting task which remains exceptionally challenging despite many recent advances. Existing video prediction models have shown promising results on simple narr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.13195","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.13195/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.13195","created_at":"2026-07-05T02:52:10.486116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.13195v1","created_at":"2026-07-05T02:52:10.486116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.13195","created_at":"2026-07-05T02:52:10.486116+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZREX7JIJ6AV5","created_at":"2026-07-05T02:52:10.486116+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZREX7JIJ6AV5MHGG","created_at":"2026-07-05T02:52:10.486116+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZREX7JIJ","created_at":"2026-07-05T02:52:10.486116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02195","citing_title":"Bridge-WA: Predicting Where and How the World Changes for Robotic Action","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29194","citing_title":"Stochastic Lifting for Generating Trajectories of Stochastic Physical Systems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02311","citing_title":"Inferring Dynamic Physical Properties from Video Foundation Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2210.02399","citing_title":"Phenaki: Variable Length Video Generation From Open Domain Textual Description","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19325","citing_title":"Long-Context Autoregressive Video Modeling with Next-Frame Prediction","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2310.10639","citing_title":"Zero-Shot Robotic Manipulation with Pretrained Image-Editing Diffusion Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2211.11018","citing_title":"MagicVideo: Efficient Video Generation With Latent Diffusion Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2310.05737","citing_title":"Language Model Beats Diffusion -- Tokenizer is Key to Visual Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2204.03458","citing_title":"Video Diffusion Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2210.02303","citing_title":"Imagen Video: High Definition Video Generation with Diffusion Models","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC","json":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC.json","graph_json":"https://pith.science/api/pith-number/ZREX7JIJ6AV5MHGGZLGEJPKFUC/graph.json","events_json":"https://pith.science/api/pith-number/ZREX7JIJ6AV5MHGGZLGEJPKFUC/events.json","paper":"https://pith.science/paper/ZREX7JIJ"},"agent_actions":{"view_html":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC","download_json":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC.json","view_paper":"https://pith.science/paper/ZREX7JIJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.13195&json=true","fetch_graph":"https://pith.science/api/pith-number/ZREX7JIJ6AV5MHGGZLGEJPKFUC/graph.json","fetch_events":"https://pith.science/api/pith-number/ZREX7JIJ6AV5MHGGZLGEJPKFUC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC/action/storage_attestation","attest_author":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC/action/author_attestation","sign_citation":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC/action/citation_signature","submit_replication":"https://pith.science/pith/ZREX7JIJ6AV5MHGGZLGEJPKFUC/action/replication_record"}},"created_at":"2026-07-05T02:52:10.486116+00:00","updated_at":"2026-07-05T02:52:10.486116+00:00"}