{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:K5Z3PMVX7AV6ELB56XBFZED7ZK","short_pith_number":"pith:K5Z3PMVX","schema_version":"1.0","canonical_sha256":"5773b7b2b7f82be22c3df5c25c907fcaa6db708cfc5ca3df77e3606a926edc07","source":{"kind":"arxiv","id":"2109.15047","version":2},"attestation_state":"computed","paper":{"title":"Deep Contextual Video Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.MM"],"primary_cat":"eess.IV","authors_text":"Bin Li, Jiahao Li, Yan Lu","submitted_at":"2021-09-30T12:14:24Z","abstract_excerpt":"Most of the existing neural video compression methods adopt the predictive coding framework, which first generates the predicted frame and then encodes its residue with the current frame. However, as for compression ratio, predictive coding is only a sub-optimal solution as it uses simple subtraction operation to remove the redundancy across frames. In this paper, we propose a deep contextual video compression framework to enable a paradigm shift from predictive coding to conditional coding. In particular, we try to answer the following questions: how to define, use, and learn condition under "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.15047","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.IV","submitted_at":"2021-09-30T12:14:24Z","cross_cats_sorted":["cs.CV","cs.MM"],"title_canon_sha256":"9855ab0aeb2376b0cd36b3a1acbc3294ea63695123ce7008c1f0c28b2bf44f6c","abstract_canon_sha256":"ab262950875da03ca96c23cf0687f8b34f6e5a55dbc7685a79ddbcdd9432914a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:40:48.326716Z","signature_b64":"GGS+soOWnx71B52FXjED+ue3B6WDPEc1Jzj1oI6PJXLHN2OQTOh1JG/Y0poXTk7VhpmEZHkwLGvUV+N48IK8CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5773b7b2b7f82be22c3df5c25c907fcaa6db708cfc5ca3df77e3606a926edc07","last_reissued_at":"2026-07-05T03:40:48.326224Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:40:48.326224Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Contextual Video Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.MM"],"primary_cat":"eess.IV","authors_text":"Bin Li, Jiahao Li, Yan Lu","submitted_at":"2021-09-30T12:14:24Z","abstract_excerpt":"Most of the existing neural video compression methods adopt the predictive coding framework, which first generates the predicted frame and then encodes its residue with the current frame. However, as for compression ratio, predictive coding is only a sub-optimal solution as it uses simple subtraction operation to remove the redundancy across frames. In this paper, we propose a deep contextual video compression framework to enable a paradigm shift from predictive coding to conditional coding. In particular, we try to answer the following questions: how to define, use, and learn condition under "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.15047","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.15047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.15047","created_at":"2026-07-05T03:40:48.326286+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.15047v2","created_at":"2026-07-05T03:40:48.326286+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.15047","created_at":"2026-07-05T03:40:48.326286+00:00"},{"alias_kind":"pith_short_12","alias_value":"K5Z3PMVX7AV6","created_at":"2026-07-05T03:40:48.326286+00:00"},{"alias_kind":"pith_short_16","alias_value":"K5Z3PMVX7AV6ELB5","created_at":"2026-07-05T03:40:48.326286+00:00"},{"alias_kind":"pith_short_8","alias_value":"K5Z3PMVX","created_at":"2026-07-05T03:40:48.326286+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.12776","citing_title":"Cross-Layer Encrypted Semantic Communication Framework for Panoramic Video Transmission","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK","json":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK.json","graph_json":"https://pith.science/api/pith-number/K5Z3PMVX7AV6ELB56XBFZED7ZK/graph.json","events_json":"https://pith.science/api/pith-number/K5Z3PMVX7AV6ELB56XBFZED7ZK/events.json","paper":"https://pith.science/paper/K5Z3PMVX"},"agent_actions":{"view_html":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK","download_json":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK.json","view_paper":"https://pith.science/paper/K5Z3PMVX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.15047&json=true","fetch_graph":"https://pith.science/api/pith-number/K5Z3PMVX7AV6ELB56XBFZED7ZK/graph.json","fetch_events":"https://pith.science/api/pith-number/K5Z3PMVX7AV6ELB56XBFZED7ZK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK/action/storage_attestation","attest_author":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK/action/author_attestation","sign_citation":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK/action/citation_signature","submit_replication":"https://pith.science/pith/K5Z3PMVX7AV6ELB56XBFZED7ZK/action/replication_record"}},"created_at":"2026-07-05T03:40:48.326286+00:00","updated_at":"2026-07-05T03:40:48.326286+00:00"}