{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:ZVCMJUCPOROS3XXC2ZHQK2MAMI","short_pith_number":"pith:ZVCMJUCP","schema_version":"1.0","canonical_sha256":"cd44c4d04f745d2ddee2d64f056980623134476656b2c1b59711aae56d299c92","source":{"kind":"arxiv","id":"1811.08383","version":3},"attestation_state":"computed","paper":{"title":"TSM: Temporal Shift Module for Efficient Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuang Gan, Ji Lin, Song Han","submitted_at":"2018-11-20T17:39:26Z","abstract_excerpt":"The explosive growth in video streaming gives rise to challenges on performing video understanding at high accuracy and low computation cost. Conventional 2D CNNs are computationally cheap but cannot capture temporal relationships; 3D CNN based methods can achieve good performance but are computationally intensive, making it expensive to deploy. In this paper, we propose a generic and effective Temporal Shift Module (TSM) that enjoys both high efficiency and high performance. Specifically, it can achieve the performance of 3D CNN but maintain 2D CNN's complexity. TSM shifts part of the channel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1811.08383","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2018-11-20T17:39:26Z","cross_cats_sorted":[],"title_canon_sha256":"8428c8e3a131de42b9928536621eb2e9c50aebe97124909ce3094ba1f9dd0b26","abstract_canon_sha256":"cb54cc862b0c4ad7f438eb2dde074ac86431f6d7f370a5a2d971af9369e8699e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:59:05.477799Z","signature_b64":"0wNxl6VY3h0b7v+bpKjyKN3GgTITMMqYKS6hA4HA0eRwB+y9b2O7SNiq63OkhdsZKgzkKAYb8e5wuAoPbG9dAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd44c4d04f745d2ddee2d64f056980623134476656b2c1b59711aae56d299c92","last_reissued_at":"2026-07-04T23:59:05.477261Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:59:05.477261Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TSM: Temporal Shift Module for Efficient Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuang Gan, Ji Lin, Song Han","submitted_at":"2018-11-20T17:39:26Z","abstract_excerpt":"The explosive growth in video streaming gives rise to challenges on performing video understanding at high accuracy and low computation cost. Conventional 2D CNNs are computationally cheap but cannot capture temporal relationships; 3D CNN based methods can achieve good performance but are computationally intensive, making it expensive to deploy. In this paper, we propose a generic and effective Temporal Shift Module (TSM) that enjoys both high efficiency and high performance. Specifically, it can achieve the performance of 3D CNN but maintain 2D CNN's complexity. TSM shifts part of the channel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1811.08383","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1811.08383/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1811.08383","created_at":"2026-07-04T23:59:05.477343+00:00"},{"alias_kind":"arxiv_version","alias_value":"1811.08383v3","created_at":"2026-07-04T23:59:05.477343+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1811.08383","created_at":"2026-07-04T23:59:05.477343+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZVCMJUCPOROS","created_at":"2026-07-04T23:59:05.477343+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZVCMJUCPOROS3XXC","created_at":"2026-07-04T23:59:05.477343+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZVCMJUCP","created_at":"2026-07-04T23:59:05.477343+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"1906.11415","citing_title":"Few-Shot Video Classification via Temporal Alignment","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17133","citing_title":"CAM-VFD: Cross-Attention Multimodal Video Forgery Detection","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20311","citing_title":"Seeing Further and Wider: Joint Spatio-Temporal Enlargement for Micro-Video Popularity Prediction","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16987","citing_title":"DVAR: Adversarial Multi-Agent Debate for Video Authenticity Detection","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI","json":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI.json","graph_json":"https://pith.science/api/pith-number/ZVCMJUCPOROS3XXC2ZHQK2MAMI/graph.json","events_json":"https://pith.science/api/pith-number/ZVCMJUCPOROS3XXC2ZHQK2MAMI/events.json","paper":"https://pith.science/paper/ZVCMJUCP"},"agent_actions":{"view_html":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI","download_json":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI.json","view_paper":"https://pith.science/paper/ZVCMJUCP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1811.08383&json=true","fetch_graph":"https://pith.science/api/pith-number/ZVCMJUCPOROS3XXC2ZHQK2MAMI/graph.json","fetch_events":"https://pith.science/api/pith-number/ZVCMJUCPOROS3XXC2ZHQK2MAMI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI/action/storage_attestation","attest_author":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI/action/author_attestation","sign_citation":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI/action/citation_signature","submit_replication":"https://pith.science/pith/ZVCMJUCPOROS3XXC2ZHQK2MAMI/action/replication_record"}},"created_at":"2026-07-04T23:59:05.477343+00:00","updated_at":"2026-07-04T23:59:05.477343+00:00"}