{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:TWFTEXCCYA4FSVFS6OVI4JSPJI","short_pith_number":"pith:TWFTEXCC","schema_version":"1.0","canonical_sha256":"9d8b325c42c0385954b2f3aa8e264f4a155aaab973f157051f2878dbb1067902","source":{"kind":"arxiv","id":"2205.13313","version":1},"attestation_state":"computed","paper":{"title":"Cross-Architecture Self-supervised Video Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bing Han, Limin Wang, Sheng Guo, Weilin Huang, Xiaobo Guo, Yujie Zhong, Zihua Xiong","submitted_at":"2022-05-26T12:41:19Z","abstract_excerpt":"In this paper, we present a new cross-architecture contrastive learning (CACL) framework for self-supervised video representation learning. CACL consists of a 3D CNN and a video transformer which are used in parallel to generate diverse positive pairs for contrastive learning. This allows the model to learn strong representations from such diverse yet meaningful pairs. Furthermore, we introduce a temporal self-supervised learning module able to predict an Edit distance explicitly between two video sequences in the temporal order. This enables the model to learn a rich temporal representation t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.13313","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-05-26T12:41:19Z","cross_cats_sorted":[],"title_canon_sha256":"ddbd74f203f4be9656faeae8a72e3183f92c1eb67f14d8fddec5dd0730639b99","abstract_canon_sha256":"d15f2820c2e4aab2c625a4e008e6a034e6c585223898fbd9eaa264e13f774244"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:26:41.949229Z","signature_b64":"HdS/oC/7KaTUfLBpLU9j8EAzWNXZOx9oZO8hC5nKcpCQ32itfkrcqLYORD3wMiS63FnCFpq6uPeY+jVmnJjBDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d8b325c42c0385954b2f3aa8e264f4a155aaab973f157051f2878dbb1067902","last_reissued_at":"2026-07-05T04:26:41.948790Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:26:41.948790Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cross-Architecture Self-supervised Video Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bing Han, Limin Wang, Sheng Guo, Weilin Huang, Xiaobo Guo, Yujie Zhong, Zihua Xiong","submitted_at":"2022-05-26T12:41:19Z","abstract_excerpt":"In this paper, we present a new cross-architecture contrastive learning (CACL) framework for self-supervised video representation learning. CACL consists of a 3D CNN and a video transformer which are used in parallel to generate diverse positive pairs for contrastive learning. This allows the model to learn strong representations from such diverse yet meaningful pairs. Furthermore, we introduce a temporal self-supervised learning module able to predict an Edit distance explicitly between two video sequences in the temporal order. This enables the model to learn a rich temporal representation t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.13313","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.13313/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.13313","created_at":"2026-07-05T04:26:41.948857+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.13313v1","created_at":"2026-07-05T04:26:41.948857+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.13313","created_at":"2026-07-05T04:26:41.948857+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWFTEXCCYA4F","created_at":"2026-07-05T04:26:41.948857+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWFTEXCCYA4FSVFS","created_at":"2026-07-05T04:26:41.948857+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWFTEXCC","created_at":"2026-07-05T04:26:41.948857+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI","json":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI.json","graph_json":"https://pith.science/api/pith-number/TWFTEXCCYA4FSVFS6OVI4JSPJI/graph.json","events_json":"https://pith.science/api/pith-number/TWFTEXCCYA4FSVFS6OVI4JSPJI/events.json","paper":"https://pith.science/paper/TWFTEXCC"},"agent_actions":{"view_html":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI","download_json":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI.json","view_paper":"https://pith.science/paper/TWFTEXCC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.13313&json=true","fetch_graph":"https://pith.science/api/pith-number/TWFTEXCCYA4FSVFS6OVI4JSPJI/graph.json","fetch_events":"https://pith.science/api/pith-number/TWFTEXCCYA4FSVFS6OVI4JSPJI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI/action/storage_attestation","attest_author":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI/action/author_attestation","sign_citation":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI/action/citation_signature","submit_replication":"https://pith.science/pith/TWFTEXCCYA4FSVFS6OVI4JSPJI/action/replication_record"}},"created_at":"2026-07-05T04:26:41.948857+00:00","updated_at":"2026-07-05T04:26:41.948857+00:00"}