{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5IXZV5BO6UBPGDBZBR6UOOIR43","short_pith_number":"pith:5IXZV5BO","schema_version":"1.0","canonical_sha256":"ea2f9af42ef502f30c390c7d473911e6c7338e4489dbbb6602d5766ce063df3a","source":{"kind":"arxiv","id":"2203.07303","version":1},"attestation_state":"computed","paper":{"title":"All in One: Exploring Unified Video-Language Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Jinpeng Wang, Guanyu Cai, Jianping Wu, Mike Zheng Shou, Rui Yan, Xiaohu Qie, Xudong Lin, Ying Shan, Yixiao Ge, Yuying Ge","submitted_at":"2022-03-14T17:06:30Z","abstract_excerpt":"Mainstream Video-Language Pre-training models \\cite{actbert,clipbert,violet} consist of three parts, a video encoder, a text encoder, and a video-text fusion Transformer. They pursue better performance via utilizing heavier unimodal encoders or multimodal fusion Transformers, resulting in increased parameters with lower efficiency in downstream tasks. In this work, we for the first time introduce an end-to-end video-language model, namely \\textit{all-in-one Transformer}, that embeds raw video and textual signals into joint representations using a unified backbone architecture. We argue that th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.07303","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-14T17:06:30Z","cross_cats_sorted":[],"title_canon_sha256":"c149a67f551a5ededac835b5cacb4db066f608637941fb05694d203428f63cbc","abstract_canon_sha256":"cb59ac69b8182311ed32a9e8b08808e6eeb45967a9d9537528d97a3d5083add9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:02.549681Z","signature_b64":"yEN8Svt05YINdetBCXgT/Or/DpiUrLSw208dA/kFzTykgYsMrHUPi6gVisNRDiwY46jz8EQEVG4HFmPERoB7DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea2f9af42ef502f30c390c7d473911e6c7338e4489dbbb6602d5766ce063df3a","last_reissued_at":"2026-07-05T04:05:02.548514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:02.548514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"All in One: Exploring Unified Video-Language Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Jinpeng Wang, Guanyu Cai, Jianping Wu, Mike Zheng Shou, Rui Yan, Xiaohu Qie, Xudong Lin, Ying Shan, Yixiao Ge, Yuying Ge","submitted_at":"2022-03-14T17:06:30Z","abstract_excerpt":"Mainstream Video-Language Pre-training models \\cite{actbert,clipbert,violet} consist of three parts, a video encoder, a text encoder, and a video-text fusion Transformer. They pursue better performance via utilizing heavier unimodal encoders or multimodal fusion Transformers, resulting in increased parameters with lower efficiency in downstream tasks. In this work, we for the first time introduce an end-to-end video-language model, namely \\textit{all-in-one Transformer}, that embeds raw video and textual signals into joint representations using a unified backbone architecture. We argue that th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.07303","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.07303/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.07303","created_at":"2026-07-05T04:05:02.549064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.07303v1","created_at":"2026-07-05T04:05:02.549064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.07303","created_at":"2026-07-05T04:05:02.549064+00:00"},{"alias_kind":"pith_short_12","alias_value":"5IXZV5BO6UBP","created_at":"2026-07-05T04:05:02.549064+00:00"},{"alias_kind":"pith_short_16","alias_value":"5IXZV5BO6UBPGDBZ","created_at":"2026-07-05T04:05:02.549064+00:00"},{"alias_kind":"pith_short_8","alias_value":"5IXZV5BO","created_at":"2026-07-05T04:05:02.549064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2212.03191","citing_title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2205.14100","citing_title":"GIT: A Generative Image-to-text Transformer for Vision and Language","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06942","citing_title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2305.06355","citing_title":"VideoChat: Chat-Centric Video Understanding","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43","json":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43.json","graph_json":"https://pith.science/api/pith-number/5IXZV5BO6UBPGDBZBR6UOOIR43/graph.json","events_json":"https://pith.science/api/pith-number/5IXZV5BO6UBPGDBZBR6UOOIR43/events.json","paper":"https://pith.science/paper/5IXZV5BO"},"agent_actions":{"view_html":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43","download_json":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43.json","view_paper":"https://pith.science/paper/5IXZV5BO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.07303&json=true","fetch_graph":"https://pith.science/api/pith-number/5IXZV5BO6UBPGDBZBR6UOOIR43/graph.json","fetch_events":"https://pith.science/api/pith-number/5IXZV5BO6UBPGDBZBR6UOOIR43/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43/action/storage_attestation","attest_author":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43/action/author_attestation","sign_citation":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43/action/citation_signature","submit_replication":"https://pith.science/pith/5IXZV5BO6UBPGDBZBR6UOOIR43/action/replication_record"}},"created_at":"2026-07-05T04:05:02.549064+00:00","updated_at":"2026-07-05T04:05:02.549064+00:00"}