{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:W3LXCXDD2IGUGZ3J6I25HIT32O","short_pith_number":"pith:W3LXCXDD","schema_version":"1.0","canonical_sha256":"b6d7715c63d20d436769f235d3a27bd397e204d972a000a33f77622f47eb5753","source":{"kind":"arxiv","id":"2102.06183","version":1},"attestation_state":"computed","paper":{"title":"Less is More: ClipBERT for Video-and-Language Learning via Sparse Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jie Lei, Jingjing Liu, Linjie Li, Luowei Zhou, Mohit Bansal, Tamara L. Berg, Zhe Gan","submitted_at":"2021-02-11T18:50:16Z","abstract_excerpt":"The canonical approach to video-and-language learning (e.g., video question answering) dictates a neural model to learn from offline-extracted dense video features from vision models and text features from language models. These feature extractors are trained independently and usually on tasks different from the target domains, rendering these fixed features sub-optimal for downstream tasks. Moreover, due to the high computational overload of dense video features, it is often difficult (or infeasible) to plug feature extractors directly into existing approaches for easy finetuning. To provide "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.06183","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-02-11T18:50:16Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a3a08cb979b627c1ea68451230784c3ef81762ddc12d3279add109ae682b6ac3","abstract_canon_sha256":"14df10f8af395e70183190b63ca4931e16716d02ce0559d1433f2f7dde5f2f4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:14:39.254539Z","signature_b64":"8qDD3AeFS3tZmpxDPLfo6oo4ruAY2w+A8jITgzXG3S73SbzWneFUbPmRv8TjSO7iuIPcFsbNhDETMnz18Qj7AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6d7715c63d20d436769f235d3a27bd397e204d972a000a33f77622f47eb5753","last_reissued_at":"2026-07-05T02:14:39.254040Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:14:39.254040Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Less is More: ClipBERT for Video-and-Language Learning via Sparse Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jie Lei, Jingjing Liu, Linjie Li, Luowei Zhou, Mohit Bansal, Tamara L. Berg, Zhe Gan","submitted_at":"2021-02-11T18:50:16Z","abstract_excerpt":"The canonical approach to video-and-language learning (e.g., video question answering) dictates a neural model to learn from offline-extracted dense video features from vision models and text features from language models. These feature extractors are trained independently and usually on tasks different from the target domains, rendering these fixed features sub-optimal for downstream tasks. Moreover, due to the high computational overload of dense video features, it is often difficult (or infeasible) to plug feature extractors directly into existing approaches for easy finetuning. To provide "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.06183","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.06183/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.06183","created_at":"2026-07-05T02:14:39.254112+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.06183v1","created_at":"2026-07-05T02:14:39.254112+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.06183","created_at":"2026-07-05T02:14:39.254112+00:00"},{"alias_kind":"pith_short_12","alias_value":"W3LXCXDD2IGU","created_at":"2026-07-05T02:14:39.254112+00:00"},{"alias_kind":"pith_short_16","alias_value":"W3LXCXDD2IGUGZ3J","created_at":"2026-07-05T02:14:39.254112+00:00"},{"alias_kind":"pith_short_8","alias_value":"W3LXCXDD","created_at":"2026-07-05T02:14:39.254112+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20838","citing_title":"USV: Towards Understanding the User-generated Short-form Videos","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21420","citing_title":"ReGATE: Learning Faster and Better with Fewer Tokens in MLLMs","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O","json":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O.json","graph_json":"https://pith.science/api/pith-number/W3LXCXDD2IGUGZ3J6I25HIT32O/graph.json","events_json":"https://pith.science/api/pith-number/W3LXCXDD2IGUGZ3J6I25HIT32O/events.json","paper":"https://pith.science/paper/W3LXCXDD"},"agent_actions":{"view_html":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O","download_json":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O.json","view_paper":"https://pith.science/paper/W3LXCXDD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.06183&json=true","fetch_graph":"https://pith.science/api/pith-number/W3LXCXDD2IGUGZ3J6I25HIT32O/graph.json","fetch_events":"https://pith.science/api/pith-number/W3LXCXDD2IGUGZ3J6I25HIT32O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O/action/storage_attestation","attest_author":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O/action/author_attestation","sign_citation":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O/action/citation_signature","submit_replication":"https://pith.science/pith/W3LXCXDD2IGUGZ3J6I25HIT32O/action/replication_record"}},"created_at":"2026-07-05T02:14:39.254112+00:00","updated_at":"2026-07-05T02:14:39.254112+00:00"}