{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:6YQPEPUD3IJXGEUS7ULNOEOV76","short_pith_number":"pith:6YQPEPUD","schema_version":"1.0","canonical_sha256":"f620f23e83da13731292fd16d711d5ff9b5eb66e54856acca3bb3af0e9395a1b","source":{"kind":"arxiv","id":"2206.03428","version":1},"attestation_state":"computed","paper":{"title":"Revealing Single Frame Bias for Video-and-Language Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jie Lei, Mohit Bansal, Tamara L. Berg","submitted_at":"2022-06-07T16:28:30Z","abstract_excerpt":"Training an effective video-and-language model intuitively requires multiple frames as model inputs. However, it is unclear whether using multiple frames is beneficial to downstream tasks, and if yes, whether the performance gain is worth the drastically-increased computation and memory costs resulting from using more frames. In this work, we explore single-frame models for video-and-language learning. On a diverse set of video-and-language tasks (including text-to-video retrieval and video question answering), we show the surprising result that, with large-scale pre-training and a proper fram"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.03428","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-06-07T16:28:30Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"9cc29a3a217807f83935265f42f2aaa38cf1d52cca0312b4de869a7f47ab4310","abstract_canon_sha256":"4adb31f8397507d704ad2215b023a6454998cd8e6a15312deee4bf1795da18b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:29:42.431349Z","signature_b64":"D2moqflow6KY2McVDei4fqr+/ivRwDkL/Pvb7pJ/9o4wibrRs3Kgp7YxKn0faik29+waqE+pACyGFoFFBMVYDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f620f23e83da13731292fd16d711d5ff9b5eb66e54856acca3bb3af0e9395a1b","last_reissued_at":"2026-07-05T04:29:42.430923Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:29:42.430923Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revealing Single Frame Bias for Video-and-Language Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jie Lei, Mohit Bansal, Tamara L. Berg","submitted_at":"2022-06-07T16:28:30Z","abstract_excerpt":"Training an effective video-and-language model intuitively requires multiple frames as model inputs. However, it is unclear whether using multiple frames is beneficial to downstream tasks, and if yes, whether the performance gain is worth the drastically-increased computation and memory costs resulting from using more frames. In this work, we explore single-frame models for video-and-language learning. On a diverse set of video-and-language tasks (including text-to-video retrieval and video question answering), we show the surprising result that, with large-scale pre-training and a proper fram"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.03428","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.03428/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.03428","created_at":"2026-07-05T04:29:42.430989+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.03428v1","created_at":"2026-07-05T04:29:42.430989+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.03428","created_at":"2026-07-05T04:29:42.430989+00:00"},{"alias_kind":"pith_short_12","alias_value":"6YQPEPUD3IJX","created_at":"2026-07-05T04:29:42.430989+00:00"},{"alias_kind":"pith_short_16","alias_value":"6YQPEPUD3IJXGEUS","created_at":"2026-07-05T04:29:42.430989+00:00"},{"alias_kind":"pith_short_8","alias_value":"6YQPEPUD","created_at":"2026-07-05T04:29:42.430989+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.23617","citing_title":"One Trajectory, One Token: Grounded Video Tokenization via Panoptic Sub-object Trajectory","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2310.01852","citing_title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","ref_index":182,"is_internal_anchor":false},{"citing_arxiv_id":"2403.00476","citing_title":"TempCompass: Do Video LLMs Really Understand Videos?","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13511","citing_title":"Adapting MLLMs for Nuanced Video Retrieval","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06942","citing_title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02713","citing_title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","ref_index":151,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76","json":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76.json","graph_json":"https://pith.science/api/pith-number/6YQPEPUD3IJXGEUS7ULNOEOV76/graph.json","events_json":"https://pith.science/api/pith-number/6YQPEPUD3IJXGEUS7ULNOEOV76/events.json","paper":"https://pith.science/paper/6YQPEPUD"},"agent_actions":{"view_html":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76","download_json":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76.json","view_paper":"https://pith.science/paper/6YQPEPUD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.03428&json=true","fetch_graph":"https://pith.science/api/pith-number/6YQPEPUD3IJXGEUS7ULNOEOV76/graph.json","fetch_events":"https://pith.science/api/pith-number/6YQPEPUD3IJXGEUS7ULNOEOV76/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76/action/storage_attestation","attest_author":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76/action/author_attestation","sign_citation":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76/action/citation_signature","submit_replication":"https://pith.science/pith/6YQPEPUD3IJXGEUS7ULNOEOV76/action/replication_record"}},"created_at":"2026-07-05T04:29:42.430989+00:00","updated_at":"2026-07-05T04:29:42.430989+00:00"}