{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:WBMFJVWSPMRTTVDZ2PB67OZGTF","short_pith_number":"pith:WBMFJVWS","schema_version":"1.0","canonical_sha256":"b05854d6d27b2339d479d3c3efbb269943f02eac7c200178178d1bcb2f0dd823","source":{"kind":"arxiv","id":"1906.03327","version":2},"attestation_state":"computed","paper":{"title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Antoine Miech, Dimitri Zhukov, Ivan Laptev, Jean-Baptiste Alayrac, Josef Sivic, Makarand Tapaswi","submitted_at":"2019-06-07T20:48:19Z","abstract_excerpt":"Learning text-video embeddings usually requires a dataset of video clips with manually provided captions. However, such datasets are expensive and time consuming to create and therefore difficult to obtain on a large scale. In this work, we propose instead to learn such embeddings from video data with readily available natural language annotations in the form of automatically transcribed narrations. The contributions of this work are three-fold. First, we introduce HowTo100M: a large-scale dataset of 136 million video clips sourced from 1.22M narrated instructional web videos depicting humans "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.03327","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-06-07T20:48:19Z","cross_cats_sorted":[],"title_canon_sha256":"c8a6f6014bcc5fc1aa871a165e494c6a125d577e69ca506d1b50f1c43a7354a3","abstract_canon_sha256":"7a926ee95a403fbc84d4d71a94d72539d1c22c6c7898c500f133dbbaac416ac6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:50:45.750604Z","signature_b64":"qWl/ef8OgZTsRYo0rtMb3+PQj4IrNOjM/3xD1xYECYuSgo8LTNcdnHa2KVsP8ty/3jnRf6Au8jZqPHXGxIRKAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b05854d6d27b2339d479d3c3efbb269943f02eac7c200178178d1bcb2f0dd823","last_reissued_at":"2026-07-04T23:50:45.750167Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:50:45.750167Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Antoine Miech, Dimitri Zhukov, Ivan Laptev, Jean-Baptiste Alayrac, Josef Sivic, Makarand Tapaswi","submitted_at":"2019-06-07T20:48:19Z","abstract_excerpt":"Learning text-video embeddings usually requires a dataset of video clips with manually provided captions. However, such datasets are expensive and time consuming to create and therefore difficult to obtain on a large scale. In this work, we propose instead to learn such embeddings from video data with readily available natural language annotations in the form of automatically transcribed narrations. The contributions of this work are three-fold. First, we introduce HowTo100M: a large-scale dataset of 136 million video clips sourced from 1.22M narrated instructional web videos depicting humans "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.03327","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.03327/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.03327","created_at":"2026-07-04T23:50:45.750221+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.03327v2","created_at":"2026-07-04T23:50:45.750221+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.03327","created_at":"2026-07-04T23:50:45.750221+00:00"},{"alias_kind":"pith_short_12","alias_value":"WBMFJVWSPMRT","created_at":"2026-07-04T23:50:45.750221+00:00"},{"alias_kind":"pith_short_16","alias_value":"WBMFJVWSPMRTTVDZ","created_at":"2026-07-04T23:50:45.750221+00:00"},{"alias_kind":"pith_short_8","alias_value":"WBMFJVWS","created_at":"2026-07-04T23:50:45.750221+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20521","citing_title":"HumanScale: Egocentric Human Video Can Outperform Real-Robot Data for Embodied Pretraining","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20561","citing_title":"TimeProVe: Propose, then Verify for Efficient Long Video Temporal Reasoning in Activities of Daily Living","ref_index":182,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30673","citing_title":"TeachObs: A Human-Validated Benchmark for Multimodal Teaching Observation and Model Evaluation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23045","citing_title":"The TIME Machine: On The Power of Motion for Efficient Perception","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":178,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06747","citing_title":"HumanNet: Scaling Human-centric Video Learning to One Million Hours","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF","json":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF.json","graph_json":"https://pith.science/api/pith-number/WBMFJVWSPMRTTVDZ2PB67OZGTF/graph.json","events_json":"https://pith.science/api/pith-number/WBMFJVWSPMRTTVDZ2PB67OZGTF/events.json","paper":"https://pith.science/paper/WBMFJVWS"},"agent_actions":{"view_html":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF","download_json":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF.json","view_paper":"https://pith.science/paper/WBMFJVWS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.03327&json=true","fetch_graph":"https://pith.science/api/pith-number/WBMFJVWSPMRTTVDZ2PB67OZGTF/graph.json","fetch_events":"https://pith.science/api/pith-number/WBMFJVWSPMRTTVDZ2PB67OZGTF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF/action/storage_attestation","attest_author":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF/action/author_attestation","sign_citation":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF/action/citation_signature","submit_replication":"https://pith.science/pith/WBMFJVWSPMRTTVDZ2PB67OZGTF/action/replication_record"}},"created_at":"2026-07-04T23:50:45.750221+00:00","updated_at":"2026-07-04T23:50:45.750221+00:00"}